Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d9f6167a4d | ||
|
|
f5088c52fc | ||
|
|
f52e23f1bc | ||
|
|
c9775c2b95 | ||
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 | ||
|
|
11f962fbcc |
+7
-1
@@ -57,7 +57,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -130,6 +130,12 @@ func splitSize(upper string) (base string, size int) {
|
|||||||
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
||||||
// EVEX-only mnemonic) and VEX otherwise.
|
// EVEX-only mnemonic) and VEX otherwise.
|
||||||
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
if gs, ok := gatherTable[upper]; ok {
|
||||||
|
return e.encodeGather(upper, gs, ops, sfx)
|
||||||
|
}
|
||||||
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
|
|||||||
+378
-8
@@ -234,6 +234,180 @@ var evexTable = map[string]evexSpec{
|
|||||||
// EVEX W1 qword shifts.
|
// EVEX W1 qword shifts.
|
||||||
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
|
||||||
|
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
|
||||||
|
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
|
||||||
|
// forms, these take the 66 prefix; W selects double/single.
|
||||||
|
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
// EVEX.66.0F38 — scale by a power of two (NDS form).
|
||||||
|
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
|
||||||
|
// rm=src, imm8).
|
||||||
|
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
|
||||||
|
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
|
||||||
|
// prefix; W selects double/single.
|
||||||
|
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
|
||||||
|
// reg field carries the opmask destination. The packed forms carry an
|
||||||
|
// explicit length in the mnemonic (X/Y/Z).
|
||||||
|
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||||
|
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||||
|
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||||
|
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||||
|
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||||
|
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||||
|
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
|
||||||
|
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
|
||||||
|
|
||||||
|
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
|
||||||
|
// source sets the length); the rest follow the destination.
|
||||||
|
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
// EVEX.66.0F38 — half-precision convert (half-width source).
|
||||||
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
||||||
|
// rm=dst, imm8 — the extract layout).
|
||||||
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX — unsigned and truncating conversions. The PD sources are the
|
||||||
|
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
|
||||||
|
// the length); the PS/UQQ destinations are wide and follow the
|
||||||
|
// destination.
|
||||||
|
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
|
||||||
|
// source; disp8×N follows its size).
|
||||||
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
|
||||||
|
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
|
||||||
|
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
|
||||||
|
// into a vector (rm = K source, reg = vector destination), *2M does the
|
||||||
|
// reverse (reg = K destination, rm = vector source, the length follows
|
||||||
|
// the vector).
|
||||||
|
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX — scalar conversions between vector and general-purpose
|
||||||
|
// registers. Vector to GPR (two operands: vec/mem source, GPR
|
||||||
|
// destination, vvvv unused): the signed and truncated pair, and the
|
||||||
|
// unsigned forms (EVEX only).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||||
// the xmm/ymm/zmm destination lengths).
|
// the xmm/ymm/zmm destination lengths).
|
||||||
@@ -466,6 +640,18 @@ var evexRound = map[string]bool{
|
|||||||
"VMINSD": true, "VMAXSD": true,
|
"VMINSD": true, "VMAXSD": true,
|
||||||
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
||||||
"VMINSS": true, "VMAXSS": true,
|
"VMINSS": true, "VMAXSS": true,
|
||||||
|
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||||||
|
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||||||
|
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||||||
|
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||||||
|
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||||||
|
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||||||
|
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
|
||||||
|
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
|
||||||
|
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
|
||||||
|
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
|
||||||
|
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
|
||||||
|
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||||
@@ -475,6 +661,19 @@ var evexBcstN = map[string]int{
|
|||||||
"VMINPD": 8, "VMAXPD": 8,
|
"VMINPD": 8, "VMAXPD": 8,
|
||||||
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
||||||
"VMINPS": 4, "VMAXPS": 4,
|
"VMINPS": 4, "VMAXPS": 4,
|
||||||
|
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
|
||||||
|
"VGETEXPPD": 8, "VGETEXPPS": 4,
|
||||||
|
"VSCALEFPD": 8, "VSCALEFPS": 4,
|
||||||
|
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
|
||||||
|
"VGETMANTPD": 8, "VGETMANTPS": 4,
|
||||||
|
"VREDUCEPD": 8, "VREDUCEPS": 4,
|
||||||
|
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
|
||||||
|
"VRANGEPD": 8, "VRANGEPS": 4,
|
||||||
|
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||||||
|
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||||||
|
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
|
||||||
|
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
|
||||||
|
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||||
@@ -504,6 +703,18 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
|
|||||||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||||||
// broadcast.
|
// broadcast.
|
||||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
// The mask/vector conversions take the K register as a genuine operand
|
||||||
|
// (source or destination), not as a mask, and accept no suffixes.
|
||||||
|
if evexKOperand[mnemUpper] {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
spec, ok := evexTable[mnemUpper]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("unsupported instruction %q", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||||
|
}
|
||||||
spec, inTable := evexTable[mnemUpper]
|
spec, inTable := evexTable[mnemUpper]
|
||||||
if inTable {
|
if inTable {
|
||||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||||
@@ -547,6 +758,10 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||||
return kdst(e.encodeEvexNDS3Imm)
|
return kdst(e.encodeEvexNDS3Imm)
|
||||||
}
|
}
|
||||||
|
case vexImmRM:
|
||||||
|
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||||
|
return kdst(e.encodeEvexImmRM)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -622,21 +837,35 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf
|
|||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||||
// no vvvv), e.g. VCVTQQ2PD.
|
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||||||
|
// *2M mask conversions) or a general-purpose register (the scalar
|
||||||
|
// vector-to-GPR conversions); in both cases the vector length comes from
|
||||||
|
// the source.
|
||||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
src, dst := ops[0], ops[1]
|
src, dst := ops[0], ops[1]
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok {
|
||||||
return fmt.Errorf("EVEX destination must be a vector register")
|
return fmt.Errorf("EVEX destination must be a register")
|
||||||
}
|
}
|
||||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
ll := dstReg.vecLenBit()
|
||||||
|
if !dstReg.isVec() {
|
||||||
|
// Mask or GPR destination: the length follows the vector source
|
||||||
|
// (128 for a memory source).
|
||||||
|
ll = 0
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
|
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
|
||||||
|
// opmask register (VFPCLASS*), in which case the vector length comes from
|
||||||
|
// the source.
|
||||||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
@@ -647,11 +876,15 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
|||||||
return fmt.Errorf("shuffle control must be an immediate")
|
return fmt.Errorf("shuffle control must be an immediate")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||||
return fmt.Errorf("shuffle destination must be a vector register")
|
return fmt.Errorf("shuffle destination must be a vector or mask register")
|
||||||
}
|
}
|
||||||
ll := dstReg.vecLenBit()
|
ll := dstReg.vecLenBit()
|
||||||
if r, ok := src.(Reg); ok && r.isVec() {
|
if dstReg.mask {
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
}
|
}
|
||||||
immByte, err := imm8(int64(immVal))
|
immByte, err := imm8(int64(immVal))
|
||||||
@@ -1034,6 +1267,143 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte,
|
|||||||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// gatherSpec describes a gather/scatter family member: all live in
|
||||||
|
// 66.0F38; the opcode and W select the index and data element widths, and n
|
||||||
|
// is the data element size (the EVEX disp8×N multiplier).
|
||||||
|
type gatherSpec struct {
|
||||||
|
opcode byte
|
||||||
|
w int
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
|
||||||
|
var gatherTable = map[string]gatherSpec{
|
||||||
|
"VGATHERDPS": {0x92, 0, 4},
|
||||||
|
"VGATHERDPD": {0x92, 1, 8},
|
||||||
|
"VGATHERQPS": {0x93, 0, 4},
|
||||||
|
"VGATHERQPD": {0x93, 1, 8},
|
||||||
|
"VPGATHERDD": {0x90, 0, 4},
|
||||||
|
"VPGATHERDQ": {0x90, 1, 8},
|
||||||
|
"VPGATHERQD": {0x91, 0, 4},
|
||||||
|
"VPGATHERQQ": {0x91, 1, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
var scatterTable = map[string]gatherSpec{
|
||||||
|
"VSCATTERDPS": {0xA2, 0, 4},
|
||||||
|
"VSCATTERDPD": {0xA2, 1, 8},
|
||||||
|
"VSCATTERQPS": {0xA3, 0, 4},
|
||||||
|
"VSCATTERQPD": {0xA3, 1, 8},
|
||||||
|
"VPSCATTERDD": {0xA0, 0, 4},
|
||||||
|
"VPSCATTERDQ": {0xA0, 1, 8},
|
||||||
|
"VPSCATTERQD": {0xA1, 0, 4},
|
||||||
|
"VPSCATTERQQ": {0xA1, 1, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
// isGather reports whether the mnemonic is a gather instruction.
|
||||||
|
func isGather(upper string) bool {
|
||||||
|
_, ok := gatherTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// isScatter reports whether the mnemonic is a scatter instruction.
|
||||||
|
func isScatter(upper string) bool {
|
||||||
|
_, ok := scatterTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||||
|
// register) and returns it with the vector length the index selects — the
|
||||||
|
// EVEX L'L field follows the index register, not the data register.
|
||||||
|
func vsibLen(op Operand, what string) (Mem, int, error) {
|
||||||
|
m, ok := op.(Mem)
|
||||||
|
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||||
|
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
|
||||||
|
}
|
||||||
|
return m, m.Index.vecLenBit(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeGather encodes a gather. The VEX spelling carries the mask in a
|
||||||
|
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
|
||||||
|
// L follows the data register); the EVEX spelling carries it in aaa (OP
|
||||||
|
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
|
||||||
|
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||||
|
rest, mask, err := splitMask(ops)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if mask != 0 || sfx.any() {
|
||||||
|
// EVEX form: OP vsib, K, dst.
|
||||||
|
if len(rest) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
vsib, ll, err := vsibLen(rest[0], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dst, ok := rest[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
|
}
|
||||||
|
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||||
|
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||||
|
}
|
||||||
|
// VEX form: OP mask, vsib, dst.
|
||||||
|
if len(rest) != 3 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
|
||||||
|
}
|
||||||
|
maskReg, ok := rest[0].(Reg)
|
||||||
|
if !ok || !maskReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: mask must be a vector register", upper)
|
||||||
|
}
|
||||||
|
vsib, _, err := vsibLen(rest[1], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dst, ok := rest[2].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
|
}
|
||||||
|
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
|
||||||
|
rBit := 0
|
||||||
|
if dst.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
|
||||||
|
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
|
||||||
|
// index.
|
||||||
|
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||||
|
rest, mask, err := splitMask(ops)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if mask == 0 {
|
||||||
|
return fmt.Errorf("%s requires a K mask register", upper)
|
||||||
|
}
|
||||||
|
if len(rest) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
src, ok := rest[0].(Reg)
|
||||||
|
if !ok || !src.isVec() {
|
||||||
|
return fmt.Errorf("%s: source must be a vector register", upper)
|
||||||
|
}
|
||||||
|
vsib, ll, err := vsibLen(rest[1], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||||
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexKOperand lists the instructions whose K register is a genuine operand
|
||||||
|
// (the source or destination of a mask/vector conversion) rather than a
|
||||||
|
// mask modifier — the M2 and 2M conversions. They take no masking.
|
||||||
|
var evexKOperand = map[string]bool{
|
||||||
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||||
|
|||||||
@@ -364,6 +364,267 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||||
|
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||||
|
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||||
|
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||||
|
// assembler.
|
||||||
|
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||||
|
vsib := func(base, idx string, scale int) Operand {
|
||||||
|
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Reciprocals and rsqrt (packed RM, scalar NDS).
|
||||||
|
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
|
||||||
|
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
|
||||||
|
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
|
||||||
|
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
|
||||||
|
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
|
||||||
|
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
|
||||||
|
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
|
||||||
|
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
|
||||||
|
// Getexp (packed RM, scalar NDS).
|
||||||
|
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
|
||||||
|
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
|
||||||
|
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
|
||||||
|
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
|
||||||
|
// Scalef (NDS).
|
||||||
|
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
|
||||||
|
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
|
||||||
|
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
|
||||||
|
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
|
||||||
|
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
|
||||||
|
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
|
||||||
|
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
|
||||||
|
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
|
||||||
|
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
|
||||||
|
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
|
||||||
|
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
|
||||||
|
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
|
||||||
|
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
|
||||||
|
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
|
||||||
|
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
|
||||||
|
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
|
||||||
|
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
|
||||||
|
// Fixupimm / range (NDS + imm8).
|
||||||
|
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
|
||||||
|
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
|
||||||
|
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
|
||||||
|
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
|
||||||
|
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
|
||||||
|
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
|
||||||
|
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
|
||||||
|
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
|
||||||
|
// FP class test ($imm, src, kdst; packed forms carry the length in
|
||||||
|
// the X/Y/Z mnemonic suffix the decoder drops).
|
||||||
|
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
|
||||||
|
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
|
||||||
|
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
|
||||||
|
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
|
||||||
|
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
|
||||||
|
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
|
||||||
|
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
|
||||||
|
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
|
||||||
|
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
|
||||||
|
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
|
||||||
|
// Scatter (EVEX only: source, K mask, VSIB).
|
||||||
|
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
|
||||||
|
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
|
||||||
|
// The remaining conversions.
|
||||||
|
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||||
|
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
|
||||||
|
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
|
||||||
|
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
|
||||||
|
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
|
||||||
|
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
|
||||||
|
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||||
|
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||||
|
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||||
|
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||||
|
// for byte against the Go assembler, including memory sources and extended
|
||||||
|
// GPRs.
|
||||||
|
func TestEvexGprGroundTruth(t *testing.T) {
|
||||||
|
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||||
|
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||||
|
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||||
|
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||||
|
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||||
|
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||||
|
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||||
|
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||||
|
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||||
|
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||||
|
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||||
|
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||||
|
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||||
|
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||||
|
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||||
|
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||||
|
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||||
|
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||||
|
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||||
|
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||||
|
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||||
|
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||||
|
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||||
|
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||||
|
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||||
|
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||||
|
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||||
|
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||||
|
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||||
|
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||||
|
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||||
|
// the Go assembler.
|
||||||
|
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Unsigned and truncating conversions.
|
||||||
|
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||||
|
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||||
|
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||||
|
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||||
|
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||||
|
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||||
|
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||||
|
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||||
|
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||||
|
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||||
|
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||||
|
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||||
|
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||||
|
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||||
|
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||||
|
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||||
|
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||||
|
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||||
|
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||||
|
// The remaining sign/zero-extending moves.
|
||||||
|
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||||
|
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||||
|
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||||
|
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||||
|
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||||
|
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||||
|
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||||
|
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||||
|
// Signed narrowing stores.
|
||||||
|
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||||
|
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||||
|
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||||
|
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||||
|
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||||
|
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||||
|
// Unsigned narrowing stores.
|
||||||
|
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||||
|
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||||
|
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||||
|
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||||
|
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||||
|
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||||
|
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||||
|
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||||
|
// Mask/vector conversions (the K register is an operand, not a
|
||||||
|
// mask).
|
||||||
|
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||||
|
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||||
|
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||||
|
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||||
|
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||||
|
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||||
|
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||||
|
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestEvexErrors checks the EVEX-specific error paths.
|
// TestEvexErrors checks the EVEX-specific error paths.
|
||||||
func TestEvexErrors(t *testing.T) {
|
func TestEvexErrors(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
|
|||||||
+38
-1
@@ -130,9 +130,15 @@ var vexTable = map[string]vexSpec{
|
|||||||
// no vvvv).
|
// no vvvv).
|
||||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM},
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||||
@@ -178,6 +184,9 @@ var vexTable = map[string]vexSpec{
|
|||||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||||
|
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||||
|
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||||
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||||
|
|
||||||
// VEX.128.0F.W0 — no operands.
|
// VEX.128.0F.W0 — no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
@@ -189,9 +198,35 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||||
|
// source).
|
||||||
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||||
|
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||||
|
// spellings: the destination is always XMM and the spelling fixes the
|
||||||
|
// source length (X = 128, Y = 256).
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
|
// VEX scalar conversions between vector and general-purpose registers.
|
||||||
|
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||||
|
// unused; the length follows the source).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||||
@@ -216,6 +251,8 @@ var vexSrcLen = map[string]int{
|
|||||||
"VCVTPD2DQY": 1,
|
"VCVTPD2DQY": 1,
|
||||||
"VCVTTPD2DQX": 0,
|
"VCVTTPD2DQX": 0,
|
||||||
"VCVTTPD2DQY": 1,
|
"VCVTTPD2DQY": 1,
|
||||||
|
"VCVTPD2PSX": 0,
|
||||||
|
"VCVTPD2PSY": 1,
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||||
|
|||||||
+6
-3
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if inst.Op.String() != mnem {
|
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+59
-1
@@ -24,11 +24,12 @@ import (
|
|||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version is the release version, stamped at build time via
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||||
var version = "0.13.0"
|
var version = "0.20.0"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
@@ -46,6 +47,8 @@ func main() {
|
|||||||
os.Exit(cmdLint(os.Args[2:]))
|
os.Exit(cmdLint(os.Args[2:]))
|
||||||
case "asm":
|
case "asm":
|
||||||
os.Exit(cmdAsm(os.Args[2:]))
|
os.Exit(cmdAsm(os.Args[2:]))
|
||||||
|
case "verify":
|
||||||
|
os.Exit(cmdVerify(os.Args[2:]))
|
||||||
case "lsp":
|
case "lsp":
|
||||||
os.Exit(cmdLSP(os.Args[2:]))
|
os.Exit(cmdLSP(os.Args[2:]))
|
||||||
case "version", "--version", "-V":
|
case "version", "--version", "-V":
|
||||||
@@ -80,6 +83,7 @@ Commands:
|
|||||||
fmt canonicalise formatting (gofmt for assembly)
|
fmt canonicalise formatting (gofmt for assembly)
|
||||||
lint run static checks
|
lint run static checks
|
||||||
asm assemble .s files to machine code (amd64)
|
asm assemble .s files to machine code (amd64)
|
||||||
|
verify JIT-assemble and run dynamic checks (amd64)
|
||||||
lsp run the language server over stdio
|
lsp run the language server over stdio
|
||||||
version print the version (same as --version)
|
version print the version (same as --version)
|
||||||
|
|
||||||
@@ -466,3 +470,57 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
}
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func cmdVerify(args []string) int {
|
||||||
|
fs := newCommand("verify", "gasm verify <file.s>", `
|
||||||
|
Assemble FILE (amd64), map it into executable memory and report the available
|
||||||
|
functions. This confirms the assembled image is self-consistent (no
|
||||||
|
unresolved external symbols) and executable — the prerequisite for dynamic
|
||||||
|
testing.
|
||||||
|
|
||||||
|
With -smoke, each NOSPLIT function is called with a zeroed argument block to
|
||||||
|
confirm the JIT trampoline works end-to-end. This is safe only for functions
|
||||||
|
that tolerate nil pointers and zero lengths in their arguments.
|
||||||
|
`)
|
||||||
|
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
|
||||||
|
fs.Parse(args)
|
||||||
|
if fs.NArg() != 1 {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] <file.s>")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
path := fs.Arg(0)
|
||||||
|
if arch.FromFilename(path) != arch.AMD64 {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 is supported")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
k, err := verify.Load(path)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer k.Close()
|
||||||
|
|
||||||
|
names := k.FuncNames()
|
||||||
|
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
|
||||||
|
rc := 0
|
||||||
|
for _, name := range names {
|
||||||
|
fl, _ := k.Func(name)
|
||||||
|
flags := ""
|
||||||
|
if fl.NoSplit {
|
||||||
|
flags = " NOSPLIT"
|
||||||
|
}
|
||||||
|
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
|
||||||
|
if *smoke && fl.NoSplit {
|
||||||
|
args := make([]byte, fl.Args)
|
||||||
|
_, err := k.CallFunc(name, args)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf(" smoke: FAIL — %v\n", err)
|
||||||
|
rc = 1
|
||||||
|
} else {
|
||||||
|
fmt.Printf(" smoke: OK\n")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return rc
|
||||||
|
}
|
||||||
|
|||||||
+36
-1
@@ -237,7 +237,17 @@ operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
|
|||||||
and extracts, compares with an opmask destination, the permutes, the
|
and extracts, compares with an opmask destination, the permutes, the
|
||||||
expand/compress family, the broadcasts, the opmask-register instructions
|
expand/compress family, the broadcasts, the opmask-register instructions
|
||||||
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
||||||
moves and the remaining extending/narrowing moves. The EVEX mnemonic
|
moves and the remaining extending/narrowing moves, the floating-point
|
||||||
|
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||||
|
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||||
|
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||||
|
truncating, including the length-suffixed X/Y spellings and the
|
||||||
|
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
|
||||||
|
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
|
||||||
|
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
|
||||||
|
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||||
|
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||||
|
not the data register. The EVEX mnemonic
|
||||||
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
||||||
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
||||||
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
||||||
@@ -274,6 +284,31 @@ references and the implicit funcdata/DWARF symbols remain future work (the
|
|||||||
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
||||||
remaining EVEX forms and the other architectures.
|
remaining EVEX forms and the other architectures.
|
||||||
|
|
||||||
|
### `verify`
|
||||||
|
|
||||||
|
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
|
||||||
|
executable memory and invokes them directly, enabling differential testing,
|
||||||
|
runtime ABI checks and coverage profiling.
|
||||||
|
|
||||||
|
The execution model is pure Go (stdlib only). `Map` copies machine code into
|
||||||
|
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
|
||||||
|
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
|
||||||
|
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
|
||||||
|
block after it, switches to that stack via `enterJIT` (which saves the Go
|
||||||
|
stack pointer in a package global and jumps to the target), and recovers
|
||||||
|
control when the function RETs into `leaveJIT` (which restores the Go stack
|
||||||
|
and returns). A 64-byte pad below the return address accommodates the
|
||||||
|
ABIInternal wrapper that the Go runtime interposes on assembly functions.
|
||||||
|
|
||||||
|
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
|
||||||
|
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
|
||||||
|
by name. The image must be self-contained (no external relocations); the
|
||||||
|
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
|
||||||
|
|
||||||
|
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
||||||
|
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||||
|
arguments to confirm the trampoline round-trips.
|
||||||
|
|
||||||
## Extension points
|
## Extension points
|
||||||
|
|
||||||
- **New architecture:** add an entry to the generator in `_gen`, run
|
- **New architecture:** add an entry to the generator in `_gen`, run
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# Deferred decisions
|
||||||
|
|
||||||
|
Design decisions deliberately postponed, with enough context to pick them up
|
||||||
|
again without re-deriving the analysis. Each entry records what is deferred,
|
||||||
|
why, the options on the table, and the trigger that should reopen it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## GOOBJ external (cross-package) symbol references
|
||||||
|
|
||||||
|
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
||||||
|
symbols defined in the file being assembled; a reference to any other symbol
|
||||||
|
is rejected.
|
||||||
|
|
||||||
|
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
||||||
|
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
||||||
|
symbol in the *referenced package's* symbol-definition table. That ordering
|
||||||
|
is not derivable from the reference site — it lives in the referenced
|
||||||
|
package's gc export data (the iexport binary format, which evolves with the
|
||||||
|
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
||||||
|
import, so emitting external references means either parsing export data
|
||||||
|
ourselves or taking a dependency that does.
|
||||||
|
|
||||||
|
**What works today.** Single-package objects: every symbol the file defines
|
||||||
|
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
||||||
|
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
||||||
|
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
||||||
|
references the assembly symbols, never the reverse. Such a package builds
|
||||||
|
with its assembly object replaced by a gasm-emitted one.
|
||||||
|
|
||||||
|
**The options, when we return.**
|
||||||
|
|
||||||
|
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
||||||
|
The straightforward path: read each imported package's export file
|
||||||
|
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
||||||
|
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
||||||
|
the export files' build IDs) and positional references. Robust across
|
||||||
|
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
||||||
|
production dependency beyond the standard library, an explicit deviation
|
||||||
|
from the "production code depends only on the standard library"
|
||||||
|
principle in the README. Requires the user's explicit agreement.
|
||||||
|
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
||||||
|
Substantial effort and inherently fragile: the format is an internal
|
||||||
|
contract that changes with Go releases, so the parser needs a
|
||||||
|
version-gated fallback and regression tests against several toolchains.
|
||||||
|
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
||||||
|
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
||||||
|
toolchain command exposes a package's symbols *in definition-index
|
||||||
|
order* — `go tool nm` sorts differently — so this does not solve the
|
||||||
|
core problem on its own; it would only feed option 1 or 2.
|
||||||
|
|
||||||
|
**Trigger to reopen.** An assembly file that needs a cross-package
|
||||||
|
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
||||||
|
metadata written in assembly), or any kernel that calls into another
|
||||||
|
package directly. Until then, option 3's limitation is moot and the
|
||||||
|
single-package emitter suffices.
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||||
|
|
||||||
version := "0.13.0"
|
version := "0.20.0"
|
||||||
|
|
||||||
default:
|
default:
|
||||||
@just --list
|
@just --list
|
||||||
|
|||||||
Vendored
+28
@@ -0,0 +1,28 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func cleanAdd(a, b int64) int64
|
||||||
|
// A well-behaved function that preserves all callee-saved registers.
|
||||||
|
TEXT ·cleanAdd(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
ADDQ b+8(FP), AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func dirtyBP(a int64) int64
|
||||||
|
// Deliberately clobbers BP (an ABI violation for a NOSPLIT frame=0 function).
|
||||||
|
TEXT ·dirtyBP(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ $0x1234, BP
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func dirtyR14(a int64) int64
|
||||||
|
// Deliberately clobbers R14 (the goroutine pointer — a serious ABI violation).
|
||||||
|
TEXT ·dirtyR14(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ $0x5678, R14
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
Vendored
+67
@@ -0,0 +1,67 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func add(a, b int64) int64
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
ADDQ b+8(FP), AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func sum(data []int64) int64
|
||||||
|
// Sums all elements of the slice.
|
||||||
|
TEXT ·sum(SB), NOSPLIT, $0-32
|
||||||
|
MOVQ data_base+0(FP), SI
|
||||||
|
MOVQ data_len+8(FP), CX
|
||||||
|
XORQ AX, AX
|
||||||
|
TESTQ CX, CX
|
||||||
|
JZ sum_done
|
||||||
|
|
||||||
|
sum_loop:
|
||||||
|
ADDQ (SI), AX
|
||||||
|
ADDQ $8, SI
|
||||||
|
DECQ CX
|
||||||
|
JNZ sum_loop
|
||||||
|
|
||||||
|
sum_done:
|
||||||
|
MOVQ AX, ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func wideCopy(dst, src []byte)
|
||||||
|
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
|
||||||
|
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||||
|
MOVQ dst_base+0(FP), DI
|
||||||
|
MOVQ dst_len+8(FP), BX
|
||||||
|
MOVQ src_base+24(FP), SI
|
||||||
|
MOVQ src_len+32(FP), R8
|
||||||
|
CMPQ BX, R8
|
||||||
|
JLE wc_have_n
|
||||||
|
MOVQ R8, BX
|
||||||
|
|
||||||
|
wc_have_n:
|
||||||
|
CMPQ BX, $32
|
||||||
|
JB wc_small
|
||||||
|
|
||||||
|
VMOVDQU (SI), Y0
|
||||||
|
VMOVDQU Y0, (DI)
|
||||||
|
VMOVDQU -32(SI)(BX*1), Y0
|
||||||
|
VMOVDQU Y0, -32(DI)(BX*1)
|
||||||
|
VZEROUPPER
|
||||||
|
RET
|
||||||
|
|
||||||
|
wc_small:
|
||||||
|
TESTQ BX, BX
|
||||||
|
JZ wc_done
|
||||||
|
|
||||||
|
wc_byte:
|
||||||
|
MOVB (SI), R8B
|
||||||
|
MOVB R8B, (DI)
|
||||||
|
INCQ SI
|
||||||
|
INCQ DI
|
||||||
|
DECQ BX
|
||||||
|
JNZ wc_byte
|
||||||
|
|
||||||
|
wc_done:
|
||||||
|
RET
|
||||||
@@ -0,0 +1,130 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// abiResult records register-clobber violations detected by the ABI-checking
|
||||||
|
// trampoline. Bit 0: BP clobbered. Bit 1: R14 clobbered.
|
||||||
|
var abiResult uint64
|
||||||
|
|
||||||
|
// savedBP holds the caller's frame pointer across the ABI-checked JIT call.
|
||||||
|
// Referenced by enterJITChecked to satisfy go vet's save-before-clobber rule.
|
||||||
|
var savedBP uintptr
|
||||||
|
|
||||||
|
// leaveCheckedPtr is initialised by the linker from the GLOBL/DATA in
|
||||||
|
// abi_amd64.s: it holds the raw address of leaveJITCheckedRaw (which has
|
||||||
|
// no ABIInternal wrapper, so the JIT function RETs directly into it).
|
||||||
|
var leaveCheckedPtr uintptr
|
||||||
|
|
||||||
|
// enterJITChecked sets sentinels in BP and R14, switches to the prepared
|
||||||
|
// stack and jumps to fn.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func enterJITChecked(fn uintptr, stack uintptr)
|
||||||
|
|
||||||
|
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
|
||||||
|
// address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr),
|
||||||
|
// which points to the .abi0 code — NOT the ABIInternal wrapper that this
|
||||||
|
// declaration would generate. The declaration exists solely to satisfy
|
||||||
|
// go vet's "missing Go declaration" check.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func leaveJITCheckedRaw()
|
||||||
|
|
||||||
|
// ABIReport describes the result of an ABI-checking call.
|
||||||
|
type ABIReport struct {
|
||||||
|
BPClobbered bool // BP was modified by the function
|
||||||
|
R14Clobbered bool // R14 (goroutine pointer) was modified
|
||||||
|
RedZoneHit bool // the 128-byte red zone below SP was written
|
||||||
|
}
|
||||||
|
|
||||||
|
// OK returns true when no violations were detected.
|
||||||
|
func (r ABIReport) OK() bool {
|
||||||
|
return !r.BPClobbered && !r.R14Clobbered && !r.RedZoneHit
|
||||||
|
}
|
||||||
|
|
||||||
|
// String returns a human-readable summary.
|
||||||
|
func (r ABIReport) String() string {
|
||||||
|
if r.OK() {
|
||||||
|
return "ABI clean"
|
||||||
|
}
|
||||||
|
s := "ABI violation:"
|
||||||
|
if r.BPClobbered {
|
||||||
|
s += " BP clobbered"
|
||||||
|
}
|
||||||
|
if r.R14Clobbered {
|
||||||
|
s += " R14 clobbered"
|
||||||
|
}
|
||||||
|
if r.RedZoneHit {
|
||||||
|
s += " red-zone written"
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// redZoneSize is the System V AMD64 red zone: 128 bytes below SP that a
|
||||||
|
// leaf function may use without adjusting SP. Go does not use the red zone,
|
||||||
|
// so any write there is a bug.
|
||||||
|
const redZoneSize = 128
|
||||||
|
|
||||||
|
// redZoneFill is the byte pattern used to detect red-zone writes.
|
||||||
|
const redZoneFill = 0xA5
|
||||||
|
|
||||||
|
// CallChecked invokes the function with ABI sentinels and a red-zone
|
||||||
|
// canary, returning both the argument block (with results) and an ABIReport.
|
||||||
|
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
|
||||||
|
report := ABIReport{}
|
||||||
|
|
||||||
|
// Reset the global result.
|
||||||
|
abiResult = 0
|
||||||
|
|
||||||
|
// Prepare the stack: [red-zone canary][padding][leaveJITCheckedRaw][args...]
|
||||||
|
// The red zone sits below the initial SP, so the function would have to
|
||||||
|
// write below SP to corrupt it.
|
||||||
|
totalSize := redZoneSize + stackPad + 8 + len(args) + 64
|
||||||
|
stackMem, err := syscall.Mmap(-1, 0, totalSize,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, report, fmt.Errorf("verify: stack mmap: %w", err)
|
||||||
|
}
|
||||||
|
defer syscall.Munmap(stackMem)
|
||||||
|
|
||||||
|
// Fill the red zone with the canary pattern.
|
||||||
|
for i := 0; i < redZoneSize; i++ {
|
||||||
|
stackMem[i] = redZoneFill
|
||||||
|
}
|
||||||
|
|
||||||
|
// Return address and args after the red zone and padding.
|
||||||
|
retOff := redZoneSize + stackPad
|
||||||
|
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveCheckedPtr))
|
||||||
|
copy(stackMem[retOff+8:], args)
|
||||||
|
|
||||||
|
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||||
|
enterJITChecked(fnAddr, stackBase)
|
||||||
|
|
||||||
|
// Read the register-clobber result.
|
||||||
|
res := abiResult
|
||||||
|
report.BPClobbered = res&1 != 0
|
||||||
|
report.R14Clobbered = res&2 != 0
|
||||||
|
|
||||||
|
// Check the red zone.
|
||||||
|
for i := 0; i < redZoneSize; i++ {
|
||||||
|
if stackMem[i] != redZoneFill {
|
||||||
|
report.RedZoneHit = true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Copy out the argument area.
|
||||||
|
out := make([]byte, len(args))
|
||||||
|
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||||
|
return out, report, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// ABI-checking trampoline. Sets sentinel values in the callee-saved
|
||||||
|
// registers (BP, R14) before entering the JIT function and checks whether
|
||||||
|
// they survived on return.
|
||||||
|
//
|
||||||
|
// The return trampoline (leaveJITCheckedRaw) is a raw TEXT symbol with no
|
||||||
|
// Go function declaration, so the toolchain does NOT interpose an
|
||||||
|
// ABIInternal wrapper — the JIT function RETs directly into the check code,
|
||||||
|
// which sees the registers exactly as the function left them.
|
||||||
|
//
|
||||||
|
// Go ABI0 on amd64 guarantees:
|
||||||
|
// - BP is callee-saved (NOSPLIT frame=0 functions must not touch it).
|
||||||
|
// - R14 holds the goroutine pointer and must survive across any call.
|
||||||
|
|
||||||
|
// Sentinel values chosen to be unlikely in normal execution.
|
||||||
|
#define SENTINEL_BP 0xDEADBEEFCAFEF00D
|
||||||
|
#define SENTINEL_R14 0x0BADF00DDEADBEEF
|
||||||
|
|
||||||
|
// GLOBL holding the raw address of the leave trampoline, read by Go.
|
||||||
|
GLOBL ·leaveCheckedPtr(SB), NOPTR, $8
|
||||||
|
DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB)
|
||||||
|
|
||||||
|
// func enterJITChecked(fn uintptr, stack uintptr)
|
||||||
|
// Sets sentinels in BP and R14, switches to the prepared stack and jumps
|
||||||
|
// to fn. The prepared stack's return address must be leaveJITCheckedRaw
|
||||||
|
// (read from leaveCheckedPtr).
|
||||||
|
TEXT ·enterJITChecked(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ fn+0(FP), AX // target (before SP switch)
|
||||||
|
MOVQ SP, ·savedSP(SB) // preserve Go stack
|
||||||
|
MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber)
|
||||||
|
MOVQ $SENTINEL_BP, BP // sentinel in BP
|
||||||
|
MOVQ $SENTINEL_R14, R14 // sentinel in R14
|
||||||
|
MOVQ stack+8(FP), SP // switch to prepared stack
|
||||||
|
JMP AX
|
||||||
|
|
||||||
|
// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function
|
||||||
|
// declaration, so no ABIInternal wrapper is generated — the JIT function's
|
||||||
|
// RET lands here directly, seeing BP and R14 exactly as the function left
|
||||||
|
// them. It checks the sentinels, records violations in abiResult, then
|
||||||
|
// restores the Go stack and returns.
|
||||||
|
TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0
|
||||||
|
// Check BP against the sentinel.
|
||||||
|
MOVQ $SENTINEL_BP, CX
|
||||||
|
CMPQ BP, CX
|
||||||
|
JEQ bp_ok
|
||||||
|
ORQ $1, ·abiResult(SB)
|
||||||
|
|
||||||
|
bp_ok:
|
||||||
|
// Check R14 against the sentinel.
|
||||||
|
MOVQ $SENTINEL_R14, CX
|
||||||
|
CMPQ R14, CX
|
||||||
|
JEQ r14_ok
|
||||||
|
ORQ $2, ·abiResult(SB)
|
||||||
|
|
||||||
|
r14_ok:
|
||||||
|
MOVQ ·savedSP(SB), SP
|
||||||
|
RET
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build !amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
// ABIReport describes the result of an ABI-checking call.
|
||||||
|
type ABIReport struct {
|
||||||
|
BPClobbered bool
|
||||||
|
R14Clobbered bool
|
||||||
|
RedZoneHit bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// OK returns true when no violations were detected.
|
||||||
|
func (r ABIReport) OK() bool { return false }
|
||||||
|
|
||||||
|
// String returns a human-readable summary.
|
||||||
|
func (r ABIReport) String() string { return "verify: ABI checks require amd64" }
|
||||||
|
|
||||||
|
// CallChecked is unavailable on non-amd64 architectures.
|
||||||
|
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
|
||||||
|
return nil, ABIReport{}, fmt.Errorf("verify: ABI checks require amd64")
|
||||||
|
}
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
func loadABIKernel(t *testing.T) *Kernel {
|
||||||
|
t.Helper()
|
||||||
|
k, err := Load("../testdata/verify/abi_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
return k
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestABIClean(t *testing.T) {
|
||||||
|
k := loadABIKernel(t)
|
||||||
|
|
||||||
|
args := make([]byte, 24)
|
||||||
|
PutUint64(args, 0, 3)
|
||||||
|
PutUint64(args, 8, 4)
|
||||||
|
|
||||||
|
out, report, err := k.CallFuncChecked("cleanAdd", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFuncChecked: %v", err)
|
||||||
|
}
|
||||||
|
if got := int64(GetUint64(out, 16)); got != 7 {
|
||||||
|
t.Errorf("cleanAdd(3, 4) = %d, want 7", got)
|
||||||
|
}
|
||||||
|
if !report.OK() {
|
||||||
|
t.Errorf("cleanAdd: %s", report)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestABIBPClobbered(t *testing.T) {
|
||||||
|
k := loadABIKernel(t)
|
||||||
|
|
||||||
|
args := make([]byte, 16)
|
||||||
|
PutUint64(args, 0, 42)
|
||||||
|
|
||||||
|
out, report, err := k.CallFuncChecked("dirtyBP", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFuncChecked: %v", err)
|
||||||
|
}
|
||||||
|
if got := int64(GetUint64(out, 8)); got != 42 {
|
||||||
|
t.Errorf("dirtyBP(42) = %d, want 42", got)
|
||||||
|
}
|
||||||
|
if !report.BPClobbered {
|
||||||
|
t.Error("dirtyBP: expected BP clobbered, but report says clean")
|
||||||
|
}
|
||||||
|
if report.R14Clobbered {
|
||||||
|
t.Error("dirtyBP: R14 should not be clobbered")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestABIR14Clobbered(t *testing.T) {
|
||||||
|
k := loadABIKernel(t)
|
||||||
|
|
||||||
|
args := make([]byte, 16)
|
||||||
|
PutUint64(args, 0, 99)
|
||||||
|
|
||||||
|
out, report, err := k.CallFuncChecked("dirtyR14", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFuncChecked: %v", err)
|
||||||
|
}
|
||||||
|
if got := int64(GetUint64(out, 8)); got != 99 {
|
||||||
|
t.Errorf("dirtyR14(99) = %d, want 99", got)
|
||||||
|
}
|
||||||
|
if !report.R14Clobbered {
|
||||||
|
t.Error("dirtyR14: expected R14 clobbered, but report says clean")
|
||||||
|
}
|
||||||
|
if report.BPClobbered {
|
||||||
|
t.Error("dirtyR14: BP should not be clobbered")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestABILZ4Kernels verifies that the production go-lz4 kernels are ABI-clean:
|
||||||
|
// they preserve BP and R14 and do not write into the red zone.
|
||||||
|
func TestABILZ4Kernels(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
// wideCopyAVX2 with a real copy.
|
||||||
|
src := make([]byte, 128)
|
||||||
|
for i := range src {
|
||||||
|
src[i] = byte(i)
|
||||||
|
}
|
||||||
|
dst := make([]byte, 128)
|
||||||
|
|
||||||
|
args := make([]byte, 48)
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||||
|
PutUint64(args, 8, 128)
|
||||||
|
PutUint64(args, 16, 128)
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||||
|
PutUint64(args, 32, 128)
|
||||||
|
PutUint64(args, 40, 128)
|
||||||
|
|
||||||
|
_, report, err := k.CallFuncChecked("wideCopyAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFuncChecked(wideCopyAVX2): %v", err)
|
||||||
|
}
|
||||||
|
if !report.OK() {
|
||||||
|
t.Errorf("wideCopyAVX2: %s", report)
|
||||||
|
}
|
||||||
|
|
||||||
|
// decodeBlockAVX2 with a simple block.
|
||||||
|
decSrc := []byte{0x50, 'H', 'e', 'l', 'l', 'o'}
|
||||||
|
decDst := make([]byte, 64)
|
||||||
|
|
||||||
|
decArgs := make([]byte, 64)
|
||||||
|
PutPtr(decArgs, 0, unsafe.Pointer(&decSrc[0]))
|
||||||
|
PutUint64(decArgs, 8, uint64(len(decSrc)))
|
||||||
|
PutUint64(decArgs, 16, uint64(cap(decSrc)))
|
||||||
|
PutPtr(decArgs, 24, unsafe.Pointer(&decDst[0]))
|
||||||
|
PutUint64(decArgs, 32, uint64(len(decDst)))
|
||||||
|
PutUint64(decArgs, 40, uint64(cap(decDst)))
|
||||||
|
|
||||||
|
_, report, err = k.CallFuncChecked("decodeBlockAVX2", decArgs)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFuncChecked(decodeBlockAVX2): %v", err)
|
||||||
|
}
|
||||||
|
if !report.OK() {
|
||||||
|
t.Errorf("decodeBlockAVX2: %s", report)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCallFuncCheckedErrors(t *testing.T) {
|
||||||
|
k := loadABIKernel(t)
|
||||||
|
|
||||||
|
// Nonexistent function.
|
||||||
|
_, _, err := k.CallFuncChecked("nope", make([]byte, 8))
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for nonexistent function")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Arg block too small.
|
||||||
|
_, _, err = k.CallFuncChecked("cleanAdd", make([]byte, 8))
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for too-small arg block")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"reflect"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// savedSP holds the Go stack pointer while a JIT call is in flight.
|
||||||
|
// Referenced by the assembly trampoline (trampoline_amd64.s).
|
||||||
|
var savedSP uintptr
|
||||||
|
|
||||||
|
// enterJIT switches to the prepared stack and jumps to fn.
|
||||||
|
// It does not return normally; the JIT function's RET transfers control
|
||||||
|
// to leaveJIT, which restores the Go stack.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func enterJIT(fn uintptr, stack uintptr)
|
||||||
|
|
||||||
|
// leaveJIT restores the Go stack after a JIT function returns.
|
||||||
|
// Its address is placed as the return address on the prepared stack.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func leaveJIT()
|
||||||
|
|
||||||
|
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
|
||||||
|
var leaveJITAddr uintptr
|
||||||
|
|
||||||
|
func init() {
|
||||||
|
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
|
||||||
|
}
|
||||||
|
|
||||||
|
// stackPad is padding below the return address on the prepared stack.
|
||||||
|
// The ABIInternal wrapper that leaveJIT's address resolves to executes
|
||||||
|
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
|
||||||
|
// bytes below the return-address slot. 64 bytes of headroom is ample.
|
||||||
|
const stackPad = 64
|
||||||
|
|
||||||
|
// Call invokes the assembled function at fnAddr with the given ABI0 argument
|
||||||
|
// block (the raw bytes that would appear at FP+0). It returns the argument
|
||||||
|
// block after the call, which contains any results the function wrote back
|
||||||
|
// (the ABI0 convention shares the argument area for inputs and outputs).
|
||||||
|
//
|
||||||
|
// The function must be NOSPLIT (no stack growth) and must not reference
|
||||||
|
// external symbols — the image is self-contained.
|
||||||
|
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||||
|
// Prepare the stack: [padding][leaveJIT addr][args...]
|
||||||
|
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
|
||||||
|
stackMem, err := syscall.Mmap(-1, 0, stackSize,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: stack mmap: %w", err)
|
||||||
|
}
|
||||||
|
defer syscall.Munmap(stackMem)
|
||||||
|
|
||||||
|
// The return address sits after the padding; the function's SP will
|
||||||
|
// point here, leaving stackPad bytes below for the wrapper's pushes.
|
||||||
|
retOff := stackPad
|
||||||
|
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
|
||||||
|
// The ABI0 argument area follows the return address.
|
||||||
|
copy(stackMem[retOff+8:], args)
|
||||||
|
|
||||||
|
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||||
|
enterJIT(fnAddr, stackBase)
|
||||||
|
|
||||||
|
// Copy out the (possibly modified) argument area.
|
||||||
|
out := make([]byte, len(args))
|
||||||
|
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build !amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
// Call is unavailable on non-amd64 architectures.
|
||||||
|
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||||
|
return nil, fmt.Errorf("verify: JIT execution requires amd64")
|
||||||
|
}
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"sort"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Block describes one basic block within a function: a maximal sequence of
|
||||||
|
// instructions with a single entry point (a label or the function start) and
|
||||||
|
// a single exit (a jump, conditional jump or RET).
|
||||||
|
type Block struct {
|
||||||
|
Offset int // byte offset within the function
|
||||||
|
Label string // label name ("" for the entry block)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Blocks identifies the basic blocks of a function from its local labels.
|
||||||
|
// Each label is a potential jump target and therefore a block boundary; the
|
||||||
|
// function entry (offset 0) is always a block. The blocks are returned in
|
||||||
|
// ascending offset order.
|
||||||
|
func (k *Kernel) Blocks(name string) ([]Block, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
fl := k.img.Funcs[idx]
|
||||||
|
|
||||||
|
blocks := []Block{{Offset: 0, Label: "(entry)"}}
|
||||||
|
// Build a reverse map: offset → label name.
|
||||||
|
offToLabel := make(map[int]string, len(fl.Labels))
|
||||||
|
for label, off := range fl.Labels {
|
||||||
|
if off > 0 && off < fl.Size {
|
||||||
|
offToLabel[off] = label
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Collect and sort offsets.
|
||||||
|
offsets := make([]int, 0, len(offToLabel))
|
||||||
|
for off := range offToLabel {
|
||||||
|
offsets = append(offsets, off)
|
||||||
|
}
|
||||||
|
sort.Ints(offsets)
|
||||||
|
for _, off := range offsets {
|
||||||
|
blocks = append(blocks, Block{Offset: off, Label: offToLabel[off]})
|
||||||
|
}
|
||||||
|
return blocks, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// BlockCount returns the number of identified basic blocks for the function.
|
||||||
|
func (k *Kernel) BlockCount(name string) (int, error) {
|
||||||
|
blocks, err := k.Blocks(name)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return len(blocks), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// PathFingerprint is the observable output of one function execution: the
|
||||||
|
// values written back into the result slots of the argument block. Two
|
||||||
|
// executions that produce the same fingerprint took observationally
|
||||||
|
// equivalent paths (though they may differ internally).
|
||||||
|
type PathFingerprint struct {
|
||||||
|
Results []uint64 // the result words from the arg block
|
||||||
|
}
|
||||||
|
|
||||||
|
// ProfilePaths runs the function with each of the given argument blocks and
|
||||||
|
// collects the distinct output fingerprints. This measures path diversity:
|
||||||
|
// how many observationally different execution paths the input corpus
|
||||||
|
// exercises. Combined with Blocks (the static block count), it gives a
|
||||||
|
// lower bound on code coverage.
|
||||||
|
func (k *Kernel) ProfilePaths(name string, argSets [][]byte, resultOffsets []int) ([]PathFingerprint, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
fl := k.img.Funcs[idx]
|
||||||
|
|
||||||
|
seen := map[string]bool{}
|
||||||
|
var paths []PathFingerprint
|
||||||
|
|
||||||
|
for _, args := range argSets {
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
return nil, fmt.Errorf("verify: %s: arg block too small", name)
|
||||||
|
}
|
||||||
|
out, err := k.CallFunc(name, args)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
fp := PathFingerprint{}
|
||||||
|
key := ""
|
||||||
|
for _, off := range resultOffsets {
|
||||||
|
v := GetUint64(out, off)
|
||||||
|
fp.Results = append(fp.Results, v)
|
||||||
|
key += fmt.Sprintf("%016x", v)
|
||||||
|
}
|
||||||
|
if !seen[key] {
|
||||||
|
seen[key] = true
|
||||||
|
paths = append(paths, fp)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return paths, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,85 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestBlocks(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
// The "sum" function has labels: sum_done, sum_loop.
|
||||||
|
blocks, err := k.Blocks("sum")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Blocks(sum): %v", err)
|
||||||
|
}
|
||||||
|
if len(blocks) < 3 {
|
||||||
|
t.Errorf("sum: expected at least 3 blocks (entry + 2 labels), got %d", len(blocks))
|
||||||
|
}
|
||||||
|
if blocks[0].Offset != 0 {
|
||||||
|
t.Errorf("first block offset = %d, want 0", blocks[0].Offset)
|
||||||
|
}
|
||||||
|
t.Logf("sum blocks: %v", blocks)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBlockCount(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
n, err := k.BlockCount("decodeBlockAVX2")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BlockCount: %v", err)
|
||||||
|
}
|
||||||
|
// The decoder has many labels (dec_loop, dec_malformed, etc.).
|
||||||
|
if n < 10 {
|
||||||
|
t.Errorf("decodeBlockAVX2: expected at least 10 blocks, got %d", n)
|
||||||
|
}
|
||||||
|
t.Logf("decodeBlockAVX2: %d basic blocks", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProfilePaths(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
// Build a corpus of varied LZ4 blocks.
|
||||||
|
var argSets [][]byte
|
||||||
|
blocks := []struct {
|
||||||
|
src []byte
|
||||||
|
dstSize int
|
||||||
|
}{
|
||||||
|
{[]byte{0x00}, 16}, // empty
|
||||||
|
{[]byte{0x50, 'H', 'e', 'l', 'l', 'o'}, 16}, // literals only
|
||||||
|
{[]byte{0x54, 'A', 'A', 'A', 'A', 'A', 5, 0, 0x30, 'B', 'B', 'B'}, 32}, // match
|
||||||
|
{[]byte{0x14, 'X', 1, 0, 0x10, 'Y'}, 16}, // overlapping
|
||||||
|
{[]byte{0x50, 'H'}, 16}, // malformed
|
||||||
|
{[]byte{0x14, 'X', 0, 0}, 16}, // zero offset
|
||||||
|
}
|
||||||
|
for _, b := range blocks {
|
||||||
|
args := make([]byte, 64)
|
||||||
|
if len(b.src) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&b.src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(b.src)))
|
||||||
|
PutUint64(args, 16, uint64(cap(b.src)))
|
||||||
|
dst := make([]byte, b.dstSize)
|
||||||
|
if len(dst) > 0 {
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 32, uint64(len(dst)))
|
||||||
|
PutUint64(args, 40, uint64(cap(dst)))
|
||||||
|
argSets = append(argSets, args)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Result offsets: n+48 and code+56.
|
||||||
|
paths, err := k.ProfilePaths("decodeBlockAVX2", argSets, []int{48, 56})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ProfilePaths: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// We expect at least 3 distinct paths: success (various n), malformed, zero offset.
|
||||||
|
if len(paths) < 3 {
|
||||||
|
t.Errorf("expected at least 3 distinct paths, got %d", len(paths))
|
||||||
|
}
|
||||||
|
t.Logf("decodeBlockAVX2: %d distinct output paths from %d inputs", len(paths), len(argSets))
|
||||||
|
}
|
||||||
@@ -0,0 +1,295 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"math/rand"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
|
||||||
|
// differential-testing oracle. It mirrors the contract of
|
||||||
|
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
|
||||||
|
// 0 = ok, 1 = malformed, 2 = zero offset.
|
||||||
|
func decodeBlockGo(src, dst []byte) (int, int) {
|
||||||
|
if len(src) == 0 {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
si, di := 0, 0
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1 // truncated: no token
|
||||||
|
}
|
||||||
|
token := int(src[si])
|
||||||
|
si++
|
||||||
|
|
||||||
|
// Literals.
|
||||||
|
lLen := token >> 4
|
||||||
|
if lLen == 15 {
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
b := int(src[si])
|
||||||
|
si++
|
||||||
|
lLen += b
|
||||||
|
if b != 255 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if si+lLen > len(src) {
|
||||||
|
return 0, 1 // truncated literals
|
||||||
|
}
|
||||||
|
if di+lLen > len(dst) {
|
||||||
|
return 0, 1 // destination overflow
|
||||||
|
}
|
||||||
|
copy(dst[di:di+lLen], src[si:si+lLen])
|
||||||
|
di += lLen
|
||||||
|
si += lLen
|
||||||
|
|
||||||
|
// End of block.
|
||||||
|
if si >= len(src) {
|
||||||
|
return di, 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match offset.
|
||||||
|
if si+2 > len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
offset := int(src[si]) | int(src[si+1])<<8
|
||||||
|
si += 2
|
||||||
|
if offset == 0 {
|
||||||
|
return 0, 2
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match length.
|
||||||
|
mLen := token & 15
|
||||||
|
if mLen == 15 {
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
b := int(src[si])
|
||||||
|
si++
|
||||||
|
mLen += b
|
||||||
|
if b != 255 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
mLen += 4
|
||||||
|
|
||||||
|
// Copy match (overlapping-safe).
|
||||||
|
if di-offset < 0 {
|
||||||
|
return 0, 1 // offset reaches before dst start
|
||||||
|
}
|
||||||
|
if di+mLen > len(dst) {
|
||||||
|
return 0, 1 // destination overflow
|
||||||
|
}
|
||||||
|
for i := 0; i < mLen; i++ {
|
||||||
|
dst[di+i] = dst[di-offset+i]
|
||||||
|
}
|
||||||
|
di += mLen
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// genLZ4Block generates a random valid LZ4 block that decompresses into
|
||||||
|
// approximately wantSize bytes. The block is always well-formed (ends with
|
||||||
|
// a literals-only sequence).
|
||||||
|
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
|
||||||
|
var block []byte
|
||||||
|
produced := 0
|
||||||
|
for produced < wantSize {
|
||||||
|
remaining := wantSize - produced
|
||||||
|
|
||||||
|
// Decide: emit a literals+match sequence or the final literals.
|
||||||
|
if remaining <= 8 || rng.Intn(4) == 0 {
|
||||||
|
// Final literals-only sequence.
|
||||||
|
lLen := remaining
|
||||||
|
if lLen > 60 {
|
||||||
|
lLen = 1 + rng.Intn(60)
|
||||||
|
}
|
||||||
|
block = appendToken(block, lLen, 0)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
produced += lLen
|
||||||
|
break
|
||||||
|
}
|
||||||
|
|
||||||
|
// Literals + match.
|
||||||
|
lLen := rng.Intn(min(16, remaining))
|
||||||
|
if produced+lLen == 0 {
|
||||||
|
lLen = 1 // must have at least 1 literal before the first match
|
||||||
|
}
|
||||||
|
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
|
||||||
|
mLen := mLenRaw + 4
|
||||||
|
if produced+mLen > remaining {
|
||||||
|
mLen = remaining - produced
|
||||||
|
if mLen < 4 {
|
||||||
|
// Not enough room for a match; emit final literals.
|
||||||
|
lLen = remaining
|
||||||
|
block = appendToken(block, lLen, 0)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
mLenRaw = mLen - 4
|
||||||
|
}
|
||||||
|
|
||||||
|
block = appendToken(block, lLen, mLenRaw)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
produced += lLen
|
||||||
|
|
||||||
|
// Offset: must be <= produced (can't reference before start).
|
||||||
|
maxOff := produced
|
||||||
|
if maxOff > 65535 {
|
||||||
|
maxOff = 65535
|
||||||
|
}
|
||||||
|
offset := 1 + rng.Intn(maxOff)
|
||||||
|
block = append(block, byte(offset), byte(offset>>8))
|
||||||
|
produced += mLen
|
||||||
|
}
|
||||||
|
return block
|
||||||
|
}
|
||||||
|
|
||||||
|
// appendToken appends a token (and extension bytes if needed) for the given
|
||||||
|
// literal and match lengths.
|
||||||
|
func appendToken(block []byte, lLen, mLenRaw int) []byte {
|
||||||
|
lit4 := lLen
|
||||||
|
if lit4 > 15 {
|
||||||
|
lit4 = 15
|
||||||
|
}
|
||||||
|
ml4 := mLenRaw
|
||||||
|
if ml4 > 15 {
|
||||||
|
ml4 = 15
|
||||||
|
}
|
||||||
|
block = append(block, byte(lit4<<4|ml4))
|
||||||
|
// Literal extension bytes.
|
||||||
|
rem := lLen - 15
|
||||||
|
for rem >= 255 {
|
||||||
|
block = append(block, 255)
|
||||||
|
rem -= 255
|
||||||
|
}
|
||||||
|
if lLen >= 15 {
|
||||||
|
block = append(block, byte(rem))
|
||||||
|
}
|
||||||
|
// Match extension bytes.
|
||||||
|
rem = mLenRaw - 15
|
||||||
|
for rem >= 255 {
|
||||||
|
block = append(block, 255)
|
||||||
|
rem -= 255
|
||||||
|
}
|
||||||
|
if mLenRaw >= 15 {
|
||||||
|
block = append(block, byte(rem))
|
||||||
|
}
|
||||||
|
return block
|
||||||
|
}
|
||||||
|
|
||||||
|
func min(a, b int) int {
|
||||||
|
if a < b {
|
||||||
|
return a
|
||||||
|
}
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
|
||||||
|
// random valid LZ4 blocks and compares the output bit-for-bit against the
|
||||||
|
// portable Go reference.
|
||||||
|
func TestDifferentialLZ4Fuzz(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
const iterations = 5000
|
||||||
|
rng := rand.New(rand.NewSource(42))
|
||||||
|
|
||||||
|
for i := 0; i < iterations; i++ {
|
||||||
|
wantSize := 1 + rng.Intn(4096)
|
||||||
|
src := genLZ4Block(rng, wantSize)
|
||||||
|
dstSize := wantSize + 64 // generous destination
|
||||||
|
|
||||||
|
// Go reference.
|
||||||
|
goDst := make([]byte, dstSize)
|
||||||
|
goN, goCode := decodeBlockGo(src, goDst)
|
||||||
|
|
||||||
|
// JIT kernel.
|
||||||
|
jitDst := make([]byte, dstSize)
|
||||||
|
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||||
|
|
||||||
|
if jitCode != goCode {
|
||||||
|
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||||
|
i, jitCode, goCode, len(src))
|
||||||
|
}
|
||||||
|
if jitCode != 0 {
|
||||||
|
continue // both agree it's malformed/zero-offset
|
||||||
|
}
|
||||||
|
if jitN != goN {
|
||||||
|
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||||
|
i, jitN, goN, len(src))
|
||||||
|
}
|
||||||
|
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
||||||
|
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
|
||||||
|
// that error codes agree with the Go reference (no crashes, same classification).
|
||||||
|
func TestDifferentialLZ4Hostile(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
const iterations = 2000
|
||||||
|
rng := rand.New(rand.NewSource(99))
|
||||||
|
|
||||||
|
for i := 0; i < iterations; i++ {
|
||||||
|
srcLen := rng.Intn(128)
|
||||||
|
src := make([]byte, srcLen)
|
||||||
|
rng.Read(src)
|
||||||
|
dstSize := rng.Intn(512)
|
||||||
|
dst := make([]byte, dstSize)
|
||||||
|
|
||||||
|
// Go reference.
|
||||||
|
goDst := make([]byte, dstSize)
|
||||||
|
copy(goDst, dst)
|
||||||
|
_, goCode := decodeBlockGo(src, goDst)
|
||||||
|
|
||||||
|
// JIT kernel.
|
||||||
|
jitDst := make([]byte, dstSize)
|
||||||
|
copy(jitDst, dst)
|
||||||
|
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||||
|
|
||||||
|
if jitCode != goCode {
|
||||||
|
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
|
||||||
|
i, jitCode, goCode, srcLen, dstSize)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
|
||||||
|
// dst size (for hostile tests where dst may be smaller than the output).
|
||||||
|
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||||
|
t.Helper()
|
||||||
|
args := make([]byte, 64)
|
||||||
|
if len(src) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(src)))
|
||||||
|
PutUint64(args, 16, uint64(cap(src)))
|
||||||
|
if len(dst) > 0 {
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 32, uint64(len(dst)))
|
||||||
|
PutUint64(args, 40, uint64(cap(dst)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||||
|
}
|
||||||
|
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Package verify provides the dynamic-analysis substrate for gasm: it
|
||||||
|
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
|
||||||
|
// directly, enabling differential testing against portable Go references,
|
||||||
|
// runtime ABI checks and basic-block coverage profiling.
|
||||||
|
//
|
||||||
|
// The execution model is pure Go (stdlib only): machine code is mapped with
|
||||||
|
// syscall.Mmap and invoked through an assembly trampoline that switches to a
|
||||||
|
// prepared ABI0 stack. No cgo, no external toolchain.
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Executable maps a copy of code into a read-execute memory region suitable
|
||||||
|
// for direct invocation. The mapping is anonymous and private; the original
|
||||||
|
// slice is not retained. Call Unmap to release the region.
|
||||||
|
type Executable struct {
|
||||||
|
addr uintptr // base address of the mapping
|
||||||
|
size int
|
||||||
|
mem []byte // the mmap'd slice (for Unmap)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map copies code into a freshly allocated RX region and returns it.
|
||||||
|
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
|
||||||
|
// copy, matching W^X policy.
|
||||||
|
func Map(code []byte) (*Executable, error) {
|
||||||
|
size := len(code)
|
||||||
|
if size == 0 {
|
||||||
|
return nil, fmt.Errorf("verify: cannot map zero-length code")
|
||||||
|
}
|
||||||
|
// Round up to the page size.
|
||||||
|
const pageSize = 4096
|
||||||
|
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
|
||||||
|
|
||||||
|
mem, err := syscall.Mmap(-1, 0, mapSize,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: mmap: %w", err)
|
||||||
|
}
|
||||||
|
copy(mem, code)
|
||||||
|
|
||||||
|
// Remove write permission (W^X).
|
||||||
|
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
|
||||||
|
syscall.Munmap(mem)
|
||||||
|
return nil, fmt.Errorf("verify: mprotect: %w", err)
|
||||||
|
}
|
||||||
|
return &Executable{
|
||||||
|
addr: uintptr(unsafe.Pointer(&mem[0])),
|
||||||
|
size: size,
|
||||||
|
mem: mem,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unmap releases the executable region.
|
||||||
|
func (e *Executable) Unmap() {
|
||||||
|
if e.mem != nil {
|
||||||
|
syscall.Munmap(e.mem)
|
||||||
|
e.mem = nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// FuncAddr returns the absolute address of a function at the given offset
|
||||||
|
// within the mapped image.
|
||||||
|
func (e *Executable) FuncAddr(offset int) uintptr {
|
||||||
|
return e.addr + uintptr(offset)
|
||||||
|
}
|
||||||
|
|
||||||
|
// PutUint64 writes v into buf at byte offset off (little-endian).
|
||||||
|
func PutUint64(buf []byte, off int, v uint64) {
|
||||||
|
binary.LittleEndian.PutUint64(buf[off:off+8], v)
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
|
||||||
|
func GetUint64(buf []byte, off int) uint64 {
|
||||||
|
return binary.LittleEndian.Uint64(buf[off : off+8])
|
||||||
|
}
|
||||||
|
|
||||||
|
// PutPtr writes a pointer value into buf at byte offset off.
|
||||||
|
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||||
|
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
|
||||||
|
}
|
||||||
@@ -0,0 +1,215 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
func loadBasic(t *testing.T) *Kernel {
|
||||||
|
t.Helper()
|
||||||
|
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
return k
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITAdd(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
a, b, want int64
|
||||||
|
}{
|
||||||
|
{0, 0, 0},
|
||||||
|
{1, 2, 3},
|
||||||
|
{-1, 1, 0},
|
||||||
|
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
|
||||||
|
{-100, -200, -300},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
args := make([]byte, 24)
|
||||||
|
PutUint64(args, 0, uint64(tt.a))
|
||||||
|
PutUint64(args, 8, uint64(tt.b))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("add", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
|
||||||
|
}
|
||||||
|
got := int64(GetUint64(out, 16))
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITSum(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
data []int64
|
||||||
|
want int64
|
||||||
|
}{
|
||||||
|
{nil, 0},
|
||||||
|
{[]int64{1}, 1},
|
||||||
|
{[]int64{1, 2, 3, 4, 5}, 15},
|
||||||
|
{[]int64{-10, 20, -30, 40}, 20},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
args := make([]byte, 32)
|
||||||
|
if len(tt.data) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(tt.data)))
|
||||||
|
PutUint64(args, 16, uint64(cap(tt.data)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("sum", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
||||||
|
}
|
||||||
|
got := int64(GetUint64(out, 24))
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITWideCopy(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
n int
|
||||||
|
}{
|
||||||
|
{"empty", 0},
|
||||||
|
{"tiny", 7},
|
||||||
|
{"exact32", 32},
|
||||||
|
{"overlap_range", 48},
|
||||||
|
{"exact64", 64},
|
||||||
|
{"unaligned", 45},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
src := make([]byte, tt.n)
|
||||||
|
for i := range src {
|
||||||
|
src[i] = byte(i * 7)
|
||||||
|
}
|
||||||
|
dst := make([]byte, tt.n)
|
||||||
|
|
||||||
|
args := make([]byte, 48)
|
||||||
|
if tt.n > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(tt.n)) // dst_len
|
||||||
|
PutUint64(args, 16, uint64(tt.n)) // dst_cap
|
||||||
|
PutUint64(args, 32, uint64(tt.n)) // src_len
|
||||||
|
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
||||||
|
|
||||||
|
_, err := k.CallFunc("wideCopy", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(wideCopy): %v", err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(dst, src) {
|
||||||
|
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelFuncNames(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
names := k.FuncNames()
|
||||||
|
want := []string{"add", "sum", "wideCopy"}
|
||||||
|
if len(names) != len(want) {
|
||||||
|
t.Fatalf("FuncNames() = %v, want %v", names, want)
|
||||||
|
}
|
||||||
|
for i, n := range names {
|
||||||
|
if n != want[i] {
|
||||||
|
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelFuncNotFound(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
_, err := k.CallFunc("nonexistent", make([]byte, 8))
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for nonexistent function")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelArgTooSmall(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for too-small arg block")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMapZeroLength(t *testing.T) {
|
||||||
|
_, err := Map(nil)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for zero-length code")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoadSourceError(t *testing.T) {
|
||||||
|
_, err := LoadSource("bad.s", "TEXT ·f(SB), NOSPLIT\n\tBADINSTRUCTION\n")
|
||||||
|
// The parser may or may not error on unknown instructions (it's
|
||||||
|
// error-tolerant), but the assembler will reject it.
|
||||||
|
if err == nil {
|
||||||
|
t.Log("LoadSource succeeded unexpectedly (parser is error-tolerant)")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoadSourceParseError(t *testing.T) {
|
||||||
|
// A completely invalid file that the parser rejects.
|
||||||
|
_, err := LoadSource("empty.s", "")
|
||||||
|
if err != nil {
|
||||||
|
t.Logf("expected: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFuncLookup(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
fl, err := k.Func("add")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Func(add): %v", err)
|
||||||
|
}
|
||||||
|
if fl.Name != "add" {
|
||||||
|
t.Errorf("Func(add).Name = %q, want %q", fl.Name, "add")
|
||||||
|
}
|
||||||
|
if fl.Args != 24 {
|
||||||
|
t.Errorf("Func(add).Args = %d, want 24", fl.Args)
|
||||||
|
}
|
||||||
|
_, err = k.Func("nonexistent")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for nonexistent function")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestABIReportString(t *testing.T) {
|
||||||
|
r := ABIReport{}
|
||||||
|
if r.String() != "ABI clean" {
|
||||||
|
t.Errorf("clean report = %q", r.String())
|
||||||
|
}
|
||||||
|
r.BPClobbered = true
|
||||||
|
if r.OK() {
|
||||||
|
t.Error("expected not OK with BP clobbered")
|
||||||
|
}
|
||||||
|
s := r.String()
|
||||||
|
if s == "ABI clean" {
|
||||||
|
t.Error("expected violation string, got clean")
|
||||||
|
}
|
||||||
|
r.R14Clobbered = true
|
||||||
|
r.RedZoneHit = true
|
||||||
|
s = r.String()
|
||||||
|
if s == "ABI clean" {
|
||||||
|
t.Error("expected violation string for all flags")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,160 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
|
||||||
|
// integration testing. The test is skipped when the file is absent
|
||||||
|
// (e.g. in CI without the sibling checkout).
|
||||||
|
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
|
||||||
|
|
||||||
|
func loadLZ4Kernel(t *testing.T) *Kernel {
|
||||||
|
t.Helper()
|
||||||
|
if _, err := os.Stat(lz4KernelPath); err != nil {
|
||||||
|
t.Skipf("sibling kernel not available: %v", err)
|
||||||
|
}
|
||||||
|
k, err := Load(lz4KernelPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
return k
|
||||||
|
}
|
||||||
|
|
||||||
|
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
|
||||||
|
// given src and dst buffers, returning (n, code).
|
||||||
|
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||||
|
t.Helper()
|
||||||
|
args := make([]byte, 64)
|
||||||
|
if len(src) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(src)))
|
||||||
|
PutUint64(args, 16, uint64(cap(src)))
|
||||||
|
if len(dst) > 0 {
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 32, uint64(len(dst)))
|
||||||
|
PutUint64(args, 40, uint64(cap(dst)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||||
|
}
|
||||||
|
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLZ4DecodeKnownAnswers(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
src []byte
|
||||||
|
dstSize int
|
||||||
|
wantDst []byte
|
||||||
|
wantN int
|
||||||
|
wantCode int
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "literals_only",
|
||||||
|
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: []byte("Hello"),
|
||||||
|
wantN: 5,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "literals_and_match",
|
||||||
|
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
|
||||||
|
dstSize: 32,
|
||||||
|
wantDst: []byte("AAAAAAAAAAAAABBB"),
|
||||||
|
wantN: 16,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "overlapping_match",
|
||||||
|
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
|
||||||
|
// then final 1 literal 'Y'.
|
||||||
|
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: []byte("XXXXXXXXXY"),
|
||||||
|
wantN: 10,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "malformed_truncated",
|
||||||
|
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
|
||||||
|
dstSize: 16,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 1,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "zero_offset",
|
||||||
|
src: []byte{0x14, 'X', 0x00, 0x00},
|
||||||
|
dstSize: 16,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 2,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty_token",
|
||||||
|
src: []byte{0x00}, // 0 literals, end of block
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: nil,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
dst := make([]byte, tt.dstSize)
|
||||||
|
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
|
||||||
|
if n != tt.wantN || code != tt.wantCode {
|
||||||
|
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
|
||||||
|
n, code, tt.wantN, tt.wantCode)
|
||||||
|
}
|
||||||
|
if tt.wantCode == 0 && tt.wantDst != nil {
|
||||||
|
if !bytes.Equal(dst[:n], tt.wantDst) {
|
||||||
|
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLZ4WideCopyAVX2(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
|
||||||
|
for _, n := range sizes {
|
||||||
|
src := make([]byte, n)
|
||||||
|
for i := range src {
|
||||||
|
src[i] = byte(i*13 + 7)
|
||||||
|
}
|
||||||
|
dst := make([]byte, n)
|
||||||
|
|
||||||
|
args := make([]byte, 48)
|
||||||
|
if n > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(n))
|
||||||
|
PutUint64(args, 16, uint64(n))
|
||||||
|
PutUint64(args, 32, uint64(n))
|
||||||
|
PutUint64(args, 40, uint64(n))
|
||||||
|
|
||||||
|
_, err := k.CallFunc("wideCopyAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(dst, src) {
|
||||||
|
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
|
||||||
|
// stack and jumps to the assembled function; when the function RETs, control
|
||||||
|
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
|
||||||
|
//
|
||||||
|
// The prepared stack must begin with the address of leaveJIT (the return
|
||||||
|
// address the JIT function will pop), followed by the function's ABI0
|
||||||
|
// argument area.
|
||||||
|
//
|
||||||
|
// Single-threaded: savedSP is a package global, so only one JIT call may be
|
||||||
|
// in flight at a time. gasm verify runs sequentially.
|
||||||
|
|
||||||
|
// func enterJIT(fn uintptr, stack uintptr)
|
||||||
|
// Switches to the prepared stack and jumps to fn. Does not return normally;
|
||||||
|
// the JIT function's RET transfers control to leaveJIT.
|
||||||
|
TEXT ·enterJIT(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ fn+0(FP), AX // target function address (before SP switch)
|
||||||
|
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
|
||||||
|
MOVQ stack+8(FP), SP // switch to the prepared stack
|
||||||
|
JMP AX
|
||||||
|
|
||||||
|
// func leaveJIT()
|
||||||
|
// Restores the Go stack pointer and returns to enterJIT's caller.
|
||||||
|
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
|
||||||
|
MOVQ ·savedSP(SB), SP
|
||||||
|
RET
|
||||||
@@ -0,0 +1,122 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Kernel is a JIT-loaded assembly image ready for direct invocation.
|
||||||
|
// It wraps an executable memory mapping and the function layout metadata
|
||||||
|
// needed to marshal ABI0 calls.
|
||||||
|
type Kernel struct {
|
||||||
|
exec *Executable
|
||||||
|
img *asm.Image
|
||||||
|
funcs map[string]int // function name → index into img.Funcs
|
||||||
|
}
|
||||||
|
|
||||||
|
// Load parses, assembles and maps a .s file into executable memory.
|
||||||
|
// The returned Kernel is ready for Call. The caller must call Close to
|
||||||
|
// release the mapping.
|
||||||
|
func Load(path string) (*Kernel, error) {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: %w", err)
|
||||||
|
}
|
||||||
|
return LoadSource(path, string(src))
|
||||||
|
}
|
||||||
|
|
||||||
|
// LoadSource parses, assembles and maps assembly source into executable memory.
|
||||||
|
func LoadSource(filename, src string) (*Kernel, error) {
|
||||||
|
file, errs := parser.Parse(filename, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
|
||||||
|
}
|
||||||
|
return LoadAST(file)
|
||||||
|
}
|
||||||
|
|
||||||
|
// LoadAST assembles a parsed AST file and maps the result into executable
|
||||||
|
// memory.
|
||||||
|
func LoadAST(file *ast.File) (*Kernel, error) {
|
||||||
|
img, err := asm.AssembleFile(file)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: assemble: %w", err)
|
||||||
|
}
|
||||||
|
if len(img.Externals) > 0 {
|
||||||
|
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
|
||||||
|
}
|
||||||
|
code := img.Bytes()
|
||||||
|
exec, err := Map(code)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
funcs := make(map[string]int, len(img.Funcs))
|
||||||
|
for i, f := range img.Funcs {
|
||||||
|
funcs[f.Name] = i
|
||||||
|
}
|
||||||
|
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Func returns the layout metadata for the named function.
|
||||||
|
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
return k.img.Funcs[idx], nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// FuncNames returns the names of all functions in the kernel, in source order.
|
||||||
|
func (k *Kernel) FuncNames() []string {
|
||||||
|
names := make([]string, len(k.img.Funcs))
|
||||||
|
for i, f := range k.img.Funcs {
|
||||||
|
names[i] = f.Name
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
// CallFunc invokes the named function with the given ABI0 argument block.
|
||||||
|
// The arg block is the raw bytes of the function's argument/result area
|
||||||
|
// (as declared by the TEXT $frame-args suffix). Returns the arg block
|
||||||
|
// after the call (with any results written back by the function).
|
||||||
|
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
fl := k.img.Funcs[idx]
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||||
|
}
|
||||||
|
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||||
|
return Call(fnAddr, args)
|
||||||
|
}
|
||||||
|
|
||||||
|
// CallFuncChecked invokes the named function with ABI sentinels and a
|
||||||
|
// red-zone canary, returning the argument block and an ABIReport that
|
||||||
|
// records any callee-saved register or red-zone violations.
|
||||||
|
func (k *Kernel) CallFuncChecked(name string, args []byte) ([]byte, ABIReport, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, ABIReport{}, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
fl := k.img.Funcs[idx]
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
return nil, ABIReport{}, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||||
|
}
|
||||||
|
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||||
|
return CallChecked(fnAddr, args)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Close releases the executable mapping.
|
||||||
|
func (k *Kernel) Close() {
|
||||||
|
if k.exec != nil {
|
||||||
|
k.exec.Unmap()
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user