Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f52e23f1bc | ||
|
|
c9775c2b95 | ||
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 | ||
|
|
11f962fbcc | ||
|
|
ee68859beb |
+28
-15
@@ -51,16 +51,17 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||
// before splitSize. A ".Z" suffix requests EVEX zeroing.
|
||||
base, zeroing, err := stripEvexSuffix(upper)
|
||||
// before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
|
||||
// off the mnemonic too.
|
||||
base, sfx, err := parseEvexSuffix(upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isVex(base) || isEvex(base) || base == "KMOVW" {
|
||||
return e.encodeVec(base, ops, zeroing)
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||
return e.encodeVec(base, ops, sfx)
|
||||
}
|
||||
if zeroing {
|
||||
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", mnem)
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
|
||||
}
|
||||
|
||||
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
|
||||
@@ -128,20 +129,32 @@ func splitSize(upper string) (base string, size int) {
|
||||
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
|
||||
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
||||
// EVEX-only mnemonic) and VEX otherwise.
|
||||
func (e *enc) encodeVec(upper string, ops []Operand, zeroing bool) error {
|
||||
if upper == "KMOVW" {
|
||||
if zeroing {
|
||||
return fmt.Errorf("KMOVW takes no .Z suffix")
|
||||
}
|
||||
return e.encodeKmovw(ops)
|
||||
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||
if gs, ok := gatherTable[upper]; ok {
|
||||
return e.encodeGather(upper, gs, ops, sfx)
|
||||
}
|
||||
if upper == "KTESTW" || !evexRequired(upper, ops) {
|
||||
if zeroing {
|
||||
if ss, ok := scatterTable[upper]; ok {
|
||||
return e.encodeScatter(upper, ss, ops, sfx)
|
||||
}
|
||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKmov(upper, ops)
|
||||
}
|
||||
if isKOp(upper) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||
}
|
||||
return e.encodeKOp(upper, ops)
|
||||
}
|
||||
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
|
||||
}
|
||||
return e.encodeVex(upper, ops)
|
||||
}
|
||||
return e.encodeEvex(upper, ops, zeroing)
|
||||
return e.encodeEvex(upper, ops, sfx)
|
||||
}
|
||||
|
||||
// --- instruction components -------------------------------------------------
|
||||
|
||||
+837
-79
File diff suppressed because it is too large
Load Diff
+386
-1
@@ -230,7 +230,11 @@ func TestEvexMasking(t *testing.T) {
|
||||
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
|
||||
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
|
||||
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
|
||||
{"unsupported suffix", "VPADDD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
|
||||
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
|
||||
}
|
||||
for _, c := range bad {
|
||||
@@ -240,6 +244,387 @@ func TestEvexMasking(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
|
||||
// logic, lane shuffles/inserts/extracts, compares with a K destination,
|
||||
// permutes, the wider integer families, expand/compress, broadcasts,
|
||||
// rotates and word shifts, the opmask instructions, the EVEX suffixes
|
||||
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
|
||||
// against the Go assembler.
|
||||
func TestEvexExtendedGroundTruth(t *testing.T) {
|
||||
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Ternary logic and lane shuffles (NDS + imm8).
|
||||
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
|
||||
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
|
||||
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
|
||||
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
|
||||
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
|
||||
// Permutes.
|
||||
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
|
||||
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
|
||||
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
|
||||
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
|
||||
// Compare with a K destination (and an immediate predicate).
|
||||
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
|
||||
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
|
||||
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
|
||||
// Rounding / SAE / broadcast suffixes.
|
||||
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
|
||||
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
|
||||
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
|
||||
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
|
||||
// Packed single arithmetic (same opcodes, no mandatory prefix) —
|
||||
// ZMM, YMM and XMM widths, rounding and broadcast.
|
||||
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
|
||||
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
|
||||
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
|
||||
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
|
||||
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
|
||||
// Compress / expand.
|
||||
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
|
||||
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
|
||||
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
|
||||
// Broadcasts.
|
||||
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
|
||||
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
|
||||
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
|
||||
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
|
||||
// Wider integer families.
|
||||
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
|
||||
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
|
||||
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
|
||||
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
|
||||
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
|
||||
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
|
||||
// Absolute values and replicating moves.
|
||||
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
|
||||
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
|
||||
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
|
||||
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
|
||||
// Rotates and word/qword shifts.
|
||||
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
|
||||
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
|
||||
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
|
||||
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
|
||||
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
|
||||
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
|
||||
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
|
||||
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
|
||||
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
|
||||
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
|
||||
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
|
||||
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
|
||||
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
|
||||
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
|
||||
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
|
||||
// Lane extract / insert.
|
||||
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
|
||||
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
|
||||
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
|
||||
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
|
||||
// Aligned moves and the scalar single move.
|
||||
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
|
||||
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
|
||||
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
|
||||
// Conversions and extending/narrowing moves.
|
||||
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
|
||||
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
|
||||
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
|
||||
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
|
||||
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
|
||||
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
if i := strings.IndexByte(want, '.'); i > 0 {
|
||||
want = want[:i]
|
||||
}
|
||||
if inst.Op.String() != want {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||
// assembler.
|
||||
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
vsib := func(base, idx string, scale int) Operand {
|
||||
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Reciprocals and rsqrt (packed RM, scalar NDS).
|
||||
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
|
||||
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
|
||||
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
|
||||
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
|
||||
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
|
||||
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
|
||||
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
|
||||
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
|
||||
// Getexp (packed RM, scalar NDS).
|
||||
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
|
||||
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
|
||||
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
|
||||
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
|
||||
// Scalef (NDS).
|
||||
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
|
||||
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
|
||||
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
|
||||
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
|
||||
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
|
||||
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
|
||||
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
|
||||
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
|
||||
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
|
||||
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
|
||||
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
|
||||
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
|
||||
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
|
||||
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
|
||||
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
|
||||
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
|
||||
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
|
||||
// Fixupimm / range (NDS + imm8).
|
||||
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
|
||||
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
|
||||
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
|
||||
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
|
||||
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
|
||||
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
|
||||
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
|
||||
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
|
||||
// FP class test ($imm, src, kdst; packed forms carry the length in
|
||||
// the X/Y/Z mnemonic suffix the decoder drops).
|
||||
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
|
||||
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
|
||||
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
|
||||
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
|
||||
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
|
||||
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
|
||||
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
|
||||
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
|
||||
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
|
||||
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
|
||||
// Scatter (EVEX only: source, K mask, VSIB).
|
||||
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
|
||||
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
|
||||
// The remaining conversions.
|
||||
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
|
||||
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
|
||||
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
|
||||
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
|
||||
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
|
||||
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||
// for byte against the Go assembler, including memory sources and extended
|
||||
// GPRs.
|
||||
func TestEvexGprGroundTruth(t *testing.T) {
|
||||
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||
// the Go assembler.
|
||||
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// Unsigned and truncating conversions.
|
||||
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||
// The remaining sign/zero-extending moves.
|
||||
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||
// Signed narrowing stores.
|
||||
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||
// Unsigned narrowing stores.
|
||||
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||
// Mask/vector conversions (the K register is an operand, not a
|
||||
// mask).
|
||||
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexErrors checks the EVEX-specific error paths.
|
||||
func TestEvexErrors(t *testing.T) {
|
||||
cases := []struct {
|
||||
|
||||
+70
-6
@@ -91,12 +91,19 @@ var vexTable = map[string]vexSpec{
|
||||
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
||||
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
||||
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
||||
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
||||
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
|
||||
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
|
||||
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
|
||||
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
|
||||
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
|
||||
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
|
||||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
|
||||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
|
||||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
|
||||
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
|
||||
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
|
||||
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
|
||||
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
|
||||
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
|
||||
@@ -123,7 +130,15 @@ var vexTable = map[string]vexSpec{
|
||||
// no vvvv).
|
||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||
@@ -169,6 +184,9 @@ var vexTable = map[string]vexSpec{
|
||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||
|
||||
// VEX.128.0F.W0 — no operands.
|
||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||
@@ -176,6 +194,45 @@ var vexTable = map[string]vexSpec{
|
||||
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
|
||||
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
|
||||
// rm=scalar memory; SD is 256-bit only).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||
// source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||
// spellings: the destination is always XMM and the spelling fixes the
|
||||
// source length (X = 128, Y = 256).
|
||||
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// VEX scalar conversions between vector and general-purpose registers.
|
||||
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||
// unused; the length follows the source).
|
||||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||
// vector source in vvvv, vector destination in reg).
|
||||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
||||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
|
||||
|
||||
// VEX.F2.0F — packed double to packed dword conversions, truncating and
|
||||
// non-truncating. The destination is always XMM; the X/Y spellings fix
|
||||
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
|
||||
@@ -194,6 +251,8 @@ var vexSrcLen = map[string]int{
|
||||
"VCVTPD2DQY": 1,
|
||||
"VCVTTPD2DQX": 0,
|
||||
"VCVTTPD2DQY": 1,
|
||||
"VCVTPD2PSX": 0,
|
||||
"VCVTPD2PSY": 1,
|
||||
}
|
||||
|
||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||
@@ -238,6 +297,11 @@ var vexMoveTable = map[string]vexMoveSpec{
|
||||
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
|
||||
// register form takes three operands and is not supported yet).
|
||||
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
|
||||
// VEX.128/256 — aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
|
||||
}
|
||||
|
||||
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
||||
|
||||
+6
-3
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != mnem {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
||||
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||
got := inst.Op.String()
|
||||
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+59
-1
@@ -24,11 +24,12 @@ import (
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||
)
|
||||
|
||||
// version is the release version, stamped at build time via
|
||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||
var version = "0.12.0"
|
||||
var version = "0.18.0"
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
@@ -46,6 +47,8 @@ func main() {
|
||||
os.Exit(cmdLint(os.Args[2:]))
|
||||
case "asm":
|
||||
os.Exit(cmdAsm(os.Args[2:]))
|
||||
case "verify":
|
||||
os.Exit(cmdVerify(os.Args[2:]))
|
||||
case "lsp":
|
||||
os.Exit(cmdLSP(os.Args[2:]))
|
||||
case "version", "--version", "-V":
|
||||
@@ -80,6 +83,7 @@ Commands:
|
||||
fmt canonicalise formatting (gofmt for assembly)
|
||||
lint run static checks
|
||||
asm assemble .s files to machine code (amd64)
|
||||
verify JIT-assemble and run dynamic checks (amd64)
|
||||
lsp run the language server over stdio
|
||||
version print the version (same as --version)
|
||||
|
||||
@@ -466,3 +470,57 @@ requires -p, the package path, and the installed Go toolchain).
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdVerify(args []string) int {
|
||||
fs := newCommand("verify", "gasm verify <file.s>", `
|
||||
Assemble FILE (amd64), map it into executable memory and report the available
|
||||
functions. This confirms the assembled image is self-consistent (no
|
||||
unresolved external symbols) and executable — the prerequisite for dynamic
|
||||
testing.
|
||||
|
||||
With -smoke, each NOSPLIT function is called with a zeroed argument block to
|
||||
confirm the JIT trampoline works end-to-end. This is safe only for functions
|
||||
that tolerate nil pointers and zero lengths in their arguments.
|
||||
`)
|
||||
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] <file.s>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
if arch.FromFilename(path) != arch.AMD64 {
|
||||
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 is supported")
|
||||
return 1
|
||||
}
|
||||
|
||||
k, err := verify.Load(path)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer k.Close()
|
||||
|
||||
names := k.FuncNames()
|
||||
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
|
||||
rc := 0
|
||||
for _, name := range names {
|
||||
fl, _ := k.Func(name)
|
||||
flags := ""
|
||||
if fl.NoSplit {
|
||||
flags = " NOSPLIT"
|
||||
}
|
||||
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
|
||||
if *smoke && fl.NoSplit {
|
||||
args := make([]byte, fl.Args)
|
||||
_, err := k.CallFunc(name, args)
|
||||
if err != nil {
|
||||
fmt.Printf(" smoke: FAIL — %v\n", err)
|
||||
rc = 1
|
||||
} else {
|
||||
fmt.Printf(" smoke: OK\n")
|
||||
}
|
||||
}
|
||||
}
|
||||
return rc
|
||||
}
|
||||
|
||||
+50
-5
@@ -228,11 +228,31 @@ explicit merging/zeroing masks — written the way Go writes them, as a K
|
||||
operand among the operands plus a `.Z` mnemonic suffix), and the compressed
|
||||
disp8×N displacement, whose multiplier follows the memory operand's size —
|
||||
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
|
||||
plus the common AVX-512 F/BW integer set and the floating-point and
|
||||
conversion set (the packed double arithmetic, the scalar SD/SS forms —
|
||||
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, and the
|
||||
width-changing conversions, including the `VCVTPD2DQ`/`VCVTTPD2DQ` family
|
||||
whose length follows the wider source operand). Every encoding is validated two ways: by
|
||||
plus the common AVX-512 F/BW integer set, the floating-point and conversion
|
||||
set (the packed double and single arithmetic, the scalar SD/SS forms —
|
||||
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the
|
||||
replicating moves, and the width-changing conversions, including the
|
||||
`VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source
|
||||
operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
|
||||
and extracts, compares with an opmask destination, the permutes, the
|
||||
expand/compress family, the broadcasts, the opmask-register instructions
|
||||
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
||||
moves and the remaining extending/narrowing moves, the floating-point
|
||||
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||
truncating, including the length-suffixed X/Y spellings and the
|
||||
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
|
||||
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
|
||||
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
|
||||
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||
not the data register. The EVEX mnemonic
|
||||
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
||||
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
||||
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
||||
and scales disp8 by the element size), and combine with the .Z zeroing
|
||||
suffix. Every encoding is validated two ways: by
|
||||
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
|
||||
the machine code the real Go assembler emits — a comparison that holds for
|
||||
whole functions: all 27 functions of both kernels assemble to exactly the Go
|
||||
@@ -264,6 +284,31 @@ references and the implicit funcdata/DWARF symbols remain future work (the
|
||||
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
||||
remaining EVEX forms and the other architectures.
|
||||
|
||||
### `verify`
|
||||
|
||||
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
|
||||
executable memory and invokes them directly, enabling differential testing,
|
||||
runtime ABI checks and coverage profiling.
|
||||
|
||||
The execution model is pure Go (stdlib only). `Map` copies machine code into
|
||||
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
|
||||
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
|
||||
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
|
||||
block after it, switches to that stack via `enterJIT` (which saves the Go
|
||||
stack pointer in a package global and jumps to the target), and recovers
|
||||
control when the function RETs into `leaveJIT` (which restores the Go stack
|
||||
and returns). A 64-byte pad below the return address accommodates the
|
||||
ABIInternal wrapper that the Go runtime interposes on assembly functions.
|
||||
|
||||
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
|
||||
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
|
||||
by name. The image must be self-contained (no external relocations); the
|
||||
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
|
||||
|
||||
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
||||
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||
arguments to confirm the trampoline round-trips.
|
||||
|
||||
## Extension points
|
||||
|
||||
- **New architecture:** add an entry to the generator in `_gen`, run
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
# Deferred decisions
|
||||
|
||||
Design decisions deliberately postponed, with enough context to pick them up
|
||||
again without re-deriving the analysis. Each entry records what is deferred,
|
||||
why, the options on the table, and the trigger that should reopen it.
|
||||
|
||||
---
|
||||
|
||||
## GOOBJ external (cross-package) symbol references
|
||||
|
||||
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
||||
symbols defined in the file being assembled; a reference to any other symbol
|
||||
is rejected.
|
||||
|
||||
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
||||
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
||||
symbol in the *referenced package's* symbol-definition table. That ordering
|
||||
is not derivable from the reference site — it lives in the referenced
|
||||
package's gc export data (the iexport binary format, which evolves with the
|
||||
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
||||
import, so emitting external references means either parsing export data
|
||||
ourselves or taking a dependency that does.
|
||||
|
||||
**What works today.** Single-package objects: every symbol the file defines
|
||||
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
||||
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
||||
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
||||
references the assembly symbols, never the reverse. Such a package builds
|
||||
with its assembly object replaced by a gasm-emitted one.
|
||||
|
||||
**The options, when we return.**
|
||||
|
||||
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
||||
The straightforward path: read each imported package's export file
|
||||
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
||||
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
||||
the export files' build IDs) and positional references. Robust across
|
||||
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
||||
production dependency beyond the standard library, an explicit deviation
|
||||
from the "production code depends only on the standard library"
|
||||
principle in the README. Requires the user's explicit agreement.
|
||||
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
||||
Substantial effort and inherently fragile: the format is an internal
|
||||
contract that changes with Go releases, so the parser needs a
|
||||
version-gated fallback and regression tests against several toolchains.
|
||||
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
||||
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
||||
toolchain command exposes a package's symbols *in definition-index
|
||||
order* — `go tool nm` sorts differently — so this does not solve the
|
||||
core problem on its own; it would only feed option 1 or 2.
|
||||
|
||||
**Trigger to reopen.** An assembly file that needs a cross-package
|
||||
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
||||
metadata written in assembly), or any kernel that calls into another
|
||||
package directly. Until then, option 3's limitation is moot and the
|
||||
single-package emitter suffices.
|
||||
@@ -3,7 +3,7 @@
|
||||
|
||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||
|
||||
version := "0.12.0"
|
||||
version := "0.18.0"
|
||||
|
||||
default:
|
||||
@just --list
|
||||
|
||||
Vendored
+67
@@ -0,0 +1,67 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func add(a, b int64) int64
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVQ a+0(FP), AX
|
||||
ADDQ b+8(FP), AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func sum(data []int64) int64
|
||||
// Sums all elements of the slice.
|
||||
TEXT ·sum(SB), NOSPLIT, $0-32
|
||||
MOVQ data_base+0(FP), SI
|
||||
MOVQ data_len+8(FP), CX
|
||||
XORQ AX, AX
|
||||
TESTQ CX, CX
|
||||
JZ sum_done
|
||||
|
||||
sum_loop:
|
||||
ADDQ (SI), AX
|
||||
ADDQ $8, SI
|
||||
DECQ CX
|
||||
JNZ sum_loop
|
||||
|
||||
sum_done:
|
||||
MOVQ AX, ret+24(FP)
|
||||
RET
|
||||
|
||||
// func wideCopy(dst, src []byte)
|
||||
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
|
||||
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||
MOVQ dst_base+0(FP), DI
|
||||
MOVQ dst_len+8(FP), BX
|
||||
MOVQ src_base+24(FP), SI
|
||||
MOVQ src_len+32(FP), R8
|
||||
CMPQ BX, R8
|
||||
JLE wc_have_n
|
||||
MOVQ R8, BX
|
||||
|
||||
wc_have_n:
|
||||
CMPQ BX, $32
|
||||
JB wc_small
|
||||
|
||||
VMOVDQU (SI), Y0
|
||||
VMOVDQU Y0, (DI)
|
||||
VMOVDQU -32(SI)(BX*1), Y0
|
||||
VMOVDQU Y0, -32(DI)(BX*1)
|
||||
VZEROUPPER
|
||||
RET
|
||||
|
||||
wc_small:
|
||||
TESTQ BX, BX
|
||||
JZ wc_done
|
||||
|
||||
wc_byte:
|
||||
MOVB (SI), R8B
|
||||
MOVB R8B, (DI)
|
||||
INCQ SI
|
||||
INCQ DI
|
||||
DECQ BX
|
||||
JNZ wc_byte
|
||||
|
||||
wc_done:
|
||||
RET
|
||||
@@ -0,0 +1,77 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build amd64
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// savedSP holds the Go stack pointer while a JIT call is in flight.
|
||||
// Referenced by the assembly trampoline (trampoline_amd64.s).
|
||||
var savedSP uintptr
|
||||
|
||||
// enterJIT switches to the prepared stack and jumps to fn.
|
||||
// It does not return normally; the JIT function's RET transfers control
|
||||
// to leaveJIT, which restores the Go stack.
|
||||
//
|
||||
//go:nosplit
|
||||
func enterJIT(fn uintptr, stack uintptr)
|
||||
|
||||
// leaveJIT restores the Go stack after a JIT function returns.
|
||||
// Its address is placed as the return address on the prepared stack.
|
||||
//
|
||||
//go:nosplit
|
||||
func leaveJIT()
|
||||
|
||||
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
|
||||
var leaveJITAddr uintptr
|
||||
|
||||
func init() {
|
||||
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
|
||||
}
|
||||
|
||||
// stackPad is padding below the return address on the prepared stack.
|
||||
// The ABIInternal wrapper that leaveJIT's address resolves to executes
|
||||
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
|
||||
// bytes below the return-address slot. 64 bytes of headroom is ample.
|
||||
const stackPad = 64
|
||||
|
||||
// Call invokes the assembled function at fnAddr with the given ABI0 argument
|
||||
// block (the raw bytes that would appear at FP+0). It returns the argument
|
||||
// block after the call, which contains any results the function wrote back
|
||||
// (the ABI0 convention shares the argument area for inputs and outputs).
|
||||
//
|
||||
// The function must be NOSPLIT (no stack growth) and must not reference
|
||||
// external symbols — the image is self-contained.
|
||||
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||
// Prepare the stack: [padding][leaveJIT addr][args...]
|
||||
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
|
||||
stackMem, err := syscall.Mmap(-1, 0, stackSize,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: stack mmap: %w", err)
|
||||
}
|
||||
defer syscall.Munmap(stackMem)
|
||||
|
||||
// The return address sits after the padding; the function's SP will
|
||||
// point here, leaving stackPad bytes below for the wrapper's pushes.
|
||||
retOff := stackPad
|
||||
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
|
||||
// The ABI0 argument area follows the return address.
|
||||
copy(stackMem[retOff+8:], args)
|
||||
|
||||
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||
enterJIT(fnAddr, stackBase)
|
||||
|
||||
// Copy out the (possibly modified) argument area.
|
||||
out := make([]byte, len(args))
|
||||
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
//go:build !amd64
|
||||
|
||||
package verify
|
||||
|
||||
import "fmt"
|
||||
|
||||
// Call is unavailable on non-amd64 architectures.
|
||||
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||
return nil, fmt.Errorf("verify: JIT execution requires amd64")
|
||||
}
|
||||
@@ -0,0 +1,295 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
|
||||
// differential-testing oracle. It mirrors the contract of
|
||||
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
|
||||
// 0 = ok, 1 = malformed, 2 = zero offset.
|
||||
func decodeBlockGo(src, dst []byte) (int, int) {
|
||||
if len(src) == 0 {
|
||||
return 0, 1
|
||||
}
|
||||
si, di := 0, 0
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1 // truncated: no token
|
||||
}
|
||||
token := int(src[si])
|
||||
si++
|
||||
|
||||
// Literals.
|
||||
lLen := token >> 4
|
||||
if lLen == 15 {
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
b := int(src[si])
|
||||
si++
|
||||
lLen += b
|
||||
if b != 255 {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if si+lLen > len(src) {
|
||||
return 0, 1 // truncated literals
|
||||
}
|
||||
if di+lLen > len(dst) {
|
||||
return 0, 1 // destination overflow
|
||||
}
|
||||
copy(dst[di:di+lLen], src[si:si+lLen])
|
||||
di += lLen
|
||||
si += lLen
|
||||
|
||||
// End of block.
|
||||
if si >= len(src) {
|
||||
return di, 0
|
||||
}
|
||||
|
||||
// Match offset.
|
||||
if si+2 > len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
offset := int(src[si]) | int(src[si+1])<<8
|
||||
si += 2
|
||||
if offset == 0 {
|
||||
return 0, 2
|
||||
}
|
||||
|
||||
// Match length.
|
||||
mLen := token & 15
|
||||
if mLen == 15 {
|
||||
for {
|
||||
if si >= len(src) {
|
||||
return 0, 1
|
||||
}
|
||||
b := int(src[si])
|
||||
si++
|
||||
mLen += b
|
||||
if b != 255 {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
mLen += 4
|
||||
|
||||
// Copy match (overlapping-safe).
|
||||
if di-offset < 0 {
|
||||
return 0, 1 // offset reaches before dst start
|
||||
}
|
||||
if di+mLen > len(dst) {
|
||||
return 0, 1 // destination overflow
|
||||
}
|
||||
for i := 0; i < mLen; i++ {
|
||||
dst[di+i] = dst[di-offset+i]
|
||||
}
|
||||
di += mLen
|
||||
}
|
||||
}
|
||||
|
||||
// genLZ4Block generates a random valid LZ4 block that decompresses into
|
||||
// approximately wantSize bytes. The block is always well-formed (ends with
|
||||
// a literals-only sequence).
|
||||
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
|
||||
var block []byte
|
||||
produced := 0
|
||||
for produced < wantSize {
|
||||
remaining := wantSize - produced
|
||||
|
||||
// Decide: emit a literals+match sequence or the final literals.
|
||||
if remaining <= 8 || rng.Intn(4) == 0 {
|
||||
// Final literals-only sequence.
|
||||
lLen := remaining
|
||||
if lLen > 60 {
|
||||
lLen = 1 + rng.Intn(60)
|
||||
}
|
||||
block = appendToken(block, lLen, 0)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
produced += lLen
|
||||
break
|
||||
}
|
||||
|
||||
// Literals + match.
|
||||
lLen := rng.Intn(min(16, remaining))
|
||||
if produced+lLen == 0 {
|
||||
lLen = 1 // must have at least 1 literal before the first match
|
||||
}
|
||||
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
|
||||
mLen := mLenRaw + 4
|
||||
if produced+mLen > remaining {
|
||||
mLen = remaining - produced
|
||||
if mLen < 4 {
|
||||
// Not enough room for a match; emit final literals.
|
||||
lLen = remaining
|
||||
block = appendToken(block, lLen, 0)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
break
|
||||
}
|
||||
mLenRaw = mLen - 4
|
||||
}
|
||||
|
||||
block = appendToken(block, lLen, mLenRaw)
|
||||
for i := 0; i < lLen; i++ {
|
||||
block = append(block, byte(rng.Intn(256)))
|
||||
}
|
||||
produced += lLen
|
||||
|
||||
// Offset: must be <= produced (can't reference before start).
|
||||
maxOff := produced
|
||||
if maxOff > 65535 {
|
||||
maxOff = 65535
|
||||
}
|
||||
offset := 1 + rng.Intn(maxOff)
|
||||
block = append(block, byte(offset), byte(offset>>8))
|
||||
produced += mLen
|
||||
}
|
||||
return block
|
||||
}
|
||||
|
||||
// appendToken appends a token (and extension bytes if needed) for the given
|
||||
// literal and match lengths.
|
||||
func appendToken(block []byte, lLen, mLenRaw int) []byte {
|
||||
lit4 := lLen
|
||||
if lit4 > 15 {
|
||||
lit4 = 15
|
||||
}
|
||||
ml4 := mLenRaw
|
||||
if ml4 > 15 {
|
||||
ml4 = 15
|
||||
}
|
||||
block = append(block, byte(lit4<<4|ml4))
|
||||
// Literal extension bytes.
|
||||
rem := lLen - 15
|
||||
for rem >= 255 {
|
||||
block = append(block, 255)
|
||||
rem -= 255
|
||||
}
|
||||
if lLen >= 15 {
|
||||
block = append(block, byte(rem))
|
||||
}
|
||||
// Match extension bytes.
|
||||
rem = mLenRaw - 15
|
||||
for rem >= 255 {
|
||||
block = append(block, 255)
|
||||
rem -= 255
|
||||
}
|
||||
if mLenRaw >= 15 {
|
||||
block = append(block, byte(rem))
|
||||
}
|
||||
return block
|
||||
}
|
||||
|
||||
func min(a, b int) int {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
|
||||
// random valid LZ4 blocks and compares the output bit-for-bit against the
|
||||
// portable Go reference.
|
||||
func TestDifferentialLZ4Fuzz(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
const iterations = 5000
|
||||
rng := rand.New(rand.NewSource(42))
|
||||
|
||||
for i := 0; i < iterations; i++ {
|
||||
wantSize := 1 + rng.Intn(4096)
|
||||
src := genLZ4Block(rng, wantSize)
|
||||
dstSize := wantSize + 64 // generous destination
|
||||
|
||||
// Go reference.
|
||||
goDst := make([]byte, dstSize)
|
||||
goN, goCode := decodeBlockGo(src, goDst)
|
||||
|
||||
// JIT kernel.
|
||||
jitDst := make([]byte, dstSize)
|
||||
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||
|
||||
if jitCode != goCode {
|
||||
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||
i, jitCode, goCode, len(src))
|
||||
}
|
||||
if jitCode != 0 {
|
||||
continue // both agree it's malformed/zero-offset
|
||||
}
|
||||
if jitN != goN {
|
||||
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||
i, jitN, goN, len(src))
|
||||
}
|
||||
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
||||
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
|
||||
// that error codes agree with the Go reference (no crashes, same classification).
|
||||
func TestDifferentialLZ4Hostile(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
const iterations = 2000
|
||||
rng := rand.New(rand.NewSource(99))
|
||||
|
||||
for i := 0; i < iterations; i++ {
|
||||
srcLen := rng.Intn(128)
|
||||
src := make([]byte, srcLen)
|
||||
rng.Read(src)
|
||||
dstSize := rng.Intn(512)
|
||||
dst := make([]byte, dstSize)
|
||||
|
||||
// Go reference.
|
||||
goDst := make([]byte, dstSize)
|
||||
copy(goDst, dst)
|
||||
_, goCode := decodeBlockGo(src, goDst)
|
||||
|
||||
// JIT kernel.
|
||||
jitDst := make([]byte, dstSize)
|
||||
copy(jitDst, dst)
|
||||
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||
|
||||
if jitCode != goCode {
|
||||
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
|
||||
i, jitCode, goCode, srcLen, dstSize)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
|
||||
// dst size (for hostile tests where dst may be smaller than the output).
|
||||
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||
t.Helper()
|
||||
args := make([]byte, 64)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if len(dst) > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||
}
|
||||
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package verify provides the dynamic-analysis substrate for gasm: it
|
||||
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
|
||||
// directly, enabling differential testing against portable Go references,
|
||||
// runtime ABI checks and basic-block coverage profiling.
|
||||
//
|
||||
// The execution model is pure Go (stdlib only): machine code is mapped with
|
||||
// syscall.Mmap and invoked through an assembly trampoline that switches to a
|
||||
// prepared ABI0 stack. No cgo, no external toolchain.
|
||||
package verify
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// Executable maps a copy of code into a read-execute memory region suitable
|
||||
// for direct invocation. The mapping is anonymous and private; the original
|
||||
// slice is not retained. Call Unmap to release the region.
|
||||
type Executable struct {
|
||||
addr uintptr // base address of the mapping
|
||||
size int
|
||||
mem []byte // the mmap'd slice (for Unmap)
|
||||
}
|
||||
|
||||
// Map copies code into a freshly allocated RX region and returns it.
|
||||
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
|
||||
// copy, matching W^X policy.
|
||||
func Map(code []byte) (*Executable, error) {
|
||||
size := len(code)
|
||||
if size == 0 {
|
||||
return nil, fmt.Errorf("verify: cannot map zero-length code")
|
||||
}
|
||||
// Round up to the page size.
|
||||
const pageSize = 4096
|
||||
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
|
||||
|
||||
mem, err := syscall.Mmap(-1, 0, mapSize,
|
||||
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: mmap: %w", err)
|
||||
}
|
||||
copy(mem, code)
|
||||
|
||||
// Remove write permission (W^X).
|
||||
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
|
||||
syscall.Munmap(mem)
|
||||
return nil, fmt.Errorf("verify: mprotect: %w", err)
|
||||
}
|
||||
return &Executable{
|
||||
addr: uintptr(unsafe.Pointer(&mem[0])),
|
||||
size: size,
|
||||
mem: mem,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Unmap releases the executable region.
|
||||
func (e *Executable) Unmap() {
|
||||
if e.mem != nil {
|
||||
syscall.Munmap(e.mem)
|
||||
e.mem = nil
|
||||
}
|
||||
}
|
||||
|
||||
// FuncAddr returns the absolute address of a function at the given offset
|
||||
// within the mapped image.
|
||||
func (e *Executable) FuncAddr(offset int) uintptr {
|
||||
return e.addr + uintptr(offset)
|
||||
}
|
||||
|
||||
// PutUint64 writes v into buf at byte offset off (little-endian).
|
||||
func PutUint64(buf []byte, off int, v uint64) {
|
||||
binary.LittleEndian.PutUint64(buf[off:off+8], v)
|
||||
}
|
||||
|
||||
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
|
||||
func GetUint64(buf []byte, off int) uint64 {
|
||||
return binary.LittleEndian.Uint64(buf[off : off+8])
|
||||
}
|
||||
|
||||
// PutPtr writes a pointer value into buf at byte offset off.
|
||||
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
|
||||
}
|
||||
@@ -0,0 +1,159 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
func loadBasic(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
func TestJITAdd(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
a, b, want int64
|
||||
}{
|
||||
{0, 0, 0},
|
||||
{1, 2, 3},
|
||||
{-1, 1, 0},
|
||||
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
|
||||
{-100, -200, -300},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
args := make([]byte, 24)
|
||||
PutUint64(args, 0, uint64(tt.a))
|
||||
PutUint64(args, 8, uint64(tt.b))
|
||||
|
||||
out, err := k.CallFunc("add", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
|
||||
}
|
||||
got := int64(GetUint64(out, 16))
|
||||
if got != tt.want {
|
||||
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJITSum(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
data []int64
|
||||
want int64
|
||||
}{
|
||||
{nil, 0},
|
||||
{[]int64{1}, 1},
|
||||
{[]int64{1, 2, 3, 4, 5}, 15},
|
||||
{[]int64{-10, 20, -30, 40}, 20},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
args := make([]byte, 32)
|
||||
if len(tt.data) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(tt.data)))
|
||||
PutUint64(args, 16, uint64(cap(tt.data)))
|
||||
|
||||
out, err := k.CallFunc("sum", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
||||
}
|
||||
got := int64(GetUint64(out, 24))
|
||||
if got != tt.want {
|
||||
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJITWideCopy(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
n int
|
||||
}{
|
||||
{"empty", 0},
|
||||
{"tiny", 7},
|
||||
{"exact32", 32},
|
||||
{"overlap_range", 48},
|
||||
{"exact64", 64},
|
||||
{"unaligned", 45},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
src := make([]byte, tt.n)
|
||||
for i := range src {
|
||||
src[i] = byte(i * 7)
|
||||
}
|
||||
dst := make([]byte, tt.n)
|
||||
|
||||
args := make([]byte, 48)
|
||||
if tt.n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(tt.n)) // dst_len
|
||||
PutUint64(args, 16, uint64(tt.n)) // dst_cap
|
||||
PutUint64(args, 32, uint64(tt.n)) // src_len
|
||||
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
||||
|
||||
_, err := k.CallFunc("wideCopy", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(wideCopy): %v", err)
|
||||
}
|
||||
if !bytes.Equal(dst, src) {
|
||||
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelFuncNames(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
names := k.FuncNames()
|
||||
want := []string{"add", "sum", "wideCopy"}
|
||||
if len(names) != len(want) {
|
||||
t.Fatalf("FuncNames() = %v, want %v", names, want)
|
||||
}
|
||||
for i, n := range names {
|
||||
if n != want[i] {
|
||||
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelFuncNotFound(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
_, err := k.CallFunc("nonexistent", make([]byte, 8))
|
||||
if err == nil {
|
||||
t.Fatal("expected error for nonexistent function")
|
||||
}
|
||||
}
|
||||
|
||||
func TestKernelArgTooSmall(t *testing.T) {
|
||||
k := loadBasic(t)
|
||||
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
|
||||
if err == nil {
|
||||
t.Fatal("expected error for too-small arg block")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMapZeroLength(t *testing.T) {
|
||||
_, err := Map(nil)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for zero-length code")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,160 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"testing"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
|
||||
// integration testing. The test is skipped when the file is absent
|
||||
// (e.g. in CI without the sibling checkout).
|
||||
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
|
||||
|
||||
func loadLZ4Kernel(t *testing.T) *Kernel {
|
||||
t.Helper()
|
||||
if _, err := os.Stat(lz4KernelPath); err != nil {
|
||||
t.Skipf("sibling kernel not available: %v", err)
|
||||
}
|
||||
k, err := Load(lz4KernelPath)
|
||||
if err != nil {
|
||||
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
|
||||
}
|
||||
t.Cleanup(k.Close)
|
||||
return k
|
||||
}
|
||||
|
||||
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
|
||||
// given src and dst buffers, returning (n, code).
|
||||
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||
t.Helper()
|
||||
args := make([]byte, 64)
|
||||
if len(src) > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(len(src)))
|
||||
PutUint64(args, 16, uint64(cap(src)))
|
||||
if len(dst) > 0 {
|
||||
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||
}
|
||||
PutUint64(args, 32, uint64(len(dst)))
|
||||
PutUint64(args, 40, uint64(cap(dst)))
|
||||
|
||||
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||
}
|
||||
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||
}
|
||||
|
||||
func TestLZ4DecodeKnownAnswers(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
src []byte
|
||||
dstSize int
|
||||
wantDst []byte
|
||||
wantN int
|
||||
wantCode int
|
||||
}{
|
||||
{
|
||||
name: "literals_only",
|
||||
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
|
||||
dstSize: 16,
|
||||
wantDst: []byte("Hello"),
|
||||
wantN: 5,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "literals_and_match",
|
||||
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
|
||||
dstSize: 32,
|
||||
wantDst: []byte("AAAAAAAAAAAAABBB"),
|
||||
wantN: 16,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "overlapping_match",
|
||||
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
|
||||
// then final 1 literal 'Y'.
|
||||
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
|
||||
dstSize: 16,
|
||||
wantDst: []byte("XXXXXXXXXY"),
|
||||
wantN: 10,
|
||||
wantCode: 0,
|
||||
},
|
||||
{
|
||||
name: "malformed_truncated",
|
||||
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
|
||||
dstSize: 16,
|
||||
wantN: 0,
|
||||
wantCode: 1,
|
||||
},
|
||||
{
|
||||
name: "zero_offset",
|
||||
src: []byte{0x14, 'X', 0x00, 0x00},
|
||||
dstSize: 16,
|
||||
wantN: 0,
|
||||
wantCode: 2,
|
||||
},
|
||||
{
|
||||
name: "empty_token",
|
||||
src: []byte{0x00}, // 0 literals, end of block
|
||||
dstSize: 16,
|
||||
wantDst: nil,
|
||||
wantN: 0,
|
||||
wantCode: 0,
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
dst := make([]byte, tt.dstSize)
|
||||
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
|
||||
if n != tt.wantN || code != tt.wantCode {
|
||||
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
|
||||
n, code, tt.wantN, tt.wantCode)
|
||||
}
|
||||
if tt.wantCode == 0 && tt.wantDst != nil {
|
||||
if !bytes.Equal(dst[:n], tt.wantDst) {
|
||||
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLZ4WideCopyAVX2(t *testing.T) {
|
||||
k := loadLZ4Kernel(t)
|
||||
|
||||
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
|
||||
for _, n := range sizes {
|
||||
src := make([]byte, n)
|
||||
for i := range src {
|
||||
src[i] = byte(i*13 + 7)
|
||||
}
|
||||
dst := make([]byte, n)
|
||||
|
||||
args := make([]byte, 48)
|
||||
if n > 0 {
|
||||
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||
}
|
||||
PutUint64(args, 8, uint64(n))
|
||||
PutUint64(args, 16, uint64(n))
|
||||
PutUint64(args, 32, uint64(n))
|
||||
PutUint64(args, 40, uint64(n))
|
||||
|
||||
_, err := k.CallFunc("wideCopyAVX2", args)
|
||||
if err != nil {
|
||||
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
|
||||
}
|
||||
if !bytes.Equal(dst, src) {
|
||||
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
|
||||
// stack and jumps to the assembled function; when the function RETs, control
|
||||
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
|
||||
//
|
||||
// The prepared stack must begin with the address of leaveJIT (the return
|
||||
// address the JIT function will pop), followed by the function's ABI0
|
||||
// argument area.
|
||||
//
|
||||
// Single-threaded: savedSP is a package global, so only one JIT call may be
|
||||
// in flight at a time. gasm verify runs sequentially.
|
||||
|
||||
// func enterJIT(fn uintptr, stack uintptr)
|
||||
// Switches to the prepared stack and jumps to fn. Does not return normally;
|
||||
// the JIT function's RET transfers control to leaveJIT.
|
||||
TEXT ·enterJIT(SB), NOSPLIT, $0-16
|
||||
MOVQ fn+0(FP), AX // target function address (before SP switch)
|
||||
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
|
||||
MOVQ stack+8(FP), SP // switch to the prepared stack
|
||||
JMP AX
|
||||
|
||||
// func leaveJIT()
|
||||
// Restores the Go stack pointer and returns to enterJIT's caller.
|
||||
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
|
||||
MOVQ ·savedSP(SB), SP
|
||||
RET
|
||||
@@ -0,0 +1,106 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package verify
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// Kernel is a JIT-loaded assembly image ready for direct invocation.
|
||||
// It wraps an executable memory mapping and the function layout metadata
|
||||
// needed to marshal ABI0 calls.
|
||||
type Kernel struct {
|
||||
exec *Executable
|
||||
img *asm.Image
|
||||
funcs map[string]int // function name → index into img.Funcs
|
||||
}
|
||||
|
||||
// Load parses, assembles and maps a .s file into executable memory.
|
||||
// The returned Kernel is ready for Call. The caller must call Close to
|
||||
// release the mapping.
|
||||
func Load(path string) (*Kernel, error) {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: %w", err)
|
||||
}
|
||||
return LoadSource(path, string(src))
|
||||
}
|
||||
|
||||
// LoadSource parses, assembles and maps assembly source into executable memory.
|
||||
func LoadSource(filename, src string) (*Kernel, error) {
|
||||
file, errs := parser.Parse(filename, src)
|
||||
if len(errs) > 0 {
|
||||
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
|
||||
}
|
||||
return LoadAST(file)
|
||||
}
|
||||
|
||||
// LoadAST assembles a parsed AST file and maps the result into executable
|
||||
// memory.
|
||||
func LoadAST(file *ast.File) (*Kernel, error) {
|
||||
img, err := asm.AssembleFile(file)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("verify: assemble: %w", err)
|
||||
}
|
||||
if len(img.Externals) > 0 {
|
||||
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
|
||||
}
|
||||
code := img.Bytes()
|
||||
exec, err := Map(code)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
funcs := make(map[string]int, len(img.Funcs))
|
||||
for i, f := range img.Funcs {
|
||||
funcs[f.Name] = i
|
||||
}
|
||||
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
|
||||
}
|
||||
|
||||
// Func returns the layout metadata for the named function.
|
||||
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
return k.img.Funcs[idx], nil
|
||||
}
|
||||
|
||||
// FuncNames returns the names of all functions in the kernel, in source order.
|
||||
func (k *Kernel) FuncNames() []string {
|
||||
names := make([]string, len(k.img.Funcs))
|
||||
for i, f := range k.img.Funcs {
|
||||
names[i] = f.Name
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// CallFunc invokes the named function with the given ABI0 argument block.
|
||||
// The arg block is the raw bytes of the function's argument/result area
|
||||
// (as declared by the TEXT $frame-args suffix). Returns the arg block
|
||||
// after the call (with any results written back by the function).
|
||||
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
|
||||
idx, ok := k.funcs[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||
}
|
||||
fl := k.img.Funcs[idx]
|
||||
if len(args) < fl.Args {
|
||||
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||
}
|
||||
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||
return Call(fnAddr, args)
|
||||
}
|
||||
|
||||
// Close releases the executable mapping.
|
||||
func (k *Kernel) Close() {
|
||||
if k.exec != nil {
|
||||
k.exec.Unmap()
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user