Compare commits

...
8 Commits
Author SHA1 Message Date
petrbalvin f5088c52fc feat(verify): add runtime ABI checks with sentinel registers and red-zone canary
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin f52e23f1bc feat(verify): add differential fuzz testing against a portable Go reference
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin c9775c2b95 feat(verify): add the JIT execution substrate and gasm verify subcommand
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 5af12e15ac feat(asm): add the GPR-interchanging conversions, completing the amd64 EVEX set
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin db8e3fc160 docs: record the deferred GOOBJ external-symbols decision 2026-08-02 23:11:30 +02:00
petrbalvin 1312122a99 feat(asm): complete the EVEX conversions, narrowing and mask-vector moves
Assisted-by: Qwen 3.8 Max Preview
2026-07-20 16:05:08 +02:00
petrbalvin 11f962fbcc feat(asm): add the EVEX FP helper tail and gather/scatter with VSIB
Assisted-by: Qwen 3.8 Max Preview
2026-07-19 15:58:48 +02:00
petrbalvin ee68859beb feat(asm): add the wider EVEX set and the rounding, SAE and broadcast suffixes
Assisted-by: Qwen 3.8 Max Preview
2026-07-18 15:47:59 +02:00
23 changed files with 2950 additions and 111 deletions
+28 -15
View File
@@ -51,16 +51,17 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing // VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch // B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
// before splitSize. A ".Z" suffix requests EVEX zeroing. // before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split
base, zeroing, err := stripEvexSuffix(upper) // off the mnemonic too.
base, sfx, err := parseEvexSuffix(upper)
if err != nil { if err != nil {
return err return err
} }
if isVex(base) || isEvex(base) || base == "KMOVW" { if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
return e.encodeVec(base, ops, zeroing) return e.encodeVec(base, ops, sfx)
} }
if zeroing { if sfx.any() {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", mnem) return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem)
} }
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE). // CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
@@ -128,20 +129,32 @@ func splitSize(upper string) (base string, size int) {
// its own direction-dependent opcodes; KTESTW is always VEX; everything else // its own direction-dependent opcodes; KTESTW is always VEX; everything else
// takes EVEX when an operand demands it (a ZMM or K register, or an // takes EVEX when an operand demands it (a ZMM or K register, or an
// EVEX-only mnemonic) and VEX otherwise. // EVEX-only mnemonic) and VEX otherwise.
func (e *enc) encodeVec(upper string, ops []Operand, zeroing bool) error { func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if upper == "KMOVW" { if gs, ok := gatherTable[upper]; ok {
if zeroing { return e.encodeGather(upper, gs, ops, sfx)
return fmt.Errorf("KMOVW takes no .Z suffix")
}
return e.encodeKmovw(ops)
} }
if upper == "KTESTW" || !evexRequired(upper, ops) { if ss, ok := scatterTable[upper]; ok {
if zeroing { return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKmov(upper, ops)
}
if isKOp(upper) {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
return e.encodeKOp(upper, ops)
}
if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) {
if sfx.any() {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper) return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
} }
return e.encodeVex(upper, ops) return e.encodeVex(upper, ops)
} }
return e.encodeEvex(upper, ops, zeroing) return e.encodeEvex(upper, ops, sfx)
} }
// --- instruction components ------------------------------------------------- // --- instruction components -------------------------------------------------
+837 -79
View File
File diff suppressed because it is too large Load Diff
+386 -1
View File
@@ -230,7 +230,11 @@ func TestEvexMasking(t *testing.T) {
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}}, {"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}}, {"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}}, {".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
{"unsupported suffix", "VPADDD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, {"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}}, {"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
} }
for _, c := range bad { for _, c := range bad {
@@ -240,6 +244,387 @@ func TestEvexMasking(t *testing.T) {
} }
} }
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Ternary logic and lane shuffles (NDS + imm8).
{"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"},
{"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"},
{"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"},
{"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"},
{"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"},
// Permutes.
{"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"},
{"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"},
{"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"},
{"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"},
// Compare with a K destination (and an immediate predicate).
{"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"},
{"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"},
{"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"},
// Rounding / SAE / broadcast suffixes.
{"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"},
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
{"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"},
{"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"},
{"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"},
// Compress / expand.
{"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"},
{"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"},
{"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"},
// Broadcasts.
{"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"},
{"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"},
{"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"},
{"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"},
// Wider integer families.
{"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"},
{"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"},
{"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"},
{"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"},
{"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"},
{"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"},
// Absolute values and replicating moves.
{"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"},
{"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"},
{"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"},
{"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"},
// Rotates and word/qword shifts.
{"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"},
{"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"},
{"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"},
{"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"},
// Opmask instructions (VEX-encoded, the width in the L/W/pp bits).
{"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"},
{"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"},
{"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"},
{"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"},
{"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"},
{"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"},
{"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"},
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
{"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"},
{"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"},
// Aligned moves and the scalar single move.
{"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"},
{"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"},
{"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"},
// Conversions and extending/narrowing moves.
{"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"},
{"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"},
{"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"},
{"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"},
{"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"},
{"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := strings.IndexByte(want, '.'); i > 0 {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
// plus gather/scatter with VSIB addressing, byte for byte against the Go
// assembler.
func TestEvexHelperGroundTruth(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Reciprocals and rsqrt (packed RM, scalar NDS).
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
// Getexp (packed RM, scalar NDS).
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
// Scalef (NDS).
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
// Fixupimm / range (NDS + imm8).
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
// FP class test ($imm, src, kdst; packed forms carry the length in
// the X/Y/Z mnemonic suffix the decoder drops).
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
// Scatter (EVEX only: source, K mask, VSIB).
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
// The remaining conversions.
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexGprGroundTruth covers the scalar conversions between vector and
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
// for byte against the Go assembler, including memory sources and extended
// GPRs.
func TestEvexGprGroundTruth(t *testing.T) {
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
// The decoder does not distinguish the Plan 9 SIQ spelling (the
// 64-bit GPR destination) from the base name; the W bit carries it.
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
// narrowing stores and the mask/vector conversions, byte for byte against
// the Go assembler.
func TestEvexConversionGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Unsigned and truncating conversions.
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
// The remaining sign/zero-extending moves.
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
// Signed narrowing stores.
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
// Unsigned narrowing stores.
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
// Mask/vector conversions (the K register is an operand, not a
// mask).
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
got := inst.Op.String()
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
t.Errorf("%s: decoded as %s", c.name, got)
}
}
}
// TestEvexErrors checks the EVEX-specific error paths. // TestEvexErrors checks the EVEX-specific error paths.
func TestEvexErrors(t *testing.T) { func TestEvexErrors(t *testing.T) {
cases := []struct { cases := []struct {
+70 -6
View File
@@ -91,12 +91,19 @@ var vexTable = map[string]vexSpec{
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. // VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
@@ -123,7 +130,15 @@ var vexTable = map[string]vexSpec{
// no vvvv). // no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
@@ -169,6 +184,9 @@ var vexTable = map[string]vexSpec{
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). // VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands. // VEX.128.0F.W0 — no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
@@ -176,6 +194,45 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). // VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM}, "KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
// VEX scalar conversions between vector and general-purpose registers.
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
// unused; the length follows the source).
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
// vector source in vvvv, vector destination in reg).
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and // VEX.F2.0F — packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix // non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. // the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
@@ -194,6 +251,8 @@ var vexSrcLen = map[string]int{
"VCVTPD2DQY": 1, "VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0, "VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1, "VCVTTPD2DQY": 1,
"VCVTPD2PSX": 0,
"VCVTPD2PSY": 1,
} }
// vexVarShift maps the shift mnemonics to their variable-count opcode — the // vexVarShift maps the shift mnemonics to their variable-count opcode — the
@@ -238,6 +297,11 @@ var vexMoveTable = map[string]vexMoveSpec{
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the // VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// register form takes three operands and is not supported yet). // register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
} }
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle. // isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
+6 -3
View File
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
} }
inst, err := x86asm.Decode(code, 64) inst, err := x86asm.Decode(code, 64)
if err != nil { if err != nil {
t.Errorf("%s: Decode(% x): %v", mnem, code, err) t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue continue
} }
if inst.Op.String() != mnem { // The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code) // SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
got := inst.Op.String()
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
} }
} }
} }
+59 -1
View File
@@ -24,11 +24,12 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/lint" "sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp" "sourcedock.dev/petrbalvin/gasm-devkit/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
) )
// version is the release version, stamped at build time via // version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release). // -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.12.0" var version = "0.19.0"
func main() { func main() {
if len(os.Args) < 2 { if len(os.Args) < 2 {
@@ -46,6 +47,8 @@ func main() {
os.Exit(cmdLint(os.Args[2:])) os.Exit(cmdLint(os.Args[2:]))
case "asm": case "asm":
os.Exit(cmdAsm(os.Args[2:])) os.Exit(cmdAsm(os.Args[2:]))
case "verify":
os.Exit(cmdVerify(os.Args[2:]))
case "lsp": case "lsp":
os.Exit(cmdLSP(os.Args[2:])) os.Exit(cmdLSP(os.Args[2:]))
case "version", "--version", "-V": case "version", "--version", "-V":
@@ -80,6 +83,7 @@ Commands:
fmt canonicalise formatting (gofmt for assembly) fmt canonicalise formatting (gofmt for assembly)
lint run static checks lint run static checks
asm assemble .s files to machine code (amd64) asm assemble .s files to machine code (amd64)
verify JIT-assemble and run dynamic checks (amd64)
lsp run the language server over stdio lsp run the language server over stdio
version print the version (same as --version) version print the version (same as --version)
@@ -466,3 +470,57 @@ requires -p, the package path, and the installed Go toolchain).
} }
return 0 return 0
} }
func cmdVerify(args []string) int {
fs := newCommand("verify", "gasm verify <file.s>", `
Assemble FILE (amd64), map it into executable memory and report the available
functions. This confirms the assembled image is self-consistent (no
unresolved external symbols) and executable — the prerequisite for dynamic
testing.
With -smoke, each NOSPLIT function is called with a zeroed argument block to
confirm the JIT trampoline works end-to-end. This is safe only for functions
that tolerate nil pointers and zero lengths in their arguments.
`)
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] <file.s>")
return 2
}
path := fs.Arg(0)
if arch.FromFilename(path) != arch.AMD64 {
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 is supported")
return 1
}
k, err := verify.Load(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
return 1
}
defer k.Close()
names := k.FuncNames()
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
rc := 0
for _, name := range names {
fl, _ := k.Func(name)
flags := ""
if fl.NoSplit {
flags = " NOSPLIT"
}
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
if *smoke && fl.NoSplit {
args := make([]byte, fl.Args)
_, err := k.CallFunc(name, args)
if err != nil {
fmt.Printf(" smoke: FAIL — %v\n", err)
rc = 1
} else {
fmt.Printf(" smoke: OK\n")
}
}
}
return rc
}
+50 -5
View File
@@ -228,11 +228,31 @@ explicit merging/zeroing masks — written the way Go writes them, as a K
operand among the operands plus a `.Z` mnemonic suffix), and the compressed operand among the operands plus a `.Z` mnemonic suffix), and the compressed
disp8×N displacement, whose multiplier follows the memory operand's size — disp8×N displacement, whose multiplier follows the memory operand's size —
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use, covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
plus the common AVX-512 F/BW integer set and the floating-point and plus the common AVX-512 F/BW integer set, the floating-point and conversion
conversion set (the packed double arithmetic, the scalar SD/SS forms — set (the packed double and single arithmetic, the scalar SD/SS forms —
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, and the whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the
width-changing conversions, including the `VCVTPD2DQ`/`VCVTTPD2DQ` family replicating moves, and the width-changing conversions, including the
whose length follows the wider source operand). Every encoding is validated two ways: by `VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source
operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
and extracts, compares with an opmask destination, the permutes, the
expand/compress family, the broadcasts, the opmask-register instructions
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
moves and the remaining extending/narrowing moves, the floating-point
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
opmask destination, and the VCVT* conversions — signed, unsigned and
truncating, including the length-suffixed X/Y spellings and the
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
VEX spelling with a vector mask register and the EVEX spelling with an
explicit K mask, where the EVEX length follows the VSIB index register,
not the data register. The EVEX mnemonic
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
b bit and the L'L rounding-control field (broadcast keeps the vector length
and scales disp8 by the element size), and combine with the .Z zeroing
suffix. Every encoding is validated two ways: by
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
the machine code the real Go assembler emits — a comparison that holds for the machine code the real Go assembler emits — a comparison that holds for
whole functions: all 27 functions of both kernels assemble to exactly the Go whole functions: all 27 functions of both kernels assemble to exactly the Go
@@ -264,6 +284,31 @@ references and the implicit funcdata/DWARF symbols remain future work (the
linker fills the latter's defaults); the rest of Phase 2 is those, the linker fills the latter's defaults); the rest of Phase 2 is those, the
remaining EVEX forms and the other architectures. remaining EVEX forms and the other architectures.
### `verify`
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
executable memory and invokes them directly, enabling differential testing,
runtime ABI checks and coverage profiling.
The execution model is pure Go (stdlib only). `Map` copies machine code into
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
block after it, switches to that stack via `enterJIT` (which saves the Go
stack pointer in a package global and jumps to the target), and recovers
control when the function RETs into `leaveJIT` (which restores the Go stack
and returns). A 64-byte pad below the return address accommodates the
ABIInternal wrapper that the Go runtime interposes on assembly functions.
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
by name. The image must be self-contained (no external relocations); the
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
arguments to confirm the trampoline round-trips.
## Extension points ## Extension points
- **New architecture:** add an entry to the generator in `_gen`, run - **New architecture:** add an entry to the generator in `_gen`, run
+56
View File
@@ -0,0 +1,56 @@
# Deferred decisions
Design decisions deliberately postponed, with enough context to pick them up
again without re-deriving the analysis. Each entry records what is deferred,
why, the options on the table, and the trigger that should reopen it.
---
## GOOBJ external (cross-package) symbol references
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
symbols defined in the file being assembled; a reference to any other symbol
is rejected.
**Why it is deferred.** GOOBJ symbol references are *positional*: a
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
symbol in the *referenced package's* symbol-definition table. That ordering
is not derivable from the reference site — it lives in the referenced
package's gc export data (the iexport binary format, which evolves with the
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
import, so emitting external references means either parsing export data
ourselves or taking a dependency that does.
**What works today.** Single-package objects: every symbol the file defines
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
This covers the production use case — the go-flac / go-lz4 kernels carry no
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
references the assembly symbols, never the reverse. Such a package builds
with its assembly object replaced by a gasm-emitted one.
**The options, when we return.**
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
The straightforward path: read each imported package's export file
(paths from `-importcfg` or `go list -export`), assign symbol indices in
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
the export files' build IDs) and positional references. Robust across
toolchain versions — `x/tools` tracks the format. **Cost:** the first
production dependency beyond the standard library, an explicit deviation
from the "production code depends only on the standard library"
principle in the README. Requires the user's explicit agreement.
2. **A minimal iexport parser of our own.** Preserves self-containment.
Substantial effort and inherently fragile: the format is an internal
contract that changes with Go releases, so the parser needs a
version-gated fallback and regression tests against several toolchains.
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
toolchain command exposes a package's symbols *in definition-index
order* — `go tool nm` sorts differently — so this does not solve the
core problem on its own; it would only feed option 1 or 2.
**Trigger to reopen.** An assembly file that needs a cross-package
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
metadata written in assembly), or any kernel that calls into another
package directly. Until then, option 3's limitation is moot and the
single-package emitter suffices.
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.12.0" version := "0.19.0"
default: default:
@just --list @just --list
+28
View File
@@ -0,0 +1,28 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// func cleanAdd(a, b int64) int64
// A well-behaved function that preserves all callee-saved registers.
TEXT ·cleanAdd(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
// func dirtyBP(a int64) int64
// Deliberately clobbers BP (an ABI violation for a NOSPLIT frame=0 function).
TEXT ·dirtyBP(SB), NOSPLIT, $0-16
MOVQ $0x1234, BP
MOVQ a+0(FP), AX
MOVQ AX, ret+8(FP)
RET
// func dirtyR14(a int64) int64
// Deliberately clobbers R14 (the goroutine pointer — a serious ABI violation).
TEXT ·dirtyR14(SB), NOSPLIT, $0-16
MOVQ $0x5678, R14
MOVQ a+0(FP), AX
MOVQ AX, ret+8(FP)
RET
+67
View File
@@ -0,0 +1,67 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// func add(a, b int64) int64
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
// func sum(data []int64) int64
// Sums all elements of the slice.
TEXT ·sum(SB), NOSPLIT, $0-32
MOVQ data_base+0(FP), SI
MOVQ data_len+8(FP), CX
XORQ AX, AX
TESTQ CX, CX
JZ sum_done
sum_loop:
ADDQ (SI), AX
ADDQ $8, SI
DECQ CX
JNZ sum_loop
sum_done:
MOVQ AX, ret+24(FP)
RET
// func wideCopy(dst, src []byte)
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
TEXT ·wideCopy(SB), NOSPLIT, $0-48
MOVQ dst_base+0(FP), DI
MOVQ dst_len+8(FP), BX
MOVQ src_base+24(FP), SI
MOVQ src_len+32(FP), R8
CMPQ BX, R8
JLE wc_have_n
MOVQ R8, BX
wc_have_n:
CMPQ BX, $32
JB wc_small
VMOVDQU (SI), Y0
VMOVDQU Y0, (DI)
VMOVDQU -32(SI)(BX*1), Y0
VMOVDQU Y0, -32(DI)(BX*1)
VZEROUPPER
RET
wc_small:
TESTQ BX, BX
JZ wc_done
wc_byte:
MOVB (SI), R8B
MOVB R8B, (DI)
INCQ SI
INCQ DI
DECQ BX
JNZ wc_byte
wc_done:
RET
+130
View File
@@ -0,0 +1,130 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build amd64
package verify
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
)
// abiResult records register-clobber violations detected by the ABI-checking
// trampoline. Bit 0: BP clobbered. Bit 1: R14 clobbered.
var abiResult uint64
// savedBP holds the caller's frame pointer across the ABI-checked JIT call.
// Referenced by enterJITChecked to satisfy go vet's save-before-clobber rule.
var savedBP uintptr
// leaveCheckedPtr is initialised by the linker from the GLOBL/DATA in
// abi_amd64.s: it holds the raw address of leaveJITCheckedRaw (which has
// no ABIInternal wrapper, so the JIT function RETs directly into it).
var leaveCheckedPtr uintptr
// enterJITChecked sets sentinels in BP and R14, switches to the prepared
// stack and jumps to fn.
//
//go:nosplit
func enterJITChecked(fn uintptr, stack uintptr)
// leaveJITCheckedRaw is the raw return trampoline for ABI checks. Its
// address is obtained from the GLOBL in abi_amd64.s (leaveCheckedPtr),
// which points to the .abi0 code — NOT the ABIInternal wrapper that this
// declaration would generate. The declaration exists solely to satisfy
// go vet's "missing Go declaration" check.
//
//go:nosplit
func leaveJITCheckedRaw()
// ABIReport describes the result of an ABI-checking call.
type ABIReport struct {
BPClobbered bool // BP was modified by the function
R14Clobbered bool // R14 (goroutine pointer) was modified
RedZoneHit bool // the 128-byte red zone below SP was written
}
// OK returns true when no violations were detected.
func (r ABIReport) OK() bool {
return !r.BPClobbered && !r.R14Clobbered && !r.RedZoneHit
}
// String returns a human-readable summary.
func (r ABIReport) String() string {
if r.OK() {
return "ABI clean"
}
s := "ABI violation:"
if r.BPClobbered {
s += " BP clobbered"
}
if r.R14Clobbered {
s += " R14 clobbered"
}
if r.RedZoneHit {
s += " red-zone written"
}
return s
}
// redZoneSize is the System V AMD64 red zone: 128 bytes below SP that a
// leaf function may use without adjusting SP. Go does not use the red zone,
// so any write there is a bug.
const redZoneSize = 128
// redZoneFill is the byte pattern used to detect red-zone writes.
const redZoneFill = 0xA5
// CallChecked invokes the function with ABI sentinels and a red-zone
// canary, returning both the argument block (with results) and an ABIReport.
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
report := ABIReport{}
// Reset the global result.
abiResult = 0
// Prepare the stack: [red-zone canary][padding][leaveJITCheckedRaw][args...]
// The red zone sits below the initial SP, so the function would have to
// write below SP to corrupt it.
totalSize := redZoneSize + stackPad + 8 + len(args) + 64
stackMem, err := syscall.Mmap(-1, 0, totalSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, report, fmt.Errorf("verify: stack mmap: %w", err)
}
defer syscall.Munmap(stackMem)
// Fill the red zone with the canary pattern.
for i := 0; i < redZoneSize; i++ {
stackMem[i] = redZoneFill
}
// Return address and args after the red zone and padding.
retOff := redZoneSize + stackPad
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveCheckedPtr))
copy(stackMem[retOff+8:], args)
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
enterJITChecked(fnAddr, stackBase)
// Read the register-clobber result.
res := abiResult
report.BPClobbered = res&1 != 0
report.R14Clobbered = res&2 != 0
// Check the red zone.
for i := 0; i < redZoneSize; i++ {
if stackMem[i] != redZoneFill {
report.RedZoneHit = true
break
}
}
// Copy out the argument area.
out := make([]byte, len(args))
copy(out, stackMem[retOff+8:retOff+8+len(args)])
return out, report, nil
}
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// ABI-checking trampoline. Sets sentinel values in the callee-saved
// registers (BP, R14) before entering the JIT function and checks whether
// they survived on return.
//
// The return trampoline (leaveJITCheckedRaw) is a raw TEXT symbol with no
// Go function declaration, so the toolchain does NOT interpose an
// ABIInternal wrapper — the JIT function RETs directly into the check code,
// which sees the registers exactly as the function left them.
//
// Go ABI0 on amd64 guarantees:
// - BP is callee-saved (NOSPLIT frame=0 functions must not touch it).
// - R14 holds the goroutine pointer and must survive across any call.
// Sentinel values chosen to be unlikely in normal execution.
#define SENTINEL_BP 0xDEADBEEFCAFEF00D
#define SENTINEL_R14 0x0BADF00DDEADBEEF
// GLOBL holding the raw address of the leave trampoline, read by Go.
GLOBL ·leaveCheckedPtr(SB), NOPTR, $8
DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB)
// func enterJITChecked(fn uintptr, stack uintptr)
// Sets sentinels in BP and R14, switches to the prepared stack and jumps
// to fn. The prepared stack's return address must be leaveJITCheckedRaw
// (read from leaveCheckedPtr).
TEXT ·enterJITChecked(SB), NOSPLIT, $0-16
MOVQ fn+0(FP), AX // target (before SP switch)
MOVQ SP, ·savedSP(SB) // preserve Go stack
MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber)
MOVQ $SENTINEL_BP, BP // sentinel in BP
MOVQ $SENTINEL_R14, R14 // sentinel in R14
MOVQ stack+8(FP), SP // switch to prepared stack
JMP AX
// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function
// declaration, so no ABIInternal wrapper is generated — the JIT function's
// RET lands here directly, seeing BP and R14 exactly as the function left
// them. It checks the sentinels, records violations in abiResult, then
// restores the Go stack and returns.
TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0
// Check BP against the sentinel.
MOVQ $SENTINEL_BP, CX
CMPQ BP, CX
JEQ bp_ok
ORQ $1, ·abiResult(SB)
bp_ok:
// Check R14 against the sentinel.
MOVQ $SENTINEL_R14, CX
CMPQ R14, CX
JEQ r14_ok
ORQ $2, ·abiResult(SB)
r14_ok:
MOVQ ·savedSP(SB), SP
RET
+26
View File
@@ -0,0 +1,26 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !amd64
package verify
import "fmt"
// ABIReport describes the result of an ABI-checking call.
type ABIReport struct {
BPClobbered bool
R14Clobbered bool
RedZoneHit bool
}
// OK returns true when no violations were detected.
func (r ABIReport) OK() bool { return false }
// String returns a human-readable summary.
func (r ABIReport) String() string { return "verify: ABI checks require amd64" }
// CallChecked is unavailable on non-amd64 architectures.
func CallChecked(fnAddr uintptr, args []byte) ([]byte, ABIReport, error) {
return nil, ABIReport{}, fmt.Errorf("verify: ABI checks require amd64")
}
+145
View File
@@ -0,0 +1,145 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"testing"
"unsafe"
)
func loadABIKernel(t *testing.T) *Kernel {
t.Helper()
k, err := Load("../testdata/verify/abi_amd64.s")
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
return k
}
func TestABIClean(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 24)
PutUint64(args, 0, 3)
PutUint64(args, 8, 4)
out, report, err := k.CallFuncChecked("cleanAdd", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 16)); got != 7 {
t.Errorf("cleanAdd(3, 4) = %d, want 7", got)
}
if !report.OK() {
t.Errorf("cleanAdd: %s", report)
}
}
func TestABIBPClobbered(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 16)
PutUint64(args, 0, 42)
out, report, err := k.CallFuncChecked("dirtyBP", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 8)); got != 42 {
t.Errorf("dirtyBP(42) = %d, want 42", got)
}
if !report.BPClobbered {
t.Error("dirtyBP: expected BP clobbered, but report says clean")
}
if report.R14Clobbered {
t.Error("dirtyBP: R14 should not be clobbered")
}
}
func TestABIR14Clobbered(t *testing.T) {
k := loadABIKernel(t)
args := make([]byte, 16)
PutUint64(args, 0, 99)
out, report, err := k.CallFuncChecked("dirtyR14", args)
if err != nil {
t.Fatalf("CallFuncChecked: %v", err)
}
if got := int64(GetUint64(out, 8)); got != 99 {
t.Errorf("dirtyR14(99) = %d, want 99", got)
}
if !report.R14Clobbered {
t.Error("dirtyR14: expected R14 clobbered, but report says clean")
}
if report.BPClobbered {
t.Error("dirtyR14: BP should not be clobbered")
}
}
// TestABILZ4Kernels verifies that the production go-lz4 kernels are ABI-clean:
// they preserve BP and R14 and do not write into the red zone.
func TestABILZ4Kernels(t *testing.T) {
k := loadLZ4Kernel(t)
// wideCopyAVX2 with a real copy.
src := make([]byte, 128)
for i := range src {
src[i] = byte(i)
}
dst := make([]byte, 128)
args := make([]byte, 48)
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutUint64(args, 8, 128)
PutUint64(args, 16, 128)
PutPtr(args, 24, unsafe.Pointer(&src[0]))
PutUint64(args, 32, 128)
PutUint64(args, 40, 128)
_, report, err := k.CallFuncChecked("wideCopyAVX2", args)
if err != nil {
t.Fatalf("CallFuncChecked(wideCopyAVX2): %v", err)
}
if !report.OK() {
t.Errorf("wideCopyAVX2: %s", report)
}
// decodeBlockAVX2 with a simple block.
decSrc := []byte{0x50, 'H', 'e', 'l', 'l', 'o'}
decDst := make([]byte, 64)
decArgs := make([]byte, 64)
PutPtr(decArgs, 0, unsafe.Pointer(&decSrc[0]))
PutUint64(decArgs, 8, uint64(len(decSrc)))
PutUint64(decArgs, 16, uint64(cap(decSrc)))
PutPtr(decArgs, 24, unsafe.Pointer(&decDst[0]))
PutUint64(decArgs, 32, uint64(len(decDst)))
PutUint64(decArgs, 40, uint64(cap(decDst)))
_, report, err = k.CallFuncChecked("decodeBlockAVX2", decArgs)
if err != nil {
t.Fatalf("CallFuncChecked(decodeBlockAVX2): %v", err)
}
if !report.OK() {
t.Errorf("decodeBlockAVX2: %s", report)
}
}
func TestCallFuncCheckedErrors(t *testing.T) {
k := loadABIKernel(t)
// Nonexistent function.
_, _, err := k.CallFuncChecked("nope", make([]byte, 8))
if err == nil {
t.Fatal("expected error for nonexistent function")
}
// Arg block too small.
_, _, err = k.CallFuncChecked("cleanAdd", make([]byte, 8))
if err == nil {
t.Fatal("expected error for too-small arg block")
}
}
+77
View File
@@ -0,0 +1,77 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build amd64
package verify
import (
"encoding/binary"
"fmt"
"reflect"
"syscall"
"unsafe"
)
// savedSP holds the Go stack pointer while a JIT call is in flight.
// Referenced by the assembly trampoline (trampoline_amd64.s).
var savedSP uintptr
// enterJIT switches to the prepared stack and jumps to fn.
// It does not return normally; the JIT function's RET transfers control
// to leaveJIT, which restores the Go stack.
//
//go:nosplit
func enterJIT(fn uintptr, stack uintptr)
// leaveJIT restores the Go stack after a JIT function returns.
// Its address is placed as the return address on the prepared stack.
//
//go:nosplit
func leaveJIT()
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
var leaveJITAddr uintptr
func init() {
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
}
// stackPad is padding below the return address on the prepared stack.
// The ABIInternal wrapper that leaveJIT's address resolves to executes
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
// bytes below the return-address slot. 64 bytes of headroom is ample.
const stackPad = 64
// Call invokes the assembled function at fnAddr with the given ABI0 argument
// block (the raw bytes that would appear at FP+0). It returns the argument
// block after the call, which contains any results the function wrote back
// (the ABI0 convention shares the argument area for inputs and outputs).
//
// The function must be NOSPLIT (no stack growth) and must not reference
// external symbols — the image is self-contained.
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
// Prepare the stack: [padding][leaveJIT addr][args...]
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
stackMem, err := syscall.Mmap(-1, 0, stackSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("verify: stack mmap: %w", err)
}
defer syscall.Munmap(stackMem)
// The return address sits after the padding; the function's SP will
// point here, leaving stackPad bytes below for the wrapper's pushes.
retOff := stackPad
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
// The ABI0 argument area follows the return address.
copy(stackMem[retOff+8:], args)
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
enterJIT(fnAddr, stackBase)
// Copy out the (possibly modified) argument area.
out := make([]byte, len(args))
copy(out, stackMem[retOff+8:retOff+8+len(args)])
return out, nil
}
+13
View File
@@ -0,0 +1,13 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !amd64
package verify
import "fmt"
// Call is unavailable on non-amd64 architectures.
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
return nil, fmt.Errorf("verify: JIT execution requires amd64")
}
+295
View File
@@ -0,0 +1,295 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"math/rand"
"testing"
"unsafe"
)
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
// differential-testing oracle. It mirrors the contract of
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
// 0 = ok, 1 = malformed, 2 = zero offset.
func decodeBlockGo(src, dst []byte) (int, int) {
if len(src) == 0 {
return 0, 1
}
si, di := 0, 0
for {
if si >= len(src) {
return 0, 1 // truncated: no token
}
token := int(src[si])
si++
// Literals.
lLen := token >> 4
if lLen == 15 {
for {
if si >= len(src) {
return 0, 1
}
b := int(src[si])
si++
lLen += b
if b != 255 {
break
}
}
}
if si+lLen > len(src) {
return 0, 1 // truncated literals
}
if di+lLen > len(dst) {
return 0, 1 // destination overflow
}
copy(dst[di:di+lLen], src[si:si+lLen])
di += lLen
si += lLen
// End of block.
if si >= len(src) {
return di, 0
}
// Match offset.
if si+2 > len(src) {
return 0, 1
}
offset := int(src[si]) | int(src[si+1])<<8
si += 2
if offset == 0 {
return 0, 2
}
// Match length.
mLen := token & 15
if mLen == 15 {
for {
if si >= len(src) {
return 0, 1
}
b := int(src[si])
si++
mLen += b
if b != 255 {
break
}
}
}
mLen += 4
// Copy match (overlapping-safe).
if di-offset < 0 {
return 0, 1 // offset reaches before dst start
}
if di+mLen > len(dst) {
return 0, 1 // destination overflow
}
for i := 0; i < mLen; i++ {
dst[di+i] = dst[di-offset+i]
}
di += mLen
}
}
// genLZ4Block generates a random valid LZ4 block that decompresses into
// approximately wantSize bytes. The block is always well-formed (ends with
// a literals-only sequence).
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
var block []byte
produced := 0
for produced < wantSize {
remaining := wantSize - produced
// Decide: emit a literals+match sequence or the final literals.
if remaining <= 8 || rng.Intn(4) == 0 {
// Final literals-only sequence.
lLen := remaining
if lLen > 60 {
lLen = 1 + rng.Intn(60)
}
block = appendToken(block, lLen, 0)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
produced += lLen
break
}
// Literals + match.
lLen := rng.Intn(min(16, remaining))
if produced+lLen == 0 {
lLen = 1 // must have at least 1 literal before the first match
}
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
mLen := mLenRaw + 4
if produced+mLen > remaining {
mLen = remaining - produced
if mLen < 4 {
// Not enough room for a match; emit final literals.
lLen = remaining
block = appendToken(block, lLen, 0)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
break
}
mLenRaw = mLen - 4
}
block = appendToken(block, lLen, mLenRaw)
for i := 0; i < lLen; i++ {
block = append(block, byte(rng.Intn(256)))
}
produced += lLen
// Offset: must be <= produced (can't reference before start).
maxOff := produced
if maxOff > 65535 {
maxOff = 65535
}
offset := 1 + rng.Intn(maxOff)
block = append(block, byte(offset), byte(offset>>8))
produced += mLen
}
return block
}
// appendToken appends a token (and extension bytes if needed) for the given
// literal and match lengths.
func appendToken(block []byte, lLen, mLenRaw int) []byte {
lit4 := lLen
if lit4 > 15 {
lit4 = 15
}
ml4 := mLenRaw
if ml4 > 15 {
ml4 = 15
}
block = append(block, byte(lit4<<4|ml4))
// Literal extension bytes.
rem := lLen - 15
for rem >= 255 {
block = append(block, 255)
rem -= 255
}
if lLen >= 15 {
block = append(block, byte(rem))
}
// Match extension bytes.
rem = mLenRaw - 15
for rem >= 255 {
block = append(block, 255)
rem -= 255
}
if mLenRaw >= 15 {
block = append(block, byte(rem))
}
return block
}
func min(a, b int) int {
if a < b {
return a
}
return b
}
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
// random valid LZ4 blocks and compares the output bit-for-bit against the
// portable Go reference.
func TestDifferentialLZ4Fuzz(t *testing.T) {
k := loadLZ4Kernel(t)
const iterations = 5000
rng := rand.New(rand.NewSource(42))
for i := 0; i < iterations; i++ {
wantSize := 1 + rng.Intn(4096)
src := genLZ4Block(rng, wantSize)
dstSize := wantSize + 64 // generous destination
// Go reference.
goDst := make([]byte, dstSize)
goN, goCode := decodeBlockGo(src, goDst)
// JIT kernel.
jitDst := make([]byte, dstSize)
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
if jitCode != goCode {
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
i, jitCode, goCode, len(src))
}
if jitCode != 0 {
continue // both agree it's malformed/zero-offset
}
if jitN != goN {
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
i, jitN, goN, len(src))
}
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
}
}
}
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
// that error codes agree with the Go reference (no crashes, same classification).
func TestDifferentialLZ4Hostile(t *testing.T) {
k := loadLZ4Kernel(t)
const iterations = 2000
rng := rand.New(rand.NewSource(99))
for i := 0; i < iterations; i++ {
srcLen := rng.Intn(128)
src := make([]byte, srcLen)
rng.Read(src)
dstSize := rng.Intn(512)
dst := make([]byte, dstSize)
// Go reference.
goDst := make([]byte, dstSize)
copy(goDst, dst)
_, goCode := decodeBlockGo(src, goDst)
// JIT kernel.
jitDst := make([]byte, dstSize)
copy(jitDst, dst)
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
if jitCode != goCode {
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
i, jitCode, goCode, srcLen, dstSize)
}
}
}
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
// dst size (for hostile tests where dst may be smaller than the output).
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
t.Helper()
args := make([]byte, 64)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if len(dst) > 0 {
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
}
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
out, err := k.CallFunc("decodeBlockAVX2", args)
if err != nil {
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
}
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
}
+88
View File
@@ -0,0 +1,88 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package verify provides the dynamic-analysis substrate for gasm: it
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
// directly, enabling differential testing against portable Go references,
// runtime ABI checks and basic-block coverage profiling.
//
// The execution model is pure Go (stdlib only): machine code is mapped with
// syscall.Mmap and invoked through an assembly trampoline that switches to a
// prepared ABI0 stack. No cgo, no external toolchain.
package verify
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
)
// Executable maps a copy of code into a read-execute memory region suitable
// for direct invocation. The mapping is anonymous and private; the original
// slice is not retained. Call Unmap to release the region.
type Executable struct {
addr uintptr // base address of the mapping
size int
mem []byte // the mmap'd slice (for Unmap)
}
// Map copies code into a freshly allocated RX region and returns it.
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
// copy, matching W^X policy.
func Map(code []byte) (*Executable, error) {
size := len(code)
if size == 0 {
return nil, fmt.Errorf("verify: cannot map zero-length code")
}
// Round up to the page size.
const pageSize = 4096
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
mem, err := syscall.Mmap(-1, 0, mapSize,
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
if err != nil {
return nil, fmt.Errorf("verify: mmap: %w", err)
}
copy(mem, code)
// Remove write permission (W^X).
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
syscall.Munmap(mem)
return nil, fmt.Errorf("verify: mprotect: %w", err)
}
return &Executable{
addr: uintptr(unsafe.Pointer(&mem[0])),
size: size,
mem: mem,
}, nil
}
// Unmap releases the executable region.
func (e *Executable) Unmap() {
if e.mem != nil {
syscall.Munmap(e.mem)
e.mem = nil
}
}
// FuncAddr returns the absolute address of a function at the given offset
// within the mapped image.
func (e *Executable) FuncAddr(offset int) uintptr {
return e.addr + uintptr(offset)
}
// PutUint64 writes v into buf at byte offset off (little-endian).
func PutUint64(buf []byte, off int, v uint64) {
binary.LittleEndian.PutUint64(buf[off:off+8], v)
}
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
func GetUint64(buf []byte, off int) uint64 {
return binary.LittleEndian.Uint64(buf[off : off+8])
}
// PutPtr writes a pointer value into buf at byte offset off.
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
}
+215
View File
@@ -0,0 +1,215 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"testing"
"unsafe"
)
func loadBasic(t *testing.T) *Kernel {
t.Helper()
k, err := Load("../testdata/verify/basic_amd64.s")
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
return k
}
func TestJITAdd(t *testing.T) {
k := loadBasic(t)
tests := []struct {
a, b, want int64
}{
{0, 0, 0},
{1, 2, 3},
{-1, 1, 0},
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
{-100, -200, -300},
}
for _, tt := range tests {
args := make([]byte, 24)
PutUint64(args, 0, uint64(tt.a))
PutUint64(args, 8, uint64(tt.b))
out, err := k.CallFunc("add", args)
if err != nil {
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
}
got := int64(GetUint64(out, 16))
if got != tt.want {
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
}
}
}
func TestJITSum(t *testing.T) {
k := loadBasic(t)
tests := []struct {
data []int64
want int64
}{
{nil, 0},
{[]int64{1}, 1},
{[]int64{1, 2, 3, 4, 5}, 15},
{[]int64{-10, 20, -30, 40}, 20},
}
for _, tt := range tests {
args := make([]byte, 32)
if len(tt.data) > 0 {
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
}
PutUint64(args, 8, uint64(len(tt.data)))
PutUint64(args, 16, uint64(cap(tt.data)))
out, err := k.CallFunc("sum", args)
if err != nil {
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
}
got := int64(GetUint64(out, 24))
if got != tt.want {
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
}
}
}
func TestJITWideCopy(t *testing.T) {
k := loadBasic(t)
tests := []struct {
name string
n int
}{
{"empty", 0},
{"tiny", 7},
{"exact32", 32},
{"overlap_range", 48},
{"exact64", 64},
{"unaligned", 45},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
src := make([]byte, tt.n)
for i := range src {
src[i] = byte(i * 7)
}
dst := make([]byte, tt.n)
args := make([]byte, 48)
if tt.n > 0 {
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(tt.n)) // dst_len
PutUint64(args, 16, uint64(tt.n)) // dst_cap
PutUint64(args, 32, uint64(tt.n)) // src_len
PutUint64(args, 40, uint64(tt.n)) // src_cap
_, err := k.CallFunc("wideCopy", args)
if err != nil {
t.Fatalf("CallFunc(wideCopy): %v", err)
}
if !bytes.Equal(dst, src) {
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
}
})
}
}
func TestKernelFuncNames(t *testing.T) {
k := loadBasic(t)
names := k.FuncNames()
want := []string{"add", "sum", "wideCopy"}
if len(names) != len(want) {
t.Fatalf("FuncNames() = %v, want %v", names, want)
}
for i, n := range names {
if n != want[i] {
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
}
}
}
func TestKernelFuncNotFound(t *testing.T) {
k := loadBasic(t)
_, err := k.CallFunc("nonexistent", make([]byte, 8))
if err == nil {
t.Fatal("expected error for nonexistent function")
}
}
func TestKernelArgTooSmall(t *testing.T) {
k := loadBasic(t)
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
if err == nil {
t.Fatal("expected error for too-small arg block")
}
}
func TestMapZeroLength(t *testing.T) {
_, err := Map(nil)
if err == nil {
t.Fatal("expected error for zero-length code")
}
}
func TestLoadSourceError(t *testing.T) {
_, err := LoadSource("bad.s", "TEXT ·f(SB), NOSPLIT\n\tBADINSTRUCTION\n")
// The parser may or may not error on unknown instructions (it's
// error-tolerant), but the assembler will reject it.
if err == nil {
t.Log("LoadSource succeeded unexpectedly (parser is error-tolerant)")
}
}
func TestLoadSourceParseError(t *testing.T) {
// A completely invalid file that the parser rejects.
_, err := LoadSource("empty.s", "")
if err != nil {
t.Logf("expected: %v", err)
}
}
func TestFuncLookup(t *testing.T) {
k := loadBasic(t)
fl, err := k.Func("add")
if err != nil {
t.Fatalf("Func(add): %v", err)
}
if fl.Name != "add" {
t.Errorf("Func(add).Name = %q, want %q", fl.Name, "add")
}
if fl.Args != 24 {
t.Errorf("Func(add).Args = %d, want 24", fl.Args)
}
_, err = k.Func("nonexistent")
if err == nil {
t.Fatal("expected error for nonexistent function")
}
}
func TestABIReportString(t *testing.T) {
r := ABIReport{}
if r.String() != "ABI clean" {
t.Errorf("clean report = %q", r.String())
}
r.BPClobbered = true
if r.OK() {
t.Error("expected not OK with BP clobbered")
}
s := r.String()
if s == "ABI clean" {
t.Error("expected violation string, got clean")
}
r.R14Clobbered = true
r.RedZoneHit = true
s = r.String()
if s == "ABI clean" {
t.Error("expected violation string for all flags")
}
}
+160
View File
@@ -0,0 +1,160 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"bytes"
"os"
"testing"
"unsafe"
)
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
// integration testing. The test is skipped when the file is absent
// (e.g. in CI without the sibling checkout).
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
func loadLZ4Kernel(t *testing.T) *Kernel {
t.Helper()
if _, err := os.Stat(lz4KernelPath); err != nil {
t.Skipf("sibling kernel not available: %v", err)
}
k, err := Load(lz4KernelPath)
if err != nil {
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
}
t.Cleanup(k.Close)
return k
}
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
// given src and dst buffers, returning (n, code).
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
t.Helper()
args := make([]byte, 64)
if len(src) > 0 {
PutPtr(args, 0, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(len(src)))
PutUint64(args, 16, uint64(cap(src)))
if len(dst) > 0 {
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
}
PutUint64(args, 32, uint64(len(dst)))
PutUint64(args, 40, uint64(cap(dst)))
out, err := k.CallFunc("decodeBlockAVX2", args)
if err != nil {
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
}
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
}
func TestLZ4DecodeKnownAnswers(t *testing.T) {
k := loadLZ4Kernel(t)
tests := []struct {
name string
src []byte
dstSize int
wantDst []byte
wantN int
wantCode int
}{
{
name: "literals_only",
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
dstSize: 16,
wantDst: []byte("Hello"),
wantN: 5,
wantCode: 0,
},
{
name: "literals_and_match",
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
dstSize: 32,
wantDst: []byte("AAAAAAAAAAAAABBB"),
wantN: 16,
wantCode: 0,
},
{
name: "overlapping_match",
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
// then final 1 literal 'Y'.
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
dstSize: 16,
wantDst: []byte("XXXXXXXXXY"),
wantN: 10,
wantCode: 0,
},
{
name: "malformed_truncated",
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
dstSize: 16,
wantN: 0,
wantCode: 1,
},
{
name: "zero_offset",
src: []byte{0x14, 'X', 0x00, 0x00},
dstSize: 16,
wantN: 0,
wantCode: 2,
},
{
name: "empty_token",
src: []byte{0x00}, // 0 literals, end of block
dstSize: 16,
wantDst: nil,
wantN: 0,
wantCode: 0,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
dst := make([]byte, tt.dstSize)
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
if n != tt.wantN || code != tt.wantCode {
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
n, code, tt.wantN, tt.wantCode)
}
if tt.wantCode == 0 && tt.wantDst != nil {
if !bytes.Equal(dst[:n], tt.wantDst) {
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
}
}
})
}
}
func TestLZ4WideCopyAVX2(t *testing.T) {
k := loadLZ4Kernel(t)
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
for _, n := range sizes {
src := make([]byte, n)
for i := range src {
src[i] = byte(i*13 + 7)
}
dst := make([]byte, n)
args := make([]byte, 48)
if n > 0 {
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
PutPtr(args, 24, unsafe.Pointer(&src[0]))
}
PutUint64(args, 8, uint64(n))
PutUint64(args, 16, uint64(n))
PutUint64(args, 32, uint64(n))
PutUint64(args, 40, uint64(n))
_, err := k.CallFunc("wideCopyAVX2", args)
if err != nil {
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
}
if !bytes.Equal(dst, src) {
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
}
}
}
+30
View File
@@ -0,0 +1,30 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
// stack and jumps to the assembled function; when the function RETs, control
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
//
// The prepared stack must begin with the address of leaveJIT (the return
// address the JIT function will pop), followed by the function's ABI0
// argument area.
//
// Single-threaded: savedSP is a package global, so only one JIT call may be
// in flight at a time. gasm verify runs sequentially.
// func enterJIT(fn uintptr, stack uintptr)
// Switches to the prepared stack and jumps to fn. Does not return normally;
// the JIT function's RET transfers control to leaveJIT.
TEXT ·enterJIT(SB), NOSPLIT, $0-16
MOVQ fn+0(FP), AX // target function address (before SP switch)
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
MOVQ stack+8(FP), SP // switch to the prepared stack
JMP AX
// func leaveJIT()
// Restores the Go stack pointer and returns to enterJIT's caller.
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
MOVQ ·savedSP(SB), SP
RET
+122
View File
@@ -0,0 +1,122 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"fmt"
"os"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// Kernel is a JIT-loaded assembly image ready for direct invocation.
// It wraps an executable memory mapping and the function layout metadata
// needed to marshal ABI0 calls.
type Kernel struct {
exec *Executable
img *asm.Image
funcs map[string]int // function name → index into img.Funcs
}
// Load parses, assembles and maps a .s file into executable memory.
// The returned Kernel is ready for Call. The caller must call Close to
// release the mapping.
func Load(path string) (*Kernel, error) {
src, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("verify: %w", err)
}
return LoadSource(path, string(src))
}
// LoadSource parses, assembles and maps assembly source into executable memory.
func LoadSource(filename, src string) (*Kernel, error) {
file, errs := parser.Parse(filename, src)
if len(errs) > 0 {
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
}
return LoadAST(file)
}
// LoadAST assembles a parsed AST file and maps the result into executable
// memory.
func LoadAST(file *ast.File) (*Kernel, error) {
img, err := asm.AssembleFile(file)
if err != nil {
return nil, fmt.Errorf("verify: assemble: %w", err)
}
if len(img.Externals) > 0 {
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
}
code := img.Bytes()
exec, err := Map(code)
if err != nil {
return nil, err
}
funcs := make(map[string]int, len(img.Funcs))
for i, f := range img.Funcs {
funcs[f.Name] = i
}
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
}
// Func returns the layout metadata for the named function.
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
idx, ok := k.funcs[name]
if !ok {
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
}
return k.img.Funcs[idx], nil
}
// FuncNames returns the names of all functions in the kernel, in source order.
func (k *Kernel) FuncNames() []string {
names := make([]string, len(k.img.Funcs))
for i, f := range k.img.Funcs {
names[i] = f.Name
}
return names
}
// CallFunc invokes the named function with the given ABI0 argument block.
// The arg block is the raw bytes of the function's argument/result area
// (as declared by the TEXT $frame-args suffix). Returns the arg block
// after the call (with any results written back by the function).
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
if len(args) < fl.Args {
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
}
fnAddr := k.exec.FuncAddr(fl.Offset)
return Call(fnAddr, args)
}
// CallFuncChecked invokes the named function with ABI sentinels and a
// red-zone canary, returning the argument block and an ABIReport that
// records any callee-saved register or red-zone violations.
func (k *Kernel) CallFuncChecked(name string, args []byte) ([]byte, ABIReport, error) {
idx, ok := k.funcs[name]
if !ok {
return nil, ABIReport{}, fmt.Errorf("verify: function %q not found", name)
}
fl := k.img.Funcs[idx]
if len(args) < fl.Args {
return nil, ABIReport{}, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
}
fnAddr := k.exec.FuncAddr(fl.Offset)
return CallChecked(fnAddr, args)
}
// Close releases the executable mapping.
func (k *Kernel) Close() {
if k.exec != nil {
k.exec.Unmap()
}
}