Compare commits

...
5 Commits
Author SHA1 Message Date
petrbalvin b914c0e390 feat(asm): add the EVEX floating-point and conversion set
Assisted-by: Qwen 3.8 Max Preview
2026-07-15 17:13:28 +02:00
petrbalvin 0f3146ff2c feat(asm): add EVEX masking, zeroing and the AVX-512 F/BW integer set
Assisted-by: Qwen 3.8 Max Preview
2026-07-14 21:03:26 +02:00
petrbalvin 9370f9c3ee feat(cli): standard --help and --version with per-command usage
Assisted-by: Qwen 3.8 Max Preview
2026-07-13 19:50:38 +02:00
petrbalvin e98680597d feat(fmt): go-fmt-style recursive formatting and canonical blank-line layout
Assisted-by: Qwen 3.8 Max Preview
2026-07-12 21:24:41 +02:00
petrbalvin 1a01870695 fix(lint): calibrate register-clobber to the Go ABI and add legacy SSE moves
Assisted-by: Qwen 3.8 Max Preview
2026-07-11 17:36:52 +02:00
18 changed files with 1456 additions and 246 deletions
+10
View File
@@ -156,6 +156,16 @@ func (t *Table) Lookup(mnemonic string) (Instr, bool) {
}
}
}
// amd64 EVEX instructions take a .Z zeroing suffix (masking is written as
// an explicit K operand rather than a suffix); strip it so the base
// instruction is still recognised.
if t.Arch == AMD64 {
if base, ok := strings.CutSuffix(key, ".Z"); ok {
if in, found := t.instrs[base]; found {
return in, true
}
}
}
return Instr{}, false
}
+20 -5
View File
@@ -51,9 +51,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
// before splitSize.
if isVex(upper) || isEvex(upper) || upper == "KMOVW" {
return e.encodeVec(upper, ops)
// before splitSize. A ".Z" suffix requests EVEX zeroing.
base, zeroing, err := stripEvexSuffix(upper)
if err != nil {
return err
}
if isVex(base) || isEvex(base) || base == "KMOVW" {
return e.encodeVec(base, ops, zeroing)
}
if zeroing {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", mnem)
}
// CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE).
@@ -93,6 +100,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
}
@@ -119,14 +128,20 @@ func splitSize(upper string) (base string, size int) {
// its own direction-dependent opcodes; KTESTW is always VEX; everything else
// takes EVEX when an operand demands it (a ZMM or K register, or an
// EVEX-only mnemonic) and VEX otherwise.
func (e *enc) encodeVec(upper string, ops []Operand) error {
func (e *enc) encodeVec(upper string, ops []Operand, zeroing bool) error {
if upper == "KMOVW" {
if zeroing {
return fmt.Errorf("KMOVW takes no .Z suffix")
}
return e.encodeKmovw(ops)
}
if upper == "KTESTW" || !evexRequired(upper, ops) {
if zeroing {
return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper)
}
return e.encodeVex(upper, ops)
}
return e.encodeEvex(upper, ops)
return e.encodeEvex(upper, ops, zeroing)
}
// --- instruction components -------------------------------------------------
+46
View File
@@ -127,6 +127,52 @@ func TestControl(t *testing.T) {
checkOp(t, x86asm.JBE, "JLS", Imm(0))
}
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
// against the Go assembler. wantOp is the decoder's name, which differs from
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
func TestSSEMoveGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
wantOp string
}{
{"MOVOU (SI),X1", "MOVOU", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "f30f6f0e", "MOVDQU"},
{"MOVOU X3,(DI)", "MOVOU", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "f30f7f1f", "MOVDQU"},
{"MOVOU X1,X2", "MOVOU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f6fd1", "MOVDQU"},
{"MOVOU (SI)(BX*4),X9", "MOVOU", []Operand{Idx(SI, BX, 4, 0, 16), vreg(t, "X9")}, "f3440f6f0c9e", "MOVDQU"},
{"MOVO (SI),X1", "MOVO", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f6f0e", "MOVDQA"},
{"MOVO X3,(DI)", "MOVO", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f7f1f", "MOVDQA"},
{"MOVUPS (SI),X1", "MOVUPS", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "0f100e", "MOVUPS"},
{"MOVAPS X3,(DI)", "MOVAPS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "0f291f", "MOVAPS"},
{"MOVUPD (SI),X1", "MOVUPD", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f100e", "MOVUPD"},
{"MOVAPD X3,(DI)", "MOVAPD", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f291f", "MOVAPD"},
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
if inst.Op.String() != c.wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
}
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
+290 -34
View File
@@ -3,14 +3,19 @@
package asm
import "fmt"
import (
"fmt"
"strings"
)
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported —
// the kernels do not use them. K-register operands (mask destinations,
// KMOVW, KTESTW) are.
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
// evexSpec describes one EVEX instruction's encoding parameters. The form
// field reuses the vexForm shapes, which carry over unchanged.
@@ -46,6 +51,31 @@ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -59,6 +89,36 @@ var evexTable = map[string]evexSpec{
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
// encodes.
"VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
@@ -74,6 +134,48 @@ var evexTable = map[string]evexSpec{
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
@@ -115,6 +217,15 @@ type evexMoveSpec struct {
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
}
@@ -149,13 +260,75 @@ func evexRequired(upper string, ops []Operand) bool {
return false
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error {
// stripEvexSuffix splits a ".Z" zeroing suffix off the mnemonic. It is the
// only EVEX suffix supported; Go writes masking as an explicit K operand, not
// a suffix.
func stripEvexSuffix(mnem string) (base string, zeroing bool, err error) {
i := strings.LastIndexByte(mnem, '.')
if i < 0 {
return mnem, false, nil
}
if mnem[i+1:] == "Z" {
return mnem[:i], true, nil
}
return "", false, fmt.Errorf("unsupported EVEX suffix %q", mnem[i+1:])
}
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) {
var rest []Operand
mask := 0
for _, op := range ops {
if r, ok := op.(Reg); ok && r.mask {
if mask != 0 {
return nil, 0, fmt.Errorf("at most one mask register operand")
}
if r.idx == 0 {
return nil, 0, fmt.Errorf("K0 is not a usable mask register")
}
mask = r.idx
continue
}
rest = append(rest, op)
}
return rest, mask, nil
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the
// operands; zeroing comes from the .Z mnemonic suffix and requires a mask.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
// Mask-destination comparisons (VPCMPEQD …, K1): the last operand is the
// destination K register, and any mask sits among the preceding operands.
if spec, ok := evexTable[mnemUpper]; ok && spec.form == vexNDS3 && len(ops) > 0 {
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
rest, mask, err := splitMask(ops[:len(ops)-1])
if err != nil {
return err
}
if zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
return e.encodeEvexNDS3(spec, append(rest, dst), mask, zeroing)
}
}
rest, mask, err := splitMask(ops)
if err != nil {
return err
}
if zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
}
ops = rest
if bs, ok := evexBcastTable[mnemUpper]; ok {
return e.encodeEvexBcast(bs, ops)
return e.encodeEvexBcast(bs, ops, mask, zeroing)
}
if ms, ok := evexMoveTable[mnemUpper]; ok {
return e.encodeEvexMove(mnemUpper, ms, ops)
return e.encodeEvexMove(mnemUpper, ms, ops, mask, zeroing)
}
spec, ok := evexTable[mnemUpper]
if !ok {
@@ -163,17 +336,21 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error {
}
switch spec.form {
case vexNDS3:
return e.encodeEvexNDS3(spec, ops)
return e.encodeEvexNDS3(spec, ops, mask, zeroing)
case vexRM:
return e.encodeEvexRM(spec, ops)
return e.encodeEvexRM(spec, ops, mask, zeroing)
case vexRMRev:
return e.encodeEvexRMRev(spec, ops)
return e.encodeEvexRMRev(spec, ops, mask, zeroing)
case vexImmRM:
return e.encodeEvexImmRM(spec, ops, mask, zeroing)
case vexShiftImm:
return e.encodeEvexShiftImm(spec, ops)
return e.encodeEvexShiftImm(spec, ops, mask, zeroing)
case vexNDS3Imm:
return e.encodeEvexNDS3Imm(spec, ops)
return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing)
case vexExtract:
return e.encodeEvexExtract(spec, ops)
return e.encodeEvexExtract(spec, ops, mask, zeroing)
case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing)
}
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
}
@@ -181,7 +358,7 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand) error {
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
// destination may be an opmask register (VPCMPEQD), in which case the vector
// length comes from the sources.
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
}
@@ -201,12 +378,12 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand) error {
ll = r.vecLenBit()
}
}
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2)
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, zeroing)
}
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
// no vvvv), e.g. VCVTQQ2PD.
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
}
@@ -215,12 +392,42 @@ func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand) error {
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
}
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 3 {
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shuffle control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shuffle destination must be a vector register")
}
ll := dstReg.vecLenBit()
if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit()
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 3 {
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
@@ -241,7 +448,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error {
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg); err != nil {
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, zeroing); err != nil {
return err
}
e.out = append(e.out, immByte)
@@ -250,7 +457,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand) error {
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
// rm=src2, imm8), e.g. VALIGND.
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 4 {
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
}
@@ -271,7 +478,7 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error {
if err != nil {
return err
}
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2); err != nil {
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2, mask, zeroing); err != nil {
return err
}
e.out = append(e.out, immByte)
@@ -280,7 +487,7 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand) error {
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
// destination, imm8), e.g. VEXTRACTI64X4.
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
}
@@ -297,7 +504,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error {
if err != nil {
return err
}
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst); err != nil {
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing); err != nil {
return err
}
e.out = append(e.out, immByte)
@@ -307,7 +514,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand) error {
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go
// assembler.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand) error {
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
@@ -336,7 +543,47 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand) error
return fmt.Errorf("%s needs a vector register operand", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm)
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing)
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("EVEX destination must be a vector register")
}
ll, err := soleLen(spec.n)
if err != nil {
return err
}
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing)
}
// soleLen returns the vector-length index of the single valid slot of n —
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
ll := -1
for i, v := range n {
if v == 0 {
continue
}
if ll >= 0 {
return 0, fmt.Errorf("ambiguous vector-length table %v", n)
}
ll = i
}
if ll < 0 {
return 0, fmt.Errorf("empty vector-length table")
}
return ll, nil
}
// memOperand reports whether op is a memory reference (including a
@@ -351,7 +598,7 @@ func memOperand(op Operand) bool {
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand) error {
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
}
@@ -360,12 +607,12 @@ func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand) error {
if !ok || !srcReg.isVec() {
return fmt.Errorf("EVEX source must be a vector register")
}
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst)
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing)
}
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
// source broadcast to every lane of the vector destination.
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand) error {
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing bool) error {
if len(ops) != 2 {
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
}
@@ -384,14 +631,15 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand) error {
default:
return fmt.Errorf("broadcast source must be a register or memory")
}
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src)
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
}
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
// vvvv register index, or -1 when unused.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand) error {
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, zeroing bool) error {
if ll > 2 {
return fmt.Errorf("invalid vector length")
}
@@ -418,7 +666,8 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand)
var sb *sbRef
switch r := rm.(type) {
case Reg:
// ModRM.mod = 11: rm[3] extends via B̄, rm[4] via X̄.
// ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX
// register-register quirk).
modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7)
sib = -1
if r.idx&8 != 0 {
@@ -427,6 +676,9 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand)
if r.idx&16 != 0 {
xBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
@@ -449,9 +701,13 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand)
return fmt.Errorf("invalid EVEX r/m operand")
}
z := 0
if zeroing {
z = 1
}
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
p2 := byte(ll<<5 | vPrimeBar<<3) // z = 0, b = 0, aaa = 0
p2 := byte(z<<7 | ll<<5 | vPrimeBar<<3 | mask) // z, L'L, b=0, V', aaa
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
+135 -1
View File
@@ -61,6 +61,27 @@ func TestEvexGroundTruth(t *testing.T) {
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
// VMOVDQU64 — the W1 qword variant.
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
// The wider AVX-512 F/BW integer set.
{"VPADDB Z1,Z2,Z3", "VPADDB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48fcd9"},
{"VPSUBW Z1,Z2,Z3", "VPSUBW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f9d9"},
{"VPANDQ Z1,Z2,Z3", "VPANDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed48dbd9"},
{"VPANDND Z1,Z2,Z3", "VPANDND", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dfd9"},
{"VPMULLW Z1,Z2,Z3", "VPMULLW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48d5d9"},
{"VPMINUB Z1,Z2,Z3", "VPMINUB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48dad9"},
{"VPMAXUQ Z1,Z2,Z3", "VPMAXUQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed483fd9"},
{"VPAVGW Z1,Z2,Z3", "VPAVGW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e3d9"},
{"VPSLLVQ Z3,Z1,Z2", "VPSLLVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54847d3"},
{"VPSRAVQ Z3,Z1,Z2", "VPSRAVQ", []Operand{vreg(t, "Z3"), vreg(t, "Z1"), vreg(t, "Z2")}, "62f2f54846d3"},
{"VPSHUFD $0x1B,Z1,Z2", "VPSHUFD", []Operand{Imm(0x1B), vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d4870d11b"},
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
// Conversions and narrowing stores (reg = wide source).
@@ -80,6 +101,32 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
{"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"},
{"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"},
{"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"},
{"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"},
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
{"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"},
// Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice),
// DQ→PD (the destination sets the length).
{"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
{"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"},
{"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"},
{"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"},
// PD→DQ conversions: the source is the wide operand and fixes the
// length (ZMM source → L'L = 10 even with an XMM destination; a
// memory source takes the length the mnemonic's spelling implies).
{"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"},
{"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"},
{"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"},
}
for _, c := range cases {
want := strings.ReplaceAll(c.want, " ", "")
@@ -106,6 +153,93 @@ func TestEvexGroundTruth(t *testing.T) {
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// Masked arithmetic: K anywhere among the operands; .Z sets the z bit.
{"VPADDD.Z merging+zeroing", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f16dcafed9"},
{"VPADDD merging", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49fed9"},
{"VADDPD.Z", "VADDPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1edca58d9"},
{"VPMINSD.Z", "VPMINSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f26dcd39d9"},
{"VPMINSQ.Z", "VPMINSQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K5"), vreg(t, "Z3")}, "62f2edcd39d9"},
// Masked immediate shift (K before the destination).
{"VPSRAD.Z", "VPSRAD.Z", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f165c972e201"},
{"VPSLLD merge", "VPSLLD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1654a72f104"},
// Masked align.
{"VALIGND", "VALIGND", []Operand{Imm(12), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f36d4b03e10c"},
// Masked conversion and extract.
{"VCVTQQ2PD.Z", "VCVTQQ2PD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f1fecae6d9"},
{"VEXTRACTI64X4", "VEXTRACTI64X4", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f3fd4a3bcb01"},
// Masked moves: K sits between the register and memory operands.
{"VMOVDQU8 store", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "K3"), Ptr(SI, 0, 64)}, "62f17f4b7f0e"},
{"VMOVDQU32 load", "VMOVDQU32", []Operand{Ptr(SI, 0, 64), vreg(t, "K4"), vreg(t, "Z1")}, "62f17e4c6f0e"},
{"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"},
// Masked comparison with a K destination: dst K1, mask K2.
{"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"},
// Masked floating point: packed double, the scalar SD/SS forms (which
// exist under EVEX only for masked and zeroing use) and conversions.
{"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"},
{"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"},
{"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"},
{"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"},
{"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"},
{"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"},
{"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"},
{"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"},
{"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"},
{"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
continue
}
want := c.mnem
if i := len(want) - 2; i > 0 && want[i:] == ".Z" {
want = want[:i]
}
if inst.Op.String() != want {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
// Error cases.
bad := []struct {
name string
mnem string
ops []Operand
}{
{"zeroing without mask", "VPADDD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}},
{"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}},
{".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}},
{"unsupported suffix", "VPADDD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}},
{"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}},
}
for _, c := range bad {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestEvexErrors checks the EVEX-specific error paths.
func TestEvexErrors(t *testing.T) {
cases := []struct {
@@ -121,7 +255,7 @@ func TestEvexErrors(t *testing.T) {
{"VPMOVDW src", "VPMOVDW", []Operand{AX, vreg(t, "Y0")}},
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VPSHUFD X16", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}},
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
+60
View File
@@ -661,6 +661,66 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
return e.emit(i)
}
// --- legacy SSE moves --------------------------------------------------------
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
// load opcode (reg = destination, rm = source) and a store opcode (the
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
type sseMove struct {
prefix byte // 0, 0x66, 0xF2 or 0xF3
load byte
store byte
}
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
op := m.store
var reg Reg
var rm Operand
switch {
case srcVec && dstVec:
op = m.load
reg, rm = dstReg, src
case srcVec:
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("SSE move: invalid destination operand")
}
reg, rm = srcReg, dst
case dstVec:
if _, ok := src.(Mem); !ok {
return fmt.Errorf("SSE move: invalid source operand")
}
op = m.load
reg, rm = dstReg, src
default:
return fmt.Errorf("SSE move needs a vector register operand")
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, reg, rm, 8); err != nil {
return err
}
return e.emit(i)
}
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
+84
View File
@@ -43,6 +43,13 @@ const (
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
// conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into
// an XMM destination, so the L bit rides with the wider source. The
// mnemonic's spelling fixes the length (X = 128, Y = 256), which also
// covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv.
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
)
@@ -86,12 +93,29 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
"VMULSD": {1, 0x59, 0, 3, -1, vexNDS3},
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3},
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
@@ -105,6 +129,19 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
@@ -138,6 +175,25 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
// the packed-double → dword conversions) to its fixed vector length:
// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for
// a memory source, matching the Go assembler's ytab.
var vexSrcLen = map[string]int{
"VCVTPD2DQX": 0,
"VCVTPD2DQY": 1,
"VCVTTPD2DQX": 0,
"VCVTTPD2DQY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
@@ -230,6 +286,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
}
@@ -295,6 +353,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, l, regField, rBit, 15, src)
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
ll, ok := vexSrcLen[mnem]
if !ok {
return fmt.Errorf("no fixed vector length for %s", mnem)
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// An unused vvvv field must be stored as all ones (v̄vvv = 1111).
return e.emitVexFields(spec, ll, regField, rBit, 15, src)
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
+103 -60
View File
@@ -160,78 +160,117 @@ func TestVexGroundTruth(t *testing.T) {
mnem string
ops []Operand
want string
wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings)
}{
// Three-operand NDS form.
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"},
{"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""},
{"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""},
{"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""},
{"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""},
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"},
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
{"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""},
{"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""},
{"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""},
{"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""},
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"},
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
{"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""},
{"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""},
{"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""},
{"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""},
{"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""},
// Immediate shifts.
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"},
{"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""},
{"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""},
// Variable-count shifts: the count lives in an XMM register or memory
// and the instruction takes the NDS form.
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0"},
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300"},
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0"},
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0"},
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0"},
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0"},
{"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""},
{"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""},
{"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""},
{"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""},
{"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""},
{"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""},
// Immediate shuffle (reg=dst, rm=src, imm8).
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"},
{"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""},
{"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""},
{"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""},
{"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""},
// Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8).
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"},
{"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""},
{"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""},
{"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""},
{"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""},
// Lane extract (reg=YMM source, rm=XMM/memory destination, imm8).
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"},
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"},
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
{"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""},
{"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""},
{"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""},
{"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""},
{"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""},
{"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""},
{"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""},
{"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""},
{"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""},
{"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""},
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
{"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""},
{"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""},
{"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""},
{"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""},
{"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""},
// Scalar double and single arithmetic (F2 / F3 pp, 128-bit only).
{"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""},
{"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""},
{"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""},
{"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""},
{"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""},
{"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""},
{"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""},
{"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""},
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
{"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""},
// PD→DQ conversions: the X/Y spellings fix the source length and the
// destination is always XMM; the decoder reports the base mnemonic.
{"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"},
{"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"},
{"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"},
{"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"},
{"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"},
// No-operand.
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877"},
{"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
@@ -251,7 +290,11 @@ func TestVexGroundTruth(t *testing.T) {
if inst.Len != len(code) {
t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code))
}
if inst.Op.String() != c.mnem {
wantOp := c.wantOp
if wantOp == "" {
wantOp = c.mnem
}
if inst.Op.String() != wantOp {
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
}
}
+156 -24
View File
@@ -11,7 +11,9 @@ import (
"flag"
"fmt"
"io"
"io/fs"
"os"
"path/filepath"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
@@ -26,7 +28,7 @@ import (
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.5.0"
var version = "0.10.0"
func main() {
if len(os.Args) < 2 {
@@ -47,30 +49,71 @@ func main() {
case "lsp":
os.Exit(cmdLSP(os.Args[2:]))
case "version", "--version", "-V":
fmt.Printf("gasm %s\n", version)
case "help", "-h", "--help":
os.Exit(cmdVersion())
case "help", "--help", "-h":
usage(os.Stdout)
default:
fmt.Fprintf(os.Stderr, "gasm: unknown command %q\n\n", os.Args[1])
usage(os.Stderr)
fmt.Fprintf(os.Stderr, "gasm: unknown command %q — run \"gasm --help\" for usage\n", os.Args[1])
os.Exit(2)
}
}
// cmdVersion prints the release version.
func cmdVersion() int {
fmt.Printf("gasm %s\n", version)
return 0
}
func usage(w io.Writer) {
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler (GAsm)
gasm bundles a lexer, parser, formatter, linter, standalone assembler and
language server for Plan 9 assembly into one self-contained binary.
Usage:
gasm tokens <file> print the lexical token stream
gasm parse <file> parse and report syntax errors
gasm fmt [-w] <file...> canonicalise formatting (-w writes in place)
gasm lint <file...> run static checks
gasm asm [-o out.bin] <file> assemble to machine code (amd64, Phase 2)
gasm lsp run the language server over stdio
gasm version print the version
gasm <command> [arguments]
gasm [flags]
Commands:
tokens print the lexical token stream
parse parse and report syntax errors
fmt canonicalise formatting (gofmt for assembly)
lint run static checks
asm assemble .s files to machine code (amd64)
lsp run the language server over stdio
version print the version (same as --version)
Flags:
-h, --help show this help
-V, --version print the version
Run "gasm <command> -h" for a command's usage and flags.
Examples:
gasm fmt reformat every .s below the current directory
gasm lint go-flac/*.s run static checks over the kernels
gasm asm -o k.bin kern_amd64.s
`, version)
}
// newCommand returns the FlagSet of a subcommand whose -h/--help prints a
// proper usage block: the one-line usage, the long description and the flag
// defaults. The flag package routes -h/--help to fs.Usage and exits 0.
func newCommand(name, usageLine, long string) *flag.FlagSet {
fs := flag.NewFlagSet(name, flag.ExitOnError)
fs.Usage = func() {
w := fs.Output()
fmt.Fprintf(w, "Usage: %s\n\n%s\n", usageLine, strings.TrimSpace(long))
hasFlags := false
fs.VisitAll(func(*flag.Flag) { hasFlags = true })
if hasFlags {
fmt.Fprintln(w, "\nFlags:")
fs.PrintDefaults()
}
}
return fs
}
// readSource returns the contents of path, or stdin when path is "-".
func readSource(path string) (string, error) {
if path == "-" {
@@ -82,7 +125,10 @@ func readSource(path string) (string, error) {
}
func cmdTokens(args []string) int {
fs := flag.NewFlagSet("tokens", flag.ExitOnError)
fs := newCommand("tokens", "gasm tokens <file>", `
Print the lexical token stream of FILE: position, token kind and text, one
token per line. FILE may be "-" to read standard input.
`)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm tokens <file>")
@@ -100,7 +146,11 @@ func cmdTokens(args []string) int {
}
func cmdParse(args []string) int {
fs := flag.NewFlagSet("parse", flag.ExitOnError)
fs := newCommand("parse", "gasm parse <file>", `
Parse FILE and report syntax errors on stderr. On success, print how many
declarations and TEXT functions the file contains. FILE may be "-" to read
standard input.
`)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm parse <file>")
@@ -130,15 +180,49 @@ func cmdParse(args []string) int {
}
func cmdFmt(args []string) int {
fs := flag.NewFlagSet("fmt", flag.ExitOnError)
fs := newCommand("fmt", "gasm fmt [-w] [path...]", `
Canonicalise the formatting of Plan 9 assembly sources: indentation, operand
spacing, per-function mnemonic alignment and blank-line layout (exactly one
blank line before each label, TEXT and GLOBL block). Formatting is
idempotent and preserves every line, comments included.
With no paths — or a directory path — every .s file below it is reformatted
in place and the changed files are listed, the way go fmt does; "." and "_"
directories are skipped. Explicit file paths print to stdout unless -w is
given.
`)
write := fs.Bool("w", false, "write result to the source file")
fs.Parse(args)
if fs.NArg() == 0 {
fmt.Fprintln(os.Stderr, "usage: gasm fmt [-w] <file...>")
return 2
// Like go fmt: with no arguments, or with a directory argument, every .s
// file below the directory is formatted in place and the names of the
// changed files are listed; explicit file arguments keep the -w / stdout
// behaviour.
paths := fs.Args()
dirMode := len(paths) == 0
if dirMode {
paths = []string{"."}
}
var files []string
for _, p := range paths {
info, err := os.Stat(p)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
if info.IsDir() {
dirMode = true
found, err := asmFiles(p)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
files = append(files, found...)
continue
}
files = append(files, p)
}
rc := 0
for _, path := range fs.Args() {
for _, path := range files {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
@@ -146,11 +230,15 @@ func cmdFmt(args []string) int {
continue
}
out := format.Source(path, src)
if *write {
if dirMode || *write {
if out != src {
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
rc = 1
continue
}
if dirMode {
fmt.Println(path)
}
}
continue
@@ -160,8 +248,40 @@ func cmdFmt(args []string) int {
return rc
}
// asmFiles collects the .s files below dir, skipping directories whose name
// starts with "." or "_" — as the go tooling does, which keeps .git and
// scratch or reference trees (e.g. _refs) untouched.
func asmFiles(dir string) ([]string, error) {
var out []string
err := filepath.WalkDir(dir, func(path string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if d.IsDir() {
if path != dir && (strings.HasPrefix(d.Name(), ".") || strings.HasPrefix(d.Name(), "_")) {
return filepath.SkipDir
}
return nil
}
if strings.HasSuffix(d.Name(), ".s") {
out = append(out, path)
}
return nil
})
return out, err
}
func cmdLint(args []string) int {
fs := flag.NewFlagSet("lint", flag.ExitOnError)
fs := newCommand("lint", "gasm lint <file...>", `
Run the static checks over the given files and print diagnostics as
"file:line:col: severity: message [code]". The exit status is non-zero when
an error-severity diagnostic is found; warnings (e.g. the register-clobber
audit) do not affect it.
Rules include unknown-instruction, operand-count, undefined-label,
duplicate-label, missing-ret, missing-textflag-include, abi-argsize,
unreachable-code, register-clobber and funcdata-pcdata.
`)
disable := fs.String("disable", "", "comma-separated rule codes to disable")
fs.Parse(args)
if fs.NArg() == 0 {
@@ -202,7 +322,13 @@ func cmdLint(args []string) int {
}
func cmdLSP(args []string) int {
fs := flag.NewFlagSet("lsp", flag.ExitOnError)
fs := newCommand("lsp", "gasm lsp", `
Run the language server over standard input/output: JSON-RPC 2.0 with
Content-Length framing. Point an LSP-capable editor at the binary and
associate it with .s files; the target architecture is inferred from the file
suffix (_amd64.s, _arm64.s, _riscv64.s, _loong64.s). Provides completion,
hover, document symbols, diagnostics and semantic-token highlighting.
`)
fs.Parse(args)
srv := lsp.New(os.Stdin, os.Stdout)
if err := srv.Run(); err != nil {
@@ -213,7 +339,13 @@ func cmdLSP(args []string) int {
}
func cmdAsm(args []string) int {
fs := flag.NewFlagSet("asm", flag.ExitOnError)
fs := newCommand("asm", "gasm asm [-o out.bin] <file>", `
Assemble FILE (amd64) without the Go toolchain: every TEXT function is
encoded to machine code — scalar, VEX/AVX2 and EVEX/AVX-512 instructions,
FP/SP frame mapping, local labels and file-local static symbols (GLOBL/DATA)
resolved RIP-relative — and printed as a hex dump. With -o the concatenated
image (functions followed by the data section) is written to a file instead.
`)
out := fs.String("o", "", "write the concatenated machine code to this file")
fs.Parse(args)
if fs.NArg() != 1 {
+70 -5
View File
@@ -52,6 +52,54 @@ func capture(fn func() int) (stdout, stderr string, code int) {
return string(ob), string(eb), code
}
// TestCmdFmtRecursive checks the go-fmt-style directory mode: with no
// arguments every .s file below the working directory is formatted in place
// ("." and "_" directories skipped), changed files are listed, and a second
// run is a no-op.
func TestCmdFmtRecursive(t *testing.T) {
tmp := t.TempDir()
t.Chdir(tmp)
unformatted := []byte("TEXT ·f(SB),NOSPLIT,$0\nRET\n")
write := func(path string) {
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, unformatted, 0o644); err != nil {
t.Fatal(err)
}
}
write("a_amd64.s")
write(filepath.Join("sub", "b_amd64.s"))
write(filepath.Join("_refs", "c_amd64.s"))
write(filepath.Join(".git", "d_amd64.s"))
out, errOut, code := capture(func() int { return cmdFmt(nil) })
if code != 0 {
t.Fatalf("code = %d (%s)", code, errOut)
}
if out != "a_amd64.s\n"+filepath.Join("sub", "b_amd64.s")+"\n" {
t.Errorf("listed files unexpected:\n%s", out)
}
for _, p := range []string{"a_amd64.s", filepath.Join("sub", "b_amd64.s")} {
b, _ := os.ReadFile(p)
if !strings.Contains(string(b), "\tRET") {
t.Errorf("%s not formatted in place:\n%s", p, b)
}
}
for _, p := range []string{filepath.Join("_refs", "c_amd64.s"), filepath.Join(".git", "d_amd64.s")} {
b, _ := os.ReadFile(p)
if string(b) != string(unformatted) {
t.Errorf("%s must not be touched:\n%s", p, b)
}
}
// Second pass: everything is canonical, nothing is listed.
out, _, code = capture(func() int { return cmdFmt(nil) })
if code != 0 || out != "" {
t.Errorf("second pass: code=%d out=%q, want a no-op", code, out)
}
}
func TestCmdTokens(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
@@ -152,15 +200,32 @@ func TestCmdFmtWrite(t *testing.T) {
func TestUsage(t *testing.T) {
var b bytes.Buffer
usage(&b)
if !strings.Contains(b.String(), "gasm") {
t.Errorf("usage text unexpected:\n%s", b.String())
out := b.String()
for _, want := range []string{
"gasm", "Commands:", "Flags:", "--help", "--version",
"tokens", "parse", "fmt", "lint", "asm", "lsp", "version",
} {
if !strings.Contains(out, want) {
t.Errorf("usage text missing %q:\n%s", want, out)
}
}
}
func TestCmdVersion(t *testing.T) {
out, _, code := capture(func() int { return cmdVersion() })
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, version) {
t.Errorf("version output %q does not mention %q", out, version)
}
}
func TestCmdArgErrors(t *testing.T) {
// Missing file arguments produce a usage error (code 2).
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
t.Errorf("cmdFmt() code = %d, want 2", code)
// A missing path is an error (code 1); cmdFmt with no arguments is the
// recursive mode now, covered by TestCmdFmtRecursive.
if _, _, code := capture(func() int { return cmdFmt([]string{"no/such/path"}) }); code != 1 {
t.Errorf("cmdFmt(missing path) code = %d, want 1", code)
}
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
t.Errorf("cmdLint() code = %d, want 2", code)
+41 -18
View File
@@ -136,13 +136,18 @@ Two deeper analyses sit on top of the AST:
control-flow graph (basic blocks split at labels and after branches, with
fall-through and jump-target edges), computes a conservative per-instruction
register def/use, and runs the standard backward liveness iteration to a fixed
point. On top of that it flags a **callee-saved register that is written but
never saved and restored** — the per-architecture callee-saved set is amd64
`BX/BP/R12–R15`, arm64 `R19–R30`, riscv64 `X1/X8/X9/X18–X27`, loong64
`R1/R22–R31`. This is an *audit*: the runtime's own assembly clobbers these
registers freely (it controls both sides of the call), so the rule is
advisory there, but in hand-written kernels called from ordinary Go code a
clobber is a genuine ABI violation. It runs only on macro-free files, where
point. On top of that it flags writes to the registers the **Go ABI** fixes
across calls that are never saved and restored — calibrated from
`cmd/compile/abi-internal.md`, *not* the platform ABI: Go's stack-based ABI0
has no System V style callee-saved registers (amd64 `BX`, `R12`–`R15` and
the like are caller-saved or permanent scratch, and hand-written kernels may
clobber them freely). The audited set is the frame pointer and the
the frame pointer, the goroutine pointer per architecture (amd64 `BP`/`R14`, arm64 `R18`/`R28`/
`R29`, riscv64 `X27`, loong64 `R22`); the goroutine pointer is reported only
when the function can reach the runtime — it is not `NOSPLIT` or makes a
call — since the ABI0 transition machinery restores it on those paths, and
NOSPLIT call-free leaves may use it (the runtime's own assembly does). It
runs only on macro-free files, where
no opaque macro can perform the save/restore.
- **`funcdata-pcdata`.** `FUNCDATA $idx, sym(SB)` and `PCDATA $idx, $val` are
checked for well-formed operands (arity, immediate index and value, symbol
@@ -152,9 +157,15 @@ Two deeper analyses sit on top of the AST:
### `format`
The formatter works on the **token stream, not the AST**, so it preserves
every line — comments and blanks included. It only normalises indentation,
operand spacing and per-function mnemonic alignment. It is idempotent and its
output always round-trips through the parser.
every line — comments and blanks included. It normalises indentation, operand
spacing, per-function mnemonic alignment and blank-line layout: a new block
(a label, `TEXT` or `GLOBL`) is preceded by exactly one blank line (comments
leading a block stay with it), runs of blanks collapse to one, and a `RET`
terminates the body so the next function's doc comment stays at column 0. It
is idempotent and its output always round-trips through the parser. With a
directory argument — or none — it reformats every `.s` file below it in
place and lists the files changed, the way `go fmt` does (`.` and `_`
directories are skipped).
### `lsp`
@@ -202,15 +213,26 @@ three-operand-plus-immediate form (`VSHUFPD`,
`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`,
`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or
memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`,
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`,
`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`,
`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD` and
`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic — the
packed double operations (`VADDPD`/`VSUBPD`/`VMULPD`/`VDIVPD`/`VMINPD`/
`VMAXPD`), the unpacks (`VUNPCKHPD`/`VUNPCKLPD`), the scalar SD and SS
operations, `VMOVDDUP`, `VXORPD`, the width-changing conversions
(`VCVTDQ2PS`, `VCVTPS2PD`, `VCVTDQ2PD`, and the `VCVTPD2DQX`/`Y` and
`VCVTTPD2DQX`/`Y` spellings, whose length follows the wider source) and
`VFMADD231PD` — and the no-operand `VZEROUPPER`, together with `VPERMD` and
the scalar families (`CMOVcc`, `SETcc`, `LZCNT`/`TZCNT`, the extending moves,
`CVTSx2SD`, `IMUL3`) and the EVEX (AVX-512) prefix — the four-byte prefix with
5-bit register fields (Z0–Z31, X/Y 16–31), opmask registers as operands and
mask destinations, and the compressed disp8×N displacement, whose multiplier
follows the memory operand's size — covering every instruction the go-flac
AVX2 and AVX-512 kernels use. Every encoding is validated two ways: by
5-bit register fields (Z0–Z31, X/Y 16–31, with the mod=11 quirk that carries
rm[4] in X̄), opmask registers (K0–K7 as operands, mask destinations and
explicit merging/zeroing masks — written the way Go writes them, as a K
operand among the operands plus a `.Z` mnemonic suffix), and the compressed
disp8×N displacement, whose multiplier follows the memory operand's size —
covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use,
plus the common AVX-512 F/BW integer set and the floating-point and
conversion set (the packed double arithmetic, the scalar SD/SS forms —
whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, and the
width-changing conversions, including the `VCVTPD2DQ`/`VCVTTPD2DQ` family
whose length follows the wider source operand). Every encoding is validated two ways: by
round-trip decoding through `golang.org/x/arch`, and byte-for-byte against
the machine code the real Go assembler emits — a comparison that holds for
whole functions: all 27 functions of both kernels assemble to exactly the Go
@@ -223,7 +245,8 @@ behind the code and resolves references to them (`mask<>(SB)`) to
RIP-relative loads whose displacements point inside the resulting image, so
the bytes are self-consistent at any base address. External (non-file-local)
symbols are rejected: they need object-file emission, which — together with
EVEX masking/zeroing and the other architectures — is the rest of Phase 2.
the remaining EVEX forms and the other architectures — is the rest of
Phase 2.
## Extension points
+87 -12
View File
@@ -27,14 +27,6 @@ func Source(path, src string) string {
mnemLen int
funcID int
}
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
infos := make([]info, len(lines))
funcID := -1
@@ -70,8 +62,8 @@ func Source(path, src string) string {
infos[i] = inf
}
// Second pass: render.
var b strings.Builder
// Second pass: render each line.
outs := make([]outLine, 0, len(lines))
inBody := false
for i, line := range lines {
inf := infos[i]
@@ -99,11 +91,94 @@ func Source(path, src string) string {
}
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that
// follow it — typically the next function's doc comment — belong
// at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") {
inBody = false
}
b.WriteString(strings.TrimRight(out, " \t"))
}
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
}
return normalizeSpacing(outs)
}
// Line classification, shared by the formatting passes.
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
// outLine is one rendered line together with its classification.
type outLine struct {
kind int
text string
}
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
// lines collapse to one, and a new block — a label, or a TEXT or GLOBL
// directive — is preceded by exactly one blank line. Comments immediately
// above a block belong to it, so the blank line is inserted before them. No
// blank line is forced at the top of the file, right after a TEXT (the
// function's first label), or between stacked labels that share an address.
func normalizeSpacing(outs []outLine) string {
blockStart := func(ol outLine) bool {
switch ol.kind {
case kLabel:
return true
case kDirective:
// TEXT and GLOBL open a block; DATA continues a GLOBL block.
return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL")
}
return false
}
insert := make([]bool, len(outs))
for i, ol := range outs {
if !blockStart(ol) {
continue
}
j := i
for j > 0 && outs[j-1].kind == kComment {
j--
}
if j == 0 {
continue // top of file
}
switch prev := outs[j-1]; {
case prev.kind == kBlank, prev.kind == kLabel:
continue // already separated, or stacked labels
case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"):
continue // the function's first label
}
insert[j] = true
}
var b strings.Builder
prevBlank := true // also suppresses leading blanks
for i, ol := range outs {
if insert[i] && !prevBlank {
b.WriteByte('\n')
}
return b.String()
if ol.kind == kBlank {
if !prevBlank {
b.WriteByte('\n')
}
prevBlank = true
continue
}
b.WriteString(ol.text)
b.WriteByte('\n')
prevBlank = false
}
out := strings.TrimRight(b.String(), "\n")
if out == "" {
return ""
}
return out + "\n"
}
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
+96
View File
@@ -39,6 +39,102 @@ func TestGolden(t *testing.T) {
}
}
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
// sits at column 0 even when another function (ending in RET) precedes it —
// the RET must terminate the previous body for indentation purposes.
func TestDocCommentIndent(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"XORQ AX, AX\n" +
"RET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"\tXORQ AX, AX\n" +
"\tRET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"\tRET\n"
got := Source("d_amd64.s", in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
// Body comments stay indented.
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
gotBody := Source("b_amd64.s", body)
if !strings.Contains(gotBody, "\t// inside the body\n") {
t.Fatalf("body comment must stay indented:\n%q", gotBody)
}
}
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
// to one, and no blank forced after TEXT, between stacked labels, or at the
// top of the file. Leading comments belong to the block they precede.
func TestBlankLines(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" + // first label: no blank after TEXT
"XORQ AX, AX\n" +
"JMP next\n" + // unlabeled glue: fmt inserts a blank before next:
"next:\n" +
"stacked:\n" + // stacked labels share an address: no blank between
"INCQ AX\n" +
"\n" +
"\n" + // two blanks collapse to one
"// separated block\n" + // comment belongs to the label below
"later:\n" +
"RET\n" +
"// func g()\n" + // doc comment: blank goes before it
"TEXT ·g(SB), NOSPLIT, $0\n" +
"RET\n" +
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
"\n" +
"\n" +
"\n" // trailing blanks dropped
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" +
"\tXORQ AX, AX\n" +
"\tJMP next\n" +
"\n" +
"next:\n" +
"stacked:\n" +
"\tINCQ AX\n" +
"\n" +
"\t// separated block\n" + // body comment before a label stays indented
"later:\n" +
"\tRET\n" +
"\n" +
"// func g()\n" +
"TEXT ·g(SB), NOSPLIT, $0\n" +
"\tRET\n" +
"\n" +
"GLOBL ·mask(SB), RODATA, $8\n" +
"DATA ·mask+0(SB)/4, $1\n"
got := Source("b_amd64.s", in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if again := Source("b_amd64.s", got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
}
func TestOperandSpacing(t *testing.T) {
cases := map[string]string{
"4(SI)": "4(SI)",
+1 -1
View File
@@ -3,7 +3,7 @@
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.5.0"
version := "0.10.0"
default:
@just --list
+61 -7
View File
@@ -242,7 +242,7 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
@@ -319,18 +319,27 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
}
// Register liveness: a callee-saved register that is written but never
// saved and restored is clobbered across the call. The check runs over the
// control-flow graph and is skipped for macro-using files, where an opaque
// macro may perform the save/restore.
// Register liveness: a register the Go ABI fixes across calls that is
// written but never saved and restored is clobbered. The check runs over
// the control-flow graph and is skipped for macro-using files, where an
// opaque macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
if len(always) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
})
}
if len(rt) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
})
}
}
@@ -341,6 +350,29 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
return out
}
// reachesRuntime reports whether a function can reach the Go runtime: it is
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
// leaf may clobber them, since the ABI0 transition restores them (the
// runtime's own assembly relies on this, e.g. R14 on amd64).
func reachesRuntime(t *ast.Text) bool {
nosplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
nosplit = true
}
}
for _, s := range t.Body {
if in, ok := s.(*ast.Instr); ok {
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
return true
}
}
}
return !nosplit
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
@@ -406,6 +438,28 @@ func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
return true
}
for _, op := range ops {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
+25 -5
View File
@@ -48,11 +48,11 @@ func TestFixtureIsClean(t *testing.T) {
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
// The fixture mirrors the go-flac kernels, which use callee-saved registers
// (BX, R13) without saving them; the register-clobber audit flags that by
// design. This test targets the other rules, so the audit is disabled here
// (it is covered by TestRegisterClobber).
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
// The fixture mirrors the go-flac kernels, which write the Go ABI0
// scratch registers (BX, R13) without saving them — legal under Go's
// stack-based ABI, so the register-clobber audit stays silent and the
// fixture must lint entirely clean.
diags := File(f, Config{Arch: arch.AMD64})
if len(diags) != 0 {
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
}
@@ -187,6 +187,26 @@ done:
}
}
// TestEvexMaskingRecognised checks that masked EVEX forms — the .Z suffix and
// an explicit K operand — are recognised and exempt from operand-count
// checks.
func TestEvexMaskingRecognised(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VPADDD.Z Z1, Z2, K2, Z3
VPMINSD Z1, Z2, K5, Z3
VMOVDQU8 Z1, K3, (SI)
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("masked EVEX must be recognised: %+v", diags)
}
if codes(diags)[CodeOperandCount] != 0 {
t.Fatalf("masked operand counts must not be flagged: %+v", diags)
}
}
func TestArm64AddressingSuffix(t *testing.T) {
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
// instruction.
+48 -47
View File
@@ -4,7 +4,6 @@
package lint
import (
"fmt"
"sort"
"strings"
@@ -248,7 +247,7 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
}
compare := isCompare(mnem)
dstIdx := dstIndex(in, a)
dstIdx := dstIndex(in)
for i, op := range in.Operands {
r := gprName(op, a)
@@ -281,13 +280,11 @@ func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
return eff
}
// dstIndex returns the operand index of the destination register: last for the
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
func dstIndex(in *ast.Instr, a arch.Arch) int {
if a == arch.AMD64 {
// dstIndex returns the operand index of the destination register: in Plan 9
// notation the destination is the last operand on every architecture Go
// supports (amd64, arm64, riscv64 and loong64 alike).
func dstIndex(in *ast.Instr) int {
return len(in.Operands) - 1
}
return 0
}
// isCompare reports whether the mnemonic only reads its operands (setting flags).
@@ -372,41 +369,37 @@ func sameSet(a, b map[string]bool) bool {
return true
}
// calleeSavedGPRs returns the general-purpose registers an assembly function
// must preserve for its caller, using the register names the assembler accepts
// for each architecture.
func calleeSavedGPRs(a arch.Arch) map[string]bool {
// goFixedGPRs returns the general-purpose registers the Go ABI designates as
// fixed across calls — the ones hand-written assembly must not permanently
// clobber. This follows cmd/compile/abi-internal.md, not the platform ABI:
// Go's stack-based ABI0 (which hand-written assembly uses) has no System V
// style callee-saved registers, so clobbering the argument and scratch
// registers (amd64 BX, R12, R13, R15, …) is legal.
//
// Two groups are returned. always holds registers whose loss is never safe.
// runtime holds registers that survive an ABI0 leaf only because the
// transition machinery restores them (on amd64 the g pointer is reloaded
// from TLS): clobbering them is safe exactly in NOSPLIT functions that make
// no calls, which is how the runtime's own assembly uses them.
func goFixedGPRs(a arch.Arch) (always, runtime map[string]bool) {
switch a {
case arch.AMD64:
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
// BP maintains the frame chain; R14 holds the current goroutine.
// R15 is scratch except in dynamically linked binaries, so it is not
// flagged.
return gprSet("BP"), gprSet("R14")
case arch.ARM64:
names := []string{"R29", "R30"} // FP, LR
for i := 19; i <= 28; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
return gprSet(names...)
// R18 is reserved for the OS on some platforms, R28 holds the current
// goroutine, R29 is the frame pointer.
return gprSet("R18", "R28", "R29"), nil
case arch.RISCV:
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
for i := 18; i <= 27; i++ {
names = append(names, fmt.Sprintf("X%d", i))
}
for i := 2; i <= 11; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
// X27 holds the current goroutine.
return gprSet("X27"), nil
case arch.LOONG64:
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
names := []string{"R1", "RA", "R22", "FP"}
for i := 23; i <= 31; i++ {
names = append(names, fmt.Sprintf("R%d", i))
// R22 holds the current goroutine.
return gprSet("R22"), nil
}
for i := 0; i <= 8; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
}
return nil
return nil, nil
}
func gprSet(names ...string) map[string]bool {
@@ -417,15 +410,16 @@ func gprSet(names ...string) map[string]bool {
return m
}
// clobberedCalleeSaved returns the callee-saved registers a function writes
// without also saving and restoring them — i.e. registers whose caller-owned
// value is lost across the call. It walks the blocks of the liveness analysis
// (so the control-flow graph is what supplies the instruction set) and
// aggregates each instruction's register effects.
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
callee := calleeSavedGPRs(a)
if len(callee) == 0 {
return nil
// clobberedGoFixed returns the Go-ABI-fixed registers a function writes
// without also saving and restoring them. The first result lists registers
// whose loss is never safe; the second lists the goroutine-pointer class,
// whose loss is reported only when reachesRuntime is true (a non-NOSPLIT
// function, or one that makes calls — the ABI0 transition machinery restores
// the g pointer only on such paths).
func clobberedGoFixed(l *liveness, a arch.Arch, reachesRuntime bool) (always, runtime []string) {
alwaysSet, runtimeSet := goFixedGPRs(a)
if len(alwaysSet) == 0 && len(runtimeSet) == 0 {
return nil, nil
}
def := map[string]bool{}
saved := map[string]bool{}
@@ -444,12 +438,19 @@ func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
}
}
}
clobbered := func(set map[string]bool) []string {
var out []string
for r := range callee {
for r := range set {
if def[r] && !(saved[r] && restored[r]) {
out = append(out, r)
}
}
sort.Strings(out)
return out
}
always = clobbered(alwaysSet)
if reachesRuntime {
runtime = clobbered(runtimeSet)
}
return always, runtime
}
+112 -16
View File
@@ -5,36 +5,132 @@ package lint
import "testing"
// TestRegisterClobber detects writes to callee-saved registers that are not
// saved and restored.
// TestRegisterClobber checks the register-clobber audit is calibrated to the
// Go ABI (cmd/compile/abi-internal.md), not the platform ABI: Go's
// stack-based ABI0 — which hand-written assembly uses — has no System V
// style callee-saved registers, so argument and scratch registers may be
// clobbered freely. Only the registers the ABI fixes across calls (the
// frame pointer, the goroutine pointer, OS-reserved registers) are audited.
func TestRegisterClobber(t *testing.T) {
// BX (callee-saved on amd64) is written but never saved → clobbered.
clob := lintSrc(t, "#include \"textflag.h\"\n"+
// amd64: BX, R12, R13 and R15 are argument/permanent-scratch registers in
// Go ABI0 — writing them unsaved is legal (a System V calibration would
// report all of these).
scratch := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ CX, BX\n"+
"\tXORL R12, R12\n"+
"\tXORL R13, R13\n"+
"\tXORL R15, R15\n"+
"\tRET\n")
if codes(clob)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
if codes(scratch)[CodeRegisterClobber] != 0 {
t.Fatalf("Go ABI0 scratch registers must not be flagged: %+v", scratch)
}
// Saved and restored → preserved.
// amd64: R14 (the goroutine pointer) in a NOSPLIT function without calls
// is the runtime's own pattern — the ABI0 transition restores it — so it
// is not flagged.
leaf := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tXORL R14, R14\n"+
"\tRET\n")
if codes(leaf)[CodeRegisterClobber] != 0 {
t.Fatalf("R14 in a NOSPLIT leaf must not be flagged: %+v", leaf)
}
// amd64: R14 in a function that makes a call is a genuine hazard.
withCall := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tXORL R14, R14\n"+
"\tCALL ·g(SB)\n"+
"\tRET\n")
if codes(withCall)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved R14 with a call should be flagged: %+v", withCall)
}
// amd64: R14 in a non-NOSPLIT function is a hazard regardless of calls.
split := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), $0\n"+
"\tMOVQ CX, R14\n"+
"\tRET\n")
if codes(split)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved R14 in a non-NOSPLIT function should be flagged: %+v", split)
}
// amd64: R14 saved and restored around the call is preserved.
saved := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $8\n"+
"\tPUSHQ BX\n"+
"\tMOVQ CX, BX\n"+
"\tPOPQ BX\n"+
"\tPUSHQ R14\n"+
"\tXORL R14, R14\n"+
"\tCALL ·g(SB)\n"+
"\tPOPQ R14\n"+
"\tRET\n")
if codes(saved)[CodeRegisterClobber] != 0 {
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
t.Fatalf("saved/restored R14 must not be flagged: %+v", saved)
}
// A caller-saved register (CX) is fine to write.
caller := lintSrc(t, "#include \"textflag.h\"\n"+
// amd64: BP maintains the frame chain and is always audited.
bp := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ $1, CX\n"+
"\tMOVQ CX, BP\n"+
"\tRET\n")
if codes(caller)[CodeRegisterClobber] != 0 {
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
if codes(bp)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved BP write should be flagged: %+v", bp)
}
// arm64: R20 is scratch; R28 (goroutine pointer) and R18 (OS-reserved)
// are fixed by the Go ABI.
armScratch := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R20\n"+
"\tRET\n")
if codes(armScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("arm64 scratch register must not be flagged: %+v", armScratch)
}
armG := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R28\n"+
"\tRET\n")
if codes(armG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved arm64 R28 write should be flagged: %+v", armG)
}
armReserved := lintSrcArch(t, "t_arm64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVD R0, R18\n"+
"\tRET\n")
if codes(armReserved)[CodeRegisterClobber] != 1 {
t.Fatalf("arm64 R18 write should be flagged: %+v", armReserved)
}
// riscv64: X27 holds the goroutine; X5–X7 are scratch.
riscScratch := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOV X5, X6\n"+
"\tRET\n")
if codes(riscScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("riscv64 scratch register must not be flagged: %+v", riscScratch)
}
riscG := lintSrcArch(t, "t_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOV X5, X27\n"+
"\tRET\n")
if codes(riscG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved riscv64 X27 write should be flagged: %+v", riscG)
}
// loong64: R22 holds the goroutine; R5–R19 are argument/scratch.
loongScratch := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVV R5, R6\n"+
"\tRET\n")
if codes(loongScratch)[CodeRegisterClobber] != 0 {
t.Fatalf("loong64 scratch register must not be flagged: %+v", loongScratch)
}
loongG := lintSrcArch(t, "t_loong64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVV R5, R22\n"+
"\tRET\n")
if codes(loongG)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved loong64 R22 write should be flagged: %+v", loongG)
}
}