diff --git a/asm/assemble.go b/asm/assemble.go index 20f43e1..dad212d 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -19,8 +19,8 @@ import ( // // Supported operands: registers, memory (real base register), immediates, // FP/SP frame-relative operands, and local-label jumps. SB (global symbol) -// operands require relocations and are not yet supported; SIMD (VEX/EVEX) -// instructions are pending. +// operands require relocations and are not yet supported; the SIMD (VEX/AVX2) +// integer and shuffle/extract/permute/move set is in. func Assemble(t *ast.Text) ([]byte, map[string]int, error) { fi := computeFrame(t) diff --git a/asm/assemble_test.go b/asm/assemble_test.go index 6276f6d..b9598e4 100644 --- a/asm/assemble_test.go +++ b/asm/assemble_test.go @@ -199,3 +199,48 @@ TEXT ·withframe(SB), NOSPLIT, $16-16 t.Errorf("frame translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } + +// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac +// kernels end with — exercising the VEX moves, shuffle and extract forms +// through the full parser → encoder path — and checks the output is +// byte-identical to the Go assembler's. +func TestAssembleVexKernel(t *testing.T) { + fn := firstText(t, ` +#include "textflag.h" +TEXT ·hsum(SB), NOSPLIT, $0 + VPADDQ Y8, Y9, Y8 + VEXTRACTI128 $1, Y8, X9 + VPADDQ X9, X8, X8 + VPSHUFD $0xEE, X8, X9 + VPADDQ X9, X8, X8 + VMOVQ X8, AX + VZEROUPPER + RET +`) + code, _, err := Assemble(fn) + if err != nil { + t.Fatalf("Assemble: %v", err) + } + // From the Go-assembled function: + // VPADDQ Y8, Y9, Y8 c44135d4c0 + // VEXTRACTI128 $1, Y8, X9 c4437d39c101 + // VPADDQ X9, X8, X8 c44139d4c1 + // VPSHUFD $0xEE, X8, X9 c4417970c8ee + // VPADDQ X9, X8, X8 c44139d4c1 + // VMOVQ X8, AX c461f97ec0 + // VZEROUPPER c5f877 + // RET c3 + want := []byte{ + 0xc4, 0x41, 0x35, 0xd4, 0xc0, + 0xc4, 0x43, 0x7d, 0x39, 0xc1, 0x01, + 0xc4, 0x41, 0x39, 0xd4, 0xc1, + 0xc4, 0x41, 0x79, 0x70, 0xc8, 0xee, + 0xc4, 0x41, 0x39, 0xd4, 0xc1, + 0xc4, 0x61, 0xf9, 0x7e, 0xc0, + 0xc5, 0xf8, 0x77, + 0xc3, + } + if hexBytes(code) != hexBytes(want) { + t.Errorf("VEX kernel mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) + } +} diff --git a/asm/encode_test.go b/asm/encode_test.go index 2b05b06..0ff28d9 100644 --- a/asm/encode_test.go +++ b/asm/encode_test.go @@ -74,6 +74,10 @@ func TestALU(t *testing.T) { checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX) checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8)) checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX) + // The Go assembler's own spelling: immediate second. + checkSyntax(t, "cmp ecx, 0x1f", "CMPL", CX, Imm(31)) + checkSyntax(t, "cmp ecx, -0x80000000", "CMPL", CX, Imm(-2147483648)) + checkSyntax(t, "cmp r9, -0x80000000", "CMPQ", Reg{idx: 9, size: 8}, Imm(-2147483648)) } func TestLea(t *testing.T) { diff --git a/asm/instrs.go b/asm/instrs.go index 91df85a..f92ba58 100644 --- a/asm/instrs.go +++ b/asm/instrs.go @@ -136,6 +136,17 @@ func (e *enc) encodeALU(op struct { return e.encodeALUImm(op.digit, dst, int64(imm), size) } + // CMP accepts the immediate in the second position too — CMPL CX, $31 is + // the form the Go assembler itself accepts — and encodes it identically + // (CMP r/m, imm sets the flags as first − second). No other ALU op takes + // an immediate destination. + if imm, ok := dst.(Imm); ok { + if op.digit != 7 { + return fmt.Errorf("immediate must be the source operand") + } + return e.encodeALUImm(op.digit, src, int64(imm), size) + } + dstReg, dstIsReg := dst.(Reg) srcReg, srcIsReg := src.(Reg) switch { diff --git a/asm/vex.go b/asm/vex.go index 16ab365..94f9569 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -7,6 +7,10 @@ import "fmt" // This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512) // support is a later increment. +// +// Every encoding choice here is validated two ways in the tests: by +// round-trip decoding through golang.org/x/arch's x86 decoder, and by +// byte-for-byte comparison against the output of the real Go assembler. // vexForm selects how an instruction's operands map onto the VEX.vvvv, // ModRM.reg and ModRM.rm fields. @@ -17,11 +21,26 @@ const ( // ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0). vexNDS3 vexForm = iota // vexRM is the two-operand form `OP src, dst` with no vvvv source: - // ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv = 1111 (unused). + // ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv unused. vexRM // vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg = // /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0. vexShiftImm + // vexImmRM is the immediate form `OP $imm, src, dst` with no vvvv source: + // ModRM.reg = dst (op2), ModRM.rm = src (op1), imm8 = op0. VPSHUFD and + // VPERMQ use this shape. + vexImmRM + // vexNDS3Imm is the three-operand plus immediate form `OP $imm, src2, + // src1, dst`: ModRM.reg = dst, VEX.vvvv = src1, ModRM.rm = src2, imm8. + // VSHUFPD, VPERM2I128 and VINSERTI128 use this shape. + vexNDS3Imm + // vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg = + // ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM + // source lives in the reg field, the destination in r/m — the PEXTR-style + // layout. VEXTRACTI128 and VEXTRACTF128 use this shape. + vexExtract + // vexZero is the no-operand form (VZEROUPPER). + vexZero ) // vexSpec describes one VEX instruction's encoding parameters. @@ -34,9 +53,9 @@ type vexSpec struct { form vexForm } -// vexTable maps an upper-case mnemonic to its VEX encoding. It covers the -// AVX2 instructions used by the go-flac kernels in the three-operand NDS form; -// it is extended incrementally. +// vexTable maps an upper-case mnemonic to its VEX encoding. It is extended +// incrementally; every entry is covered by a byte-for-byte ground-truth test +// against the Go assembler. var vexTable = map[string]vexSpec{ // VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare. "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3}, @@ -52,12 +71,26 @@ var vexTable = map[string]vexSpec{ "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3}, "VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3}, + // VEX.256.66.0F38.W0 — dword permute (three-operand NDS form). + "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F38.WIG. "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3}, "VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3}, "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, + // VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. + "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, + "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, + "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, + "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, + // VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed + // opcodes with an F2 pp). + "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, + "VMULSD": {1, 0x59, 0, 3, -1, vexNDS3}, + // VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form). + "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, + // VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src, // no vvvv). "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, @@ -65,6 +98,9 @@ var vexTable = map[string]vexSpec{ "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, + // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion + // (reg=dst, rm=src, no vvvv; the length follows the destination). + "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, // VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src). "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) @@ -75,16 +111,75 @@ var vexTable = map[string]vexSpec{ "VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm}, "VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm}, "VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm}, + + // VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8). + "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, + // VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8). + "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, + + // VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2, + // imm8). + "VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm}, + // VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is + // the XMM or memory source). + "VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm}, + "VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm}, + + // VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). + "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, + "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, + + // VEX.128.0F.W0 — no operands. + "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, +} + +// vexMoveSpec describes a VEX move, which takes different opcodes (and +// sometimes a different VEX.W) per operand direction. The Go assembler +// encodes a vector→vector move with the store-form opcode (reg = source, +// rm = destination), so regReg defaults to store when zero. +type vexMoveSpec struct { + mapSel int + pp int + load byte // r/m → vector: reg=dst, rm=src + store byte // vector → r/m: reg=src, rm=dst + loadW int + storeW int + regReg byte // vector → vector opcode; 0 uses store + regW int + vecOK bool // the non-fixed operand may be a vector register + gprOK bool // the non-fixed operand may be a general-purpose register + xmmOnly bool // YMM registers are rejected +} + +// vexMoveTable maps an upper-case move mnemonic to its encoding. +var vexMoveTable = map[string]vexMoveSpec{ + // VEX.128/256.F3.0F.WIG — unaligned integer move. + "VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false}, + // VEX.128/256.66.0F.WIG — unaligned packed double move. + "VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false}, + // VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM. + "VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true}, + // VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm). + "VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true}, + // VEX.128.F2.0F.WIG — scalar double move, memory operands only (the + // register form takes three operands and is not supported yet). + "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, } // isVex reports whether the mnemonic is a VEX-encoded instruction we handle. func isVex(mnemUpper string) bool { - _, ok := vexTable[mnemUpper] + if _, ok := vexTable[mnemUpper]; ok { + return true + } + _, ok := vexMoveTable[mnemUpper] return ok } // encodeVex encodes a VEX instruction with operands in Plan 9 order. func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { + if ms, ok := vexMoveTable[mnemUpper]; ok { + return e.encodeVexMove(mnemUpper, ms, ops) + } spec := vexTable[mnemUpper] switch spec.form { case vexNDS3: @@ -93,6 +188,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { return e.encodeVexRM(spec, ops) case vexShiftImm: return e.encodeVexShiftImm(spec, ops) + case vexImmRM: + return e.encodeVexImmRM(spec, ops) + case vexNDS3Imm: + return e.encodeVexNDS3Imm(spec, ops) + case vexExtract: + return e.encodeVexExtract(spec, ops) + case vexZero: + return e.encodeVexZero(mnemUpper, spec, ops) } return fmt.Errorf("unhandled VEX form for %s", mnemUpper) } @@ -151,7 +254,9 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error { l = srcReg.vecLenBit() } - return e.emitVexFields(spec, l, regField, rBit, 0, src) // vvvv unused → vvvvBar=0 + // An unused vvvv field must be stored as all ones (v̄vvv = 1111); the + // hardware raises #UD on any other value. + return e.emitVexFields(spec, l, regField, rBit, 15, src) } // encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst. @@ -176,27 +281,223 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error { } vvvvBar := 15 - (dstReg.idx & 15) - l := dstReg.vecLenBit() - rmField := srcReg.idx & 7 - bBit := 0 - if srcReg.idx >= 8 { - bBit = 1 + if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil { + return err } - modrm := 0xC0 | spec.opdigit<<3 | rmField - - if spec.mapSel == 1 && bBit == 0 && spec.w == 0 { - e.out = append(e.out, 0xC5, byte(1<<7|vvvvBar<<3|l<<2|spec.pp)) - } else { - e.out = append(e.out, 0xC4, - byte(1<<7|1<<6|(1-bBit)<<5|spec.mapSel), - byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp)) + immByte, err := imm8(int64(immVal)) + if err != nil { + return err } - e.out = append(e.out, spec.opcode, byte(modrm), byte(int8(immVal))) + e.out = append(e.out, immByte) return nil } +// imm8 range-checks an immediate for an 8-bit field. Shuffle controls are +// unsigned bit masks, but the negative spelling ($-1 = all bits set) is +// accepted, so the accepted span is -128..255. +func imm8(v int64) (byte, error) { + if v < -128 || v > 255 { + return 0, fmt.Errorf("immediate $%d does not fit in 8 bits", v) + } + return byte(v), nil +} + +// encodeVexImmRM encodes an immediate form with no vvvv source: OP $imm, src, +// dst (VPSHUFD, VPERMQ). ModRM.reg = dst, ModRM.rm = src, imm8 appended. +func (e *enc) encodeVexImmRM(spec vexSpec, ops []Operand) error { + if len(ops) != 3 { + return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) + } + imm, src, dst := ops[0], ops[1], ops[2] + immVal, ok := imm.(Imm) + if !ok { + return fmt.Errorf("shuffle control must be an immediate") + } + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("shuffle destination must be a vector register") + } + + // The vector length follows the source when it is a vector register, + // otherwise the destination (a memory source carries no length). + l := dstReg.vecLenBit() + if srcReg, ok := src.(Reg); ok && srcReg.isVec() { + l = srcReg.vecLenBit() + } + regField := dstReg.idx & 7 + rBit := 0 + if dstReg.idx >= 8 { + rBit = 1 + } + if err := e.emitVexFields(spec, l, regField, rBit, 15, src); err != nil { + return err + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + e.out = append(e.out, immByte) + return nil +} + +// encodeVexNDS3Imm encodes the three-operand plus immediate form: OP $imm, +// src2, src1, dst (VSHUFPD, VPERM2I128, VINSERTI128). ModRM.reg = dst, +// VEX.vvvv = src1, ModRM.rm = src2, imm8 appended. +func (e *enc) encodeVexNDS3Imm(spec vexSpec, ops []Operand) error { + if len(ops) != 4 { + return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) + } + imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] + immVal, ok := imm.(Imm) + if !ok { + return fmt.Errorf("shuffle control must be an immediate") + } + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("destination must be a vector register") + } + vvvvReg, ok := src1.(Reg) + if !ok || !vvvvReg.isVec() { + return fmt.Errorf("second source must be a vector register") + } + + regField := dstReg.idx & 7 + rBit := 0 + if dstReg.idx >= 8 { + rBit = 1 + } + vvvvBar := 15 - (vvvvReg.idx & 15) + if err := e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2); err != nil { + return err + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + e.out = append(e.out, immByte) + return nil +} + +// encodeVexExtract encodes a lane extract: OP $imm, ysrc, xdst +// (VEXTRACTI128, VEXTRACTF128). The YMM source occupies ModRM.reg and the +// XMM (or memory) destination ModRM.rm; imm8 selects the lane. +func (e *enc) encodeVexExtract(spec vexSpec, ops []Operand) error { + if len(ops) != 3 { + return fmt.Errorf("extract expects 3 operands ($imm, ysrc, xdst), got %d", len(ops)) + } + imm, src, dst := ops[0], ops[1], ops[2] + immVal, ok := imm.(Imm) + if !ok { + return fmt.Errorf("extract lane must be an immediate") + } + srcReg, ok := src.(Reg) + if !ok || !srcReg.isVec() { + return fmt.Errorf("extract source must be a vector register") + } + + regField := srcReg.idx & 7 + rBit := 0 + if srcReg.idx >= 8 { + rBit = 1 + } + if err := e.emitVexFields(spec, srcReg.vecLenBit(), regField, rBit, 15, dst); err != nil { + return err + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + e.out = append(e.out, immByte) + return nil +} + +// encodeVexZero encodes a no-operand instruction (VZEROUPPER). +func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error { + if len(ops) != 0 { + return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops)) + } + // 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 0. + e.out = append(e.out, 0xC5, byte(1<<7|15<<3|spec.pp), spec.opcode) + return nil +} + +// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, +// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector +// move uses the store-form layout (reg = source, rm = destination), matching +// the Go assembler. +func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error { + if len(ops) != 2 { + return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops)) + } + src, dst := ops[0], ops[1] + srcReg, srcIsVec := vecReg(src) + dstReg, dstIsVec := vecReg(dst) + + var reg Reg + var rm Operand + op, w := ms.store, ms.storeW + switch { + case srcIsVec && dstIsVec: + if !ms.vecOK { + return fmt.Errorf("%s does not take two vector registers", mnem) + } + if ms.xmmOnly && (srcReg.size == 32 || dstReg.size == 32) { + return fmt.Errorf("%s operates on XMM registers only", mnem) + } + if ms.regReg != 0 { + op, w = ms.regReg, ms.regW + } + reg, rm = srcReg, dst // store form: reg = source, rm = destination. + case srcIsVec: + // vector → memory, or → GPR (VMOVD/VMOVQ only). + if !validMoveOther(ms, dst) { + return fmt.Errorf("%s: invalid destination operand", mnem) + } + reg, rm = srcReg, dst + case dstIsVec: + // memory → vector, or GPR → vector (VMOVD/VMOVQ only). + if !validMoveOther(ms, src) { + return fmt.Errorf("%s: invalid source operand", mnem) + } + op, w = ms.load, ms.loadW + reg, rm = dstReg, src + default: + return fmt.Errorf("%s needs a vector register operand", mnem) + } + if ms.xmmOnly && reg.size == 32 { + return fmt.Errorf("%s operates on XMM registers only", mnem) + } + + regField := reg.idx & 7 + rBit := 0 + if reg.idx >= 8 { + rBit = 1 + } + spec := vexSpec{mapSel: ms.mapSel, opcode: op, w: w, pp: ms.pp, opdigit: -1} + return e.emitVexFields(spec, reg.vecLenBit(), regField, rBit, 15, rm) +} + +// vecReg extracts a vector register from an operand. +func vecReg(op Operand) (Reg, bool) { + r, ok := op.(Reg) + return r, ok && r.isVec() +} + +// validMoveOther reports whether the non-vector operand of a move is +// acceptable: memory always is, a GPR only for VMOVD/VMOVQ. +func validMoveOther(ms vexMoveSpec, op Operand) bool { + switch o := op.(type) { + case Mem: + return true + case Reg: + return ms.gprOK && !o.isVec() + } + return false +} + // emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for -// the given precomputed fields. It is shared by the NDS and RM forms. +// the given precomputed fields. It is shared by every register/rm VEX form; +// immediate bytes are appended by the caller. func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error { var modrm, sib int var disp []byte diff --git a/asm/vex_test.go b/asm/vex_test.go index b81dd7a..a9bd8b9 100644 --- a/asm/vex_test.go +++ b/asm/vex_test.go @@ -4,6 +4,7 @@ package asm import ( + "strings" "testing" "golang.org/x/arch/x86/x86asm" @@ -26,7 +27,12 @@ func TestVexNDS3(t *testing.T) { if spec.form != vexNDS3 { continue } - code, err := Encode(mnem, vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")) + // Scalar (F2/F3 pp) instructions exist only in the 128-bit form. + vec := "Y" + if spec.pp >= 2 { + vec = "X" + } + code, err := Encode(mnem, vreg(t, vec+"0"), vreg(t, vec+"1"), vreg(t, vec+"2")) if err != nil { t.Errorf("%s: Encode: %v", mnem, err) continue @@ -72,7 +78,7 @@ func TestVexXMM(t *testing.T) { if inst.Op != x86asm.VPXOR { t.Fatalf("decoded %s, want VPXOR", inst.Op) } - // vpxor xmm7, xmm7, xmm7 → C5 C9 EF FF (2-byte VEX, L=0). + // vpxor xmm7, xmm7, xmm7 → C5 C1 EF FF (2-byte VEX, L=0). if code[0] != 0xC5 { t.Errorf("expected 2-byte VEX (C5), got % x", code) } @@ -140,3 +146,179 @@ func TestVexShiftImm(t *testing.T) { t.Fatalf("VPSRAD decoded %v (err %v), want VPSRAD", inst.Op, err) } } + +// TestVexGroundTruth checks byte-for-byte agreement with the real Go +// assembler. The expected bytes were extracted from the machine code the Go +// toolchain produced for exactly these instructions (go build + a .text +// section dump of the resulting binary), never from a disassembler's +// rendering. This locks the v̄vvv = 1111 rule for unused vvvv fields (a +// value the hardware rejects with #UD and the x86 decoder silently ignores) +// as well as every new operand form. +func TestVexGroundTruth(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + // Three-operand NDS form. + {"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"}, + {"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"}, + {"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"}, + {"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"}, + {"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"}, + {"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"}, + {"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"}, + // Floating point (packed and scalar) and FMA — same NDS form, the pp + // bits and map select the operation. + {"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"}, + {"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"}, + {"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"}, + {"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"}, + {"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"}, + {"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"}, + {"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"}, + {"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"}, + {"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"}, + // Two-operand reg/rm form (v̄vvv must be 1111). + {"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"}, + {"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"}, + {"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"}, + {"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"}, + {"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"}, + {"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"}, + {"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"}, + // Immediate shifts. + {"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"}, + {"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"}, + // Immediate shuffle (reg=dst, rm=src, imm8). + {"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"}, + {"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"}, + {"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"}, + {"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"}, + // Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8). + {"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"}, + {"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"}, + {"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"}, + {"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"}, + // Lane extract (reg=YMM source, rm=XMM/memory destination, imm8). + {"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"}, + {"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"}, + {"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"}, + // Moves — each direction picks its own opcode and VEX.W. + {"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"}, + {"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"}, + {"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"}, + {"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"}, + {"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"}, + {"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"}, + {"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"}, + {"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"}, + {"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"}, + {"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"}, + {"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"}, + {"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"}, + {"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"}, + {"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"}, + {"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"}, + {"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"}, + // No-operand. + {"VZEROUPPER", "VZEROUPPER", nil, "c5f877"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := strings.ReplaceAll(hexBytes(code), " ", ""); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + continue + } + inst, err := x86asm.Decode(code, 64) + if err != nil { + t.Errorf("%s: Decode(% x): %v", c.name, code, err) + continue + } + if inst.Len != len(code) { + t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code)) + } + if inst.Op.String() != c.mnem { + t.Errorf("%s: decoded as %s", c.name, inst.Op.String()) + } + } +} + +// TestVexNewFormsSyntax checks the decoded Intel-syntax rendering of the new +// SIMD forms (operand order is the decoder's, confirming the fields landed). +func TestVexNewFormsSyntax(t *testing.T) { + checkSyntax(t, "vpshufd xmm9, xmm8, 0xee", "VPSHUFD", Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")) + checkSyntax(t, "vpermq ymm2, ymm1, 0x1b", "VPERMQ", Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")) + checkSyntax(t, "vextracti128 xmm9, ymm8, 0x1", "VEXTRACTI128", Imm(1), vreg(t, "Y8"), vreg(t, "X9")) + checkSyntax(t, "vinserti128 ymm2, ymm1, xmm5, 0x1", "VINSERTI128", Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")) + checkSyntax(t, "vperm2i128 ymm3, ymm2, ymm1, 0x31", "VPERM2I128", Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")) + checkSyntax(t, "vpermd ymm3, ymm2, ymm1", "VPERMD", vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")) + checkSyntax(t, "vshufpd xmm3, xmm2, xmm1, 0x1", "VSHUFPD", Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")) + checkSyntax(t, "vmovq rax, xmm8", "VMOVQ", vreg(t, "X8"), AX) + checkSyntax(t, "vmovq xmm9, rax", "VMOVQ", AX, vreg(t, "X9")) + checkSyntax(t, "vmovdqu ymm1, ymmword ptr [rsi]", "VMOVDQU", Ptr(SI, 0, 32), vreg(t, "Y1")) + checkSyntax(t, "vmovdqu ymmword ptr [rdi], ymm3", "VMOVDQU", vreg(t, "Y3"), Ptr(DI, 0, 32)) + checkSyntax(t, "vzeroupper", "VZEROUPPER") +} + +// TestVexMemoryForms round-trips the new forms with memory sources/destinations, +// covering the SIB/indexed path through the VEX prefix emitter. +func TestVexMemoryForms(t *testing.T) { + checkSyntax(t, "vpshufd ymm1, ymmword ptr [rsi], 0x4e", "VPSHUFD", Imm(0x4E), Ptr(SI, 0, 32), vreg(t, "Y1")) + checkSyntax(t, "vinserti128 ymm2, ymm1, xmmword ptr [rdi], 0x1", "VINSERTI128", Imm(1), Ptr(DI, 0, 16), vreg(t, "Y1"), vreg(t, "Y2")) + checkSyntax(t, "vmovdqu ymm1, ymmword ptr [rax+4*rbx]", "VMOVDQU", Idx(AX, BX, 4, 0, 32), vreg(t, "Y1")) + checkSyntax(t, "vpermq ymm2, ymmword ptr [rsi], 0x1b", "VPERMQ", Imm(0x1B), Ptr(SI, 0, 32), vreg(t, "Y2")) + checkSyntax(t, "vfmadd231pd ymm8, ymm12, ymm14", "VFMADD231PD", vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")) + checkSyntax(t, "vcvtdq2pd ymm12, xmmword ptr [rsi]", "VCVTDQ2PD", Ptr(SI, 0, 16), vreg(t, "Y12")) + // The top and bottom of the accepted imm8 span: $255 and $-1 both encode + // an all-bits-set control. + checkSyntax(t, "vpshufd xmm1, xmm0, 0xff", "VPSHUFD", Imm(255), vreg(t, "X0"), vreg(t, "X1")) + checkSyntax(t, "vpshufd xmm1, xmm0, 0xff", "VPSHUFD", Imm(-1), vreg(t, "X0"), vreg(t, "X1")) +} + +// TestVexErrors checks that invalid operand shapes are rejected. +func TestVexErrors(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + }{ + {"VPSHUFD arity", "VPSHUFD", []Operand{vreg(t, "X0"), vreg(t, "X1")}}, + {"VPSHUFD non-imm control", "VPSHUFD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}}, + {"VPSHUFD gpr dst", "VPSHUFD", []Operand{Imm(1), vreg(t, "X0"), AX}}, + {"VEXTRACTI128 arity", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y0")}}, + {"VEXTRACTI128 non-vec src", "VEXTRACTI128", []Operand{Imm(1), AX, vreg(t, "X0")}}, + {"VINSERTI128 arity", "VINSERTI128", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "Y1")}}, + {"VINSERTI128 non-vec vvvv", "VINSERTI128", []Operand{Imm(1), vreg(t, "X0"), AX, vreg(t, "Y1")}}, + {"VPERM2I128 non-imm control", "VPERM2I128", []Operand{AX, vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}}, + {"VMOVSD reg-reg", "VMOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}}, + {"VMOVD reg-reg", "VMOVD", []Operand{vreg(t, "X1"), vreg(t, "X2")}}, + {"VMOVQ ymm", "VMOVQ", []Operand{vreg(t, "Y1"), AX}}, + {"VMOVQ mixed X/Y", "VMOVQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}}, + {"VMOVQ no vector", "VMOVQ", []Operand{AX, BX}}, + {"VMOVDQU gpr", "VMOVDQU", []Operand{AX, vreg(t, "Y1")}}, + {"VMOVUPD gpr", "VMOVUPD", []Operand{vreg(t, "X1"), AX}}, + {"VZEROUPPER operands", "VZEROUPPER", []Operand{AX}}, + {"VPSLLD non-vec dst", "VPSLLD", []Operand{Imm(1), vreg(t, "Y0"), AX}}, + {"VPSLLD non-imm count", "VPSLLD", []Operand{AX, vreg(t, "Y0"), vreg(t, "Y1")}}, + {"VPSLLD non-vec src", "VPSLLD", []Operand{Imm(1), AX, vreg(t, "Y1")}}, + {"VPADDD non-vec vvvv", "VPADDD", []Operand{vreg(t, "Y0"), AX, vreg(t, "Y1")}}, + {"VEXTRACTI128 non-imm lane", "VEXTRACTI128", []Operand{AX, vreg(t, "Y0"), vreg(t, "X0")}}, + {"VMOVQ imm operand", "VMOVQ", []Operand{Imm(1), vreg(t, "X0")}}, + {"VPSHUFD imm rm", "VPSHUFD", []Operand{Imm(1), Imm(2), vreg(t, "X0")}}, + {"VPSHUFD imm range", "VPSHUFD", []Operand{Imm(256), vreg(t, "X0"), vreg(t, "X1")}}, + {"VPERMQ imm range", "VPERMQ", []Operand{Imm(300), vreg(t, "Y0"), vreg(t, "Y1")}}, + {"VEXTRACTI128 imm range", "VEXTRACTI128", []Operand{Imm(256), vreg(t, "Y0"), vreg(t, "X0")}}, + {"VPSLLD imm range", "VPSLLD", []Operand{Imm(-129), vreg(t, "Y0"), vreg(t, "Y1")}}, + } + for _, c := range cases { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } +} diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 00d9bae..91c8ded 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -26,7 +26,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.1.0" +var version = "0.2.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 2fc58f5..e21e6c0 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -188,16 +188,27 @@ registers are translated onto the hardware stack pointer — `x+N(FP)` becomes `(N+8)(SP)` for a zero-frame function and `(N+frame+16)(SP)` once a frame pointer is set up, with the matching Go prologue/epilogue generated — so the output is byte-identical to the Go assembler for these cases. SIMD is handled -SIMD is handled by a VEX (AVX/AVX2) encoder — the two- and three-byte VEX prefixes with XMM/YMM -registers — across three operand forms (the three-operand NDS form, the - two-operand reg/rm form, and the immediate-shift form), together covering the -bulk of the integer SIMD set; each encoding is validated by round-trip - decoding. This increment covers register / memory / immediate / FP-frame -operands, local-label jumps and these VEX SIMD forms; the remaining SIMD forms -(shuffles, extract/insert, permute, moves), EVEX / AVX-512, `SB` (global -symbol) operands (relocations) and object-file emission are the rest of -Phase 2. +registers — across seven operand forms: the three-operand NDS form, the +two-operand reg/rm form, the immediate-shift form, the immediate shuffle form +(`VPSHUFD`, `VPERMQ`), the three-operand-plus-immediate form (`VSHUFPD`, +`VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`, +lane-extract form (`VEXTRACTI128`, +`VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or +memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`, +`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`, +`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`, +`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD`, +covering every integer, shuffle and FP instruction the go-flac AVX2 kernels +use. Every encoding is validated two ways: by round-trip decoding +through `golang.org/x/arch`, and byte-for-byte against the machine code the +real Go assembler emits (which also locks the v̄vvv = 1111 rule for unused +vvvv fields — a value the hardware rejects with #UD and the decoder silently +ignores). This increment covers register / memory / immediate / FP-frame +operands, local-label jumps and these VEX SIMD forms; EVEX / AVX-512, `SB` +(global symbol) operands (relocations), a handful of scalar gaps the kernels +hit (`CMOVcc`, `SETcc`, `LZCNT`, `MOVSX`/`MOVZX`) and object-file emission +are the rest of Phase 2. ## Extension points diff --git a/justfile b/justfile index fa391f5..875ced6 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.1.0" +version := "0.2.0" default: @just --list