From 6e49cbd0996d4a0aa11d63bef874fa82439471c9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 20:20:06 +0200 Subject: [PATCH] feat(asm): assemble the extended instruction layer on amd64 Assisted-by: GLM 5.3 Flash --- asm/amd64_ext.go | 300 ++++++++++++++++++++++++++++++++++++++ asm/amd64_ext_asm_test.go | 284 ++++++++++++++++++++++++++++++++++++ asm/assemble.go | 20 +++ 3 files changed, 604 insertions(+) create mode 100644 asm/amd64_ext.go create mode 100644 asm/amd64_ext_asm_test.go diff --git a/asm/amd64_ext.go b/asm/amd64_ext.go new file mode 100644 index 0000000..e9ca6c7 --- /dev/null +++ b/asm/amd64_ext.go @@ -0,0 +1,300 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// The assembler's side of the amd64 extension layer: this file turns a parsed +// amd64 statement into the operand form arch.ExtInstr.Encode consumes and +// routes statements only the layer can encode through the registry. It sits +// beside the main amd64 encoders, never inside them: the generated table, the +// legacy SSE paths and the VEX and EVEX mechanisms are untouched, and a +// statement reaches this file only when the mnemonic is registered in the +// extension layer and the scalar paths cannot encode it. +// +// The spellings are the layer's own Plan 9 forms, the ones its metadata +// documents: the vector registers carry the house names X0, Y0 and Z0 (the +// EVEX 128, 256 and 512-bit classes, registers 16 to 31 included), the general +// registers the width their spelling fixes (RAX through R15, EAX through EDI +// and R8D through R15D), the opmask registers K0 through K7, and the memory +// operand the base-relative form off(base) with the optional scaled index +// off(base)(index*scale) the SIB byte carries. The decorations ride the +// operand in braces: the write mask {k1} through {k7} and zeroing {z} on the +// destination, the {1toN} broadcast on the memory source, and {sae} and +// {rn-sae} through {rz-sae} beside the rounding-capable destinations. The +// imm8-control forms take their control byte as the leading $ immediate the +// reference listings write first. +// +// The Feature field stays metadata at assembly time: the assembler has no CPU, +// the toolchain does not gate assembly on CPU features, and every registered +// feature assembles, the behaviour the arm64 wiring established +// (asm/arm64_ext.go). The field remains for the linter and the listing. + +package asm + +import ( + "fmt" + "strconv" + "strings" + + "sourcedock.dev/petrbalvin/gasm-sdk/arch" + "sourcedock.dev/petrbalvin/gasm-sdk/ast" +) + +// amd64ExtStatement converts one instruction's operands into the extended +// layer's operand form. pinned reports that the statement belongs to the +// layer: the mnemonic is registered in the amd64 registry and the scalar +// paths cannot encode it. A pinned statement can only encode through the +// layer, so every operand is read here and its diagnostic replaces whatever +// the scalar paths would have said about operands they cannot read; err is +// non-nil for a pinned statement whose operands the layer refuses, and +// extops is complete only when err is nil. Unpinned means the statement is +// nobody's: the caller falls through to the ordinary amd64 encoders, which +// keep their exact behaviour for every statement they knew before. +func amd64ExtStatement(mnem string, ops []*ast.Operand) (extops []arch.ExtOperand, pinned bool, err error) { + if _, ok := LookupExtension(arch.AMD64, mnem); !ok { + return nil, false, nil + } + if Encodable(mnem) { + // A mnemonic the main encoder knows is never the layer's, whatever + // the registry carries: the scalar paths keep the statement. No + // registered mnemonic trips this today (the layer is sealed by + // test), but the guard keeps the fall-through promise exact should + // the toolchain ever learn one of these names. + return nil, false, nil + } + out := make([]arch.ExtOperand, 0, len(ops)) + for i, op := range ops { + ext, convErr := amd64ExtOperand(mnem, op, i+1) + if convErr != nil { + return nil, true, convErr + } + out = append(out, ext) + } + return out, true, nil +} + +// amd64ExtOperand converts one parsed operand into the layer's form: a $ immediate, +// a vector, general or opmask register, or a base-relative memory operand, each +// with the brace decorations the spelling carries. +func amd64ExtOperand(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, error) { + if op.Kind == ast.OpImmediate { + return amd64ExtImmediate(mnem, op, pos) + } + body, dec, err := amd64ExtDecorations(mnem, op, pos) + if err != nil { + return arch.ExtOperand{}, err + } + if strings.ContainsRune(body, '(') { + ext, ok := amd64ExtMemory(mnem, op, pos) + if !ok { + return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an extended-layer operand: want a base-relative memory operand, off(base)(index*scale) shape", mnem, pos, op.Raw) + } + ext.Broadcast = dec.broadcast + if dec.hasMask { + ext.Mask, ext.HasMask = dec.mask, true + } + ext.Zeroing = dec.zeroing + ext.Round = dec.round + return ext, nil + } + ext, ok := amd64ExtRegister(mnem, op, pos, body) + if !ok { + return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an extended-layer operand: want a vector, general or opmask register, a base-relative memory operand or an immediate", mnem, pos, op.Raw) + } + if dec.hasMask { + ext.Mask, ext.HasMask = dec.mask, true + } + ext.Zeroing = dec.zeroing + ext.Round = dec.round + return ext, nil +} + +// amd64ExtImmediate converts a $ immediate into the layer's form. The +// parser folds a parenthesised constant expression in full and reads a bare +// literal greedily, dropping any trailing operator tokens: $255<<8 parses as +// 255 with the shift silently gone. Encoding that silent prefix would +// assemble what the text did not say, so an unparenthesised immediate is +// accepted only when its whole text reads back as one integer carrying the +// parser's value. +func amd64ExtImmediate(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, error) { + if !op.Imm.HasVal { + return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an immediate the layer can read", mnem, pos, op.Raw) + } + text := strings.Join(strings.Fields(strings.TrimPrefix(op.Raw, "$")), "") + if !strings.HasPrefix(text, "(") { + if _, parseErr := strconv.ParseInt(text, 0, 64); parseErr != nil { + return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an immediate the layer can read", mnem, pos, op.Raw) + } + } + v := op.Imm.Val + if op.Imm.Neg { + v = -v + } + return arch.ExtOperand{Kind: arch.ExtImm, Imm: v}, nil +} + +// amd64ExtRegister parses a register operand off a normalised operand body: +// the vector classes X0-X31, Y0-Y31 and Z0-Z31, the width-fixed general +// spellings RAX through R15 and EAX through R15D, and the opmask registers +// K0-K7. The register ranges are left to the encoding, whose diagnostics +// name them. +func amd64ExtRegister(mnem string, op *ast.Operand, pos int, body string) (arch.ExtOperand, bool) { + if body == "" { + return arch.ExtOperand{}, false + } + r, ok := ParseReg(body) + if !ok { + return arch.ExtOperand{}, false + } + switch { + case r.mask: + return arch.ExtOperand{Kind: arch.ExtKReg, Reg: r.idx}, true + case r.isVec(): + kind := arch.ExtXMM + switch r.size { + case 32: + kind = arch.ExtYMM + case 64: + kind = arch.ExtZMM + } + return arch.ExtOperand{Kind: kind, Reg: r.idx}, true + case r.size == 8: + return arch.ExtOperand{Kind: arch.ExtR64, Reg: r.idx}, true + case r.size == 4: + return arch.ExtOperand{Kind: arch.ExtR32, Reg: r.idx}, true + } + return arch.ExtOperand{}, false +} + +// amd64ExtMemory parses a base-relative memory operand off the parsed +// address: off(base) and off(base)(index*scale), the SIB shapes the layer's +// entries carry. The base and the index are general registers spelled in any +// width the house names offer, the displacement the leading signed term, and +// a group whose scale is not written scales by one, the choice the main +// amd64 paths make for the same spelling. Vector, opmask and segment +// registers are refused as base and index, and so is every frame form: the +// layer's memory operand is hardware addressing alone. +func amd64ExtMemory(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, bool) { + a := op.Addr + if a.Range != nil || a.Base == "" { + return arch.ExtOperand{}, false + } + if a.Sym != nil && a.Sym.Pseudo != "" { + return arch.ExtOperand{}, false + } + base, ok := amd64ExtGprNumber(a.Base) + if !ok { + return arch.ExtOperand{}, false + } + ext := arch.ExtOperand{Kind: arch.ExtMem, Reg: base, Imm: a.Offset} + if a.Index != "" { + index, ok := amd64ExtGprNumber(a.Index) + if !ok { + return arch.ExtOperand{}, false + } + scale := a.Scale + if scale == 0 { + scale = 1 + } + ext.Index, ext.Scale, ext.HasIndex = index, scale, true + } + return ext, true +} + +// amd64ExtGprNumber resolves one general-register spelling to its number: +// whatever the register table carries for indices 0-15, the vector, opmask, +// x87, MMX, segment and control-debug classes refused, so a vector register +// in a base or index position names itself rather than encoding as its +// same-numbered general register. +func amd64ExtGprNumber(name string) (int, bool) { + r, ok := ParseReg(name) + if !ok || r.mask || r.fp || r.mmx || r.seg != 0 || r.ctl != 0 || r.size > 8 { + return 0, false + } + return r.idx, true +} + +// amd64ExtDecorations splits the brace decorations off a normalised operand +// text and returns the body before the first brace and the decorations they +// spell: the write mask {k1} through {k7}, zeroing {z}, the {1toN} broadcast +// and the rounding controls {sae} and {rn-sae} through {rz-sae}, matched +// case-insensitively the way the register spellings are. The mask, zeroing +// and rounding fields land on the operand the conversion builds; whether the +// position takes them is the encoding's judgement, whose diagnostics name the +// entry. The broadcast factor N is checked as a number and otherwise left to +// the entry: the layer's model carries the spelling, not the lane count. +func amd64ExtDecorations(mnem string, op *ast.Operand, pos int) (body string, dec amd64ExtDecor, err error) { + compact := strings.Join(strings.Fields(op.Raw), "") + i := strings.IndexByte(compact, '{') + if i < 0 { + return compact, dec, nil + } + body = compact[:i] + for i < len(compact) { + if compact[i] != '{' { + return "", dec, fmt.Errorf("%s: operand %d (%s): text between brace decorations", mnem, pos, op.Raw) + } + end := strings.IndexByte(compact[i:], '}') + if end < 0 { + return "", dec, fmt.Errorf("%s: operand %d (%s): brace decoration without a closing brace", mnem, pos, op.Raw) + } + content := strings.ToUpper(compact[i+1 : i+end]) + switch { + case content == "Z": + if dec.zeroing { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two zeroing decorations", mnem, pos, op.Raw) + } + dec.zeroing = true + case content == "SAE": + if dec.round != arch.ExtRoundNone { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw) + } + dec.round = arch.ExtRoundSAE + case content == "RN-SAE": + if dec.round != arch.ExtRoundNone { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw) + } + dec.round = arch.ExtRoundNearest + case content == "RD-SAE": + if dec.round != arch.ExtRoundNone { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw) + } + dec.round = arch.ExtRoundDown + case content == "RU-SAE": + if dec.round != arch.ExtRoundNone { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw) + } + dec.round = arch.ExtRoundUp + case content == "RZ-SAE": + if dec.round != arch.ExtRoundNone { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw) + } + dec.round = arch.ExtRoundTruncate + case strings.HasPrefix(content, "K") && content != "K": + n, convErr := strconv.Atoi(content[1:]) + if convErr != nil || n < 0 { + return "", dec, fmt.Errorf("%s: operand %d (%s): %q is not a mask decoration, want {k1} through {k7}", mnem, pos, op.Raw, content) + } + if dec.hasMask { + return "", dec, fmt.Errorf("%s: operand %d (%s) carries two write masks", mnem, pos, op.Raw) + } + dec.mask, dec.hasMask = n, true + case strings.HasPrefix(content, "1TO"): + if _, convErr := strconv.Atoi(content[3:]); convErr != nil { + return "", dec, fmt.Errorf("%s: operand %d (%s): %q is not a broadcast decoration, want {1toN}", mnem, pos, op.Raw, content) + } + dec.broadcast = true + default: + return "", dec, fmt.Errorf("%s: operand %d (%s): {%s} is not a decoration the layer reads: want {k1} through {k7}, {z}, {1toN}, {sae} or {rn-sae} through {rz-sae}", mnem, pos, op.Raw, content) + } + i += end + 1 + } + return body, dec, nil +} + +// amd64ExtDecor carries the brace decorations one operand's spelling names. +type amd64ExtDecor struct { + mask int + hasMask bool + zeroing bool + broadcast bool + round arch.ExtRounding +} diff --git a/asm/amd64_ext_asm_test.go b/asm/amd64_ext_asm_test.go new file mode 100644 index 0000000..4c38737 --- /dev/null +++ b/asm/amd64_ext_asm_test.go @@ -0,0 +1,284 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "encoding/hex" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-sdk/arch" + "sourcedock.dev/petrbalvin/gasm-sdk/parser" +) + +// assembleAmd64ExtBody parses src, assembles it for amd64 and returns the +// first function's body. Every statement must encode: a failure is the +// test's. +func assembleAmd64ExtBody(t *testing.T, src string) []byte { + t.Helper() + f, errs := parser.Parse("ext_amd64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFile(f) + if err != nil { + t.Fatalf("assemble: %v", err) + } + if len(img.Funcs) != 1 { + t.Fatalf("got %d functions, want 1", len(img.Funcs)) + } + return img.Code[img.Funcs[0].Offset:][:img.Funcs[0].Size] +} + +// assembleAmd64ExtError parses and assembles src and returns the assembler's +// error text. +func assembleAmd64ExtError(t *testing.T, src string) string { + t.Helper() + f, errs := parser.Parse("ext_amd64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + _, err := AssembleFile(f) + if err == nil { + t.Fatal("assembled, want an error") + } + return err.Error() +} + +const amd64ExtProbeHead = "#include \"textflag.h\"\nTEXT ·t(SB), NOSPLIT, $0\n" + +// amd64ExtProbe assembles one statement alone and returns the function body: +// exactly the statement's bytes, no trailing RET. +func amd64ExtProbe(t *testing.T, stmt string) []byte { + t.Helper() + return assembleAmd64ExtBody(t, amd64ExtProbeHead+"\t"+stmt+"\n") +} + +// TestAmd64AssembleExtensionGolden drives the wired layer through the full +// assembler: text in, machine bytes out. One statement per family, the +// decorations the layer spells beside them, and the memory mechanism's +// canonical choices; each want is the byte string the registry's golden +// vectors in arch/amd64_ext_test.go and arch/amd64_ext_mem_test.go already +// pin, so these prove the text-to-bytes path lands on the same encoding the +// metadata layer produces. +func TestAmd64AssembleExtensionGolden(t *testing.T) { + tests := []struct { + stmt string + want string + }{ + // AVX512-BF16: the two converts and the dot product, the write mask + // riding the destination in braces. + {"VCVTNE2PS2BF16 Z5, Z4, Z6", "62f2574872f4"}, + {"VCVTNE2PS2BF16 Z21, Z20, Z23", "62a2574072fc"}, + {"VCVTNEPS2BF16 Z5, Y6", "62f27e4872f5"}, + {"VCVTNEPS2BF16 Y5, X6{K6}", "62f27e2e72f5"}, + {"VDPBF16PS Z5, Z4, Z6{K5}", "62f2564d52f4"}, + // AVX512-VP2INTERSECT: sources first, the opmask destination last. + {"VP2INTERSECTD Y2, Y1, K2", "62f26f2868d1"}, + // AVX512-FP16, the scalar core: high registers, memory and the + // general-register pairs in both directions. + {"VADDSH X29, X28, X30", "6205160058f4"}, + {"VMINSH X5, X4, X6{SAE}", "62f556185df4"}, + {"VMOVSH (R9), X30", "62457e081031"}, + {"VCVTSH2SI (R9), R12", "6255fe082d21"}, + {"VCVTSH2SI X30, EDX", "62957e082dd6"}, + {"VCVTSI2SH X29, R12, X30", "624596002af4"}, + {"VMOVW R12, X30", "62457d086ef4"}, + {"VCMPSH $0x7b, X29, X28, K5", "62931600c2ec7b"}, + {"VGETMANTSH $0x0b, X29, X28, X30", "6203140027f40b"}, + // The packed FP16 arithmetic and the embedded rounding: {sae} and the + // four rounding modes compose with the write mask and zeroing. + {"VADDPH Z5, Z4, Z6{RN-SAE}", "62f5541858f4"}, + {"VADDPH Z5, Z4, Z6{RD-SAE}", "62f5543858f4"}, + {"VADDPH Z5, Z4, Z6{RU-SAE}", "62f5545858f4"}, + {"VADDPH Z5, Z4, Z6{RZ-SAE}", "62f5547858f4"}, + {"VADDPH Z29, Z28, Z30{K7}{Z}", "620514c758f4"}, + {"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "62f5547f58f4"}, + {"VADDPH Z28, (R9), Z30{K7}{Z}", "62451cc75831"}, + {"VSQRTPH Z29, Z30{K3}{Z}", "62057ccb51f5"}, + {"VFMADD132PH Z29, Z28, Z30", "6206154098f4"}, + // The packed conversions: full-width sources and the {1toN} broadcast + // over the integer sources. + {"VCVTPH2W Z5, Z6", "62f57d487df5"}, + {"VCVTPH2QQ X5, Z6{RZ-SAE}", "62f57d787bf5"}, + {"VCVTPH2PD X5, Z6", "62f57c485af5"}, + {"VCVTDQ2PH (R9){1TO8}, Y30", "62457c585b31"}, + {"VRNDSCALEPH $0x7b, Z5, Z6", "62f37c4808f57b"}, + // The complex families and the imm8 minimum-or-maximum pair. + {"VFCMULCPH Z29, Z28, Z30", "62061740d6f4"}, + {"VFMADDCPH Z29, Z28, Z30", "6206164056f4"}, + {"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "62f6571856f4"}, + {"VFMADDCSH X29, (R9), X30", "624616005731"}, + {"VMINMAXPH $0x88, Z29, (R9), Z30", "62431440523188"}, + {"VMINMAXSH $0x88, X28, (R9), X29", "62431c00532988"}, + // AVX-VNNI-INT16: the VEX word, its 256-bit length and the memory + // source. + {"VPDPWSUD X2, X1, X3", "c4e26ad2d9"}, + {"VPDPWUSDS Y10, Y15, Y8", "c4422dd3c7"}, + {"VPDPWSUD X2, 127(RCX), X1", "c4e26ad2497f"}, + // The memory mechanism through the ext statements: the disp8 and + // disp32 choices, the RSP-base SIB byte, RBP's forced displacement + // and the scaled index with its EVEX.X handling. + {"VMOVSH 127(RCX), X30", "62657e0810717f"}, + {"VMOVSH 8128(RDX), X30", "62657e0810b2c01f0000"}, + {"VMOVSH (R12), X30", "62457e08103424"}, + {"VMOVSH (RBP), X30", "62657e08107500"}, + {"VADDPH Z29, (RCX)(DX*1), Z30", "62651440583411"}, + {"VADDPH Z29, (RCX)(R12*2), Z30", "62251440583461"}, + {"VADDPH Z29, (RBP)(R14*8), Z30", "622514405874f500"}, + } + for _, tt := range tests { + body := amd64ExtProbe(t, tt.stmt) + if got := hex.EncodeToString(body); got != tt.want { + t.Errorf("%s:\n got %s\n want %s", tt.stmt, got, tt.want) + } + } +} + +// TestAmd64AssembleExtensionRegistryParity pins the layer's contract over a +// wider slice: every statement here assembles to exactly the bytes +// EncodeExtension produces for the model operands the statement spells, so +// the front end and the registry cannot drift apart unnoticed. +func TestAmd64AssembleExtensionRegistryParity(t *testing.T) { + tests := []struct { + stmt string + mnem string + ops []arch.ExtOperand + }{ + {"VCVTNE2PS2BF16 Z5, Z4, Z6", "VCVTNE2PS2BF16", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtZmm(6)}}, + {"VCVTNEPS2BF16 Y5, X6{K6}", "VCVTNEPS2BF16", + []arch.ExtOperand{arch.ExtYmm(5), arch.ExtWriteMasked(arch.ExtXmm(6), 6, false)}}, + {"VDPBF16PS Z5, Z4, Z6{K5}", "VDPBF16PS", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtWriteMasked(arch.ExtZmm(6), 5, false)}}, + {"VP2INTERSECTD Y2, Y1, K2", "VP2INTERSECTD", + []arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)}}, + {"VADDSH X29, X28, X30", "VADDSH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}}, + {"VMINSH X5, X4, X6{SAE}", "VMINSH", + []arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)}}, + {"VCVTSI2SH X29, R12, X30", "VCVTSI2SH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtGpr64(12), arch.ExtXmm(30)}}, + {"VCVTSH2SI X30, EDX", "VCVTSH2SI", + []arch.ExtOperand{arch.ExtXmm(30), arch.ExtGpr32(2)}}, + {"VCMPSH $0x7b, X29, X28, K5", "VCMPSH", + []arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtMask(5)}}, + {"VGETMANTSH $0x0b, X29, X28, X30", "VGETMANTSH", + []arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}}, + {"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "VADDPH", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), + arch.ExtRounded(arch.ExtWriteMasked(arch.ExtZmm(6), 7, false), arch.ExtRoundTruncate)}}, + {"VADDPH Z28, (R9), Z30{K7}{Z}", "VADDPH", + []arch.ExtOperand{arch.ExtZmm(28), arch.ExtMemory(9, 0), + arch.ExtWriteMasked(arch.ExtZmm(30), 7, true)}}, + {"VCVTDQ2PH (R9){1TO8}, Y30", "VCVTDQ2PH", + []arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)}}, + {"VFMADD132PH Z29, Z28, Z30", "VFMADD132PH", + []arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)}}, + {"VFMADD231PH Y5, (RCX){1TO8}, Y6", "VFMADD231PH", + []arch.ExtOperand{arch.ExtYmm(5), arch.ExtBroadcast(1, 0), arch.ExtYmm(6)}}, + {"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "VFCMADDCPH", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)}}, + {"VFMADDCSH X29, (R9), X30", "VFMADDCSH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)}}, + {"VMINMAXPH $0x88, Z29, (R9), Z30", "VMINMAXPH", + []arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)}}, + {"VPDPWSUD X2, 127(RCX), X1", "VPDPWSUD", + []arch.ExtOperand{arch.ExtXmm(2), arch.ExtMemory(1, 127), arch.ExtXmm(1)}}, + {"VPDPWSUD X2, (RCX)(R12*2), X1", "VPDPWSUD", + []arch.ExtOperand{arch.ExtXmm(2), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtXmm(1)}}, + {"VMOVSH X30, (R9)", "VMOVSH", + []arch.ExtOperand{arch.ExtXmm(30), arch.ExtMemory(9, 0)}}, + {"VPDPWUSDS Y10, Y15, Y8", "VPDPWUSDS", + []arch.ExtOperand{arch.ExtYmm(10), arch.ExtYmm(15), arch.ExtYmm(8)}}, + {"VMOVSH 8128(RDX), X30", "VMOVSH", + []arch.ExtOperand{arch.ExtMemory(2, 8128), arch.ExtXmm(30)}}, + {"VADDPH Z29, (RCX)(R12*2), Z30", "VADDPH", + []arch.ExtOperand{arch.ExtZmm(29), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtZmm(30)}}, + } + for _, tt := range tests { + body := amd64ExtProbe(t, tt.stmt) + want, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) + if err != nil { + t.Fatalf("%s: registry encode: %v", tt.stmt, err) + } + if !bytes.Equal(body, want) { + t.Errorf("%s:\n got %x\n want %x (the registry encoding)", tt.stmt, body, want) + } + } +} + +// TestAmd64AssembleExtensionRefusals pins the diagnostics a pinned statement +// gets from the layer instead of a scalar path's complaint, and the +// conversion's own diagnostics for spellings the layer cannot read. Where +// the Go toolchain knows a family member the shape of the message is its +// rejection bar; the FP16, BF16, VP2INTERSECT and VNNI-INT16 families take +// the GNU assembler's. +func TestAmd64AssembleExtensionRefusals(t *testing.T) { + tests := []struct { + stmt string + want string + }{ + {"VADDPH Z33, Z1, Z2", "not an extended-layer operand"}, + {"VADDPH Z5, Z4, Z6{K0}", "outside the masking registers k1-k7"}, + {"VADDPH Z5, Z4, Z6{Z}", "zeroing without a write mask"}, + {"VADDPH Y5, Y4, Y6{RZ-SAE}", "wants a ZMM register"}, + {"VADDPH Z5, Z4, Z6{SAE}", "spells {sae} without a mode"}, + {"VADDPH Z29, (RCX){RZ-SAE}, Z30", "the memory operand takes none"}, + {"VFMULCPH Z5, (RCX){1TO8}, Z6", "the entry's memory operand takes none"}, + {"VADDPH Z29, Z28, Z30{BOGUS}", "is not a decoration the layer reads"}, + {"VADDPH Z29, Z28, Z30{K7}{K3}", "carries two write masks"}, + {"VADDSH X5, X4, X6{K3}", "the entry's destination takes none"}, + {"VCMPSH $300, X29, X28, K5", "outside the unsigned byte range"}, + {"VGETMANTSH $0x20, X29, X28, X30", "the upper nibble of the mantissa control is reserved"}, + {"VCVTSI2SH X29, X12, X30", "general register"}, + {"VADDPH Z5, Z4", "got 2 operands"}, + {"VMOVSH (Z4), X30", "not an extended-layer operand"}, + {"VMOVSH foo+4(SB), X30", "not an extended-layer operand"}, + {"VMOVSH 8(RCX)(DX*3), X30", "outside the byte multipliers"}, + {"VMOVSH (K1), X30", "not an extended-layer operand"}, + } + for _, tt := range tests { + got := assembleAmd64ExtError(t, amd64ExtProbeHead+"\t"+tt.stmt+"\n") + if !strings.Contains(got, tt.want) { + t.Errorf("%s: error %q does not name %q", tt.stmt, got, tt.want) + } + } +} + +// TestAmd64AssembleExtensionLabelOffsets proves pass 1 and pass 2 agree on a +// function mixing two ext statements with a backward jump: the label sits +// exactly where the laid-down bytes put it, so the JMP's rel8 reaches it. +func TestAmd64AssembleExtensionLabelOffsets(t *testing.T) { + body := assembleAmd64ExtBody(t, amd64ExtProbeHead+` + VADDPH Z1, Z2, Z3 +loop: + VFMADD132PH Z1, Z2, Z3 + JMP loop +`) + // Two six-byte EVEX words, then the short JMP whose displacement + // measures from its own end (14) back to the label (6). + if len(body) != 14 { + t.Fatalf("body is %d bytes, want 14", len(body)) + } + if body[12] != 0xEB || body[13] != 0xF8 { + t.Errorf("JMP encoded % x, want ebf8", body[12:14]) + } +} + +// TestAmd64AssembleExtensionLeavesTheMainEncoderAlone pins the non-invasion +// promise on the amd64 side: VPOPCNTD, an AVX-512 instruction the toolchain +// knows and the layer deliberately does not carry, encodes through the main +// EVEX path, byte for byte what that path produces on its own. +func TestAmd64AssembleExtensionLeavesTheMainEncoderAlone(t *testing.T) { + body := amd64ExtProbe(t, "VPOPCNTD Z1, Z2") + e := &enc{} + if err := e.encode("VPOPCNTD", []Operand{Reg{idx: 1, size: 64}, Reg{idx: 2, size: 64}}); err != nil { + t.Fatalf("main encoder: %v", err) + } + if !bytes.Equal(body, e.out) { + t.Errorf("VPOPCNTD Z1, Z2: got %x, want the main encoder's %x", body, e.out) + } +} diff --git a/asm/assemble.go b/asm/assemble.go index 1e879ef..d1f6993 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -8,6 +8,7 @@ import ( "strconv" "strings" + "sourcedock.dev/petrbalvin/gasm-sdk/arch" "sourcedock.dev/petrbalvin/gasm-sdk/ast" ) @@ -929,6 +930,25 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { mnemUpper := strings.ToUpper(s.Mnemonic.Text) + + // The extended-instruction layer: a statement whose mnemonic is + // registered in the extension registry and that the scalar paths cannot + // encode goes through the registry, before operandFromAST would reject + // operands only the layer reads (asm/amd64_ext.go). The registry's + // Feature field stays metadata at assembly time: the assembler has no + // CPU, and the toolchain does not gate assembly on CPU features, so + // every registered feature assembles. + if extops, pinned, convErr := amd64ExtStatement(mnemUpper, s.Operands); pinned { + if convErr != nil { + return nil, nil, nil, convErr + } + code, encErr := EncodeExtension(arch.AMD64, mnemUpper, extops...) + if encErr != nil { + return nil, nil, nil, encErr + } + return code, nil, nil, nil + } + if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" { code, err := encodeBookkeeping(mnemUpper, s) if err != nil {