// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "bytes" "encoding/hex" "strings" "testing" "sourcedock.dev/petrbalvin/gasm-sdk/arch" "sourcedock.dev/petrbalvin/gasm-sdk/parser" ) // assembleAmd64ExtBody parses src, assembles it for amd64 and returns the // first function's body. Every statement must encode: a failure is the // test's. func assembleAmd64ExtBody(t *testing.T, src string) []byte { t.Helper() f, errs := parser.Parse("ext_amd64.s", src) if len(errs) > 0 { t.Fatalf("parse: %v", errs) } img, err := AssembleFile(f) if err != nil { t.Fatalf("assemble: %v", err) } if len(img.Funcs) != 1 { t.Fatalf("got %d functions, want 1", len(img.Funcs)) } return img.Code[img.Funcs[0].Offset:][:img.Funcs[0].Size] } // assembleAmd64ExtError parses and assembles src and returns the assembler's // error text. func assembleAmd64ExtError(t *testing.T, src string) string { t.Helper() f, errs := parser.Parse("ext_amd64.s", src) if len(errs) > 0 { t.Fatalf("parse: %v", errs) } _, err := AssembleFile(f) if err == nil { t.Fatal("assembled, want an error") } return err.Error() } const amd64ExtProbeHead = "#include \"textflag.h\"\nTEXT ·t(SB), NOSPLIT, $0\n" // amd64ExtProbe assembles one statement alone and returns the function body: // exactly the statement's bytes, no trailing RET. func amd64ExtProbe(t *testing.T, stmt string) []byte { t.Helper() return assembleAmd64ExtBody(t, amd64ExtProbeHead+"\t"+stmt+"\n") } // TestAmd64AssembleExtensionGolden drives the wired layer through the full // assembler: text in, machine bytes out. One statement per family, the // decorations the layer spells beside them, and the memory mechanism's // canonical choices; each want is the byte string the registry's golden // vectors in arch/amd64_ext_test.go and arch/amd64_ext_mem_test.go already // pin, so these prove the text-to-bytes path lands on the same encoding the // metadata layer produces. func TestAmd64AssembleExtensionGolden(t *testing.T) { tests := []struct { stmt string want string }{ // AVX512-BF16: the two converts and the dot product, the write mask // riding the destination in braces. {"VCVTNE2PS2BF16 Z5, Z4, Z6", "62f2574872f4"}, {"VCVTNE2PS2BF16 Z21, Z20, Z23", "62a2574072fc"}, {"VCVTNEPS2BF16 Z5, Y6", "62f27e4872f5"}, {"VCVTNEPS2BF16 Y5, X6{K6}", "62f27e2e72f5"}, {"VDPBF16PS Z5, Z4, Z6{K5}", "62f2564d52f4"}, // AVX512-VP2INTERSECT: sources first, the opmask destination last. {"VP2INTERSECTD Y2, Y1, K2", "62f26f2868d1"}, // AVX512-FP16, the scalar core: high registers, memory and the // general-register pairs in both directions. {"VADDSH X29, X28, X30", "6205160058f4"}, {"VMINSH X5, X4, X6{SAE}", "62f556185df4"}, {"VMOVSH (R9), X30", "62457e081031"}, {"VCVTSH2SI (R9), R12", "6255fe082d21"}, {"VCVTSH2SI X30, EDX", "62957e082dd6"}, {"VCVTSI2SH X29, R12, X30", "624596002af4"}, {"VMOVW R12, X30", "62457d086ef4"}, {"VCMPSH $0x7b, X29, X28, K5", "62931600c2ec7b"}, {"VGETMANTSH $0x0b, X29, X28, X30", "6203140027f40b"}, // The packed FP16 arithmetic and the embedded rounding: {sae} and the // four rounding modes compose with the write mask and zeroing. {"VADDPH Z5, Z4, Z6{RN-SAE}", "62f5541858f4"}, {"VADDPH Z5, Z4, Z6{RD-SAE}", "62f5543858f4"}, {"VADDPH Z5, Z4, Z6{RU-SAE}", "62f5545858f4"}, {"VADDPH Z5, Z4, Z6{RZ-SAE}", "62f5547858f4"}, {"VADDPH Z29, Z28, Z30{K7}{Z}", "620514c758f4"}, {"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "62f5547f58f4"}, {"VADDPH Z28, (R9), Z30{K7}{Z}", "62451cc75831"}, {"VSQRTPH Z29, Z30{K3}{Z}", "62057ccb51f5"}, {"VFMADD132PH Z29, Z28, Z30", "6206154098f4"}, // The packed conversions: full-width sources and the {1toN} broadcast // over the integer sources. {"VCVTPH2W Z5, Z6", "62f57d487df5"}, {"VCVTPH2QQ X5, Z6{RZ-SAE}", "62f57d787bf5"}, {"VCVTPH2PD X5, Z6", "62f57c485af5"}, {"VCVTDQ2PH (R9){1TO8}, Y30", "62457c585b31"}, {"VRNDSCALEPH $0x7b, Z5, Z6", "62f37c4808f57b"}, // The complex families and the imm8 minimum-or-maximum pair. {"VFCMULCPH Z29, Z28, Z30", "62061740d6f4"}, {"VFMADDCPH Z29, Z28, Z30", "6206164056f4"}, {"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "62f6571856f4"}, {"VFMADDCSH X29, (R9), X30", "624616005731"}, {"VMINMAXPH $0x88, Z29, (R9), Z30", "62431440523188"}, {"VMINMAXSH $0x88, X28, (R9), X29", "62431c00532988"}, // AVX-VNNI-INT16: the VEX word, its 256-bit length and the memory // source. {"VPDPWSUD X2, X1, X3", "c4e26ad2d9"}, {"VPDPWUSDS Y10, Y15, Y8", "c4422dd3c7"}, {"VPDPWSUD X2, 127(RCX), X1", "c4e26ad2497f"}, // The memory mechanism through the ext statements: the disp8 and // disp32 choices, the RSP-base SIB byte, RBP's forced displacement // and the scaled index with its EVEX.X handling. {"VMOVSH 127(RCX), X30", "62657e0810717f"}, {"VMOVSH 8128(RDX), X30", "62657e0810b2c01f0000"}, {"VMOVSH (R12), X30", "62457e08103424"}, {"VMOVSH (RBP), X30", "62657e08107500"}, {"VADDPH Z29, (RCX)(DX*1), Z30", "62651440583411"}, {"VADDPH Z29, (RCX)(R12*2), Z30", "62251440583461"}, {"VADDPH Z29, (RBP)(R14*8), Z30", "622514405874f500"}, } for _, tt := range tests { body := amd64ExtProbe(t, tt.stmt) if got := hex.EncodeToString(body); got != tt.want { t.Errorf("%s:\n got %s\n want %s", tt.stmt, got, tt.want) } } } // TestAmd64AssembleExtensionRegistryParity pins the layer's contract over a // wider slice: every statement here assembles to exactly the bytes // EncodeExtension produces for the model operands the statement spells, so // the front end and the registry cannot drift apart unnoticed. func TestAmd64AssembleExtensionRegistryParity(t *testing.T) { tests := []struct { stmt string mnem string ops []arch.ExtOperand }{ {"VCVTNE2PS2BF16 Z5, Z4, Z6", "VCVTNE2PS2BF16", []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtZmm(6)}}, {"VCVTNEPS2BF16 Y5, X6{K6}", "VCVTNEPS2BF16", []arch.ExtOperand{arch.ExtYmm(5), arch.ExtWriteMasked(arch.ExtXmm(6), 6, false)}}, {"VDPBF16PS Z5, Z4, Z6{K5}", "VDPBF16PS", []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtWriteMasked(arch.ExtZmm(6), 5, false)}}, {"VP2INTERSECTD Y2, Y1, K2", "VP2INTERSECTD", []arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)}}, {"VADDSH X29, X28, X30", "VADDSH", []arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}}, {"VMINSH X5, X4, X6{SAE}", "VMINSH", []arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)}}, {"VCVTSI2SH X29, R12, X30", "VCVTSI2SH", []arch.ExtOperand{arch.ExtXmm(29), arch.ExtGpr64(12), arch.ExtXmm(30)}}, {"VCVTSH2SI X30, EDX", "VCVTSH2SI", []arch.ExtOperand{arch.ExtXmm(30), arch.ExtGpr32(2)}}, {"VCMPSH $0x7b, X29, X28, K5", "VCMPSH", []arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtMask(5)}}, {"VGETMANTSH $0x0b, X29, X28, X30", "VGETMANTSH", []arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}}, {"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "VADDPH", []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtWriteMasked(arch.ExtZmm(6), 7, false), arch.ExtRoundTruncate)}}, {"VADDPH Z28, (R9), Z30{K7}{Z}", "VADDPH", []arch.ExtOperand{arch.ExtZmm(28), arch.ExtMemory(9, 0), arch.ExtWriteMasked(arch.ExtZmm(30), 7, true)}}, {"VCVTDQ2PH (R9){1TO8}, Y30", "VCVTDQ2PH", []arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)}}, {"VFMADD132PH Z29, Z28, Z30", "VFMADD132PH", []arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)}}, {"VFMADD231PH Y5, (RCX){1TO8}, Y6", "VFMADD231PH", []arch.ExtOperand{arch.ExtYmm(5), arch.ExtBroadcast(1, 0), arch.ExtYmm(6)}}, {"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "VFCMADDCPH", []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)}}, {"VFMADDCSH X29, (R9), X30", "VFMADDCSH", []arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)}}, {"VMINMAXPH $0x88, Z29, (R9), Z30", "VMINMAXPH", []arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)}}, {"VPDPWSUD X2, 127(RCX), X1", "VPDPWSUD", []arch.ExtOperand{arch.ExtXmm(2), arch.ExtMemory(1, 127), arch.ExtXmm(1)}}, {"VPDPWSUD X2, (RCX)(R12*2), X1", "VPDPWSUD", []arch.ExtOperand{arch.ExtXmm(2), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtXmm(1)}}, {"VMOVSH X30, (R9)", "VMOVSH", []arch.ExtOperand{arch.ExtXmm(30), arch.ExtMemory(9, 0)}}, {"VPDPWUSDS Y10, Y15, Y8", "VPDPWUSDS", []arch.ExtOperand{arch.ExtYmm(10), arch.ExtYmm(15), arch.ExtYmm(8)}}, {"VMOVSH 8128(RDX), X30", "VMOVSH", []arch.ExtOperand{arch.ExtMemory(2, 8128), arch.ExtXmm(30)}}, {"VADDPH Z29, (RCX)(R12*2), Z30", "VADDPH", []arch.ExtOperand{arch.ExtZmm(29), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtZmm(30)}}, } for _, tt := range tests { body := amd64ExtProbe(t, tt.stmt) want, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil { t.Fatalf("%s: registry encode: %v", tt.stmt, err) } if !bytes.Equal(body, want) { t.Errorf("%s:\n got %x\n want %x (the registry encoding)", tt.stmt, body, want) } } } // TestAmd64AssembleExtensionRefusals pins the diagnostics a pinned statement // gets from the layer instead of a scalar path's complaint, and the // conversion's own diagnostics for spellings the layer cannot read. Where // the Go toolchain knows a family member the shape of the message is its // rejection bar; the FP16, BF16, VP2INTERSECT and VNNI-INT16 families take // the GNU assembler's. func TestAmd64AssembleExtensionRefusals(t *testing.T) { tests := []struct { stmt string want string }{ {"VADDPH Z33, Z1, Z2", "not an extended-layer operand"}, {"VADDPH Z5, Z4, Z6{K0}", "outside the masking registers k1-k7"}, {"VADDPH Z5, Z4, Z6{Z}", "zeroing without a write mask"}, {"VADDPH Y5, Y4, Y6{RZ-SAE}", "wants a ZMM register"}, {"VADDPH Z5, Z4, Z6{SAE}", "spells {sae} without a mode"}, {"VADDPH Z29, (RCX){RZ-SAE}, Z30", "the memory operand takes none"}, {"VFMULCPH Z5, (RCX){1TO8}, Z6", "the entry's memory operand takes none"}, {"VADDPH Z29, Z28, Z30{BOGUS}", "is not a decoration the layer reads"}, {"VADDPH Z29, Z28, Z30{K7}{K3}", "carries two write masks"}, {"VADDSH X5, X4, X6{K3}", "the entry's destination takes none"}, {"VCMPSH $300, X29, X28, K5", "outside the unsigned byte range"}, {"VGETMANTSH $0x20, X29, X28, X30", "the upper nibble of the mantissa control is reserved"}, {"VCVTSI2SH X29, X12, X30", "general register"}, {"VADDPH Z5, Z4", "got 2 operands"}, {"VMOVSH (Z4), X30", "not an extended-layer operand"}, {"VMOVSH foo+4(SB), X30", "not an extended-layer operand"}, {"VMOVSH 8(RCX)(DX*3), X30", "outside the byte multipliers"}, {"VMOVSH (K1), X30", "not an extended-layer operand"}, } for _, tt := range tests { got := assembleAmd64ExtError(t, amd64ExtProbeHead+"\t"+tt.stmt+"\n") if !strings.Contains(got, tt.want) { t.Errorf("%s: error %q does not name %q", tt.stmt, got, tt.want) } } } // TestAmd64AssembleExtensionLabelOffsets proves pass 1 and pass 2 agree on a // function mixing two ext statements with a backward jump: the label sits // exactly where the laid-down bytes put it, so the JMP's rel8 reaches it. func TestAmd64AssembleExtensionLabelOffsets(t *testing.T) { body := assembleAmd64ExtBody(t, amd64ExtProbeHead+` VADDPH Z1, Z2, Z3 loop: VFMADD132PH Z1, Z2, Z3 JMP loop `) // Two six-byte EVEX words, then the short JMP whose displacement // measures from its own end (14) back to the label (6). if len(body) != 14 { t.Fatalf("body is %d bytes, want 14", len(body)) } if body[12] != 0xEB || body[13] != 0xF8 { t.Errorf("JMP encoded % x, want ebf8", body[12:14]) } } // TestAmd64AssembleExtensionLeavesTheMainEncoderAlone pins the non-invasion // promise on the amd64 side: VPOPCNTD, an AVX-512 instruction the toolchain // knows and the layer deliberately does not carry, encodes through the main // EVEX path, byte for byte what that path produces on its own. func TestAmd64AssembleExtensionLeavesTheMainEncoderAlone(t *testing.T) { body := amd64ExtProbe(t, "VPOPCNTD Z1, Z2") e := &enc{} if err := e.encode("VPOPCNTD", []Operand{Reg{idx: 1, size: 64}, Reg{idx: 2, size: 64}}); err != nil { t.Fatalf("main encoder: %v", err) } if !bytes.Equal(body, e.out) { t.Errorf("VPOPCNTD Z1, Z2: got %x, want the main encoder's %x", body, e.out) } }