// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "bytes" "encoding/hex" "os" "path/filepath" "testing" ) // The byte forms of the suffixed scalar families answer to the register // operands as much as to the mnemonic: the toolchain's own disassembly // prints them with the L suffix or none at all (the rendered suffix rides // the operand-size attribute), the register names carrying the width. The // tests here pin that reconciliation: the renderer's spellings encode the // byte form byte for byte, the W and Q spellings never ride a byte // register, and the classic names stay size-agnostic. // TestOperandWidthByteForms pins the renderer's spellings: an L suffix or // no suffix with a byte-spelled register encodes the 8-bit form, every // register joining at its low byte. Every row's bytes were cross-checked // against `go tool asm -S` over the B-suffixed spelling of the same // operands (the toolchain rejects the L spelling itself), and against the // toolchain's objdump text for the byte encodings. func TestOperandWidthByteForms(t *testing.T) { r11 := Reg{idx: 11, size: 8} r8 := Reg{idx: 8, size: 1} r9 := Reg{idx: 9, size: 1} for _, tt := range []struct { name string ops []Operand want string }{ // The atomics pair: XADD and CMPXCHG drop to the 0F C0/0F B0 byte // opcodes the L spelling would otherwise widen past. {"XADDL DL, DL", []Operand{DL, DL}, "0fc0d2"}, {"XADDB DL, DL", []Operand{DL, DL}, "0fc0d2"}, {"XADDL R8B, R9B", []Operand{r8, r9}, "450fc0c1"}, {"CMPXCHGL DL, DL", []Operand{DL, DL}, "0fb0d2"}, {"XCHGL DL, DL", []Operand{DL, DL}, "86d2"}, {"XCHGL DL, 0(BX)", []Operand{DL, Ptr(BX, 0, 1)}, "8613"}, // The ALU immediates: the accumulator short form for the AL // spelling, the generic 0x80 /digit elsewhere. {"CMPL AL, $7", []Operand{AL, Imm(7)}, "3c07"}, {"ADDL $7, AL", []Operand{Imm(7), AL}, "0407"}, {"SUBL $7, AL", []Operand{Imm(7), AL}, "2c07"}, {"ANDL $7, AL", []Operand{Imm(7), AL}, "2407"}, {"SBBL $7, AL", []Operand{Imm(7), AL}, "1c07"}, {"SBBL $7, DL", []Operand{Imm(7), DL}, "80da07"}, {"ADDB $3, AX", []Operand{Imm(3), AX}, "80c003"}, {"ORB $7, AX", []Operand{Imm(7), AX}, "80c807"}, {"SBBB $7, AX", []Operand{Imm(7), AX}, "80d807"}, // The register forms, the extended register riding REX.B and the // byte source in the reg field. {"SBBL DL, R11", []Operand{DL, r11}, "4118d3"}, {"TESTL R11, DL", []Operand{r11, DL}, "4484da"}, {"TESTB $7, AX", []Operand{Imm(7), AX}, "f6c007"}, {"TESTB R11, DL", []Operand{r11, DL}, "4484da"}, // CRC32 keeps the F0 byte opcode under the unsuffixed spelling. {"CRC32 DL, R11", []Operand{DL, r11}, "f2440f38f0da"}, {"CRC32B DL, R11", []Operand{DL, r11}, "f2440f38f0da"}, {"CRC32B AX, CX", []Operand{AX, CX}, "f20f38f0c8"}, // The move and unary families follow the same rule. {"MOVL $7, DL", []Operand{Imm(7), DL}, "b207"}, {"MOVB $7, DL", []Operand{Imm(7), DL}, "b207"}, {"MOVB AX, AL", []Operand{AX, AL}, "88c0"}, {"INCL DL", []Operand{DL}, "fec2"}, {"NEGL DL", []Operand{DL}, "f6da"}, {"IMULL DL", []Operand{DL}, "f6ea"}, {"SHLL $2, DL", []Operand{Imm(2), DL}, "c0e202"}, {"ROLL CL, DL", []Operand{CL, DL}, "d2c2"}, // The shift count never narrows the shifted value. {"RCLW CL, 0(R11)", []Operand{CL, Ptr(r11, 0, 2)}, "6641d313"}, {"RORQ CL, AX", []Operand{CL, AX}, "48d3c8"}, // The size-agnostic spellings stay width-free: the L and Q forms // of the same families are untouched by the reconciliation. {"XADDL AX, CX", []Operand{AX, CX}, "0fc1c1"}, {"ADDL $7, AX", []Operand{Imm(7), AX}, "83c007"}, {"ADDL $256, AX", []Operand{Imm(256), AX}, "0500010000"}, {"CRC32L AX, CX", []Operand{AX, CX}, "f20f38f1c8"}, } { t.Run(tt.name, func(t *testing.T) { got, err := Encode(mnemonicOf(tt.name), tt.ops...) if err != nil { t.Fatalf("%s: %v", tt.name, err) } if want := unhex(tt.want); !bytes.Equal(got, want) { t.Errorf("%s: % x, want % x", tt.name, got, want) } }) } } // TestOperandWidthConflicts pins the refusals: the W and Q spellings never // ride a byte register (go tool asm rejects MOVQ AL, AX and its siblings // outright), and neither does the MOVD alias of the quad move. func TestOperandWidthConflicts(t *testing.T) { for _, tt := range []struct { name string ops []Operand }{ {"MOVQ AL, AX", []Operand{AL, AX}}, {"MOVQ AX, AL", []Operand{AX, AL}}, {"MOVQ DL, DL", []Operand{DL, DL}}, {"MOVW $7, DL", []Operand{Imm(7), DL}}, {"MOVD AL, AX", []Operand{AL, AX}}, {"XADDQ DL, DL", []Operand{DL, DL}}, {"CMPXCHGQ DL, DL", []Operand{DL, DL}}, {"XCHGQ DL, DL", []Operand{DL, DL}}, {"SHLQ $2, DL", []Operand{Imm(2), DL}}, {"INCQ DL", []Operand{DL}}, {"IMULQ DL", []Operand{DL}}, {"TESTQ R11, DL", []Operand{Reg{idx: 11, size: 8}, DL}}, {"CRC32Q DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}}, {"CRC32W DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}}, } { t.Run(tt.name, func(t *testing.T) { if _, err := Encode(mnemonicOf(tt.name), tt.ops...); err == nil { t.Errorf("%s: encoded, want the byte-register conflict refused", tt.name) } }) } } // TestOperandWidthDifferential is the byte-parity oracle for the same // reconciliation: the B-suffixed spellings, which go tool asm accepts, must // encode identically through both assemblers, the agnostic names included // (their low byte joins the byte form) and the accumulator division (AL // short, AX generic) with them. func TestOperandWidthDifferential(t *testing.T) { kernel := "#include \"textflag.h\"\n" + "TEXT \u00b7bytewidth(SB), NOSPLIT, $0\n" + "\tXADDB DL, DL\n" + "\tXADDB R8B, R9B\n" + "\tXADDL AX, CX\n" + "\tCMPXCHGB DL, DL\n" + "\tXCHGB DL, DL\n" + "\tXCHGB DL, 0(BX)\n" + "\tCMPB AL, $7\n" + "\tADDB $7, AL\n" + "\tADDB $3, AX\n" + "\tSUBB $7, AL\n" + "\tANDB $7, AL\n" + "\tSBBB $7, AL\n" + "\tSBBB $7, DL\n" + "\tSBBB DL, R11\n" + "\tORB $7, AX\n" + "\tTESTB R11, DL\n" + "\tTESTB $7, AX\n" + "\tCRC32B DL, R11\n" + "\tCRC32B AX, CX\n" + "\tCRC32B R8B, CX\n" + "\tCRC32L AX, CX\n" + "\tMOVB $7, DL\n" + "\tMOVB $3, AX\n" + "\tMOVB AX, AL\n" + "\tINCB DL\n" + "\tNEGB DL\n" + "\tIMULB DL\n" + "\tMULB CL\n" + "\tSHLB $2, DL\n" + "\tROLB CL, DL\n" + "\tRET\n" path := filepath.Join(t.TempDir(), "bytewidth_amd64.s") if err := os.WriteFile(path, []byte(kernel), 0o644); err != nil { t.Fatal(err) } gt := oracleFuncCode(t, toolAsmObject(t, path, "")) goCode, ok := gt["bytewidth.bytewidth"] if !ok { t.Fatalf("oracle: bytewidth function missing (%d functions)", len(gt)) } gasmCode := code(path, kernel) if gasmCode == nil { t.Fatal("gasm: assemble failed") } if !bytes.Equal(gasmCode, goCode) { t.Errorf("byte width kernel:\ngasm %x\ngo %x", gasmCode, goCode) } } // mnemonicOf returns the first whitespace-free token of a rendered row. func mnemonicOf(text string) string { for i := 0; i < len(text); i++ { if text[i] == ' ' || text[i] == ',' { return text[:i] } } return text } // unhex decodes a hex string, failing the test on malformed input. func unhex(s string) []byte { out, err := hex.DecodeString(s) if err != nil { panic("unhex: " + err.Error()) } return out }