From 8bded3ea39523d80947f0cb443a703fcc75ba76c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 12:58:34 +0200 Subject: [PATCH] test(asm): pin the byte-form width reconciliation and its oracle Table-driven rows for the renderer's spellings (the L suffix or none with a byte register encodes the byte form, every register joining at its low byte), the refusals (W, Q and the MOVD alias take no byte register) and a differential kernel of the B-suffixed spellings assembled through both gasm and go tool asm, byte for byte. Assisted-by: GLM 5.3 --- asm/amd64_width_test.go | 202 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 202 insertions(+) create mode 100644 asm/amd64_width_test.go diff --git a/asm/amd64_width_test.go b/asm/amd64_width_test.go new file mode 100644 index 0000000..06306f1 --- /dev/null +++ b/asm/amd64_width_test.go @@ -0,0 +1,202 @@ +// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "encoding/hex" + "os" + "path/filepath" + "testing" +) + +// The byte forms of the suffixed scalar families answer to the register +// operands as much as to the mnemonic: the toolchain's own disassembly +// prints them with the L suffix or none at all (the rendered suffix rides +// the operand-size attribute), the register names carrying the width. The +// tests here pin that reconciliation: the renderer's spellings encode the +// byte form byte for byte, the W and Q spellings never ride a byte +// register, and the classic names stay size-agnostic. + +// TestOperandWidthByteForms pins the renderer's spellings: an L suffix or +// no suffix with a byte-spelled register encodes the 8-bit form, every +// register joining at its low byte. Every row's bytes were cross-checked +// against `go tool asm -S` over the B-suffixed spelling of the same +// operands (the toolchain rejects the L spelling itself), and against the +// toolchain's objdump text for the byte encodings. +func TestOperandWidthByteForms(t *testing.T) { + r11 := Reg{idx: 11, size: 8} + r8 := Reg{idx: 8, size: 1} + r9 := Reg{idx: 9, size: 1} + for _, tt := range []struct { + name string + ops []Operand + want string + }{ + // The atomics pair: XADD and CMPXCHG drop to the 0F C0/0F B0 byte + // opcodes the L spelling would otherwise widen past. + {"XADDL DL, DL", []Operand{DL, DL}, "0fc0d2"}, + {"XADDB DL, DL", []Operand{DL, DL}, "0fc0d2"}, + {"XADDL R8B, R9B", []Operand{r8, r9}, "450fc0c1"}, + {"CMPXCHGL DL, DL", []Operand{DL, DL}, "0fb0d2"}, + {"XCHGL DL, DL", []Operand{DL, DL}, "86d2"}, + {"XCHGL DL, 0(BX)", []Operand{DL, Ptr(BX, 0, 1)}, "8613"}, + // The ALU immediates: the accumulator short form for the AL + // spelling, the generic 0x80 /digit elsewhere. + {"CMPL AL, $7", []Operand{AL, Imm(7)}, "3c07"}, + {"ADDL $7, AL", []Operand{Imm(7), AL}, "0407"}, + {"SUBL $7, AL", []Operand{Imm(7), AL}, "2c07"}, + {"ANDL $7, AL", []Operand{Imm(7), AL}, "2407"}, + {"SBBL $7, AL", []Operand{Imm(7), AL}, "1c07"}, + {"SBBL $7, DL", []Operand{Imm(7), DL}, "80da07"}, + {"ADDB $3, AX", []Operand{Imm(3), AX}, "80c003"}, + {"ORB $7, AX", []Operand{Imm(7), AX}, "80c807"}, + {"SBBB $7, AX", []Operand{Imm(7), AX}, "80d807"}, + // The register forms, the extended register riding REX.B and the + // byte source in the reg field. + {"SBBL DL, R11", []Operand{DL, r11}, "4118d3"}, + {"TESTL R11, DL", []Operand{r11, DL}, "4484da"}, + {"TESTB $7, AX", []Operand{Imm(7), AX}, "f6c007"}, + {"TESTB R11, DL", []Operand{r11, DL}, "4484da"}, + // CRC32 keeps the F0 byte opcode under the unsuffixed spelling. + {"CRC32 DL, R11", []Operand{DL, r11}, "f2440f38f0da"}, + {"CRC32B DL, R11", []Operand{DL, r11}, "f2440f38f0da"}, + {"CRC32B AX, CX", []Operand{AX, CX}, "f20f38f0c8"}, + // The move and unary families follow the same rule. + {"MOVL $7, DL", []Operand{Imm(7), DL}, "b207"}, + {"MOVB $7, DL", []Operand{Imm(7), DL}, "b207"}, + {"MOVB AX, AL", []Operand{AX, AL}, "88c0"}, + {"INCL DL", []Operand{DL}, "fec2"}, + {"NEGL DL", []Operand{DL}, "f6da"}, + {"IMULL DL", []Operand{DL}, "f6ea"}, + {"SHLL $2, DL", []Operand{Imm(2), DL}, "c0e202"}, + {"ROLL CL, DL", []Operand{CL, DL}, "d2c2"}, + // The shift count never narrows the shifted value. + {"RCLW CL, 0(R11)", []Operand{CL, Ptr(r11, 0, 2)}, "6641d313"}, + {"RORQ CL, AX", []Operand{CL, AX}, "48d3c8"}, + // The size-agnostic spellings stay width-free: the L and Q forms + // of the same families are untouched by the reconciliation. + {"XADDL AX, CX", []Operand{AX, CX}, "0fc1c1"}, + {"ADDL $7, AX", []Operand{Imm(7), AX}, "83c007"}, + {"ADDL $256, AX", []Operand{Imm(256), AX}, "0500010000"}, + {"CRC32L AX, CX", []Operand{AX, CX}, "f20f38f1c8"}, + } { + t.Run(tt.name, func(t *testing.T) { + got, err := Encode(mnemonicOf(tt.name), tt.ops...) + if err != nil { + t.Fatalf("%s: %v", tt.name, err) + } + if want := unhex(tt.want); !bytes.Equal(got, want) { + t.Errorf("%s: % x, want % x", tt.name, got, want) + } + }) + } +} + +// TestOperandWidthConflicts pins the refusals: the W and Q spellings never +// ride a byte register (go tool asm rejects MOVQ AL, AX and its siblings +// outright), and neither does the MOVD alias of the quad move. +func TestOperandWidthConflicts(t *testing.T) { + for _, tt := range []struct { + name string + ops []Operand + }{ + {"MOVQ AL, AX", []Operand{AL, AX}}, + {"MOVQ AX, AL", []Operand{AX, AL}}, + {"MOVQ DL, DL", []Operand{DL, DL}}, + {"MOVW $7, DL", []Operand{Imm(7), DL}}, + {"MOVD AL, AX", []Operand{AL, AX}}, + {"XADDQ DL, DL", []Operand{DL, DL}}, + {"CMPXCHGQ DL, DL", []Operand{DL, DL}}, + {"XCHGQ DL, DL", []Operand{DL, DL}}, + {"SHLQ $2, DL", []Operand{Imm(2), DL}}, + {"INCQ DL", []Operand{DL}}, + {"IMULQ DL", []Operand{DL}}, + {"TESTQ R11, DL", []Operand{Reg{idx: 11, size: 8}, DL}}, + {"CRC32Q DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}}, + {"CRC32W DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}}, + } { + t.Run(tt.name, func(t *testing.T) { + if _, err := Encode(mnemonicOf(tt.name), tt.ops...); err == nil { + t.Errorf("%s: encoded, want the byte-register conflict refused", tt.name) + } + }) + } +} + +// TestOperandWidthDifferential is the byte-parity oracle for the same +// reconciliation: the B-suffixed spellings, which go tool asm accepts, must +// encode identically through both assemblers, the agnostic names included +// (their low byte joins the byte form) and the accumulator division (AL +// short, AX generic) with them. +func TestOperandWidthDifferential(t *testing.T) { + kernel := "#include \"textflag.h\"\n" + + "TEXT \u00b7bytewidth(SB), NOSPLIT, $0\n" + + "\tXADDB DL, DL\n" + + "\tXADDB R8B, R9B\n" + + "\tXADDL AX, CX\n" + + "\tCMPXCHGB DL, DL\n" + + "\tXCHGB DL, DL\n" + + "\tXCHGB DL, 0(BX)\n" + + "\tCMPB AL, $7\n" + + "\tADDB $7, AL\n" + + "\tADDB $3, AX\n" + + "\tSUBB $7, AL\n" + + "\tANDB $7, AL\n" + + "\tSBBB $7, AL\n" + + "\tSBBB $7, DL\n" + + "\tSBBB DL, R11\n" + + "\tORB $7, AX\n" + + "\tTESTB R11, DL\n" + + "\tTESTB $7, AX\n" + + "\tCRC32B DL, R11\n" + + "\tCRC32B AX, CX\n" + + "\tCRC32B R8B, CX\n" + + "\tCRC32L AX, CX\n" + + "\tMOVB $7, DL\n" + + "\tMOVB $3, AX\n" + + "\tMOVB AX, AL\n" + + "\tINCB DL\n" + + "\tNEGB DL\n" + + "\tIMULB DL\n" + + "\tMULB CL\n" + + "\tSHLB $2, DL\n" + + "\tROLB CL, DL\n" + + "\tRET\n" + path := filepath.Join(t.TempDir(), "bytewidth_amd64.s") + if err := os.WriteFile(path, []byte(kernel), 0o644); err != nil { + t.Fatal(err) + } + gt := oracleFuncCode(t, toolAsmObject(t, path, "")) + goCode, ok := gt["bytewidth.bytewidth"] + if !ok { + t.Fatalf("oracle: bytewidth function missing (%d functions)", len(gt)) + } + gasmCode := code(path, kernel) + if gasmCode == nil { + t.Fatal("gasm: assemble failed") + } + if !bytes.Equal(gasmCode, goCode) { + t.Errorf("byte width kernel:\ngasm %x\ngo %x", gasmCode, goCode) + } +} + +// mnemonicOf returns the first whitespace-free token of a rendered row. +func mnemonicOf(text string) string { + for i := 0; i < len(text); i++ { + if text[i] == ' ' || text[i] == ',' { + return text[:i] + } + } + return text +} + +// unhex decodes a hex string, failing the test on malformed input. +func unhex(s string) []byte { + out, err := hex.DecodeString(s) + if err != nil { + panic("unhex: " + err.Error()) + } + return out +}