Files
gasm-sdk/asm/amd64_width_test.go
T
petrbalvin 8bded3ea39 test(asm): pin the byte-form width reconciliation and its oracle
Table-driven rows for the renderer's spellings (the L suffix or none with
a byte register encodes the byte form, every register joining at its low
byte), the refusals (W, Q and the MOVD alias take no byte register) and a
differential kernel of the B-suffixed spellings assembled through both
gasm and go tool asm, byte for byte.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00

203 lines
7.2 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"os"
"path/filepath"
"testing"
)
// The byte forms of the suffixed scalar families answer to the register
// operands as much as to the mnemonic: the toolchain's own disassembly
// prints them with the L suffix or none at all (the rendered suffix rides
// the operand-size attribute), the register names carrying the width. The
// tests here pin that reconciliation: the renderer's spellings encode the
// byte form byte for byte, the W and Q spellings never ride a byte
// register, and the classic names stay size-agnostic.
// TestOperandWidthByteForms pins the renderer's spellings: an L suffix or
// no suffix with a byte-spelled register encodes the 8-bit form, every
// register joining at its low byte. Every row's bytes were cross-checked
// against `go tool asm -S` over the B-suffixed spelling of the same
// operands (the toolchain rejects the L spelling itself), and against the
// toolchain's objdump text for the byte encodings.
func TestOperandWidthByteForms(t *testing.T) {
r11 := Reg{idx: 11, size: 8}
r8 := Reg{idx: 8, size: 1}
r9 := Reg{idx: 9, size: 1}
for _, tt := range []struct {
name string
ops []Operand
want string
}{
// The atomics pair: XADD and CMPXCHG drop to the 0F C0/0F B0 byte
// opcodes the L spelling would otherwise widen past.
{"XADDL DL, DL", []Operand{DL, DL}, "0fc0d2"},
{"XADDB DL, DL", []Operand{DL, DL}, "0fc0d2"},
{"XADDL R8B, R9B", []Operand{r8, r9}, "450fc0c1"},
{"CMPXCHGL DL, DL", []Operand{DL, DL}, "0fb0d2"},
{"XCHGL DL, DL", []Operand{DL, DL}, "86d2"},
{"XCHGL DL, 0(BX)", []Operand{DL, Ptr(BX, 0, 1)}, "8613"},
// The ALU immediates: the accumulator short form for the AL
// spelling, the generic 0x80 /digit elsewhere.
{"CMPL AL, $7", []Operand{AL, Imm(7)}, "3c07"},
{"ADDL $7, AL", []Operand{Imm(7), AL}, "0407"},
{"SUBL $7, AL", []Operand{Imm(7), AL}, "2c07"},
{"ANDL $7, AL", []Operand{Imm(7), AL}, "2407"},
{"SBBL $7, AL", []Operand{Imm(7), AL}, "1c07"},
{"SBBL $7, DL", []Operand{Imm(7), DL}, "80da07"},
{"ADDB $3, AX", []Operand{Imm(3), AX}, "80c003"},
{"ORB $7, AX", []Operand{Imm(7), AX}, "80c807"},
{"SBBB $7, AX", []Operand{Imm(7), AX}, "80d807"},
// The register forms, the extended register riding REX.B and the
// byte source in the reg field.
{"SBBL DL, R11", []Operand{DL, r11}, "4118d3"},
{"TESTL R11, DL", []Operand{r11, DL}, "4484da"},
{"TESTB $7, AX", []Operand{Imm(7), AX}, "f6c007"},
{"TESTB R11, DL", []Operand{r11, DL}, "4484da"},
// CRC32 keeps the F0 byte opcode under the unsuffixed spelling.
{"CRC32 DL, R11", []Operand{DL, r11}, "f2440f38f0da"},
{"CRC32B DL, R11", []Operand{DL, r11}, "f2440f38f0da"},
{"CRC32B AX, CX", []Operand{AX, CX}, "f20f38f0c8"},
// The move and unary families follow the same rule.
{"MOVL $7, DL", []Operand{Imm(7), DL}, "b207"},
{"MOVB $7, DL", []Operand{Imm(7), DL}, "b207"},
{"MOVB AX, AL", []Operand{AX, AL}, "88c0"},
{"INCL DL", []Operand{DL}, "fec2"},
{"NEGL DL", []Operand{DL}, "f6da"},
{"IMULL DL", []Operand{DL}, "f6ea"},
{"SHLL $2, DL", []Operand{Imm(2), DL}, "c0e202"},
{"ROLL CL, DL", []Operand{CL, DL}, "d2c2"},
// The shift count never narrows the shifted value.
{"RCLW CL, 0(R11)", []Operand{CL, Ptr(r11, 0, 2)}, "6641d313"},
{"RORQ CL, AX", []Operand{CL, AX}, "48d3c8"},
// The size-agnostic spellings stay width-free: the L and Q forms
// of the same families are untouched by the reconciliation.
{"XADDL AX, CX", []Operand{AX, CX}, "0fc1c1"},
{"ADDL $7, AX", []Operand{Imm(7), AX}, "83c007"},
{"ADDL $256, AX", []Operand{Imm(256), AX}, "0500010000"},
{"CRC32L AX, CX", []Operand{AX, CX}, "f20f38f1c8"},
} {
t.Run(tt.name, func(t *testing.T) {
got, err := Encode(mnemonicOf(tt.name), tt.ops...)
if err != nil {
t.Fatalf("%s: %v", tt.name, err)
}
if want := unhex(tt.want); !bytes.Equal(got, want) {
t.Errorf("%s: % x, want % x", tt.name, got, want)
}
})
}
}
// TestOperandWidthConflicts pins the refusals: the W and Q spellings never
// ride a byte register (go tool asm rejects MOVQ AL, AX and its siblings
// outright), and neither does the MOVD alias of the quad move.
func TestOperandWidthConflicts(t *testing.T) {
for _, tt := range []struct {
name string
ops []Operand
}{
{"MOVQ AL, AX", []Operand{AL, AX}},
{"MOVQ AX, AL", []Operand{AX, AL}},
{"MOVQ DL, DL", []Operand{DL, DL}},
{"MOVW $7, DL", []Operand{Imm(7), DL}},
{"MOVD AL, AX", []Operand{AL, AX}},
{"XADDQ DL, DL", []Operand{DL, DL}},
{"CMPXCHGQ DL, DL", []Operand{DL, DL}},
{"XCHGQ DL, DL", []Operand{DL, DL}},
{"SHLQ $2, DL", []Operand{Imm(2), DL}},
{"INCQ DL", []Operand{DL}},
{"IMULQ DL", []Operand{DL}},
{"TESTQ R11, DL", []Operand{Reg{idx: 11, size: 8}, DL}},
{"CRC32Q DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}},
{"CRC32W DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}},
} {
t.Run(tt.name, func(t *testing.T) {
if _, err := Encode(mnemonicOf(tt.name), tt.ops...); err == nil {
t.Errorf("%s: encoded, want the byte-register conflict refused", tt.name)
}
})
}
}
// TestOperandWidthDifferential is the byte-parity oracle for the same
// reconciliation: the B-suffixed spellings, which go tool asm accepts, must
// encode identically through both assemblers, the agnostic names included
// (their low byte joins the byte form) and the accumulator division (AL
// short, AX generic) with them.
func TestOperandWidthDifferential(t *testing.T) {
kernel := "#include \"textflag.h\"\n" +
"TEXT \u00b7bytewidth(SB), NOSPLIT, $0\n" +
"\tXADDB DL, DL\n" +
"\tXADDB R8B, R9B\n" +
"\tXADDL AX, CX\n" +
"\tCMPXCHGB DL, DL\n" +
"\tXCHGB DL, DL\n" +
"\tXCHGB DL, 0(BX)\n" +
"\tCMPB AL, $7\n" +
"\tADDB $7, AL\n" +
"\tADDB $3, AX\n" +
"\tSUBB $7, AL\n" +
"\tANDB $7, AL\n" +
"\tSBBB $7, AL\n" +
"\tSBBB $7, DL\n" +
"\tSBBB DL, R11\n" +
"\tORB $7, AX\n" +
"\tTESTB R11, DL\n" +
"\tTESTB $7, AX\n" +
"\tCRC32B DL, R11\n" +
"\tCRC32B AX, CX\n" +
"\tCRC32B R8B, CX\n" +
"\tCRC32L AX, CX\n" +
"\tMOVB $7, DL\n" +
"\tMOVB $3, AX\n" +
"\tMOVB AX, AL\n" +
"\tINCB DL\n" +
"\tNEGB DL\n" +
"\tIMULB DL\n" +
"\tMULB CL\n" +
"\tSHLB $2, DL\n" +
"\tROLB CL, DL\n" +
"\tRET\n"
path := filepath.Join(t.TempDir(), "bytewidth_amd64.s")
if err := os.WriteFile(path, []byte(kernel), 0o644); err != nil {
t.Fatal(err)
}
gt := oracleFuncCode(t, toolAsmObject(t, path, ""))
goCode, ok := gt["bytewidth.bytewidth"]
if !ok {
t.Fatalf("oracle: bytewidth function missing (%d functions)", len(gt))
}
gasmCode := code(path, kernel)
if gasmCode == nil {
t.Fatal("gasm: assemble failed")
}
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("byte width kernel:\ngasm %x\ngo %x", gasmCode, goCode)
}
}
// mnemonicOf returns the first whitespace-free token of a rendered row.
func mnemonicOf(text string) string {
for i := 0; i < len(text); i++ {
if text[i] == ' ' || text[i] == ',' {
return text[:i]
}
}
return text
}
// unhex decodes a hex string, failing the test on malformed input.
func unhex(s string) []byte {
out, err := hex.DecodeString(s)
if err != nil {
panic("unhex: " + err.Error())
}
return out
}