Files
gasm-sdk/disasm/naming_amd64.go
T
petrbalvin dd356b9e6a fix(disasm): name the amd64 families x/arch decodes to the zero opcode
x/arch reports the ADCX, ADOX, RDSEED, RDPID, TPAUSE, UMONITOR, UMWAIT
and ENDBR families with no error but the degenerate zero instruction,
which GoSyntax renders as Op(0) under its prefix decoration and with a
length of one.  A supplementary naming table keyed by the opcode
pattern restores the toolchain's own spellings and lengths; the parity
fixtures pin all 41 corpus rows (ENDBR32 alone, which the toolchain
cannot spell, pins as bytes and text in the focused naming test).

Assisted-by: GLM 5.3 Flash
2026-10-07 02:27:51 +02:00

321 lines
8.9 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package disasm
import "fmt"
// golang.org/x/arch decodes a handful of amd64 opcode families to no error
// and the degenerate zero instruction: no opcode, no operands, one byte.
// GoSyntax then renders the placeholder "Op(0)", decorated with whatever
// prefixes it saw, and reports a length of one. The families are the ADCX
// and ADOX carry-propagation pair, RDSEED, RDPID, the WAITPKG trio TPAUSE,
// UMONITOR and UMWAIT, and the ENDBR pair.
//
// nameAMD64Degenerate restores the names the Go toolchain itself carries
// for these families. The toolchain's assembler test corpus
// machine-checks the encodings (amd64enc.s's "ADCXL DX, DX // 660f38f6d2"
// and its kin), and the spellings below are the corpus's own; the parity
// fixtures pin every corpus row the table names, so a decoder bump that
// changes a length or a register reading fails there rather than silently
// moving the text. The table is keyed by the opcode pattern: legacy
// prefixes, opcode bytes and the ModR/M shape, so it names every encoding
// of a family, not only the corpus rows.
// amd64GPRNames holds the Plan 9 spelling of the general-purpose registers
// by index. The names are width-independent: the mnemonic's suffix carries
// the size, so RDX reads DX in both ADCXL and ADCXQ.
var amd64GPRNames = [16]string{
"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI",
"R8", "R9", "R10", "R11", "R12", "R13", "R14", "R15",
}
// amd64Prefixes is the legacy prefix reading of one instruction: the
// operand-size override, the rep and repne not-really-prefixes, and REX.
// n is the number of bytes the prefixes consumed.
type amd64Prefixes struct {
osz bool // 66
rep bool // f3
repne bool // f2
rexW, rexR, rexX, rexB bool
n int
}
// scanAMD64Prefixes reads the legacy prefixes at the start of code. It
// stops at the first byte that is not one, which is where the opcode
// begins.
func scanAMD64Prefixes(code []byte) (amd64Prefixes, bool) {
var p amd64Prefixes
for ; p.n < len(code); p.n++ {
switch b := code[p.n]; {
case b == 0x66:
p.osz = true
case b == 0xf3:
p.rep = true
case b == 0xf2:
p.repne = true
case b >= 0x40 && b <= 0x4f:
p.rexW = b&0x8 != 0
p.rexR = b&0x4 != 0
p.rexX = b&0x2 != 0
p.rexB = b&0x1 != 0
default:
return p, true
}
}
return p, false // prefixes with no opcode behind them
}
// amd64RM is the ModR/M (and SIB) reading of one operand, in the shape the
// named families use: one register or one memory reference, never an
// immediate. n counts the ModR/M byte, an SIB byte and the displacement.
type amd64RM struct {
regForm bool // mod == 11: the operand is the r/m register
reg int // reg field with REX.R applied
rm int // r/m field with REX.B applied, register form
base int // memory base register, -1 when absent
index int // memory index register, -1 when absent
scale int
disp int32
rip bool // mod == 00, r/m == 101: RIP-relative
n int
}
// decodeAMD64RM reads the ModR/M byte at the start of code, with the SIB
// byte and displacement that mod 00 and 01 may carry behind it. rexR,
// rexX and rexB extend the register fields to R8 through R15.
func decodeAMD64RM(code []byte, rexR, rexX, rexB bool) (amd64RM, bool) {
if len(code) == 0 {
return amd64RM{}, false
}
b := code[0]
var rm amd64RM
rm.n = 1
rm.reg = int(b>>3) & 7
if rexR {
rm.reg += 8
}
mod := b >> 6
rmb := int(b & 7)
if mod == 3 {
rm.regForm = true
rm.rm = rmb
if rexB {
rm.rm += 8
}
return rm, true
}
rm.base, rm.index, rm.scale = -1, -1, 1
switch rmb {
case 4: // SIB byte follows
if len(code) < rm.n+1 {
return amd64RM{}, false
}
sib := code[rm.n]
rm.n++
rm.scale = 1 << (sib >> 6)
idx := int(sib>>3) & 7
if rexX {
idx += 8
}
if idx%8 != 4 { // index 100 is the no-index encoding
rm.index = idx
}
bs := int(sib & 7)
if rexB {
bs += 8
}
if mod == 0 && bs%8 == 5 {
// base 101 with no displacement byte is disp32 alone
} else {
rm.base = bs
}
case 5:
if mod == 0 {
rm.rip = true
} else {
rm.base = rmb
if rexB {
rm.base += 8
}
}
default:
rm.base = rmb
if rexB {
rm.base += 8
}
}
switch mod {
case 1:
if len(code) < rm.n+1 {
return amd64RM{}, false
}
rm.disp = int32(int8(code[rm.n]))
rm.n++
case 2:
if len(code) < rm.n+4 {
return amd64RM{}, false
}
rm.disp = int32(uint32(code[rm.n]) | uint32(code[rm.n+1])<<8 |
uint32(code[rm.n+2])<<16 | uint32(code[rm.n+3])<<24)
rm.n += 4
}
return rm, true
}
// text renders the operand the way the GoSyntax renderer prints a memory
// reference: the displacement in hex (a zero displacement printed as 0),
// then the base, then the scaled index.
func (rm amd64RM) text() string {
if rm.regForm {
return amd64GPRNames[rm.rm]
}
s := "0"
if rm.disp != 0 {
s = fmt.Sprintf("%#x", rm.disp)
}
if rm.rip {
// The renderer names the instruction pointer IP in a memory
// reference.
return s + "(IP)"
}
if rm.base >= 0 {
s += "(" + amd64GPRNames[rm.base] + ")"
}
if rm.index >= 0 {
s += fmt.Sprintf("(%s*%d)", amd64GPRNames[rm.index], rm.scale)
}
return s
}
// nameAMD64Degenerate names an amd64 encoding the decoder returned as the
// degenerate zero instruction. It reports the rendered text, the
// instruction's length in bytes and whether the bytes matched a family.
// Unmatched bytes keep the renderer's own placeholder output.
func nameAMD64Degenerate(code []byte) (string, int, bool) {
p, ok := scanAMD64Prefixes(code)
if !ok || len(code) < p.n+3 || code[p.n] != 0x0f {
return "", 0, false
}
tail := code[p.n+1:]
switch tail[0] {
case 0x38:
return nameAMD64Carry(p, tail[1:])
case 0xc7:
return nameAMD64RNG(p, tail[1:])
case 0xae:
return nameAMD64Wait(p, tail[1:])
case 0x1e:
return nameAMD64Endbr(p, tail[1:])
}
return "", 0, false
}
// nameAMD64Carry names the ADCX and ADOX pair, 0F 38 F6 /r. The operand
// size override carries ADCX and rep carries ADOX; the length suffix
// follows REX.W. The destination is the reg field and the source the
// r/m operand, printed source first.
func nameAMD64Carry(p amd64Prefixes, tail []byte) (string, int, bool) {
var name string
switch {
case p.osz && !p.rep && !p.repne:
name = "ADCX"
case p.rep && !p.osz && !p.repne:
name = "ADOX"
default:
return "", 0, false
}
if len(tail) < 2 || tail[0] != 0xf6 {
return "", 0, false
}
rm, ok := decodeAMD64RM(tail[1:], p.rexR, p.rexX, p.rexB)
if !ok {
return "", 0, false
}
suffix := "L"
if p.rexW {
suffix = "Q"
}
text := name + suffix + " " + rm.text() + ", " + amd64GPRNames[rm.reg]
return text, p.n + 3 + rm.n, true
}
// nameAMD64RNG names RDSEED, 0F C7 /7, and its rep-prefixed sibling
// RDPID. Both take the destination register in the r/m field and exist
// only in the register form; the memory forms of the same opcode are
// CLFLUSH and its descendants, which the decoder names itself. The size
// suffix follows REX.W and the operand-size override.
func nameAMD64RNG(p amd64Prefixes, tail []byte) (string, int, bool) {
var name string
switch {
case p.rep && !p.osz && !p.repne:
name = "RDPID"
case !p.rep && !p.repne:
name = "RDSEED"
default:
return "", 0, false
}
if len(tail) < 1 {
return "", 0, false
}
rm, ok := decodeAMD64RM(tail, p.rexR, p.rexX, p.rexB)
if !ok || !rm.regForm || rm.reg != 7 {
return "", 0, false
}
if name == "RDPID" {
return name + " " + amd64GPRNames[rm.rm], p.n + 2 + rm.n, true
}
suffix := "L"
switch {
case p.rexW:
suffix = "Q"
case p.osz:
suffix = "W"
}
return name + suffix + " " + amd64GPRNames[rm.rm], p.n + 2 + rm.n, true
}
// nameAMD64Wait names the WAITPKG and monitor trio on 0F AE /6: TPAUSE
// under the operand-size override, UMONITOR under rep and UMWAIT under
// repne. Each takes one 32-bit register in the r/m field.
func nameAMD64Wait(p amd64Prefixes, tail []byte) (string, int, bool) {
var name string
switch {
case p.osz && !p.rep && !p.repne:
name = "TPAUSE"
case p.rep && !p.osz && !p.repne:
name = "UMONITOR"
case p.repne && !p.osz && !p.rep:
name = "UMWAIT"
default:
return "", 0, false
}
if len(tail) < 1 {
return "", 0, false
}
rm, ok := decodeAMD64RM(tail, p.rexR, p.rexX, p.rexB)
if !ok || !rm.regForm || rm.reg != 6 {
return "", 0, false
}
return name + " " + amd64GPRNames[rm.rm], p.n + 2 + rm.n, true
}
// nameAMD64Endbr names the ENDBR pair, F3 0F 1E with the fixed ModR/M
// bytes FA for the 64-bit variant and FB for the 32-bit one. The
// instructions take no operands.
func nameAMD64Endbr(p amd64Prefixes, tail []byte) (string, int, bool) {
if !p.rep || p.osz || p.repne {
return "", 0, false
}
if len(tail) < 1 {
return "", 0, false
}
switch tail[0] {
case 0xfa:
return "ENDBR64", p.n + 3, true
case 0xfb:
return "ENDBR32", p.n + 3, true
}
return "", 0, false
}