Compare commits

...
6 Commits
Author SHA1 Message Date
petrbalvin 97dfaa7526 docs: changelog for macro expansion and the corrected corpus audit
Test / test (push) Successful in 2m14s
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 66aa4dbc8b test(verify): register the campaign kernels in the ground-truth suites
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin dce5d31462 feat(amd64): LOCK and REP prefixes, literal data pseudo-ops and ADJSP
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9dc3987e02 feat(riscv64,loong64): PCALIGN, branch relaxation and operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9b238a525a feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin ad82aac663 feat(parser): macro expansion, conditionals and include splicing with -I
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
40 changed files with 6425 additions and 303 deletions
+15
View File
@@ -9,6 +9,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added ### Added
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at
the point of use, `#undef` and the `#ifdef`/`#ifndef`/`#else`/
`#endif` family select branches, `#include` splices headers resolved
through the source directory and the new repeatable `-I` flag, `;`
separates statements, and constant expressions left in operands
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the - **The GOROOT instruction wave, part 1.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked, instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the byte-verified against `go tool asm`: on amd64 the carry ALU, the
@@ -129,6 +139,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Fixed ### Fixed
- **The corpus audit attempts fewer files that no build would compile.**
Files named for Go ports gasm does not target (arm, 386, s390x, ...)
are reported as other-port and never attempted, the headline rate is
computed over attemptable files, and the audit searches the
toolchain's shipped headers (funcdata.h and friends) automatically.
- **riscv64 JALR silently jumped to the wrong register.** The trampoline - **riscv64 JALR silently jumped to the wrong register.** The trampoline
form `JALR X0, 0(X5)` read the memory operand's base as the destination, form `JALR X0, 0(X5)` read the memory operand's base as the destination,
encoding a jump to X0 with no diagnostic; the destination is the first encoding a jump to X0 with no diagnostic; the destination is the first
+566 -105
View File
@@ -5,6 +5,7 @@ package asm
import ( import (
"fmt" "fmt"
"math/bits"
"strconv" "strconv"
"strings" "strings"
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD": "FMOVS", "FMOVD":
return arm64MovSize(mnem, ops, fi) return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW": case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
"ADDS", "ADDSW", "SUBS", "SUBSW":
if len(ops) >= 2 && isImmOperand(ops[0]) { if len(ops) >= 2 && isImmOperand(ops[0]) {
v := arm64Imm64(ops[0]) // Size exactly as the encoder will emit: a single imm12 word, the
// Small immediate (0..4095 or -2048..-1) fits in one instruction. // two-word ADDCON2 split, or a materialisation into REGTMP plus
if v >= 0 && v <= 0xFFF { // the register form. Anything else would desynchronise the label
return 4 // offsets of pass 1 from the bytes pass 2 lays down.
if v, ok := arm64ImmOperandValue(ops[0]); ok {
rn, rd := 0, 0
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
rd = n
} }
if v >= -2048 && v < 0 { if len(ops) == 3 {
return 4 if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
rn = n
} }
// Larger immediates need MOV materialisation + op. }
return 8 if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
return 4 * len(ws)
}
}
return 4
} }
} }
return 4 return 4
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Bitfield(mnem, enc.op, ops) return encodeARM64Bitfield(mnem, enc.op, ops)
} }
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
}
// EXTR. // EXTR.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR { if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
return encodeARM64Extr(mnem, enc.op, ops) return encodeARM64Extr(mnem, enc.op, ops)
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so // VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below. // this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
if spec, ok := a64SimdVTable[mnem]; ok { // only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops) return encodeARM64SimdV(mnem, spec, ops)
} }
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
} }
// SIMD table lookup. // SIMD table lookup.
if mnem == "VTBL" { if mnem == "VTBL" || mnem == "VTBX" {
return encodeARM64VTBL(ops) return encodeARM64VTBL(mnem, ops)
} }
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R, // SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
} }
op := ops[0] op := ops[0]
// Branch to the program counter itself: JMP (PC) spins forever. The // Branch to the program counter: JMP (PC) spins forever, and a spelled
// toolchain encodes it as an unconditional branch with a zero offset. // offset (CALL -1(PC), the return stub) rides the imm26 field in word
// units. The toolchain encodes both as a plain branch of that offset.
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
if link { rel := op.Addr.Offset
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem) if rel < -(1<<25) || rel >= (1<<25) {
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
} }
return a64wordLE(a64Branch(0, 0)), nil bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, int32(rel))), nil
} }
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The // Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
// the ADD/SUB-with-flags family an add/sub immediate. // the ADD/SUB-with-flags family an add/sub immediate.
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW" isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
mnem == "MVN" || mnem == "MVNW" ||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating // Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the // bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
var logical bool var logical bool
switch mnem { switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW", case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW": "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
"TST", "TSTW":
logical = true logical = true
} }
if logical { if logical {
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
} }
inverted := false inverted := false
switch mnem { switch mnem {
case "BIC", "BICW", "BICS", "BICSW": case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
inverted = true inverted = true
} }
if inverted { if inverted {
@@ -700,8 +727,38 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
} }
n, immr, imms, ok := a64LogicalImm(v, width) n, immr, imms, ok := a64LogicalImm(v, width)
if !ok { if !ok {
// Beyond the bitmask immediates the toolchain materialises
// the constant into REGTMP (R27) and uses the register form
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
// value, so the materialisation uses v before any inversion.
written := v
if inverted {
written = ^v
}
width := mnem
if strings.HasSuffix(mnem, "W") {
width = "MOVW"
} else {
width = "MOVD"
}
mw, merr := encodeARM64LoadImm(27, written, width)
var rn, rd int
switch len(ops) {
case 3:
rn = arm64RegNum(operandRegName(ops[1]))
rd = arm64RegNum(operandRegName(ops[2]))
default:
rd = arm64RegNum(operandRegName(ops[1]))
rn = rd
}
if isCmp {
rd = 31
}
if merr != nil || rn < 0 || rd < 0 {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " ")) return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
} }
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
}
opc := (baseOp >> 29) & 7 opc := (baseOp >> 29) & 7
sf := (baseOp >> 31) & 1 sf := (baseOp >> 31) & 1
var rn, rd int var rn, rd int
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem) return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
} }
if !isExtend { if !isExtend {
if amount < 0 || amount > 63 { // ROR rides the shifted-register field only for the logical
// group; the toolchain reports "unsupported shift operator" for
// the arithmetic forms, whose shift=11 encoding is unallocated.
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
}
// The imm6 field is 5 bits and truncates at the 32-bit width.
limit := 63
if strings.HasSuffix(mnem, "W") {
limit = 31
}
if amount < 0 || amount > limit {
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount) return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
} }
// SP-based ADD/SUB have no shifted-register encoding: the // SP-based ADD/SUB have no shifted-register encoding: the
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
switch len(ops) { switch len(ops) {
case 3: case 3:
// The carry family carries an immediate spelling in three operands
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
// operand, the same shape the two-operand form takes.
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
if v := arm64Imm64(ops[0]); v != 0 {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// OP Rm, Rn, Rd // OP Rm, Rn, Rd
rm := arm64RegNum(operandRegName(ops[0])) rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[1]))
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
} }
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
// shifted-register group, the only forms whose register operand accepts the
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
func arm64LogicalShifted(mnem string) bool {
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
"TST", "TSTW", "MVN", "MVNW":
return true
}
return false
}
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
// data-processing instructions the toolchain accepts an immediate $0
// operand spelling for.
func arm64CarryOp(mnem string) bool {
switch mnem {
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
return true
}
return false
}
// arm64RegMod reports whether a register operand carries the shifted-register // arm64RegMod reports whether a register operand carries the shifted-register
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend // or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
// option (R0.UXTW, R3.SXTW<<2). // option (R0.UXTW, R3.SXTW<<2).
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
// arm64RegModifier resolves a modified register operand: the register number, // arm64RegModifier resolves a modified register operand: the register number,
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend // the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
// option (UXTB=0..SXTX=7) with its shift amount. // option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
// before the operator match.
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) { func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
name := operandRegName(op) name := operandRegName(op)
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
if before, after, ok0 := strings.Cut(name, "."); ok0 { if before, after, ok0 := strings.Cut(name, "."); ok0 {
switch strings.ToUpper(strings.TrimSpace(after)) { switch strings.ToUpper(strings.TrimSpace(after)) {
case "UXTB": case "UXTB":
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
if rm < 0 { if rm < 0 {
return 0, 0, 0, false, 0, false return 0, 0, 0, false, 0, false
} }
amount, ok = arm64ShiftAmount(op.Addr.Shift) amount, ok = arm64ShiftAmount(shift)
if !ok || amount < 0 || amount > 4 { if !ok || amount < 0 || amount > 4 {
return 0, 0, 0, false, 0, false return 0, 0, 0, false, 0, false
} }
return rm, 0, extendOpt, true, amount, true return rm, 0, extendOpt, true, amount, true
} }
shiftKind = 0 // LSL shiftKind = 0 // LSL
shift := strings.TrimSpace(op.Addr.Shift)
switch { switch {
case strings.HasPrefix(shift, "<<"): case strings.HasPrefix(shift, "<<"):
shiftKind = 0 shiftKind = 0
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
default: default:
return 0, 0, 0, false, 0, false return 0, 0, 0, false, 0, false
} }
amount, ok = arm64ShiftAmount(op.Addr.Shift) amount, ok = arm64ShiftAmount(shift)
if !ok { if !ok {
return 0, 0, 0, false, 0, false return 0, 0, 0, false, 0, false
} }
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the // operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
// MUL mnemonic, not to these. // MUL mnemonic, not to these.
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// The widening three-operand forms (SMULL, UMNEGL, …) read the
// accumulate register as ZR, already preset in the table's base word.
if len(ops) == 3 {
switch mnem {
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
default:
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
if len(ops) != 4 { if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
} }
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// ---- ADD/SUB immediate ---- // ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction. // encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
// following the toolchain's immediate classification (asm7.go conclass and
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
// otherwise a constant materialisation into REGTMP (R27) followed by the
// register form.
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 { if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
} }
v := arm64Imm64(ops[0]) v, ok := arm64ImmOperandValue(ops[0])
if !ok {
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
}
rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd rn := rd
if len(ops) == 3 { if len(ops) == 3 {
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 || rd < 0 { if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem) return nil, fmt.Errorf("invalid register operand in %s", mnem)
} }
// CMP/CMN discard the destination.
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
sf := uint32(1) // 64-bit
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
sf = 0 // 32-bit
}
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
rd = 31 // ZR rd = 31 // ZR
} }
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
return a64WordsLE(ws...), nil
}
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
// their W forms. rn and rd are resolved register numbers (a comparison
// discards rd, so the caller passes 31).
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
w := strings.HasSuffix(mnem, "W")
sf := uint32(1) // 64-bit
d := v
if w {
sf = 0 // 32-bit
// The W forms classify the 32-bit value (asm7.go con32class).
d = int64(uint32(v))
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
op := uint32(0) // ADD op := uint32(0) // ADD
S := uint32(0) S := uint32(0)
if isSub { if isSub {
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if isS { if isS {
S = 1 S = 1
} }
single := func(sh, imm12 uint32) []uint32 {
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
}
if v >= 0 && v <= 0xFFF { // imm12: plain, then the one-shifted-by-12 form.
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil if d >= 0 && d <= 0xFFF {
return single(0, uint32(d)), nil
} }
if v >= -2048 && v < 0 { if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
// Encode as the opposite operation with positive immediate. return single(1, uint32(d>>12)), nil
opp := op ^ 1
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
} }
// Try with shift by 12.
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 { // ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil // split into two imm12 instructions, low half first (asm7.go case 48).
// The encoding is complete in itself: no REGTMP, no register form. The S
// forms must not break addition/subtraction, so the toolchain
// reclassifies them and falls through to the materialisation below.
dm := ^d
if w {
dm = ^d & 0xFFFFFFFF
} }
// The imm12 field cannot carry the value; rejecting (rather than _, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
// truncating) matches the toolchain, which reports the same shape. if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v) return []uint32{
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
}, nil
}
// Constant into REGTMP (R27), then the register form. The first word
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
// the complement form, the bitmask ORR otherwise, and the full
// omovlconst sequence when no single word carries the value.
var seq []uint32
switch s := arm64Movcon(d); {
case s >= 0:
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
case arm64Movcon(dm) >= 0:
s := arm64Movcon(dm)
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
case isBitcon:
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
default:
seq = arm64MovLConst(d, sf)
}
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
for i := range seq[:len(seq)-1] {
seq[i] |= 27 // REGTMP
}
return seq, nil
}
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
// returned with the destination field clear so the caller can OR its own
// register in. movcon and movcon-of-complement must fail for d before this
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
func arm64MovLConst(d int64, sf uint32) []uint32 {
if sf == 0 {
// omovlconst AMOVW: both 16-bit halves, low first.
return []uint32{
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
}
}
dn := ^d
var immh [4]uint64
zero, neg := 0, 0
for i := range immh {
immh[i] = uint64(d>>(i*16)) & 0xFFFF
switch immh[i] {
case 0:
zero++
case 0xFFFF:
neg++
}
}
mw := func(opc uint32, val int64, chunk int) uint32 {
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
}
var os []uint32
switch {
case zero == 2:
// one MOVZ and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg == 2:
// one MOVN and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
// a single MOVK (constants from strength-reduced division).
if zero == 0 && neg == 0 {
for i := range 4 {
mask := uint64(0xFFFF) << (i * 16)
for period := 2; period <= 32; period *= 2 {
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
os = append(os, mw(3, d, i))
return os
}
}
}
}
switch {
case zero >= 1:
// one MOVZ and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg >= 1:
// one MOVN and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// one MOVZ and three MOVKs
os = append(os, mw(2, d, 0))
for i := 1; i < 4; i++ {
os = append(os, mw(3, d, i))
}
}
}
return os
} }
// ---- MOV pseudo-instruction ---- // ---- MOV pseudo-instruction ----
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
} }
} }
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk. // Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
var ws []uint32 // for the first special 16-bit chunk, then MOVK per remaining one, with
first := true // the bitmask-plus-fixup shortcut for strength-reduced constants).
for i := range 4 { ws := arm64MovLConst(d, sf)
chunk := (d >> uint(i*16)) & 0xFFFF for i := range ws {
if chunk == 0 { ws[i] |= uint32(rd)
continue
}
if first {
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
first = false
} else {
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
}
}
if len(ws) == 0 {
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
} }
return a64WordsLE(ws...), nil return a64WordsLE(ws...), nil
} }
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
} }
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
// with imms = lsb+width-1.
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
}
lsb := arm64Imm64(ops[0])
width := arm64Imm64(ops[2])
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
bits := int64(32) << (baseOp >> 31 & 1)
if lsb < 0 || lsb >= bits {
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
}
if width < 1 || width > bits || lsb+width > bits {
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
}
var immr, imms int64
switch mnem {
case "BFXIL", "BFXILW":
immr, imms = lsb, lsb+width-1
default: // BFI, SBFIZ, UBFIZ
immr, imms = (-lsb)%bits, width-1
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The // encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
// third field carries Rm or a 5-bit immediate in the same bits, at the // third field carries Rm or a 5-bit immediate in the same bits, at the
// toolchain's choice of register or immediate operand. // toolchain's choice of register or immediate operand.
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// MRS <sysreg>, Rd MSR $imm4, <sysreg> // MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op> // PRFM (Rn), $imm|<op>
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(w), nil
}
switch mnem { switch mnem {
case "BRK", "SVC": case "BRK", "SVC":
base := uint32(0xd4200000) base := uint32(0xd4200000)
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
} }
return a64wordLE(base | uint32(v)<<5), nil return a64wordLE(base | uint32(v)<<5), nil
case "DMB", "DSB", "ISB": case "DMB", "DSB", "ISB", "CLREX":
if len(ops) != 1 || !isImmOperand(ops[0]) { if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem) return nil, fmt.Errorf("%s expects $immediate", mnem)
} }
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if v < 0 || v > 15 { if v < 0 || v > 15 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v) return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
} }
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem] base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil return a64wordLE(base | uint32(v)<<8), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 127 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
}
return a64wordLE(0xd503201f | uint32(v)<<5), nil
case "BTI":
op := operandRegName(ops[0])
base, ok := map[string]uint32{"C": 0xd503245f}[op]
if !ok {
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
}
return a64wordLE(base), nil
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 0xFFFF {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
"DCPS3": 0xd4a00003}[mnem]
return a64wordLE(base | uint32(v)<<5), nil
case "DC": case "DC":
if len(ops) != 2 { if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn") return nil, fmt.Errorf("DC expects <op>, Rn")
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
} }
return a64wordLE(base | uint32(rd)&31), nil return a64wordLE(base | uint32(rd)&31), nil
case "MSR": case "MSR":
if len(ops) != 2 || !isImmOperand(ops[0]) { if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>") return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(base | uint32(rs)&31), nil
} }
base, ok := a64MSROps[operandRegName(ops[1])] base, ok := a64MSROps[operandRegName(ops[1])]
if !ok { if !ok {
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index. // specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
func specBit(i int) uint16 { return 1 << uint(i) } func specBit(i int) uint16 { return 1 << uint(i) }
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
return true
}
if !strings.HasPrefix(mnem, "VFCM") {
return false
}
s := strings.Join(strings.Fields(op.Raw), "")
s = strings.TrimPrefix(s, "$")
s = strings.Trim(s, "()")
return s == "0" || s == "0.0"
}
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD // encodeARM64SimdV encodes an arrangement-aware three-register SIMD
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a // instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
// zero immediate takes its compare-against-zero form instead, and the // compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
// polynomial multiplies read the arrangement from their source operands // their compare-against-zero form instead, and the polynomial multiplies read
// alone, the result spelling (H8, Q1) riding no encoding bits. // the arrangement from their source operands alone, the result spelling
// (H8, Q1) riding no encoding bits.
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) { func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
if mnem == "VPMULL" || mnem == "VPMULL2" { if mnem == "VPMULL" || mnem == "VPMULL2" {
if len(ops) != 3 { if len(ops) != 3 {
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
rd, _ := arm64VecOf(ops[2]) rd, _ := arm64VecOf(ops[2])
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
} }
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) { if len(ops) == 3 && isImmOperand(ops[0]) {
if arm64Imm64(ops[0]) != 0 { base, ok := a64SimdVZero[mnem]
if !ok {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
if !arm64SimdZeroImm(mnem, ops[0]) {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
} }
vn, ok1 := arm64VecOf(ops[1]) vn, ok1 := arm64VecOf(ops[1])
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx { if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
return nil, fmt.Errorf("invalid register operand in %s", mnem) return nil, fmt.Errorf("invalid register operand in %s", mnem)
} }
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f) allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil { if err != nil {
return nil, err return nil, err
} }
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
} }
if len(ops) != 3 { if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if spec.fixed { if spec.fixed {
arrBits = 0 arrBits = 0
} }
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
} }
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil { if err != nil {
return nil, err return nil, err
} }
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
} }
// VUADDLV spells its arrangement on the source alone; the rest take it // VUADDLV spells its arrangement on the source alone; the rest take it
// on both. // on both.
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil { if err != nil {
return nil, err return nil, err
} }
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
} }
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX: // encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
// index register rides bits 19:16, the first table register bits 9:5, the // index register rides bits 19:16, the first table register bits 9:5, the
// destination bits 4:0 and the table length (registers minus one) bits // destination bits 4:0 and the table length (registers minus one) bits
// 14:13. The table registers must be consecutive. // 14:13. The table registers must be consecutive.
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) { func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) < 3 { if len(ops) < 3 {
return nil, fmt.Errorf("VTBL expects index, table list and destination") return nil, fmt.Errorf("VTBL expects index, table list and destination")
} }
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
default: default:
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr) return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
} }
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil base := uint32(0x0e000000)
if mnem == "VTBX" {
base |= 1 << 12
}
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
} }
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with // encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
} }
var immval int64 var immval int64
switch mnem { switch mnem {
case "VSHL": case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
if sh < 0 || sh >= esize { if sh < 0 || sh >= esize {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1) return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
} }
immval = esize + sh immval = esize + sh
default: // VUSHR, VSRI default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
if sh < 1 || sh > esize { if sh < 1 || sh > esize {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize) return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
} }
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
sym *ast.Symbol // the parsed frame-relative reference (when mem) sym *ast.Symbol // the parsed frame-relative reference (when mem)
} }
aliases := map[string]alias{} aliases := map[string]alias{}
raws := map[string]string{}
for _, d := range f.Decls { for _, d := range f.Decls {
pre, ok := d.(*ast.Preproc) pre, ok := d.(*ast.Preproc)
if !ok { if !ok {
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") { strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
continue continue
} }
raws[name] = body
}
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
// transitively until nothing changes, bounded against cycles.
for range 8 {
changed := false
for name, body := range raws {
if next, ok := raws[body]; ok && next != body {
raws[name] = next
changed = true
}
}
if !changed {
break
}
}
for name, body := range raws {
isReg := func(s string) bool { isReg := func(s string) bool {
if arm64RegNum(s) >= 0 { if arm64RegNum(s) >= 0 {
return true return true
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
return return
} }
// replace rewrites whole-word occurrences of the alias names in s.
replace := func(s string) string {
if s == "" {
return s
}
out := strings.Fields(s)
for i, w := range out {
if a, ok := aliases[w]; ok {
out[i] = a.raw
}
}
if len(out) == 0 {
return s
}
return strings.Join(out, " ")
}
// replaceToken rewrites an operand whose whole text is one alias use // replaceToken rewrites an operand whose whole text is one alias use
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix // possibly followed by syntax (POLY.D[0]): the alias must be a prefix
// ending at a non-identifier character. // ending at a non-identifier character.
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
return s, false return s, false
} }
// replaceScan rewrites alias uses inside a composite operand (a
// parenthesised memory operand or a bracketed register list): every
// identifier run of word and dot characters is matched against the alias
// names, everything else copies verbatim. The whitespace-split replace
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
// punctuation attached.
replaceScan := func(s string) string {
var b strings.Builder
for i := 0; i < len(s); {
if isAliasWordByte(s[i]) || s[i] == '.' {
j := i
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
j++
}
if nn, ok := replaceToken(s[i:j]); ok {
b.WriteString(nn)
} else {
b.WriteString(s[i:j])
}
i = j
continue
}
b.WriteByte(s[i])
i++
}
return b.String()
}
for _, d := range f.Decls { for _, d := range f.Decls {
t, ok := d.(*ast.Text) t, ok := d.(*ast.Text)
if !ok { if !ok {
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
op.Raw = a.raw op.Raw = a.raw
continue continue
} }
op.Addr.Sym.Name = nn op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
op.Addr.Sym.Raw = nn // The span shape depends on what trailed the
// name: an element or arrangement selector
// (POLY.D[0], POLY.B16) rides in Shift and folds
// back onto the rewritten token; a shift
// operator stays in Shift while the span carries
// the bare register; a split list keeps its
// closing bracket, so the rewrite goes through
// the scan.
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
switch {
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
// Element or arrangement selectors and the
// closing bracket of a split list belong to
// the token text.
op.Raw = nn + sfx
op.Addr.Shift = ""
case op.Addr.Shift != "":
op.Raw = nn op.Raw = nn
default:
op.Raw = replaceScan(op.Raw)
}
continue continue
} }
} }
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
// [V0.B16, V1.B16] with aliased members. // [V0.B16, V1.B16] with aliased members.
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") || if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") { strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
op.Raw = replace(op.Raw) op.Raw = replaceScan(op.Raw)
} }
} }
} }
+238 -1
View File
@@ -308,6 +308,8 @@ const (
a64CondLT = 0xb a64CondLT = 0xb
a64CondGT = 0xc a64CondGT = 0xc
a64CondLE = 0xd a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
) )
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. // arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
"LT": a64CondLT, "LT": a64CondLT,
"GT": a64CondGT, "GT": a64CondGT,
"LE": a64CondLE, "LE": a64CondLE,
"AL": a64CondAL,
"NV": a64CondNV,
} }
// ---- instruction format tags ---- // ---- instruction format tags ----
@@ -343,6 +347,7 @@ const (
a64FADR // ADR/ADRP a64FADR // ADR/ADRP
a64FEXTR // EXTR a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
@@ -487,6 +492,16 @@ func init() {
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24} a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15} a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15} a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// The widening multiplies: a 64-bit result riding the same layout, the
// three-operand forms reading the accumulate register as ZR.
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
// ---- move wide ---- // ---- move wide ----
// MOVZ/MOVN/MOVK // MOVZ/MOVN/MOVK
@@ -536,6 +551,15 @@ func init() {
// ---- bitfield ---- // ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22} a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22} a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
@@ -716,6 +740,13 @@ func init() {
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800, "RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400, "REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400, "RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
// the source width.
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
"REV16W": 0x5ac00400,
} }
for m, op := range dp1 { for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op} a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
@@ -734,7 +765,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000} a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ---- // ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} { for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys} a64InstrTable[m] = a64Enc{format: a64FSys}
} }
@@ -785,6 +816,73 @@ func init() {
for m, op := range lse { for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
} }
// The remaining width and ordering spellings of the same shapes, and the
// CAS compare-and-swap family, word-verified against go tool asm.
lseMore := map[string]uint32{
"LDADDAB": 0x38a00000,
"LDADDAH": 0x78a00000,
"LDADDALB": 0x38e00000,
"LDADDALH": 0x78e00000,
"LDADDLB": 0x38600000,
"LDADDLD": 0xf8600000,
"LDADDLH": 0x78600000,
"LDADDLW": 0xb8600000,
"LDCLRAB": 0x38a01000,
"LDCLRAH": 0x78a01000,
"LDCLRALH": 0x78e01000,
"LDCLRB": 0x38201000,
"LDCLRD": 0xf8201000,
"LDCLRH": 0x78201000,
"LDCLRLB": 0x38601000,
"LDCLRLD": 0xf8601000,
"LDCLRLH": 0x78601000,
"LDCLRLW": 0xb8601000,
"LDCLRW": 0xb8201000,
"LDEORAB": 0x38a02000,
"LDEORAD": 0xf8a02000,
"LDEORAH": 0x78a02000,
"LDEORALB": 0x38e02000,
"LDEORALH": 0x78e02000,
"LDEORAW": 0xb8a02000,
"LDEORB": 0x38202000,
"LDEORD": 0xf8202000,
"LDEORH": 0x78202000,
"LDEORLB": 0x38602000,
"LDEORLD": 0xf8602000,
"LDEORLH": 0x78602000,
"LDEORLW": 0xb8602000,
"LDEORW": 0xb8202000,
"LDORAB": 0x38a03000,
"LDORAD": 0xf8a03000,
"LDORAH": 0x78a03000,
"LDORALH": 0x78e03000,
"LDORAW": 0xb8a03000,
"LDORB": 0x38203000,
"LDORD": 0xf8203000,
"LDORH": 0x78203000,
"LDORLB": 0x38603000,
"LDORLD": 0xf8603000,
"LDORLH": 0x78603000,
"LDORLW": 0xb8603000,
"LDORW": 0xb8203000,
"SWPAB": 0x38a08000,
"SWPAD": 0xf8a08000,
"SWPAH": 0x78a08000,
"SWPALH": 0x78e08000,
"SWPAW": 0xb8a08000,
"SWPB": 0x38208000,
"SWPH": 0x78208000,
"SWPLB": 0x38608000,
"SWPLD": 0xf8608000,
"SWPLH": 0x78608000,
"SWPLW": 0xb8608000,
"CASAD": 0xc8e07c00,
"CASALB": 0x08e0fc00,
"CASLW": 0x88a0fc00,
}
for m, op := range lseMore {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ---- // ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate // MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
@@ -794,6 +892,11 @@ func init() {
"ADCS": 0xba000000, "ADCSW": 0x3a000000, "ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000, "SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000, "SBCS": 0xfa000000, "SBCSW": 0x7a000000,
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
"NGC": 0xda000000, "NGCW": 0x5a000000,
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
"NEGSW": 0x6b000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00, "MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00, "SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
} }
@@ -835,6 +938,12 @@ func init() {
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10} a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10} a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10} a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST} a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
return false return false
} }
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
var a64SimdQOnly = map[string]bool{
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
"VFABS": true, "VFNEG": true, "VFSQRT": true,
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
"VFMAXV": true, "VFMAXNMV": true,
}
// a64ArrBits carries the fixed bits an arrangement contributes to the // a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit // three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base. // flag at bit 30. Bit 29 belongs to the instruction's own base.
@@ -928,19 +1054,129 @@ var a64SimdVTable = map[string]a64SimdVSpec{
"VZIP1": {0x0e003800, 0x7f, false}, "VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false}, "VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false}, "VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only "VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false}, "VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false}, "VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
} }
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
// combination for them, so they stay out and the encoder rejects the shape.
var a64SimdVZero = map[string]uint32{
"VCMEQ": 0x0e209800,
"VCMGT": 0x0e208800,
"VCMGE": 0x2e208800,
"VCMLT": 0x0e20a800,
"VCMLE": 0x2e209800,
// FP compares against (0.0): the register forms above carry the U and op
// bits; the zero forms reshape them.
"VFCMEQ": 0x0ea0d800,
"VFCMGE": 0x2ea0c800,
"VFCMGT": 0x0ea0c800,
"VFCMLE": 0x2ea0d800,
"VFCMLT": 0x0ea0e800,
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions // a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with // (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm. // the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{ var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false}, "VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false}, "VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false}, "VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false}, "VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
} }
// a64CryptoArr is the arrangement each crypto instruction's operands must // a64CryptoArr is the arrangement each crypto instruction's operands must
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
// MSR Rn, <sysreg>; the source register rides bits 4:0. // MSR Rn, <sysreg>; the source register rides bits 4:0.
var a64MSRRegOps = map[string]uint32{ var a64MSRRegOps = map[string]uint32{
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420, "NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
"ELR_EL1": 0xd5184020,
} }
// a64MSROps maps the system register names GOROOT writes to their fixed // a64MSROps maps the system register names GOROOT writes to their fixed
+161 -10
View File
@@ -4,6 +4,7 @@
package asm package asm
import ( import (
"strings"
"testing" "testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
} }
} }
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are // TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
// rejected instead of wrapping through int32. // applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
func TestArm64AddSubImmRange(t *testing.T) { // into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
for _, body := range []string{ // and the MOVZ/MOVN/MOVK materialisations followed by the register form.
"\tADD $0x100000000, R0, R1\n", // Comparisons never split, and the W forms classify the 32-bit value. Every
"\tSUB $-0x100000000, R0, R1\n", // word is go tool asm's own for the same source.
"\tCMP $0x100000000, R0\n", func TestArm64AddSubImmWide(t *testing.T) {
} { got := arm64Words(t, strings.Join([]string{
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") "\tADD $0xaaaaaa, R2, R3",
"\tSUB $0xaaaaaa, R2",
"\tADD $0x186a0, R2, R5",
"\tADD $0x1ffe00, R2, R3",
"\tADD $0x3fffffffc000, R5",
"\tADD $-100000, R2, R3",
"\tADD $-2048, R2, R3",
"\tCMP $0xaaaaaa, R2",
"\tCMP $0xffffffffffa0, R3",
"\tCMPW $27745, R2",
"\tCMPW $0x60060, R2",
"\tADDS $0xaaaaaa, R2, R3",
"\tADD $0x12345678, R2, R3",
"\tADDW $0x60060, R2",
"\tSUB $0xe7791f700, R3, R1",
"\tADDW $0x12345678, R2, R3",
"\tCMN $0x1000000, R2",
}, "\n")+"\n")
want := []uint32{
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
0x11018042, 0x11418042, // ADDW $0x60060: W split
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
// three operands, the ROR shift on the logical group (and its rejection for
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
// alias with an element selector. Words are go tool asm's own.
func TestArm64CarryShiftAlias(t *testing.T) {
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
want := []uint32{
0x9a1f0294, // ADC ZR, R20, R20
0x9a1f0284, // ADC ZR, R20, R4
0xfa1f008c, // SBCS ZR, R4, R12
0xfa0f008c, // SBCS R15, R4, R12
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
0x8ac18443, // AND R1 ROR 33, R2, R3
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
0xda0203e7, // SBC ZR, R7, R2
0x9b0eff77, // MSUB ZR, R27, R14, R23
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
}
}
// ROR on an arithmetic form is unallocated: the toolchain reports an
// unsupported shift operator.
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
if len(errs) > 0 { if len(errs) > 0 {
t.Fatalf("parse: %v", errs) t.Fatalf("parse: %v", errs)
} }
if _, err := AssembleFileARM64(f); err == nil { if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body) t.Error("ADD R1@>33: expected an error, got none")
}
}
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
// operand with an element selector and inside a split register list: the
// aliases resolve textually where the parser carries the selector apart from
// the name. Words are go tool asm's own.
func TestArm64VecAliasElement(t *testing.T) {
src := `#include "textflag.h"
#define POLY V15
#define ACC0 V8
#define ACC1 V9
TEXT ·f(SB), NOSPLIT, $0-0
VMOV R1, POLY.D[0]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [ACC0.B16]
VLD1.P (R0), [ACC0.B16, ACC1.B16]
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
want := []uint32{
0x4e081c2f, // INS V15.D[0], R1
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
0x4c407008, // VLD1 (R0), [V8.B16]
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
// once the value leaves every imm12 form: a constant sequence into REGTMP
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
// tool asm's own.
func TestArm64AddSubImmBeyond32(t *testing.T) {
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
want := []uint32{
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0x8b1b0001, // ADD R27, R0, R1
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
0xcb1b0001, // SUB R27, R0, R1
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0xeb1b001f, // CMP R27, R0
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
} }
} }
} }
+55
View File
@@ -67,6 +67,9 @@ type spadjStep struct {
// patch sites (for the file-level layout to resolve), the label table and the // patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries. // stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) { func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
if err := checkAdjspBalance(t); err != nil {
return nil, nil, nil, nil, nil, err
}
fi := computeFrame(t) fi := computeFrame(t)
chain := jumpChain(t) chain := jumpChain(t)
resolve := func(name string) string { resolve := func(name string) string {
@@ -203,6 +206,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{guardLen + len(fi.prologue), 8 + fi.size}, spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
) )
} }
// frameBase is the SP delta the prologue leaves: 8 for the saved base
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
// statements' straight-line sum, so a mid-body step's value is the
// frame base plus what the body has opened so far.
frameBase, bodyDelta := 0, 0
if fi.useFP {
frameBase = 8 + fi.size
}
pos := guardLen + len(fi.prologue) pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body { for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr) s, ok := stmt.(*ast.Instr)
@@ -230,6 +241,16 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
ps[k].kind = RelCall ps[k].kind = RelCall
} }
} }
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
// The statement shifted SP mid-body: record the new running
// delta as the value in effect from just past the instruction.
v := s.Operands[0].Imm.Val
if s.Operands[0].Imm.Neg {
v = -v
}
bodyDelta += int(v)
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
}
patches = append(patches, ps...) patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line}) lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...) out = append(out, code...)
@@ -412,6 +433,40 @@ func hasCall(t *ast.Text) bool {
return false return false
} }
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
// shifts SP away from the entry state and every RET must see the shifts
// closed. The assembler's own prologue and epilogue contribute matching
// deltas on both sides, so the statements' straight-line sum must be zero
// at each RET; branches do not reset the walk, which runs over the program
// list in source order. go tool asm reports an offender as "unbalanced
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
// $16/$-16 pair, per-RET rather than per-function).
func checkAdjspBalance(t *ast.Text) error {
delta := 0
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "ADJSP":
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
continue // reported during emission
}
v := in.Operands[0].Imm.Val
if in.Operands[0].Imm.Neg {
v = -v
}
delta += int(v)
case "RET":
if delta != 0 {
return fmt.Errorf("unbalanced PUSH/POP")
}
}
}
return nil
}
// guardLen returns the byte length of the stack-split guard prefix. The // guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the // final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form. // short form and 6 in the long form.
+129
View File
@@ -439,3 +439,132 @@ func TestSubSPEncodings(t *testing.T) {
} }
} }
} }
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
// asserts the three behaviours the toolchain shows: each prefix statement is
// a standalone byte with a PC of its own (so a label placed on the LOCK
// points at the F0), the data pseudo-ops write their literal bytes inline,
// and END terminates nothing (the statements after it still belong to the
// function and carry no trace of it).
func TestAssemblePseudoStatements(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pseudo(SB), NOSPLIT, $0-0
pfx:
LOCK
CMPXCHGQ AX, (BX)
REP
MOVSQ
BYTE $0x0f
BYTE $0x1f
WORD $0x1234
END
BYTE $0x02
RET
`)
code, labels, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
want := []byte{
0xf0,
0x48, 0x0f, 0xb1, 0x03,
0xf3, 0x48, 0xa5,
0x0f, 0x1f, 0x34, 0x12,
0x02, 0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// The label sits on the LOCK byte, exactly where the toolchain's PC
// listing puts it.
if off := labels["pfx"]; off != 0 {
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
}
// The trailing BYTE lands where the layout says: after the 8 bytes of
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
// none.
if code[12] != 0x02 {
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
}
}
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
// ADJSP: the straight-line sum of the adjustments must be zero at each
// RET, branches in between counting for nothing (verified against go tool
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
// $16/$-16 pair with a JMP in between assembles).
func TestAssembleAdjspBalance(t *testing.T) {
// Balanced pair with a branch in between, bytes pinned from go tool asm.
fn := firstText(t, `
#include "textflag.h"
TEXT ·adjsp(SB), NOSPLIT, $0-0
ADJSP $16
JMP body
body:
ADJSP $-16
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
if hexBytes(code) != hexBytes(want) {
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// Unbalanced at the RET: the toolchain diagnoses, so must we.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·unbalanced(SB), NOSPLIT, $0-0
ADJSP $16
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// The check runs per RET: a closed pair before the first RET does not
// excuse an open adjustment before the second.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·tworet(SB), NOSPLIT, $0-0
ADJSP $8
ADJSP $-8
RET
mid:
ADJSP $8
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// A framed function: the assembler's own prologue and epilogue
// contribute matching deltas, so the pair in the body still balances,
// and the bytes match go tool asm end to end.
fn = firstText(t, `
#include "textflag.h"
TEXT ·framed(SB), $16-8
ADJSP $8
ADJSP $-8
RET
`)
code, _, err = Assemble(fn)
if err != nil {
t.Fatalf("Assemble framed: %v", err)
}
want = []byte{
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
0xC3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
+4 -1
View File
@@ -19,7 +19,10 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix). // Fixed-name instructions (no size suffix).
switch upper { switch upper {
case "RET", "NOP", "CALL", "JMP", case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2": "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the
// SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
return true return true
} }
if _, ok := noOperandTable[upper]; ok { if _, ok := noOperandTable[upper]; ok {
+77
View File
@@ -92,6 +92,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// SHA256RNDS2 carries the round constant in a literal X0 first operand. // SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2": case "SHA256RNDS2":
return e.encodeSha256rnds2(ops) return e.encodeSha256rnds2(ops)
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
// the SUBQ and ADDQ forms.
case "BYTE", "WORD", "LONG", "QUAD":
return e.encodeData(upper, ops)
case "END":
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
} }
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing // VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
@@ -241,6 +251,73 @@ var prefetchVariant = map[string]int{
"PREFETCHT2": 3, "PREFETCHT2": 3,
} }
// dataWidth is the literal byte count of each data-emission pseudo-op.
var dataWidth = map[string]int{
"BYTE": 1,
"WORD": 2,
"LONG": 4,
"QUAD": 8,
}
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
// little-endian, with no opcode lookup. The value is truncated to the
// width rather than range-checked, exactly as go tool asm behaves (BYTE
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
// exactly one immediate is accepted: the toolchain rejects a list such as
// BYTE $1, $2, $3.
func (e *enc) encodeData(mnem string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s requires an integer immediate", mnem)
}
width := dataWidth[mnem]
out := make([]byte, width)
u := uint64(imm)
for i := range width {
out[i] = byte(u >> (8 * i))
}
e.out = append(e.out, out...)
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
// encoded body carries no trace of it, whatever operands follow the name
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
func (e *enc) encodeEnd(ops []Operand) error {
return nil
}
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
// picks (the same selection subSP and addSP make for the frame). go tool
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
// statement's effect on the SP balance is checked by the function-level
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
func (e *enc) encodeAdjsp(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("ADJSP requires an integer immediate")
}
switch v := int(imm); {
case v > 0:
e.out = append(e.out, subSP(v)...)
case v < 0:
e.out = append(e.out, addSP(-v)...)
default:
return fmt.Errorf("ADJSP $0 has no encoding")
}
return nil
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic. // splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) { func splitSize(upper string) (base string, size int) {
if upper == "" { if upper == "" {
+139
View File
@@ -925,3 +925,142 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
} }
} }
} }
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
// prefix field merged into the following instruction, and it validates
// nothing about the pairing (LOCK before NOP assembles). The prefixed
// atomic and string shapes are the bytes the runtime's own kernels need.
func TestPrefixStatements(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"LOCK", "LOCK", nil, "f0"},
{"REP", "REP", nil, "f3"},
{"REPN", "REPN", nil, "f2"},
// LOCK; CMPXCHGQ AX, (BX)
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
// REP; MOVSQ
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
// REPN; MOVSB
{"REPN MOVSB", "MOVSB", nil, "a4"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The prefix statements take no operands, as the toolchain reports for
// LOCK AX.
if _, err := Encode("LOCK", AX); err == nil {
t.Error("LOCK AX assembled, want an error")
}
if _, err := Encode("REP", Imm(1)); err == nil {
t.Error("REP $1 assembled, want an error")
}
}
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked (go tool asm
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
func TestDataEmission(t *testing.T) {
cases := []struct {
name string
mnem string
imm Imm
want string
}{
{"BYTE", "BYTE", 0x0f, "0f"},
{"BYTE negative", "BYTE", -1, "ff"},
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
{"WORD", "WORD", 0x1234, "3412"},
{"WORD negative", "WORD", -1, "ffff"},
{"WORD truncated", "WORD", 0x12345, "4523"},
{"LONG", "LONG", 0x11223344, "44332211"},
{"LONG negative", "LONG", -1, "ffffffff"},
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
// register or a missing operand is no immediate at all.
if _, err := Encode("BYTE"); err == nil {
t.Error("BYTE with no operand assembled, want an error")
}
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
t.Error("BYTE $1, $2 assembled, want an error")
}
if _, err := Encode("WORD", AX); err == nil {
t.Error("WORD AX assembled, want an error")
}
}
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
// encodes to zero bytes and takes any operands without complaint (the
// toolchain accepts END $0 and END AX alike).
func TestEndIgnored(t *testing.T) {
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
code, err := Encode("END", ops...)
if err != nil {
t.Errorf("END: %v", err)
continue
}
if len(code) != 0 {
t.Errorf("END = %x, want no bytes", code)
}
}
}
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
// has no encoding (go tool asm refuses ADJSP $0 outright).
func TestAdjsp(t *testing.T) {
cases := []struct {
name string
imm Imm
want string
}{
{"imm8", 112, "4883ec70"},
{"imm8 negative", -112, "4883c470"},
{"imm32", 200, "4881ecc8000000"},
{"imm32 negative", -200, "4881c4c8000000"},
{"small", 8, "4883ec08"},
}
for _, c := range cases {
code, err := Encode("ADJSP", c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
}
}
if _, err := Encode("ADJSP", Imm(0)); err == nil {
t.Error("ADJSP $0 assembled, want an error")
}
if _, err := Encode("ADJSP"); err == nil {
t.Error("ADJSP with no operand assembled, want an error")
}
if _, err := Encode("ADJSP", AX); err == nil {
t.Error("ADJSP AX assembled, want an error")
}
}
+12
View File
@@ -65,6 +65,15 @@ var bitTestOp = map[string]int{
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The // noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
// fence names carry their opcode inside the 0F AE /digit group spelled out in // fence names carry their opcode inside the 0F AE /digit group spelled out in
// full (E8/F0/F8), and PAUSE is F3 90. // full (E8/F0/F8), and PAUSE is F3 90.
//
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
// respectively), not as a prefix field merged into the next instruction: the
// statement that follows is encoded unaware of it, and nothing validates
// that the pairing is a legal one (LOCK before NOP assembles without
// complaint, each byte pinned against the toolchain). Because the bytes
// land in the stream before the following statement anyway, a LOCKed
// CMPXCHGQ encodes identically to a prefixed form.
var noOperandTable = map[string][]byte{ var noOperandTable = map[string][]byte{
"CPUID": {0x0F, 0xA2}, "CPUID": {0x0F, 0xA2},
"RDTSC": {0x0F, 0x31}, "RDTSC": {0x0F, 0x31},
@@ -78,6 +87,9 @@ var noOperandTable = map[string][]byte{
"MFENCE": {0x0F, 0xAE, 0xF0}, "MFENCE": {0x0F, 0xAE, 0xF0},
"SFENCE": {0x0F, 0xAE, 0xF8}, "SFENCE": {0x0F, 0xAE, 0xF8},
"UNDEF": {0x0F, 0x0B}, "UNDEF": {0x0F, 0x0B},
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
} }
// --- MOV -------------------------------------------------------------------- // --- MOV --------------------------------------------------------------------
+387 -40
View File
@@ -46,30 +46,70 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize}) spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
} }
// Pass 1: label offsets from the instruction sizes. // The toolchain's parser counts N(PC) displacements over the source
offsets := map[string]int{} // instructions at a uniform 4 bytes each, so a PC-relative branch
pos := guardLen + len(prologue) // resolves to the instruction N slots away in body order; the resolved
// target then participates in layout and loop-head padding like any
// branch target.
instrs := make([]*ast.Instr, 0, len(t.Body))
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" {
case *ast.Label: instrs = append(instrs, in)
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += loong64InstrSize(s, fi)
} }
} }
parseIndex := make(map[*ast.Instr]int, len(instrs))
for i, in := range instrs {
parseIndex[in] = i
}
pcRelTarget := make(map[*ast.Instr]*ast.Instr)
for _, in := range instrs {
off, ok := loong64PCRelOffset(in)
if !ok {
continue
}
tgt := parseIndex[in] + off
if tgt < 0 || tgt >= len(instrs) {
continue
}
pcRelTarget[in] = instrs[tgt]
}
// Pass 2: encode. The guard prefix precedes the prologue; its branches // Pass 1: label offsets from the instruction sizes. PCALIGN contributes
// target the morestack block at the end of the function, which the first // only its padding. On top of the explicit PCALIGNs, the toolchain pads
// pass has sized. // every backward-branch target (loop head) to a 16-byte boundary, so the
bodyLen := 0 // layout runs to a fixpoint over the alignment set.
{ loopAligns := map[string]bool{}
p := guardLen + len(prologue) alignInstrs := map[*ast.Instr]bool{}
for _, stmt := range t.Body { for {
if in, ok := stmt.(*ast.Instr); ok { offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
p += loong64InstrSize(in, fi) changed := false
for _, in := range instrs {
// A backward PC-relative target is the resolved instruction.
if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] {
alignInstrs[tgt] = true
changed = true
}
target, ok := loong64BranchTarget(in)
if !ok {
continue
}
tOff, ok := offsets[target]
if !ok || tOff >= pcs[in] || loopAligns[target] {
continue
}
loopAligns[target] = true
changed = true
}
if !changed {
break
} }
} }
bodyLen = p - (guardLen + len(prologue)) // Final layout with the complete alignment set.
offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
bodyLen := bodyEnd - (guardLen + len(prologue))
pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget))
for in, tgt := range pcRelTarget {
pcRelPcs[in] = pcs[tgt]
} }
var out []byte var out []byte
if fi.needSplit { if fi.needSplit {
@@ -84,7 +124,20 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
if !ok { if !ok {
continue continue
} }
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve) // PCALIGN pads to the requested boundary with andi $0, $0, 0, the
// architecture's NOP, and encodes to nothing itself.
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
pad := loong64PCAlignPad(pc, in)
out = append(out, loong64PadBytes(pad)...)
pc += pad
continue
}
// Loop-head alignment padding precedes the instruction.
if pad := alignPad[in]; pad > 0 {
out = append(out, loong64PadBytes(pad)...)
pc += pad
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
} }
@@ -115,6 +168,115 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
return out, offsets, relocs, lines, spadj, nil return out, offsets, relocs, lines, spadj, nil
} }
// loong64PCRelOffset reports the N of a branch operand spelled N(PC): the
// displacement counted in source instructions from the branch itself.
func loong64PCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
branch := false
switch mnem {
case "JMP":
branch = len(instr.Operands) == 1
case "JAL", "CALL", "BL":
branch = len(instr.Operands) == 1 || len(instr.Operands) == 2
case "BFPT", "BFPF":
branch = len(instr.Operands) == 1
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
branch = len(instr.Operands) >= 2
}
if !branch {
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// loong64Layout walks the function body once and returns the label offsets,
// the loop-alignment padding due before each instruction (a pad of 0 needs
// nothing), the pc each instruction starts at (its padding included) and the
// first pc past the body. Explicit PCALIGN pads, the alignment pads for the
// labels in aligns and those for the instructions in alignInstrs (backward
// PC-relative targets) all contribute, mirroring the toolchain's layout
// pass.
func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) {
offsets := map[string]int{}
alignPad := map[*ast.Instr]int{}
pcs := map[*ast.Instr]int{}
pos := start
pendingAlign := false
var pendingNames []string
explicit := false
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
if aligns[s.Name.Text] {
pendingAlign = true
}
pendingNames = append(pendingNames, s.Name.Text)
// Provisional: a branch to the label lands here unless a loop
// alignment pad follows, in which case the label resolves to the
// padded instruction (the toolchain's labels bind to the branch
// target instruction, which the padding pass precedes).
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pos += loong64PCAlignPad(pos, s)
explicit = true
continue
}
if pendingAlign {
pendingAlign = false
if pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
}
if alignInstrs[s] && pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
if !explicit {
for _, n := range pendingNames {
offsets[n] = pos + alignPad[s]
}
}
pendingNames = nil
explicit = false
pcs[s] = pos + alignPad[s]
pos += alignPad[s] + loong64InstrSize(s, fi)
}
}
return offsets, alignPad, pcs, pos
}
// loong64BranchTarget reports the local label a branch-like instruction
// transfers to, the loop-head signal the toolchain derives from backward
// branch targets.
func loong64BranchTarget(instr *ast.Instr) (string, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var op *ast.Operand
switch {
case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF":
if len(ops) != 1 {
return "", false
}
op = ops[0]
case mnem == "TEQ" || mnem == "TNE":
return "", false
case len(ops) >= 2:
op = ops[len(ops)-1]
default:
return "", false
}
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's // loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional // branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go // local jump redirects its own jumpers to the ultimate target. The Go
@@ -177,21 +339,51 @@ func l64LabelOK(op *ast.Operand) (string, bool) {
return "", false return "", false
} }
// l64SubToAdd rewrites the SUB family with an immediate first operand onto
// its ADD counterpart with the negated immediate: LoongArch has no
// subtract-immediate instructions, and the toolchain folds SUB $v into the
// ADD immediate form through the same optab matching (the $0 fold into 3R
// and the large-constant materialisations included). The negation is the
// second result; the operand is left untouched because the size pass
// normalises the same instruction.
func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch mnem {
case "SUB":
return "ADD", true
case "SUBW":
return "ADDW", true
case "SUBV", "SUBVU":
return "ADDV", true
}
}
return mnem, false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for // loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions. // most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int { func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text) mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands ops := instr.Operands
var neg bool
mnem, neg = l64SubToAdd(mnem, ops)
if mnem == "RET" { if mnem == "RET" {
return len(loong64Return(fi)) return len(loong64Return(fi))
} }
switch mnem { switch mnem {
case "TEQ", "TNE":
return 8 // bne/beq over the BREAK, then BREAK
case "PRELDX":
return 20 // the four-instruction constant materialisation + preldx
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi) return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU": case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) { if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0]) v := l64Imm64(ops[0])
if neg {
v = -v
}
if v == 0 { if v == 0 {
return 4 // folds into the 3R form (rk = R0) return 4 // folds into the 3R form (rk = R0)
} }
@@ -229,11 +421,50 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
return 4 return 4
} }
// loong64PCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func loong64PCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0
// (the architecture's NOP) for every full 4 bytes of pad.
func loong64PadBytes(pad int) []byte {
nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0))
out := make([]byte, 0, pad/4*len(nop))
for i := 0; i < pad/4; i++ {
out = append(out, nop...)
}
return out
}
// encodeLOONG64Instr encodes a single LoongArch instruction. // encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text) mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands ops := instr.Operands
// The SUB family with an immediate first operand folds onto the ADD
// immediate form with the negated immediate; the negation happens on a
// copy of the operand, never on the shared syntax tree.
mnem, neg := l64SubToAdd(mnem, ops)
if neg {
c := *ops[0]
c.Imm.Val = -c.Imm.Val
ops2 := make([]*ast.Operand, len(ops))
ops2[0] = &c
copy(ops2[1:], ops[1:])
ops = ops2
}
// Pseudo-instructions and the branches first. // Pseudo-instructions and the branches first.
switch mnem { switch mnem {
case "RET": case "RET":
@@ -249,10 +480,80 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
} }
return l64wordLE(uint32(immFromOperand(ops[0]))), nil return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "NEGW", "NEGV":
// The integer negation pseudo is a subtract from zero:
// NEGW src, dst → sub.w r0, src, dst.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := l64Reg(ops[0]), l64Reg(ops[1])
if src < 0 || dst < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sub := l64InstrTable["SUBW"].op
if mnem == "NEGV" {
sub = l64InstrTable["SUBV"].op
}
return l64wordLE(l64rrr(sub, src, 0, dst)), nil
case "TEQ", "TNE":
// The trap pseudo expands to two instructions: bne/beq rj, rd over
// the BREAK (offset 2 instruction units), then BREAK $code.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
code := int(immFromOperand(ops[0]))
rj, rd := 0, l64Reg(ops[len(ops)-1])
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
bop := l64branchTable["BNE"]
if mnem == "TNE" {
bop = l64branchTable["BEQ"]
}
return l64WordsLE(
l64irr16(bop, 2, rj, rd),
l64i15(l64InstrTable["BREAK"].op, code),
), nil
case "PRELDX":
// preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs
// (addrSeq, blockSize, blockNums, stride); the constant v built from
// it materialises in R30 across four instructions, then the preldx.
if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint")
}
rj := loong64RegNum(ops[0].Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
n := uint64(l64Imm64(ops[1]))
hint := int(l64Imm64(ops[2]))
addrSeq := (n >> 0) & 0x1
blkSize := (n >> 1) & 0x7ff
blkNums := (n >> 12) & 0x1ff
stride := (n >> 21) & 0xffff
v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 +
((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
ori = 0x00e << 22
preldx = 0x7058 << 15
)
return l64WordsLE(
l64ir(lu12iw, int(uint32(v>>12)), 30),
l64irr(ori, int(uint32(v)), 30, 30),
l64ir(lu32id, int(uint32(v>>32)), 30),
l64irr(lu52id, int(uint32(v>>52)), 30, 30),
l64rrr(preldx, 30, rj, hint),
), nil
case "JMP", "B": case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs) return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
case "JAL", "CALL", "BL": case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs) return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs) return encodeLOONG64Mov(instr, mnem, fi, relocs)
} }
@@ -262,12 +563,12 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
if mnem == "JIRL" { if mnem == "JIRL" {
return encodeLOONG64Jirl(op, ops) return encodeLOONG64Jirl(op, ops)
} }
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve) return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
} }
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ, // Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0). // BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok { if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve) return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
} }
// B/BL aliases reached only via JMP/JAL above. // B/BL aliases reached only via JMP/JAL above.
@@ -533,11 +834,23 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
// //
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0 // JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0 // JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) { func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(instr.Operands) != 1 { if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands)) return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
} }
op := instr.Operands[0] op := instr.Operands[0]
// PC-relative displacement: N(PC) resolves to the instruction N slots
// away in source order (the toolchain's parse-time count), and the field
// carries the final pc distance in instruction units.
if op.Addr.Sym == nil && op.Addr.Base == "PC" {
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset)
}
v := (targetPc - pc) >> 2
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil { if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl. // Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base) rj := loong64RegNum(op.Addr.Base)
@@ -620,16 +933,28 @@ func l64offsetOperand(op *ast.Operand) (int32, bool) {
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU): // encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain // INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand. // turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 { if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
} }
target := resolve(l64Label(ops[len(ops)-1])) var target string
var v int
lastOp := ops[len(ops)-1]
if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target = resolve(l64Label(lastOp))
targetOff, ok := offsets[target] targetOff, ok := offsets[target]
if !ok { if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
v := (targetOff - pc) >> 2 v = (targetOff - pc) >> 2
}
if len(ops) == 2 { if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/ // Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field. // BGEZ-family aliases encoded with rj in the rj field.
@@ -690,33 +1015,55 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while // BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field // BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately. // and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 2 { isBF := mnem == "BFPT" || mnem == "BFPF"
if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
} }
target := resolve(l64Label(ops[1])) var rj int
targetOff, ok := offsets[target] tgtOp := ops[len(ops)-1]
if !ok { if isBF {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) // BFPT/BFPF test an FCC condition register, defaulting to FCC0 when
} // spelled without one.
v := (targetOff - pc) >> 2 rj = 0
rj := 0 // BFPT/BFPF default to FCC0 if len(ops) == 2 {
if mnem != "BFPT" && mnem != "BFPF" {
rj = l64Reg(ops[0]) rj = l64Reg(ops[0])
if rj < 0 { if rj < 0 {
return nil, fmt.Errorf("invalid register operand") return nil, fmt.Errorf("invalid register operand")
} }
} }
} else {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
var v int
if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target := resolve(l64Label(tgtOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
if mnem == "BGTZ" || mnem == "BLEZ" { if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the // The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form. // 16-bit offset form.
if (v<<16)>>16 != v { if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) return nil, fmt.Errorf("branch %d too far (16-bit range)", v)
} }
return l64wordLE(l64irr16(op, v, 0, rj)), nil return l64wordLE(l64irr16(op, v, 0, rj)), nil
} }
if (v<<11)>>11 != v { if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) return nil, fmt.Errorf("branch %d too far (21-bit range)", v)
} }
return l64wordLE(l64ir21(op, v, rj)), nil return l64wordLE(l64ir21(op, v, rj)), nil
} }
+557 -38
View File
@@ -4,7 +4,9 @@
package asm package asm
import ( import (
"errors"
"fmt" "fmt"
"slices"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -31,32 +33,54 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
} }
// Pass 1: collect instructions and compute label offsets assuming 4 bytes // Pass 1: collect instructions and compute label offsets assuming 4 bytes
// per instruction (or 8 for MOV $large-imm). No encoding yet. // per instruction (or 8 for MOV $large-imm). No encoding yet. PCALIGN
// contributes only its padding, which is attached to the following
// instruction and emitted ahead of it. A relaxed branch carries the
// inverted condition and is followed by an inserted JMP rec (jmpTo set)
// that carries the original target.
type instrRec struct { type instrRec struct {
instr *ast.Instr instr *ast.Instr
compressed bool compressed bool
code []byte code []byte
pad int
relaxed bool
jmpTo string
} }
var recs []instrRec var recs []instrRec
offsets := map[string]int{} offsets := map[string]int{}
pos := guardLen + len(prologue) pos := guardLen + len(prologue)
pendingPad := 0
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
recs = append(recs, instrRec{instr: s}) if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pendingPad += riscvPCAlignPad(pos, s)
pos += riscvPCAlignPad(pos, s)
continue
}
recs = append(recs, instrRec{instr: s, pad: pendingPad})
pendingPad = 0
pos += riscvInstrSize(s, fi) pos += riscvInstrSize(s, fi)
} }
} }
// Pass 2: encode each instruction using Pass-1 offsets. // Pass 2: encode each instruction using Pass-1 offsets. A branch or
// jump the offsets prove overlong encodes to a 4-byte placeholder: the
// relaxation pass rewrites it before the final encoding. pcRelPcs is
// unavailable this early, so the N(PC) forms take the same placeholder
// path.
pc := len(prologue) pc := len(prologue)
for i := range recs { for i := range recs {
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2 branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
if err != nil { code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil) // no relocs in Pass 2
if err != nil && !(branchLike && riscvIsRangeError(err)) {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err) return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
} }
if err != nil {
code = make([]byte, 4)
}
recs[i].code = code recs[i].code = code
pc += len(code) pc += len(code)
} }
@@ -72,20 +96,100 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// Pass 4: recompute offsets with actual sizes. recs holds the // Pass 4: recompute offsets with actual sizes. recs holds the
// instructions in emission order, so an index into it walks t.Body in // instructions in emission order, so an index into it walks t.Body in
// lockstep (the same single pass Pass 1 uses) instead of rescanning the // lockstep (the same single pass Pass 1 uses) instead of rescanning the
// whole slice per statement. // whole slice per statement. PCALIGN padding is recomputed here, since
// compression has shifted instruction sizes since Pass 1.
offsets = map[string]int{} offsets = map[string]int{}
pos = guardLen + len(prologue) pos = guardLen + len(prologue)
ri := 0 ri := 0
pendingPad = 0
for _, stmt := range t.Body { for _, stmt := range t.Body {
switch s := stmt.(type) { switch s := stmt.(type) {
case *ast.Label: case *ast.Label:
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pad := riscvPCAlignPad(pos, s)
pendingPad += pad
pos += pad
continue
}
recs[ri].pad = pendingPad
pendingPad = 0
pos += len(recs[ri].code) pos += len(recs[ri].code)
ri++ ri++
} }
} }
// Pass 4b: relax overlong conditional branches exactly as the toolchain
// does: invert the branch condition, point it at the instruction after an
// inserted JMP, let the JMP carry the original target, and re-layout until
// a pass inserts nothing. Inserted JMP recs share their branch's source
// line and trail it in emission order, so the body walk flushes them
// before every statement and at the end.
var pcRelPcs map[*ast.Instr]int
for {
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
pcs := make([]int, len(recs))
flushJmps := func() {
for ri < len(recs) && recs[ri].jmpTo != "" {
pcs[ri] = pos
pos += 4
ri++
}
}
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
flushJmps()
offsets[s.Name.Text] = pos
case *ast.Instr:
flushJmps()
if ri >= len(recs) {
continue
}
pcs[ri] = pos + recs[ri].pad
pos += recs[ri].pad + len(recs[ri].code)
ri++
}
}
flushJmps()
changed := false
for i := range recs {
r := &recs[i]
if r.relaxed || r.jmpTo != "" {
continue
}
mnem := strings.ToUpper(r.instr.Mnemonic.Text)
if !riscvIsCondBranch(mnem) || len(r.instr.Operands) == 0 {
continue
}
target := labelFromOperand(r.instr.Operands[len(r.instr.Operands)-1])
targetOff, ok := offsets[target]
if !ok {
continue
}
if delta := int32(targetOff - pcs[i]); delta < -4096 || delta >= 4096 {
r.relaxed = true
recs = slices.Insert(recs, i+1, instrRec{instr: r.instr, jmpTo: target})
changed = true
}
}
if !changed {
// Capture the final pcs for the N(PC) branch forms: their target
// is the instruction N source slots away, resolved by index.
pcRelPcs = map[*ast.Instr]int{}
for i := range recs {
if _, ok := riscvPCRelOffset(recs[i].instr); ok {
pcRelPcs[recs[i].instr] = pcs[i]
}
}
break
}
}
// Pass 5: re-encode branches with corrected offsets. Record relocations // Pass 5: re-encode branches with corrected offsets. Record relocations
// during this final pass (relocation offsets are relative to instruction // during this final pass (relocation offsets are relative to instruction
// start). The guard prefix precedes the prologue; its branches target // start). The guard prefix precedes the prologue; its branches target
@@ -104,12 +208,40 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
preCount := len(relocs) preCount := len(relocs)
var lines []LineEntry var lines []LineEntry
for _, r := range recs { for _, r := range recs {
// PCALIGN padding precedes the instruction it was attached to.
if r.pad > 0 {
out = append(out, riscvPadBytes(r.pad)...)
pc += r.pad
}
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line}) lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) { var code []byte
out = append(out, r.code...) switch {
pc += len(r.code) case r.jmpTo != "":
} else { // The JMP a relaxation inserted: JAL X0 to the original target.
code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs) targetOff, ok := offsets[r.jmpTo]
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
return nil, nil, nil, nil, nil, err
}
word := riscvJType(0, offset)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.relaxed:
// The inverted half of a relaxed branch: it targets the inserted
// JMP, always the very next instruction (offset 4).
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
}
word := riscvBType(enc, rs1, rs2, 4)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.compressed && !isBranchLike(r.instr.Mnemonic.Text):
code = r.code
default:
var err error
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, err return nil, nil, nil, nil, nil, err
} }
@@ -131,10 +263,10 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 { if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0}) spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
} }
}
out = append(out, code...) out = append(out, code...)
pc += len(code) pc += len(code)
} }
}
if fi.needSplit { if fi.needSplit {
relocs = append(relocs, guardReloc) relocs = append(relocs, guardReloc)
} }
@@ -150,6 +282,8 @@ var riscvImmAlias = map[string]string{
"AND": "ANDI", "AND": "ANDI",
"OR": "ORI", "OR": "ORI",
"XOR": "XORI", "XOR": "XORI",
"SLT": "SLTI",
"SLTU": "SLTIU",
"SLL": "SLLI", "SLL": "SLLI",
"SRL": "SRLI", "SRL": "SRLI",
"SRA": "SRAI", "SRA": "SRAI",
@@ -178,6 +312,35 @@ func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
return mnem, false return mnem, false
} }
// riscvPCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func riscvPCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// riscvPadBytes renders PCALIGN padding: 4-byte NOPs (addi $0, X0, X0) with a
// trailing 2-byte compressed NOP when the pad is 2 mod 4, exactly as the
// toolchain lays the bytes down.
func riscvPadBytes(pad int) []byte {
out := make([]byte, 0, pad)
for ; pad >= 4; pad -= 4 {
out = append(out, 0x13, 0x00, 0x00, 0x00)
}
if pad == 2 {
out = append(out, 0x01, 0x00)
}
return out
}
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
// Most instructions are 4 bytes; MOV with a large immediate and I-type // Most instructions are 4 bytes; MOV with a large immediate and I-type
// arithmetic with a large immediate expand to several (possibly compressed) // arithmetic with a large immediate expand to several (possibly compressed)
@@ -224,6 +387,10 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
} }
return riscvItypeImmediateSize(mnem, imm) return riscvItypeImmediateSize(mnem, imm)
} }
// BYTE lays down one raw byte per operand.
if mnem == "BYTE" {
return len(ops)
}
// The toolchain's synthesised instructions: some emit one word, others // The toolchain's synthesised instructions: some emit one word, others
// expand to a fixed sequence. // expand to a fixed sequence.
return riscvExtendedSize(mnem, ops) return riscvExtendedSize(mnem, ops)
@@ -318,12 +485,171 @@ func isBranchLike(mnem string) bool {
return false return false
} }
// riscvIsCondBranch reports whether m is a conditional branch, the only
// instruction class branch relaxation rewrites.
func riscvIsCondBranch(mnem string) bool {
switch mnem {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
return true
}
return false
}
// riscvCSRNames maps the standard CSR mnemonics the assembler accepts onto
// their addresses.
var riscvCSRNames = map[string]int32{
"FFLAGS": 0x001,
"FRM": 0x002,
"FCSR": 0x003,
"VSTART": 0x008,
"VXSAT": 0x009,
"VXRM": 0x00A,
"VCSR": 0x00F,
"CYCLE": 0xC00,
"TIME": 0xC01,
"INSTRET": 0xC02,
"CYCLEH": 0xC80,
"TIMEH": 0xC81,
"INSTRETH": 0xC82,
"VL": 0xC20,
"VLENB": 0xC22,
}
// riscvCSRAddress resolves a CSR operand: an integer immediate or one of the
// standard CSR names.
func riscvCSRAddress(op *ast.Operand) (int32, bool) {
if isImmOperand(op) {
return immFromOperand(op), true
}
if op.Addr.Sym != nil {
if v, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
return v, true
}
}
return 0, false
}
// riscvPCRelOffset reports the N of a branch or jump operand spelled N(PC):
// the displacement counted in source instructions from the branch itself.
func riscvPCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
switch mnem {
case "JMP":
if len(instr.Operands) != 1 {
return 0, false
}
case "JAL":
if len(instr.Operands) != 1 && len(instr.Operands) != 2 {
return 0, false
}
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
if len(instr.Operands) < 2 {
return 0, false
}
default:
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// riscvPCRelTargetOff resolves the target displacement of a branch whose last
// operand is N(PC): the toolchain's parser counts the source instructions at
// a uniform 4 bytes, so the target is the instruction N slots away, and the
// displacement tracks that instruction's final pc. A nil pcRelPcs (the
// layout passes) yields a placeholder range error; the caller tolerates it
// for branch-like instructions.
func riscvPCRelTargetOff(instr *ast.Instr, pc int, pcRelPcs map[*ast.Instr]int) (int, bool, error) {
off, ok := riscvPCRelOffset(instr)
if !ok {
return 0, false, nil
}
if pcRelPcs == nil {
return 0, true, &riscvRangeError{"pc-relative placeholder"}
}
targetPc, ok := pcRelPcs[instr]
if !ok {
return 0, true, fmt.Errorf("PC-relative target %d out of range", off)
}
return targetPc, true, nil
}
// riscvInvertedBranchEnc returns the encoding of mnem's inverted condition
// for the given operands: InvertBranch's table applied at the encoding level.
// The register operands are already in position for the inverted form.
func riscvInvertedBranchEnc(mnem string, ops []*ast.Operand) (riscvEnc, int, int, bool) {
reg := func(i int) int { return regFromOperand(ops[i]) }
switch mnem {
case "BEQ": // → BNE rs1, rs2
return riscvEnc{0x63, 0x1, 0x00}, reg(0), reg(1), true
case "BNE": // → BEQ rs1, rs2
return riscvEnc{0x63, 0x0, 0x00}, reg(0), reg(1), true
case "BLT": // → BGE rs1, rs2
return riscvEnc{0x63, 0x5, 0x00}, reg(0), reg(1), true
case "BGE": // → BLT rs1, rs2
return riscvEnc{0x63, 0x4, 0x00}, reg(0), reg(1), true
case "BLTU": // → BGEU rs1, rs2
return riscvEnc{0x63, 0x7, 0x00}, reg(0), reg(1), true
case "BGEU": // → BLTU rs1, rs2
return riscvEnc{0x63, 0x6, 0x00}, reg(0), reg(1), true
case "BEQZ": // → BNEZ rs, X0
return riscvEnc{0x63, 0x1, 0x00}, reg(0), 0, true
case "BNEZ": // → BEQZ rs, X0
return riscvEnc{0x63, 0x0, 0x00}, reg(0), 0, true
case "BLTZ": // → BGEZ rs, X0
return riscvEnc{0x63, 0x5, 0x00}, reg(0), 0, true
case "BGEZ": // → BLTZ rs, X0
return riscvEnc{0x63, 0x4, 0x00}, reg(0), 0, true
case "BLEZ": // → BGTZ: blt X0, rs
return riscvEnc{0x63, 0x4, 0x00}, 0, reg(0), true
case "BGTZ": // → BLEZ: bge X0, rs
return riscvEnc{0x63, 0x5, 0x00}, 0, reg(0), true
case "BGT": // → BLE: bge rs2, rs1
return riscvEnc{0x63, 0x5, 0x00}, reg(1), reg(0), true
case "BLE": // → BGT: blt rs2, rs1
return riscvEnc{0x63, 0x4, 0x00}, reg(1), reg(0), true
case "BGTU": // → BLEU: bgeu rs2, rs1
return riscvEnc{0x63, 0x7, 0x00}, reg(1), reg(0), true
case "BLEU": // → BGTU: bltu rs2, rs1
return riscvEnc{0x63, 0x6, 0x00}, reg(1), reg(0), true
}
return riscvEnc{}, 0, 0, false
}
// riscvRangeError reports a branch or jump displacement beyond its
// architecture limit. The layout passes tolerate it (the relaxation pass
// rewrites overlong conditional branches before the final encoding); a range
// error reaching the final pass is a real failure.
type riscvRangeError struct{ msg string }
func (e *riscvRangeError) Error() string { return e.msg }
// riscvIsRangeError reports whether err is a displacement-range rejection.
func riscvIsRangeError(err error) bool {
var re *riscvRangeError
return errors.As(err, &re)
}
// riscvRoundModes maps the rounding-mode suffixes onto their funct7 codes.
var riscvRoundModes = map[string]uint32{
"RNE": 0,
"RTZ": 1,
"RDN": 2,
"RUP": 3,
"RMM": 4,
}
// riscvCheckBranchOffset rejects a B-type displacement outside its signed // riscvCheckBranchOffset rejects a B-type displacement outside its signed
// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an // 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an
// out-of-range offset would otherwise wrap to a wrong target. // out-of-range offset would otherwise wrap to a wrong target.
func riscvCheckBranchOffset(target string, off int32) error { func riscvCheckBranchOffset(target string, off int32) error {
if off < -4096 || off > 4094 { if off < -4096 || off > 4094 {
return fmt.Errorf("branch to %q too far (13-bit range)", target) return &riscvRangeError{fmt.Sprintf("branch to %q too far (13-bit range)", target)}
} }
return nil return nil
} }
@@ -332,13 +658,13 @@ func riscvCheckBranchOffset(target string, off int32) error {
// 21-bit span [-1048576, 1048574]. // 21-bit span [-1048576, 1048574].
func riscvCheckJumpOffset(target string, off int32) error { func riscvCheckJumpOffset(target string, off int32) error {
if off < -1048576 || off > 1048574 { if off < -1048576 || off > 1048574 {
return fmt.Errorf("jump to %q too far (21-bit range)", target) return &riscvRangeError{fmt.Sprintf("jump to %q too far (21-bit range)", target)}
} }
return nil return nil
} }
// encodeRISCVInstr encodes a single RISC-V instruction. // encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := instr.Mnemonic.Text mnem := instr.Mnemonic.Text
ops := instr.Operands ops := instr.Operands
var immNeg bool var immNeg bool
@@ -351,6 +677,27 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// RET = epilogue (restore LR and close the frame when present) + // RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET). // uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil return riscvReturn(fi), nil
case "WORD":
// WORD $w lays down a raw 32-bit little-endian word.
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
w := int64(immFromOperand(ops[0]))
if w < 0 || w > 0xFFFFFFFF {
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
}
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}, nil
case "BYTE":
// BYTE $b lays down one raw byte per operand.
var out []byte
for _, op := range ops {
b := int64(immFromOperand(op))
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "CALL": case "CALL":
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL // CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
// relocation. The Go assembler rejects CALL to a local branch label. // relocation. The Go assembler rejects CALL to a local branch label.
@@ -404,6 +751,17 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0) word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
} }
targetOff, ok := offsets[target] targetOff, ok := offsets[target]
if !ok { if !ok {
@@ -424,6 +782,18 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} else if len(ops) == 1 { } else if len(ops) == 1 {
target = labelFromOperand(ops[0]) target = labelFromOperand(ops[0])
} }
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff := off
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
targetOff, ok := offsets[target] targetOff, ok := offsets[target]
if !ok { if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
@@ -456,11 +826,21 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if rs < 0 { if rs < 0 {
return nil, fmt.Errorf("%s: invalid register", mnem) return nil, fmt.Errorf("%s: invalid register", mnem)
} }
target := labelFromOperand(ops[1]) targetOff := 0
targetOff, ok := offsets[target] target := ""
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
target = labelFromOperand(ops[1])
var ok bool
targetOff, ok = offsets[target]
if !ok { if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
}
var enc riscvEnc var enc riscvEnc
rs1, rs2 := rs, 0 rs1, rs2 := rs, 0
switch mnem { switch mnem {
@@ -484,18 +864,25 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// System instructions with no operands. // System instructions with no operands.
case "FENCE", "ECALL", "EBREAK": case "FENCE", "ECALL", "EBREAK", "FENCE.TSO", "PAUSE":
enc, ok := riscvInstrTable[mnem] enc, ok := riscvInstrTable[mnem]
if !ok { if !ok {
return nil, fmt.Errorf("unsupported system instruction %q", mnem) return nil, fmt.Errorf("unsupported system instruction %q", mnem)
} }
// The bare FENCE expands to fence iorw, iorw: the predecessor and // The bare FENCE expands to fence iorw, iorw: the predecessor and
// successor fields both carry 0xF in the I-type immediate // successor fields both carry 0xF in the I-type immediate
// (the toolchain's encodeFenceOperand TYPE_NONE default). // (the toolchain's encodeFenceOperand TYPE_NONE default). FENCE.TSO
// carries the TSO fence mode with RW predecessor and successor.
imm := int32(0) imm := int32(0)
if mnem == "FENCE" { if mnem == "FENCE" {
imm = 0x0FF imm = 0x0FF
} }
if mnem == "FENCE.TSO" {
imm = 0x833
}
if mnem == "PAUSE" {
imm = 0x010
}
word = riscvIType(enc, 0, 0, imm) word = riscvIType(enc, 0, 0, imm)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
@@ -516,6 +903,29 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
// low three bits of the funct7 field.
if i := strings.IndexByte(mnem, '.'); i > 0 {
if base, ok := riscvCvtTable[mnem[:i]]; ok {
rm, ok := riscvRoundModes[mnem[i+1:]]
if !ok {
return nil, fmt.Errorf("unsupported rounding mode in %q", mnem)
}
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
base.funct7 = (base.funct7 &^ 7) | rm
word := riscvCvtType(base, rd, rs1)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
// R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last). // R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last).
if fmaEnc, ok := riscvFmaTable[mnem]; ok { if fmaEnc, ok := riscvFmaTable[mnem]; ok {
if len(ops) != 4 { if len(ops) != 4 {
@@ -532,29 +942,103 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last). // CSR instructions: INSTR csr, rs1|uimm, rd (destination last). The
if csrEnc, ok := riscvCsrTable[mnem]; ok { // write-only pseudos (CSRS/CSRC/CSRW and their immediate forms) spell
if len(ops) != 3 { // the source first, the CSR second, and read the destination as X0; the
// immediate or register variant follows the source operand's kind.
csrMnem := mnem
csrPseudo := false
csrRead := false
csrFix := int32(0)
switch mnem {
case "CSRS", "CSRW", "CSRC", "CSRSI", "CSRWI", "CSRCI":
csrMnem = map[string]string{
"CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC",
"CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI",
}[mnem]
csrPseudo = true
// CSRR csr, rd is CSRRS rd, csr, X0; the read pseudos RDCYCLE/RDTIME/
// RDINSTRET fix the CSR to cycle/time/instret.
case "CSRR":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
case "RDCYCLE", "RDTIME", "RDINSTRET":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
csrFix = map[string]int32{"RDCYCLE": 0xC00, "RDTIME": 0xC01, "RDINSTRET": 0xC02}[mnem]
}
if csrEnc, ok := riscvCsrTable[csrMnem]; ok {
if csrRead && len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
if csrPseudo && !csrRead && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if !csrPseudo && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
} }
csr := immFromOperand(ops[0]) // CSR address (12-bit) csrOp := ops[0]
srcOp := ops[0]
rdOp := ops[len(ops)-1]
switch {
case csrRead:
// CSRR csr, rd (or the fixed-CSR read pseudos with only rd).
case csrPseudo:
// src, csr.
if len(ops) > 1 {
csrOp, srcOp = ops[1], ops[0]
}
rdOp = nil
default:
// Either src, csr, rd or csr, src, rd: a CSR *name* in the
// second operand marks the toolchain's order.
srcOp = ops[1]
if op := ops[1]; op.Addr.Sym != nil {
if _, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
csrOp, srcOp = ops[1], ops[0]
}
}
}
csr, ok := riscvCSRAddress(csrOp)
if !ok && csrFix == 0 {
return nil, fmt.Errorf("%s: unknown CSR %q", mnem, csrOp.Raw)
}
if csrFix != 0 {
csr = csrFix
}
if csr < 0 || csr > 0xFFF { if csr < 0 || csr > 0xFFF {
return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr) return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr)
} }
rd := regFromOperand(ops[2]) // destination register rd := 0
if !csrPseudo {
rd = regFromOperand(rdOp) // destination register
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem) return nil, fmt.Errorf("invalid destination register in %s", mnem)
} }
}
if csrRead {
rd = regFromOperand(rdOp)
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
}
var src int var src int
if csrEnc.imm { switch {
// Immediate variant: ops[1] is a 5-bit unsigned immediate. case csrRead:
src = int(immFromOperand(ops[1])) // CSRR reads with rs1 = X0: src stays zero.
case isImmOperand(srcOp):
// Immediate variant: the source is a 5-bit unsigned immediate.
src = int(immFromOperand(srcOp))
if src < 0 || src > 31 { if src < 0 || src > 31 {
return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem) return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem)
} }
} else { case csrEnc.imm:
// Register variant: ops[1] is a register. return nil, fmt.Errorf("%s expects an immediate source", mnem)
src = regFromOperand(ops[1]) default:
// Register variant: the source is a register.
src = regFromOperand(srcOp)
if src < 0 { if src < 0 {
return nil, fmt.Errorf("invalid source register in %s", mnem) return nil, fmt.Errorf("invalid source register in %s", mnem)
} }
@@ -738,15 +1222,35 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvSType(enc, rs1, rs2, imm) word = riscvSType(enc, rs1, rs2, imm)
// Branches: rs1, rs2, label. // Branches: rs1, rs2, label. BGT/BLE/BGTU/BLEU are the swapped-spelling
// forms of BLT/BGE/BLTU/BGEU (bgt rs1, rs2 is blt rs2, rs1).
case len(ops) == 3 && isBranchInstr(mnem): case len(ops) == 3 && isBranchInstr(mnem):
rs1 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1]) rs2 := regFromOperand(ops[1])
target := labelFromOperand(ops[2]) target := labelFromOperand(ops[2])
targetOff, ok := offsets[target] switch mnem {
case "BGT":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, rs2, rs1 // blt rs2, rs1
case "BLE":
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, rs2, rs1 // bge rs2, rs1
case "BGTU":
enc, rs1, rs2 = riscvEnc{0x63, 0x6, 0x00}, rs2, rs1 // bltu rs2, rs1
case "BLEU":
enc, rs1, rs2 = riscvEnc{0x63, 0x7, 0x00}, rs2, rs1 // bgeu rs2, rs1
}
targetOff := 0
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
var ok bool
targetOff, ok = offsets[target]
if !ok { if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
}
offset := int32(targetOff - pc) offset := int32(targetOff - pc)
if rs1 < 0 || rs2 < 0 { if rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
@@ -758,10 +1262,15 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ. // The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
word = riscvBType(enc, rs1, rs2, offset) word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm. // U-type: rd, imm (or the toolchain testdata's INSTR $imm, rd).
case len(ops) == 2 && isUTypeInstr(mnem): case len(ops) == 2 && isUTypeInstr(mnem):
rd := regFromOperand(ops[0]) var rd int
imm := immFromOperand(ops[1]) var imm int32
if isImmOperand(ops[0]) {
imm, rd = immFromOperand(ops[0]), regFromOperand(ops[1])
} else {
rd, imm = regFromOperand(ops[0]), immFromOperand(ops[1])
}
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
@@ -1226,6 +1735,11 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := riscvCompressMnem(instr) mnem := riscvCompressMnem(instr)
ops := instr.Operands ops := instr.Operands
// The immediate aliases fold onto their I-type mnemonics before
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly
// as it compresses the spelling ADDI.
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
switch mnem { switch mnem {
case "LD", "MOV": case "LD", "MOV":
@@ -1289,6 +1803,9 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
case "ADDI": case "ADDI":
rd, rs1, imm := extractITypeParams(instr) rd, rs1, imm := extractITypeParams(instr)
if immNeg {
imm = -imm
}
if rd == -1 || rs1 == -1 { if rd == -1 || rs1 == -1 {
return 0, false return 0, false
} }
@@ -2045,7 +2562,8 @@ func isRTypeInstr(m string) bool {
case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND", case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND",
"ADDW", "SUBW", "SLLW", "SRLW", "SRAW", "ADDW", "SUBW", "SLLW", "SRLW", "SRAW",
"MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU", "MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU",
"MULW", "DIVW", "DIVUW", "REMW", "REMUW": "MULW", "DIVW", "DIVUW", "REMW", "REMUW",
"CZEROEQZ", "CZERONEZ":
return true return true
} }
return false return false
@@ -2085,7 +2603,7 @@ func isStoreInstr(m string) bool {
func isBranchInstr(m string) bool { func isBranchInstr(m string) bool {
switch m { switch m {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU": case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU":
return true return true
} }
return false return false
@@ -2111,7 +2629,8 @@ func isFPArithInstr(m string) bool {
switch m { switch m {
case "FADDS", "FSUBS", "FMULS", "FDIVS", case "FADDS", "FSUBS", "FMULS", "FDIVS",
"FADDD", "FSUBD", "FMULD", "FDIVD", "FADDD", "FSUBD", "FMULD", "FDIVD",
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD": "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
"FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX":
return true return true
} }
return false return false
+22
View File
@@ -219,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVUW": {0x3B, 0x5, 0x01}, "DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01}, "REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01}, "REMUW": {0x3B, 0x7, 0x01},
// Zicond conditional zeroing.
"CZEROEQZ": {0x33, 0x5, 0x07},
"CZERONEZ": {0x33, 0x7, 0x07},
// RV64I, I-type arithmetic. // RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00}, "ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00}, "ADDIW": {0x1B, 0x0, 0x00},
@@ -247,6 +250,12 @@ var riscvInstrTable = map[string]riscvEnc{
"BGE": {0x63, 0x5, 0x00}, "BGE": {0x63, 0x5, 0x00},
"BLTU": {0x63, 0x6, 0x00}, "BLTU": {0x63, 0x6, 0x00},
"BGEU": {0x63, 0x7, 0x00}, "BGEU": {0x63, 0x7, 0x00},
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
// with the register operands swapped.
"BGT": {0x63, 0x4, 0x00},
"BLE": {0x63, 0x5, 0x00},
"BGTU": {0x63, 0x6, 0x00},
"BLEU": {0x63, 0x7, 0x00},
// U-type. // U-type.
"LUI": {0x37, 0x0, 0x00}, "LUI": {0x37, 0x0, 0x00},
"AUIPC": {0x17, 0x0, 0x00}, "AUIPC": {0x17, 0x0, 0x00},
@@ -254,6 +263,8 @@ var riscvInstrTable = map[string]riscvEnc{
"ECALL": {0x73, 0x0, 0x00}, "ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00}, "EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00}, "FENCE": {0x0F, 0x0, 0x00},
"FENCE.TSO": {0x0F, 0x0, 0x00},
"PAUSE": {0x0F, 0x0, 0x00},
// JALR, indirect jump/call (I-type). // JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00}, "JALR": {0x67, 0x0, 0x00},
@@ -304,6 +315,13 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXD": {0x53, 0x1, 0x15}, "FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source. // FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11}, "FSGNJD": {0x53, 0x0, 0x11},
"FSGNJS": {0x53, 0x0, 0x10},
"FSGNJX": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x0, 0x15},
"FSGNJXS": {0x53, 0x0, 0x14},
"FSGNJND": {0x53, 0x1, 0x11},
"FSGNJNS": {0x53, 0x1, 0x10},
"FSGNJNX": {0x53, 0x1, 0x14},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03). // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
// The toolchain gives LR acquire ordering (aq = 1) and SC release // The toolchain gives LR acquire ordering (aq = 1) and SC release
@@ -375,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move) "FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move) "FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move) "FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
// The toolchain's W/D suffix spellings of the same moves.
"FMVXS": {0x70, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x53},
"FMVSX": {0x79, 0x0, 0x53},
} }
// riscvCvtType encodes an FP conversion instruction. // riscvCvtType encodes an FP conversion instruction.
+19 -6
View File
@@ -868,7 +868,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
t.Helper() t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src) fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr) instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil) return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil)
} }
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span // TestRISCVBranchJumpRange checks that displacements beyond the B-type span
@@ -905,9 +905,10 @@ func TestRISCVBranchJumpRange(t *testing.T) {
} }
} }
// TestRISCVBranchFarBody drives the range check through the full two-pass // TestRISCVBranchFarBody drives the relaxation pass through the full
// assembler: a forward branch over a body larger than the B-type span must // assembler: a forward branch over a body larger than the B-type span is
// error rather than wrap. // rewritten as an inverted branch over an inserted JMP, the same layout the
// toolchain produces, instead of wrapping to a wrong target.
func TestRISCVBranchFarBody(t *testing.T) { func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n") sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
@@ -916,8 +917,20 @@ func TestRISCVBranchFarBody(t *testing.T) {
} }
sb.WriteString("done:\n\tRET\n") sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String()) fn := firstTextRISCV(t, sb.String())
if _, _, _, _, _, err := assembleRISCV(fn); err == nil { out, _, _, _, _, err := assembleRISCV(fn)
t.Error("expected a branch-out-of-range error, got none") if err != nil {
t.Fatalf("unexpected error: %v", err)
}
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
if !bytes.Equal(out[0:4], wantBranch) {
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
}
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
wantJmp := wordLE(riscvJType(0, 4404))
if !bytes.Equal(out[4:8], wantJmp) {
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
} }
} }
+30 -7
View File
@@ -37,7 +37,7 @@ import (
// construction and are excluded from the diff; the other architectures list // construction and are excluded from the diff; the other architectures list
// their conditional branches outright. // their conditional branches outright.
func cmdAuditInstructions(args []string) error { func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", ` fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -57,11 +57,13 @@ per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order. the encodability backlog by frequency rather than by table order.
`) `)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons") corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return err return err
} }
if *corpus { if *corpus {
return cmdAuditCorpus(fs.Args()) return cmdAuditCorpus(fs.Args(), dirs)
} }
archName := "amd64" archName := "amd64"
switch n := len(fs.Args()); { switch n := len(fs.Args()); {
@@ -395,8 +397,11 @@ func (t *corpusTally) fail(path, reason string) {
} }
} }
// cmdAuditCorpus implements audit-instructions --corpus. // cmdAuditCorpus implements audit-instructions --corpus. The include
func cmdAuditCorpus(args []string) error { // directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error {
if len(args) > 1 { if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")} return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
} }
@@ -410,7 +415,25 @@ func cmdAuditCorpus(args []string) error {
} }
root = filepath.Join(strings.TrimSpace(string(out)), "src") root = filepath.Join(strings.TrimSpace(string(out)), "src")
} }
stats, err := runCorpusAudit(root) // The toolchain's shipped headers (funcdata.h and friends) define the
// macros GOROOT files include; a corpus audit measures those files, so
// the header directory joins the search path automatically. go_asm.h
// is compiler-generated per package and stays unresolvable on purpose.
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
seen := false
for _, d := range dirs {
if d == pkgInclude {
seen = true
}
}
if !seen {
dirs = append(dirs, pkgInclude)
}
}
}
stats, err := runCorpusAudit(root, dirs)
if err != nil { if err != nil {
return err return err
} }
@@ -454,7 +477,7 @@ func otherPortFile(path string) bool {
return false return false
} }
func runCorpusAudit(root string) (*corpusStats, error) { func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root) files, err := asmFiles(root)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -479,7 +502,7 @@ func runCorpusAudit(root string) (*corpusStats, error) {
if err != nil { if err != nil {
return nil, err return nil, err
} }
f, errs := parser.Parse(path, src) f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
var wanted []int // indexes into targets var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown { if a := arch.FromFilename(path); a != arch.Unknown {
+26 -11
View File
@@ -240,6 +240,16 @@ func readSource(path string) (string, error) {
return string(b), err return string(b), err
} }
// includeDirs collects repeatable -I flags: the directories searched for
// #include files during macro expansion and include splicing.
type includeDirs []string
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
func (d *includeDirs) Set(v string) error {
*d = append(*d, v)
return nil
}
func cmdTokens(args []string) int { func cmdTokens(args []string) int {
fs := newCommand("tokens", "gasm tokens <file>", ` fs := newCommand("tokens", "gasm tokens <file>", `
Print the lexical token stream of FILE: position, token kind and text, one Print the lexical token stream of FILE: position, token kind and text, one
@@ -476,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
} }
func cmdAsm(args []string) int { func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", ` fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>", `
Assemble FILE without the Go toolchain: every TEXT function is encoded to Assemble FILE without the Go toolchain: every TEXT function is encoded to
machine code and printed as a hex dump. Supported architectures: amd64 machine code and printed as a hex dump. Supported architectures: amd64
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP, (including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
@@ -498,9 +508,11 @@ and the format version from go version).
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)") format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)") pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)") archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
fs.Parse(args) fs.Parse(args)
if fs.NArg() != 1 { if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>") fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>")
return 2 return 2
} }
// The format is validated before anything else, so a bogus value exits 2 // The format is validated before anything else, so a bogus value exits 2
@@ -526,7 +538,7 @@ and the format version from go version).
fmt.Fprintln(os.Stderr, "gasm:", err) fmt.Fprintln(os.Stderr, "gasm:", err)
return 1 return 1
} }
f, errs := parser.Parse(path, src) f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
@@ -634,7 +646,7 @@ and the format version from go version).
// cmdDiff compares the machine code of two assembly files. // cmdDiff compares the machine code of two assembly files.
func cmdDiff(args []string) int { func cmdDiff(args []string) int {
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", ` set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
Compare the machine code produced by assembling two files. Compare the machine code produced by assembling two files.
Shows which functions differ and the byte-level differences. Shows which functions differ and the byte-level differences.
Useful for verifying that two implementations produce identical code, Useful for verifying that two implementations produce identical code,
@@ -645,9 +657,11 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
`) `)
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names") mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64") archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
var dirs includeDirs
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
set.Parse(args) set.Parse(args)
if set.NArg() != 2 { if set.NArg() != 2 {
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>") fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
return 2 return 2
} }
path1, path2 := set.Arg(0), set.Arg(1) path1, path2 := set.Arg(0), set.Arg(1)
@@ -675,12 +689,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
} }
// Assemble both files. // Assemble both files.
img1, err := assemblePath(path1, forced) img1, err := assemblePath(path1, forced, dirs)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err) fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
return 1 return 1
} }
img2, err := assemblePath(path2, forced) img2, err := assemblePath(path2, forced, dirs)
if err != nil { if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err) fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
return 1 return 1
@@ -755,14 +769,15 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
} }
} }
// assemblePath reads, parses and assembles a file (used by cmdDiff). A // assemblePath reads, preprocesses, parses and assembles a file (used by
// non-Unknown forced architecture overrides the file-name suffix. // cmdDiff). A non-Unknown forced architecture overrides the file-name
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) { // suffix.
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
src, err := readSource(path) src, err := readSource(path)
if err != nil { if err != nil {
return nil, err return nil, err
} }
f, errs := parser.Parse(path, src) f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs { for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
} }
+1 -1
View File
@@ -403,7 +403,7 @@ func TestRunCorpusAudit(t *testing.T) {
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n") write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n") write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
stats, err := runCorpusAudit(dir) stats, err := runCorpusAudit(dir, nil)
if err != nil { if err != nil {
t.Fatalf("runCorpusAudit: %v", err) t.Fatalf("runCorpusAudit: %v", err)
} }
+109
View File
@@ -0,0 +1,109 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"path/filepath"
"strings"
"testing"
)
// writeTree writes a directory of files and returns its root.
func writeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
// in-file parameterised macro and an include resolved through -I, and checks
// the assembled bytes came from the expansion (the loop body counts six
// increments, two per expanded iteration).
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"inc/consts.h": "#define NITER 3\n",
"main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"consts.h\"\n" +
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
"\tXORQ AX, AX\n" +
"\tMOVQ $NITER, CX\n" +
"loop:\n" +
"\tSTEP(AX)\n" +
"\tDECQ CX\n" +
"\tJNZ loop\n" +
"\tMOVQ AX, ret+0(FP)\n" +
"\tRET\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// The macro expanded to two ADDQ $1 encodings in the static body; the
// iteration count lives in the runtime loop.
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
}
}
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
// including file's directory wins over the -I directories.
func TestAsmIncludeResolutionOrder(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"src/main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"vals.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ $VAL, AX\n" +
"\tRET\n",
"src/vals.h": "#define VAL 1\n",
"late/vals.h": "#define VAL 2\n",
"early/vals.h": "#define VAL 3\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// VAL came from src/vals.h, not from either -I directory: the image
// loads the immediate 1.
if !strings.Contains(stdout, "b8 01 00 00 00") {
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
}
}
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
// resolves nowhere on the assembly path.
func TestAsmMissingIncludeIsAnError(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
if code == 0 {
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
}
if !strings.Contains(stderr, `#include "nothere.h"`) {
t.Errorf("stderr does not name the failing include: %s", stderr)
}
}
+12 -3
View File
@@ -141,12 +141,13 @@ gasm lint kernel_amd64.s
## asm ## asm
```text ```text
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file> Usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>
``` ```
| Flag | Default | Effect | | Flag | Default | Effect |
|---|---|---| |---|---|---|
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) | | `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols | | `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
| `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix | | `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix |
| `-o` | empty | write the output to this file instead of a hex dump on stdout | | `-o` | empty | write the output to this file instead of a hex dump on stdout |
@@ -162,6 +163,13 @@ system toolchain; `goobj` emits the Go toolchain's own object format, which
installed: the object preamble is captured from `go tool asm` and the format installed: the object preamble is captured from `go tool asm` and the format
version from `go version`. `raw` and `elf` need no toolchain at all. version from `go version`. `raw` and `elf` need no toolchain at all.
Assembly preprocessing matches the toolchain's: `#define` macros (object and
parameterised) expand at the point of use, `#undef`, `#ifdef`, `#ifndef`,
`#else` and `#endif` behave as in `go tool asm`, `;` separates statements,
and `#include "file"` splices the named file in, resolved against the source
directory and then each `-I` directory in order. `textflag.h` is the one
header that is not spliced: gasm consumes its flag names natively.
```sh ```sh
gasm asm hello_amd64.s gasm asm hello_amd64.s
``` ```
@@ -305,12 +313,13 @@ gasm debug --func add --cover hello_amd64.s
## diff ## diff
```text ```text
Usage: gasm diff [-GOARCH arch] <file1.s> <file2.s> Usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>
``` ```
| Flag | Default | Effect | | Flag | Default | Effect |
|---|---|---| |---|---|---|
| `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes | | `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names | | `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
Functions are paired by exact name unless `--map` says otherwise, so Functions are paired by exact name unless `--map` says otherwise, so
@@ -348,7 +357,7 @@ add: 16 bytes, args=24, frame=0 NOSPLIT
## audit-instructions ## audit-instructions
```text ```text
Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64] Usage: gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]
``` ```
Compare the gasm encoder for the given architecture (default amd64) against the Compare the gasm encoder for the given architecture (default amd64) against the
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME .SH NAME
gasm-asm \- assemble Plan 9 assembly without the Go toolchain gasm-asm \- assemble Plan 9 assembly without the Go toolchain
.SH SYNOPSIS .SH SYNOPSIS
.B gasm asm [\-\-format raw|elf|goobj] [\-p pkg] [\-GOARCH arch] [\-o out] <file> .B gasm asm [\-\-format raw|elf|goobj] [\-I dir] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
.SH DESCRIPTION .SH DESCRIPTION
Assemble FILE without the Go toolchain: every TEXT function is encoded Assemble FILE without the Go toolchain: every TEXT function is encoded
to machine code and printed as a hex dump. Supported architectures: to machine code and printed as a hex dump. Supported architectures:
@@ -47,6 +47,10 @@ functions link too.
.B \-\-format \fIraw|elf|goobj\fR .B \-\-format \fIraw|elf|goobj\fR
Output format; the default is raw. Output format; the default is raw.
.TP .TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-p \fIpkg\fR .B \-p \fIpkg\fR
Package path for --format goobj, qualifying the exported symbols. Package path for --format goobj, qualifying the exported symbols.
.TP .TP
+7 -1
View File
@@ -2,7 +2,7 @@
.SH NAME .SH NAME
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
.SH SYNOPSIS .SH SYNOPSIS
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [amd64|arm64|riscv64|loong64] .B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-I dir] [amd64|arm64|riscv64|loong64]
.SH DESCRIPTION .SH DESCRIPTION
Compare the gasm encoder for the given architecture (default amd64) Compare the gasm encoder for the given architecture (default amd64)
against against
@@ -38,6 +38,12 @@ second.
.B \-\-corpus [\fIdir\fR] .B \-\-corpus [\fIdir\fR]
Assemble a corpus of .s files and report pass rates and failure Assemble a corpus of .s files and report pass rates and failure
reasons. reasons.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory. A corpus run whose files include
toolchain headers (such as GOROOT/pkg/include) needs it, the same -I a
toolchain comparison takes.
.SH EXIT STATUS .SH EXIT STATUS
The mnemonic-diff mode reports through its output and exits 0; a failed The mnemonic-diff mode reports through its output and exits 0; a failed
probe or an unknown architecture exits non-zero. probe or an unknown architecture exits non-zero.
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME .SH NAME
gasm-diff \- compare the machine code of two assembly files gasm-diff \- compare the machine code of two assembly files
.SH SYNOPSIS .SH SYNOPSIS
.B gasm diff [\-GOARCH arch] <file1.s> <file2.s> .B gasm diff [\-GOARCH arch] [\-I dir] <file1.s> <file2.s>
.SH DESCRIPTION .SH DESCRIPTION
Compare the machine code produced by assembling two files. Shows which Compare the machine code produced by assembling two files. Shows which
functions differ and the byte-level differences. Useful for verifying functions differ and the byte-level differences. Useful for verifying
@@ -20,6 +20,10 @@ pairs two variants regardless of suffix.
Target architecture for both files: amd64, arm64, riscv64 or loong64; Target architecture for both files: amd64, arm64, riscv64 or loong64;
overrides the file-name suffixes. overrides the file-name suffixes.
.TP .TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-\-map \fIspec\fR .B \-\-map \fIspec\fR
Comma-separated old=new pairs to match functions with different names. Comma-separated old=new pairs to match functions with different names.
.SH EXIT STATUS .SH EXIT STATUS
+44 -1
View File
@@ -119,7 +119,9 @@ func (l *Lexer) Next() token.Token {
// is a C-preprocessor line continuation (used by #define macros in the // is a C-preprocessor line continuation (used by #define macros in the
// runtime .s files): splice the lines together by consuming both, so // runtime .s files): splice the lines together by consuming both, so
// the whole macro becomes one logical line that the parser treats as an // the whole macro becomes one logical line that the parser treats as an
// opaque preprocessor directive. // opaque preprocessor directive. The backslash may also reach its
// newline across whitespace and a trailing comment ("…; \ // note\n"),
// which the toolchain's scanner skips the same way.
for { for {
c := l.cur() c := l.cur()
if c == ' ' || c == '\t' || c == '\r' { if c == ' ' || c == '\t' || c == '\r' {
@@ -136,6 +138,16 @@ func (l *Lexer) Next() token.Token {
} }
continue continue
} }
if c == '\\' && l.continuationAhead() {
l.advance() // backslash, then the runes the scan saw
for !l.atEnd() && l.cur() != '\n' {
l.advance()
}
if !l.atEnd() {
l.advance() // the newline that closes the continuation
}
continue
}
break break
} }
@@ -185,6 +197,28 @@ func (l *Lexer) Next() token.Token {
} }
} }
// continuationAhead reports, without consuming anything, whether the
// backslash at the current position closes onto a newline through nothing
// but horizontal whitespace and one line comment. Positions after the
// backslash are inspected directly on the rune slice so a non-match leaves
// the scanner state untouched.
func (l *Lexer) continuationAhead() bool {
i := l.i + 1
for i < len(l.src) {
switch r := l.src[i]; {
case r == ' ' || r == '\t' || r == '\r':
i++
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
for i < len(l.src) && l.src[i] != '\n' {
i++
}
default:
return r == '\n'
}
}
return false
}
// lineComment consumes a // comment up to, but not including, the newline. A // lineComment consumes a // comment up to, but not including, the newline. A
// trailing run of \r, spaces and tabs is line-ending whitespace rather than // trailing run of \r, spaces and tabs is line-ending whitespace rather than
// comment content, so it never enters the token text. Trimming only a \r // comment content, so it never enters the token text. Trimming only a \r
@@ -403,6 +437,15 @@ func (l *Lexer) punct(start token.Position) token.Token {
case '|': case '|':
l.advance() l.advance()
return l.make(token.Pipe, start, "|") return l.make(token.Pipe, start, "|")
case ';':
l.advance()
return l.make(token.Semicolon, start, ";")
case '&':
l.advance()
return l.make(token.Ampersand, start, "&")
case '~':
l.advance()
return l.make(token.Tilde, start, "~")
default: default:
// Unknown rune: emit it as Illegal and move on. // Unknown rune: emit it as Illegal and move on.
l.advance() l.advance()
+146
View File
@@ -0,0 +1,146 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Constant-expression folding for operands. The toolchain's assembler
// evaluates arithmetic in every operand position, and macro-heavy GOROOT
// sources lean on it: parameterised bodies carry offsets like
// ((index*4)+0)(base), immediates like $(32-shift) and masks like
// $~63 or $(1<<0|1<<9). Substituting the parameters textually therefore
// leaves constant arithmetic behind, and the parser folds it here, keeping
// the operand AST identical to what the same literals written out would
// produce. Anything that is not a closed integer expression fails to fold
// and falls through to the ordinary operand paths.
package parser
import (
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// foldExpr evaluates the constant integer expression at the head of ts and
// returns its value together with the unconsumed tokens. ok is false when
// the tokens do not form an expression, which is the callers' signal to use
// the ordinary parsing paths.
func foldExpr(ts []token.Token) (val int64, rest []token.Token, ok bool) {
v, rest, ok := foldAdd(ts)
if !ok {
return 0, ts, false
}
return v, rest, true
}
// foldAdd parses addition-level expressions: +, - and | bind loosest, the
// Plan 9 convention that makes x<<1|3 read as (x<<1)|3.
func foldAdd(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldMul(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
kind := rest[0].Kind
if kind != token.Plus && kind != token.Minus && kind != token.Pipe {
return v, rest, true
}
w, r2, ok := foldMul(rest[1:])
if !ok {
return v, rest, true
}
switch kind {
case token.Plus:
v += w
case token.Minus:
v -= w
case token.Pipe:
v |= w
}
rest = r2
}
return v, rest, true
}
// foldMul parses multiplication-level expressions: *, / and the bit
// operators &, << and >>.
func foldMul(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldFactor(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
switch rest[0].Kind {
case token.Star:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v *= w
rest = r2
case token.Slash:
w, r2, ok := foldFactor(rest[1:])
if !ok || w == 0 {
return v, rest, true
}
v /= w
rest = r2
case token.Ampersand:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v &= w
rest = r2
case token.LShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v <<= uint(w)
rest = r2
case token.RShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v >>= uint(w)
rest = r2
default:
return v, rest, true
}
}
return v, rest, true
}
// foldFactor parses a number, a parenthesised expression, or a unary sign
// or complement.
func foldFactor(ts []token.Token) (int64, []token.Token, bool) {
if len(ts) == 0 {
return 0, ts, false
}
switch ts[0].Kind {
case token.Number:
v, ok := tryInt(ts[0].Text)
if !ok {
return 0, ts, false
}
return v, ts[1:], true
case token.LParen:
v, rest, ok := foldAdd(ts[1:])
if !ok || len(rest) == 0 || rest[0].Kind != token.RParen {
return 0, ts, false
}
return v, rest[1:], true
case token.Minus:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return -v, rest, true
case token.Plus:
return foldFactor(ts[1:])
case token.Tilde:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return ^v, rest, true
}
return 0, ts, false
}
+23
View File
@@ -408,6 +408,18 @@ func parseImmediate(g []token.Token) ast.Immediate {
return imm return imm
} }
} }
// A constant expression introduced by '(' or '~'. Textual macro
// substitution leaves arithmetic such as $(32-shift) and $~63 behind,
// and the toolchain evaluates it in place; only shapes the ordinary
// paths below cannot read reach the folder, so every existing form
// keeps its exact parse.
if g[0].Kind == token.LParen || g[0].Kind == token.Tilde {
if v, rest, ok := foldExpr(g); ok && len(rest) == 0 {
imm.Val = v
imm.HasVal = true
return imm
}
}
i := 0 i := 0
if g[i].Kind == token.Minus { if g[i].Kind == token.Minus {
imm.Neg = true imm.Neg = true
@@ -459,6 +471,17 @@ func parseAddress(g []token.Token) ast.Address {
} }
i := 0 i := 0
// A parenthesised constant expression as the displacement: substituted
// macro bodies carry ((index*4)+0)(base) shapes. As with the signed
// number path below, the value is committed only when a base group
// follows.
if i < len(g) && g[i].Kind == token.LParen {
if v, rest, ok := foldExpr(g[i:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen {
addr.Offset = v
addr.HasOff = true
i = len(g) - len(rest)
}
}
// Optional leading displacement before a '(' base group. A sign pushes // Optional leading displacement before a '(' base group. A sign pushes
// the parenthesis one token further out: -4(DX) has it at i+2. // the parenthesis one token further out: -4(DX) has it at i+2.
if isSignedNumber(g, i) { if isSignedNumber(g, i) {
+475
View File
@@ -0,0 +1,475 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The preprocessor turns #define and #include directives into the token
// stream the parser really sees, the way the Go toolchain's assembler does:
// object and parameterised macros expand at the point of use, and an
// #include splices the named file's lines in place of the directive. The
// pass runs only on the assembly path (gasm asm, diff, the corpus audit),
// where the result is machine code; parsing for the linter, formatter and
// language server keeps the raw file so their view of #define lines, and
// therefore their macro-aware behaviour, is unchanged.
package parser
import (
"fmt"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Options controls the optional preprocessing applied before a file is
// parsed. The zero value reproduces Parse exactly.
type Options struct {
// IncludeDirs lists the -I directories searched for #include files,
// in order, after the including file's own directory.
IncludeDirs []string
// Expand enables macro expansion, include splicing and the
// statement-separator reading of ';' that the expanded bodies rely on.
Expand bool
}
// ParseWithOptions parses src like Parse, optionally preprocessing it first.
// The returned file is usable even when errors is non-empty.
func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) {
tokens := lexer.Tokenize(src)
var lines [][]token.Token
var errs []error
if opts.Expand {
pp := &preproc{opts: opts, macros: map[string]*macroDef{}}
lines = pp.fileLines(path, tokens, token.Position{})
errs = pp.errs
} else {
lines = splitLines(tokens)
}
p := &state{path: path}
p.parse(lines)
return p.file, append(errs, p.errs...)
}
// maxExpansionDepth bounds recursive macro expansion; the toolchain's
// assembler gives up after 100 nested invocations without producing a token.
const maxExpansionDepth = 100
// textflagHeader names the one header gasm does not splice: its flag macros
// (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser,
// encoders and linter, and expanding them to their numeric constants would
// leave every consumer blind to them.
const textflagHeader = "textflag.h"
// macroDef is one #define. A nil args slice is an object macro; a non-nil
// (possibly empty) one is parameterised, the C distinction between
// "#define A(x)" and "#define A (x)".
type macroDef struct {
name string
args []string
body []token.Token
}
// preproc carries the state of one expansion pass: the live macro table, the
// chain of files currently being read, for cycle detection, and the
// conditional-inclusion stack of #ifdef regions.
type preproc struct {
opts Options
macros map[string]*macroDef
errs []error
stack []string // absolute paths of files being read, innermost last
ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth
}
// enabled reports whether the position being read is inside a live
// conditional branch. Directives inside a disabled branch contribute
// nothing, and its content lines are dropped, exactly as the toolchain's
// input stack does.
func (pp *preproc) enabled() bool {
return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1]
}
func (pp *preproc) errorf(pos token.Position, format string, args ...any) {
pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
}
// fileLines tokenizes and preprocesses one file into logical lines.
// Directive lines are kept (the parser records them for the tooling);
// #include lines are replaced by the included file's lines. includePos is
// the position of the #include that pulled this file in, zero for the
// top-level file, and only serves cycle diagnostics.
func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token {
abs, err := filepath.Abs(path)
if err != nil {
abs = filepath.Clean(path)
}
if slices.Contains(pp.stack, abs) {
if includePos.IsValid() {
pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path))
}
return nil
}
pp.stack = append(pp.stack, abs)
var out [][]token.Token
for _, line := range splitLines(tokens) {
if len(line) == 0 {
out = append(out, line)
continue
}
if line[0].Kind == token.Hash {
out = append(out, pp.directive(line, filepath.Dir(path))...)
continue
}
if !pp.enabled() {
continue
}
out = append(out, splitOnSemicolons(pp.expandTokens(line))...)
}
pp.stack = pp.stack[:len(pp.stack)-1]
if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 {
// The stack is per-input, shared across includes, so only the
// top-level file's end can decide the input was left unclosed.
pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef")
}
return out
}
// directive processes one '#' line and returns the lines to keep in the
// stream: every directive line is kept as-is for the parser (which records
// it), except #include, which is replaced by the spliced content.
// Conditionals are tracked on every line; every other directive is inert
// inside a disabled branch.
func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token {
if len(line) < 2 || line[1].Kind != token.Ident {
return [][]token.Token{line}
}
switch line[1].Text {
case "ifdef", "ifndef":
pp.ifdef(line, line[1].Text == "ifndef")
case "else":
pp.elseBranch(line)
case "endif":
pp.endif(line)
case "define":
if pp.enabled() {
pp.define(line)
}
case "undef":
if pp.enabled() {
pp.undef(line)
}
case "include":
if pp.enabled() {
return pp.include(line, dir)
}
default:
// #line and unknown directives are recorded but not interpreted:
// conservative support keeps the parser's view intact and files
// using them fail on their content, not silently.
}
return [][]token.Token{line}
}
// ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth
// onto the conditional stack. A branch opened inside a disabled region is
// itself disabled, however the name resolves.
func (pp *preproc) ifdef(line []token.Token, inverted bool) {
truth := false
if len(line) >= 3 && line[2].Kind == token.Ident {
_, defined := pp.macros[line[2].Text]
truth = defined != inverted
} else {
pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text)
}
if !pp.enabled() {
truth = false
}
pp.ifdefStack = append(pp.ifdefStack, truth)
}
// elseBranch flips the innermost conditional's truth, but only when the
// region enclosing it is itself live: the toolchain keeps outer overrides.
func (pp *preproc) elseBranch(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #else")
return
}
if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] {
pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1]
}
}
// endif closes the innermost conditional.
func (pp *preproc) endif(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #endif")
return
}
pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1]
}
// define parses "#define NAME[(formals)] body" into the macro table. The
// body runs to the end of the logical line (the lexer has already spliced
// backslash continuations) and stops at a comment, which never expands.
func (pp *preproc) define(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
name := line[2]
args := []string(nil)
body := line[3:]
// The definition is parameterised only when '(' follows the name
// directly; the toolchain separates "#define A(x)" from
// "#define A (x)" by adjacency, and so does the column check here.
if len(body) > 0 && body[0].Kind == token.LParen &&
body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) {
args = []string{}
i := 1
for i < len(body) && body[i].Kind != token.RParen {
if body[i].Kind == token.Ident {
args = append(args, body[i].Text)
}
i++
}
if i < len(body) {
body = body[i+1:]
} else {
body = nil
}
}
if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 {
body = body[:i]
}
if _, exists := pp.macros[name.Text]; exists {
// The toolchain refuses redefinition, so a file the oracle accepts
// never redefines; failing here keeps that contract visible.
pp.errorf(name.Pos, "redefinition of macro %s", name.Text)
}
pp.macros[name.Text] = &macroDef{name: name.Text, args: args, body: pp.bodyWithBreaks(body)}
}
// bodyWithBreaks records the statement boundaries the continuations carry.
// The lexer splices backslash-continued lines into one logical line, but the
// toolchain keeps the newline as a token in the stored body, which is how a
// multi-instruction body without semicolons (the arm64 style) still splits
// into statements on expansion. A line change inside the logical line is
// exactly a continuation, so the boundary is restored from the positions.
func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token {
out := make([]token.Token, 0, len(body))
for i, t := range body {
if i > 0 && t.Pos.Line != body[i-1].Pos.Line {
out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos})
}
out = append(out, t)
}
return out
}
// undef handles "#undef NAME", which the toolchain honours and requires to
// name a defined macro.
func (pp *preproc) undef(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
if _, ok := pp.macros[line[2].Text]; !ok {
pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text)
return
}
delete(pp.macros, line[2].Text)
}
// include resolves and splices "#include \"file\"". A header that cannot be
// read keeps the directive line in the stream, with a diagnostic.
func (pp *preproc) include(line []token.Token, dir string) [][]token.Token {
if len(line) < 3 || line[2].Kind != token.String {
return [][]token.Token{line}
}
header := line[2]
name, err := strconv.Unquote(header.Text)
if err != nil {
pp.errorf(header.Pos, "unquoting include file name: %v", err)
return [][]token.Token{line}
}
if filepath.Base(name) == textflagHeader {
// Flag macros are handled natively (see textflagHeader); the
// directive stays so tools still see the include.
return [][]token.Token{line}
}
resolved, ok := pp.resolve(name, dir)
if !ok {
searched := append([]string{dir}, pp.opts.IncludeDirs...)
pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", "))
return [][]token.Token{line}
}
src, err := os.ReadFile(resolved)
if err != nil {
pp.errorf(header.Pos, "#include %q: %v", name, err)
return [][]token.Token{line}
}
return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos)
}
// resolve looks an include name up the way the toolchain does: as written
// (relative to the working directory), then relative to the including
// file's directory, then in each -I directory in order.
func (pp *preproc) resolve(name, dir string) (string, bool) {
candidates := []string{name}
if !filepath.IsAbs(name) {
candidates = append(candidates, filepath.Join(dir, name))
for _, d := range pp.opts.IncludeDirs {
candidates = append(candidates, filepath.Join(d, name))
}
}
for _, c := range candidates {
if st, err := os.Stat(c); err == nil && !st.IsDir() {
return c, true
}
}
return "", false
}
// expandTokens expands every macro invocation in a token sequence,
// recursively, with a depth guard. A body is spliced into the sequence in
// place and rescanned, the way the toolchain's input stack re-reads pushed
// tokens: an object macro may name a parameterised one, and the argument
// list of the expansion may then come from the tokens that follow.
func (pp *preproc) expandTokens(in []token.Token) []token.Token {
s := in
i := 0
consecutive := 0
for i < len(s) {
t := s[i]
if t.Kind != token.Ident {
i++
consecutive = 0
continue
}
def := pp.macros[t.Text]
if def == nil {
i++
consecutive = 0
continue
}
// The guard mirrors the toolchain's: 100 nested invocations in a
// row without a plain token between them means recursion.
consecutive++
if consecutive > maxExpansionDepth {
pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth)
return nil
}
if def.args == nil {
s = append(s[:i], append(restamp(def.body, t.Pos), s[i+1:]...)...)
continue
}
// A parameterised macro invoked without its parentheses stands
// unexpanded, naming itself, as in the toolchain.
if i+1 >= len(s) || s[i+1].Kind != token.LParen {
i++
consecutive = 0
continue
}
args, next := pp.collectArgs(s, i+1, t)
if args == nil {
return nil
}
// A zero-argument macro may be invoked as NAME().
if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 {
args = nil
}
if len(args) != len(def.args) {
pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args))
i = next
consecutive = 0
continue
}
sub := make([]token.Token, 0, len(def.body))
for _, bt := range def.body {
if bt.Kind == token.Ident {
if k := slices.Index(def.args, bt.Text); k >= 0 {
sub = append(sub, restamp(args[k], t.Pos)...)
continue
}
}
sub = append(sub, bt)
}
s = append(s[:i], append(sub, s[next:]...)...)
}
return s
}
// collectArgs reads the actual argument tokens of an invocation; the opening
// parenthesis is at start. Commas separate arguments except inside nested
// parentheses. A nil result means the list was unterminated, which is a
// diagnostic.
func (pp *preproc) collectArgs(in []token.Token, start int, name token.Token) ([][]token.Token, int) {
var args [][]token.Token
var cur []token.Token
nesting := 0
for i := start + 1; i < len(in); i++ {
t := in[i]
switch t.Kind {
case token.LParen:
nesting++
cur = append(cur, t)
case token.RParen:
if nesting == 0 {
return append(args, cur), i + 1
}
nesting--
cur = append(cur, t)
case token.Comma:
if nesting == 0 {
args = append(args, cur)
cur = nil
continue
}
cur = append(cur, t)
case token.Comment:
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, i
default:
cur = append(cur, t)
}
}
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, len(in)
}
// restamp copies body tokens to the invocation's position, so diagnostics
// and the line table point where the macro was used, as the toolchain's
// input stack does.
func restamp(body []token.Token, pos token.Position) []token.Token {
out := make([]token.Token, len(body))
for i, t := range body {
t.Pos, t.End = pos, pos
out[i] = t
}
return out
}
// splitOnSemicolons breaks a token sequence at ';' statement separators and
// at the Newline markers that record continuation boundaries inside macro
// bodies, producing the logical lines the parser expects. The separators
// carry no meaning beyond the break, so the pieces are exactly what the same
// statements on separate lines would produce.
func splitOnSemicolons(ts []token.Token) [][]token.Token {
var out [][]token.Token
start := 0
for i, t := range ts {
if t.Kind == token.Semicolon || t.Kind == token.Newline {
if i > start {
out = append(out, ts[start:i])
}
start = i + 1
}
}
if start < len(ts) {
out = append(out, ts[start:])
}
return out
}
+552
View File
@@ -0,0 +1,552 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// expand parses src with preprocessing enabled and returns the first TEXT's
// body instructions as "MNEMONIC operand|operand" strings, the shape the
// expansion assertions below compare against. Runs of spaces are
// collapsed: Raw renders a token group as its tokens joined with single
// spaces, so "$(32-7)" arrives as "$ ( 32 - 7 )" and the comparison must
// not depend on that spelling.
func expand(t *testing.T, src string) (*ast.File, []string) {
t.Helper()
f, errs := ParseWithOptions("t_amd64.s", src, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
ts := texts(f)
if len(ts) == 0 {
t.Fatalf("no TEXT in:\n%s", src)
}
var got []string
for _, s := range ts[0].Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
var ops []string
for _, op := range in.Operands {
ops = append(ops, op.Raw)
}
line := in.Mnemonic.Text + " " + strings.Join(ops, ", ")
got = append(got, strings.ReplaceAll(line, " ", ""))
}
return f, got
}
func wantLines(t *testing.T, got []string, want ...string) {
t.Helper()
strip := func(lines []string) string {
var out []string
for _, l := range lines {
out = append(out, strings.ReplaceAll(l, " ", ""))
}
return strings.Join(out, "\n")
}
if strip(got) != strip(want) {
t.Errorf("expanded body:\n %s\nwant:\n %s", strings.Join(got, "\n "), strings.Join(want, "\n "))
}
}
func TestObjectMacroExpandsAtUse(t *testing.T) {
_, got := expand(t, `
#define REGTMP CX
#define TWICE ADDQ CX, AX; ADDQ CX, AX
TEXT ·f(SB), NOSPLIT, $0
MOVQ 8(SP), REGTMP
TWICE
RET
`)
wantLines(t, got,
"MOVQ 8(SP), CX",
"ADDQ CX, AX",
"ADDQ CX, AX",
"RET",
)
}
func TestParameterisedMacroSubstitutesArguments(t *testing.T) {
f, errs := ParseWithOptions("t_amd64.s", `
#define ROUND1(a, index, const, shift) \
ADDQ $const, a; \
MOVW (index*4)(SP), a; \
RORQ $(32-shift), a
TEXT ·f(SB), NOSPLIT, $0
ROUND1(AX, 3, 0xd76aa478, 7)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
add := body[0].(*ast.Instr)
if add.Mnemonic.Text != "ADDQ" || !add.Operands[0].Imm.HasVal ||
add.Operands[0].Imm.Val != 0xd76aa478 || add.Operands[1].Addr.Sym == nil ||
add.Operands[1].Addr.Sym.Name != "AX" {
t.Errorf("ADDQ operands substituted wrong: %+v %+v", add.Operands[0].Imm, add.Operands[1].Addr)
}
mov := body[1].(*ast.Instr)
if addr := mov.Operands[0].Addr; !addr.HasOff || addr.Offset != 12 {
t.Errorf("MOVW offset = %+v, want 12 from 3*4", addr)
}
ror := body[2].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25 from (32-7)", ror.Operands[0].Imm)
}
}
func TestMacroArgumentsKeepCommasInParens(t *testing.T) {
// An argument may itself be an unparenthesised expression: the tokens
// substitute verbatim and the parser folds the result, as the
// toolchain's parser does.
f, errs := ParseWithOptions("t_amd64.s", `
#define LOAD(dst, off) MOVQ off(SP), dst
TEXT ·f(SB), NOSPLIT, $0
LOAD(AX, 1*8)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
addr := in.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 {
t.Errorf("offset = %+v, want 8", addr)
}
if sym := in.Operands[1].Addr.Sym; sym == nil || sym.Name != "AX" {
t.Errorf("destination = %+v, want AX", in.Operands[1].Addr)
}
}
func TestNestedMacroInvocations(t *testing.T) {
// An object macro naming a parameterised one, and a parameterised body
// invoking another parameterised macro: the toolchain's input stack
// rescans substituted tokens, and so does expansion here.
_, got := expand(t, `
#define DOUBLE(x) ADDQ x, x
#define TWICE2 DOUBLE
#define FOUR(a, b) DOUBLE(a); DOUBLE(b)
TEXT ·f(SB), NOSPLIT, $0
TWICE2(AX)
FOUR(AX, CX)
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestMultiLineBodySplitsWithoutSemicolons(t *testing.T) {
// The arm64 style: backslash-continued lines with no semicolons. The
// continuation newline is a statement boundary, as in the toolchain.
_, got := expand(t, `
#define PAIR \
ADDQ AX, AX \
MOVQ AX, CX
TEXT ·f(SB), NOSPLIT, $0
PAIR
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"MOVQ AX, CX",
"RET",
)
}
func TestZeroArgumentMacro(t *testing.T) {
_, got := expand(t, `
#define BARRIER()
TEXT ·f(SB), NOSPLIT, $0
BARRIER()
RET
`)
wantLines(t, got, "RET")
}
func TestParameterisedWithoutParensStandsAsName(t *testing.T) {
// A parameterised macro invoked without its parentheses names itself,
// which the parser then reports as an unknown instruction rather than
// silently expanding nothing.
f, errs := ParseWithOptions("t_amd64.s", `
#define M(x) ADDQ x, x
TEXT ·f(SB), NOSPLIT, $0
M
RET
`, Options{Expand: true})
if len(errs) != 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
if len(fn.Body) == 0 {
t.Fatal("body empty")
}
in, ok := fn.Body[0].(*ast.Instr)
if !ok || in.Mnemonic.Text != "M" {
t.Fatalf("bare parameterised macro did not stand as its name: %+v", fn.Body[0])
}
}
func TestDefinitionScoping(t *testing.T) {
// A definition applies from its point onward: the use before the
// #define stays untouched.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
SPECIAL
#define SPECIAL ADDQ AX, AX
SPECIAL
RET
`)
wantLines(t, got,
"SPECIAL",
"ADDQ AX, AX",
"RET",
)
}
func TestUndefRemovesMacro(t *testing.T) {
_, got := expand(t, `
#define TEMP AX
TEXT ·f(SB), NOSPLIT, $0
TEMP
#undef TEMP
TEMP
RET
`)
wantLines(t, got,
"AX",
"TEMP",
"RET",
)
}
func TestUndefUndefinedMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#undef NOSUCH\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "undefined macro NOSUCH") {
t.Fatalf("#undef of an undefined macro: got %v, want an error naming it", errs)
}
}
func TestRedefinitionIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A X\n#define A Y\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "redefinition of macro A") {
t.Fatalf("redefinition: got %v, want an error", errs)
}
}
func TestRecursiveMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A B\n#define B A\nTEXT ·f(SB), NOSPLIT, $0\n\tA\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "recursive macro invocation") {
t.Fatalf("recursion: got %v, want a recursive-macro error, not a hang", errs)
}
}
func TestWrongArgumentCountIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define M(a, b) ADDQ a, b\nTEXT ·f(SB), NOSPLIT, $0\n\tM(AX)\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "wrong arg count for macro M") {
t.Fatalf("arg count: got %v, want an error", errs)
}
}
func TestConditionalsSelectOneBranch(t *testing.T) {
_, got := expand(t, `
#define MODE2
TEXT ·f(SB), NOSPLIT, $0
#ifdef MODE2
ADDQ AX, AX
#else
SUBQ AX, AX
#endif
#ifndef MODE2
SUBQ CX, CX
#else
ADDQ CX, CX
#endif
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestConditionalsHideDefinitionsAndIncludes(t *testing.T) {
// A definition inside a disabled branch must not exist, and an
// unresolvable include there must not be followed.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
#ifdef NOTDEFINED
#define HIDEN ADDQ AX, AX
#include "nowhere.h"
#endif
HIDEN
RET
`)
wantLines(t, got, "HIDEN", "RET")
}
func TestUnclosedConditionalIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#ifdef X\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unclosed #ifdef") {
t.Fatalf("unclosed conditional: got %v, want an error", errs)
}
}
func TestUnmatchedConditionalDelimitersAreErrors(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#endif\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #endif") {
t.Fatalf("unmatched #endif: got %v, want an error", errs)
}
_, errs = ParseWithOptions("t_amd64.s", "#else\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #else") {
t.Fatalf("unmatched #else: got %v, want an error", errs)
}
}
// includeTree writes a directory of include files and returns its path.
func includeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
func TestIncludeSplicesAndDefinesAreShared(t *testing.T) {
dir := includeTree(t, map[string]string{
"consts.h": "#define KONST $42\n",
})
f, errs := ParseWithOptions("t_amd64.s", `
#include "consts.h"
TEXT ·f(SB), NOSPLIT, $0
MOVQ KONST, AX
RET
`, Options{Expand: true, IncludeDirs: []string{dir}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if in.Mnemonic.Text != "MOVQ" || strings.ReplaceAll(in.Operands[0].Raw, " ", "") != "$42" {
t.Fatalf("include splicing failed: %+v", in)
}
}
func TestIncludeResolutionOrder(t *testing.T) {
// The including file's directory wins over the -I list, and the -I list
// is searched in order.
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"inc/which.h": "#define WHO ONE\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
// The header exists in the including file's directory and in two -I
// directories; the source-directory copy must win.
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
found := false
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO ONE") {
found = true
}
}
if !found {
t.Error("the including file's directory did not win include resolution")
}
}
func TestIncludeSearchesIncludeDirsInOrder(t *testing.T) {
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO THREE") {
t.Error("the second -I directory was searched before the first")
}
}
}
func TestIncludeCycleIsDetected(t *testing.T) {
src := includeTree(t, map[string]string{
"a.s": "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"b.s": "#include \"a.s\"\n",
})
_, errs := ParseWithOptions(filepath.Join(src, "a.s"), "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "include cycle") {
t.Fatalf("include cycle: got %v, want a cycle diagnostic, not a hang", errs)
}
}
func TestUnresolvableIncludeIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true, IncludeDirs: []string{t.TempDir()}})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), `#include "nothere.h"`) {
t.Fatalf("missing include: got %v, want a clear diagnostic", errs)
}
}
func TestTextflagHeaderIsNeverSpliced(t *testing.T) {
// textflag.h resolves nowhere here, yet the file must parse: the flag
// names are consumed natively and the include stays in the tree.
f, errs := ParseWithOptions("t_amd64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
hasInclude := false
for _, d := range f.Decls {
if _, ok := d.(*ast.Include); ok {
hasInclude = true
}
}
if !hasInclude {
t.Error("textflag.h include was dropped from the tree")
}
}
func TestSemicolonSplitsRawLinesToo(t *testing.T) {
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
BYTE $0x0f; BYTE $0x1f
RET
`)
wantLines(t, got, "BYTE $0x0f", "BYTE $0x1f", "RET")
}
func TestParseUnchangedWithoutExpand(t *testing.T) {
// Without Expand the preprocessor must not exist: a macro invocation
// stays an unexpanded instruction line and ';' keeps the old parse.
f, errs := Parse("t_amd64.s", `
#define TWICE ADDQ AX, AX
TEXT ·f(SB), NOSPLIT, $0
TWICE
BYTE $0x0f; BYTE $0x1f
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
var mnemonics []string
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
mnemonics = append(mnemonics, in.Mnemonic.Text)
}
}
if strings.Join(mnemonics, " ") != "TWICE BYTE RET" {
t.Errorf("non-expanding parse changed: %v", mnemonics)
}
}
func TestConstantExpressionFolding(t *testing.T) {
// The shapes substituted macro bodies leave behind: parenthesised
// arithmetic in immediates and displacements, tilde complements. The
// assertions read the semantic fields; Raw keeps the operand's tokens
// in the canonicalised rendering, not the folded values.
f, errs := ParseWithOptions("t_amd64.s", `
TEXT ·f(SB), NOSPLIT, $0
RORQ $(32-7), AX
ANDQ $~63, AX
MOVQ ((2*4)+0)(SP), AX
MOVQ $((1<<3)|(1<<1)), AX
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
ror := body[0].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25", ror.Operands[0].Imm)
}
and := body[1].(*ast.Instr)
if !and.Operands[0].Imm.HasVal || and.Operands[0].Imm.Val != -64 {
t.Errorf("ANDQ immediate = %+v, want -64", and.Operands[0].Imm)
}
mov := body[2].(*ast.Instr)
addr := mov.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 || addr.Base != "SP" {
t.Errorf("MOVQ address = %+v, want 8(SP)", addr)
}
mov2 := body[3].(*ast.Instr)
if !mov2.Operands[0].Imm.HasVal || mov2.Operands[0].Imm.Val != 10 {
t.Errorf("MOVQ immediate = %+v, want 10", mov2.Operands[0].Imm)
}
}
func TestConstantExpressionFoldsWithoutExpand(t *testing.T) {
// Folding is a parser capability, not a preprocessing one: a
// hand-written $(32-7) folds the same way with expansion off.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tRORQ $(32-7), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if !in.Operands[0].Imm.HasVal || in.Operands[0].Imm.Val != 25 {
t.Errorf("Imm = %+v, want 25", in.Operands[0].Imm)
}
}
func TestNotAnExpressionFallsBack(t *testing.T) {
// Symbol immediates and floats must keep their ordinary parse.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ $1.5, AX\n\tMOVQ $·sym(SB), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
mov1 := fn.Body[0].(*ast.Instr)
if mov1.Operands[0].Imm.HasVal || mov1.Operands[0].Imm.Float != "1.5" {
t.Errorf("float immediate parsed as %+v", mov1.Operands[0].Imm)
}
mov2 := fn.Body[1].(*ast.Instr)
if mov2.Operands[0].Imm.Sym == nil {
t.Errorf("symbol immediate parsed as %+v", mov2.Operands[0].Imm)
}
}
File diff suppressed because it is too large Load Diff
+48
View File
@@ -0,0 +1,48 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Carry arithmetic, logical shifts, register aliases with element selectors
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
#include "textflag.h"
#define acc0 V8
#define acc1 V9
#define const0 R15
#define POLY V15
// carry pins the ADC/SBC family: the $0 spellings in two and three
// operands, and the register-carry forms.
TEXT ·carry(SB), NOSPLIT, $0-0
ADC $0, R20
ADC $0, R20, R4
SBCS $0, R4
SBCS $0, R4, R12
SBCS R15, R4, R12
SBC $0, R1
ADCSW $0, R2, R3
RET
// shift pins the shifted-register forms including ROR, which only the
// logical family accepts.
TEXT ·shift(SB), NOSPLIT, $0-0
ANDW R9@>7, R19, R26
AND R1@>33, R2, R3
ADD R1<<11, R2, R3
SUB R1->33, R2
ORR R5<<2, R6, R7
RET
// vecalias pins the vector aliases with element selectors and the
// structure loads with aliased members.
TEXT ·vecalias(SB), NOSPLIT, $0-0
MOVD $0xC2, R1
VMOV R1, POLY.D[0]
VMOV R0, POLY.D[1]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [acc0.B16]
VLD1.P (R0), [acc0.B16, acc1.B16]
VST1 [acc0.B16, acc1.B16], (R1)
VST1.P [acc0.B16, acc1.B16], 32(R1)
RET
+51
View File
@@ -0,0 +1,51 @@
// The subtract-immediate fold, the TEQ/TNE trap pseudos, PRELDX, the FP
// condition branches and the N(PC) branch spellings, against the toolchain.
#include "textflag.h"
// func SubFold(x int64) int64
TEXT ·SubFold(SB), NOSPLIT, $0-16
MOVV x+0(FP), R8
SUBV $0, R8
SUBV $4, R9, R10
SUBV $4096, R11
SUBV $-4, R12
SUB $1, R13
SUBVU $4, R14
SUBV $1048576, R15
MOVV R8, ret+8(FP)
RET
// func Traps(x int64) int64
TEXT ·Traps(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
TEQ $4, R4, R5
TEQ $4, R4
TNE $6, R5, R6
MOVV R4, ret+8(FP)
RET
// func Prefetch(x int64) int64
TEXT ·Prefetch(SB), NOSPLIT, $0-16
MOVV x+0(FP), R7
PRELDX 0(R7), $0x80001021, $0
PRELDX -1(R7), $0x1021, $2
MOVV R7, ret+8(FP)
RET
// func BranchForms(x int64) int64
TEXT ·BranchForms(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
l1:
BFPT l1
BFPT FCC3, l1
BFPF l1
JMP -4(PC)
JAL 1(PC)
JAL (R4)
loop:
ADDV $1, R4
BEQ R4, R5, loop
BNE R4, l1
RET
+33
View File
@@ -0,0 +1,33 @@
// PCALIGN padding on loong64: andi $0, $0, 0 (the architecture's NOP), plus
// the automatic loop-head alignment to a 16-byte boundary.
#include "textflag.h"
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $16
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $32
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func LoopAlign(x int64) int64
TEXT ·LoopAlign(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
MOVV $10, R5
loop:
BEQ R4, R5, done
ADDV $1, R4
JMP loop
done:
MOVV R4, ret+8(FP)
RET
+35
View File
@@ -0,0 +1,35 @@
// PCALIGN padding on riscv64: 4-byte NOPs with a 2-byte compressed NOP when
// the pad is 2 mod 4, exactly as the toolchain lays the bytes down.
#include "textflag.h"
// func Pad8(x int64) int64
TEXT ·Pad8(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $16
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $32
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func PadAfterOdd(x int64) int64
TEXT ·PadAfterOdd(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
+103
View File
@@ -0,0 +1,103 @@
// Instruction prefixes: LOCK, REP and REPN. go tool asm encodes each
// statement as a standalone one-byte instruction with a PC of its own (F0,
// F3 and F2 respectively); the statement that follows is encoded unaware of
// it, and nothing validates the pairing. The shapes are the runtime's
// atomic read-modify-write family and the string moves, every result folded
// back.
#include "textflag.h"
// func cas64(ptr *uint64, old, new uint64) bool
TEXT ·cas64(SB), NOSPLIT, $0-25
MOVQ ptr+0(FP), BX
MOVQ old+8(FP), AX
MOVQ new+16(FP), CX
LOCK
CMPXCHGQ CX, 0(BX)
SETEQ ret+24(FP)
RET
// func casloop(addr *uint64, v uint64) uint64
// The runtime's Or64 shape: a LOCK inside a branch loop, the backward jump
// measuring over the prefix statement's own byte.
TEXT ·casloop(SB), NOSPLIT, $0-24
MOVQ addr+0(FP), BX
MOVQ v+8(FP), CX
loop:
MOVQ CX, DX
MOVQ (BX), AX
ORQ AX, DX
LOCK
CMPXCHGQ DX, (BX)
JNZ loop
MOVQ AX, ret+16(FP)
RET
// func xadd64(p *uint64, v uint64) uint64
TEXT ·xadd64(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
LOCK
XADDQ BX, (AX)
MOVQ AX, ret+16(FP)
RET
// func xaddw(p *uint16, v uint16) uint16
TEXT ·xaddw(SB), NOSPLIT, $0-12
MOVQ p+0(FP), AX
MOVW v+8(FP), BX
LOCK
XADDW BX, (AX)
MOVW AX, ret+8(FP)
RET
// func lockarith(p *uint64)
TEXT ·lockarith(SB), NOSPLIT, $0-8
MOVQ p+0(FP), AX
LOCK
ORQ CX, (AX)
LOCK
ANDL CX, (AX)
LOCK
INCQ (AX)
LOCK
DECQ (AX)
LOCK
ORB BX, (AX)
RET
// func repstring(dst, src *byte, n int)
// The memmove shapes: forward copy by quadwords, backward tails.
TEXT ·repstring(SB), NOSPLIT, $0-24
MOVQ dst+0(FP), DI
MOVQ src+8(FP), SI
REP
MOVSQ
REP
MOVSB
REPN
MOVSB
REP
STOSQ
REP
STOSB
RET
// func pfxlabel()
// Labels pinned on prefix statements' own bytes: pfx: sits on the LOCK,
// mid: on the REPN.
TEXT ·pfxlabel(SB), NOSPLIT, $0-0
pfx:
LOCK
XCHGL BX, (AX)
JMP done
mid:
REPN
MOVSB
done:
REP
STOSB
RET
+62
View File
@@ -0,0 +1,62 @@
// Literal data emission: BYTE, WORD, LONG and QUAD write the immediate
// into the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked; END is
// accepted and ignored, contributing no bytes and ending nothing. The
// shapes mirror the runtime's hand-laid markers
// (crypto/internal/boring/sig/sig_amd64.s) and its syscall stubs
// (runtime/sys_linux_amd64.s).
#include "textflag.h"
// func marker()
// A boring/crypto-style marker: a hand-laid forward branch whose skip
// distance is patched at runtime. One BYTE per statement, as the
// runtime's own file spells it: the semicolon-separated one-liner the
// sys_linux_amd64.s stub uses does not survive gasm fmt, which drops the
// statement separators.
TEXT ·marker(SB), NOSPLIT, $0-0
BYTE $0xEB
BYTE $0x1D
BYTE $0xF4
BYTE $0x48
BYTE $0xF4
BYTE $0x4B
BYTE $0xC3
RET
// func stub()
// The sys_linux_amd64.s stub bytes: the sign-extended
// "48 c7 c0 0f 00 00 00" form of MOVQ $rt_sigreturn, AX.
TEXT ·stub(SB), NOSPLIT, $0-0
BYTE $0x48
BYTE $0xc7
BYTE $0xc0
BYTE $0x0f
BYTE $0x00
BYTE $0x00
BYTE $0x00
RET
// func words()
// The wider literals, and an END that ends nothing: the WORD after it
// still lands in this function.
TEXT ·words(SB), NOSPLIT, $0-0
WORD $0x1234
WORD $-1
LONG $0x11223344
LONG $-1
QUAD $0x1122334455667788
QUAD $-2
END
WORD $0xBEEF
RET
// func trunc()
// Truncation, not a range check: each literal keeps its low bytes, exactly
// as go tool asm emits them.
TEXT ·trunc(SB), NOSPLIT, $0-0
BYTE $0x1FF
WORD $0x12345
LONG $0x123456789
QUAD $-2
RET
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Wide-immediate arithmetic: every classification band of the ADD/SUB
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
// their materialised fallback. Byte-for-byte against go tool asm.
#include "textflag.h"
// imm12 covers the plain and shifted-by-12 imm12 forms.
TEXT ·imm12(SB), NOSPLIT, $0-0
ADD $1, R2, R3
ADD $0x000aaa, R2, R3
ADD $0xaaa000, R2
SUB $0x000aaa, R2, R3
SUB $0xaaa000, R2
ADDW $40960, R0
CMP $40960, R0
CMPW $40960, R0
RET
// split pins the ADDCON2 band: two imm12 instructions, low half first.
TEXT ·split(SB), NOSPLIT, $0-0
ADD $0xaaaaaa, R2, R3
SUB $0xaaaaaa, R2
ADD $0x186a0, R2, R5
SUB $0x186a0, R2, R3
ADDW $0x60060, R2
RET
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
// MOVN for the complement form, the bitmask ORR otherwise.
TEXT ·regtmp(SB), NOSPLIT, $0-0
ADD $0x1ffe00, R2, R3
ADD $0x3fffffffc000, R5
ADD $-2048, R2, R3
ADD $-100000, R2, R3
CMP $0x1000000, R2
CMP $0x100000000, R0
SUB $-0x100000000, R0, R1
RET
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
// compare forms that never split.
TEXT ·movseq(SB), NOSPLIT, $0-0
ADD $0x12345678, R2, R3
SUB $0xe7791f700, R3, R1
CMP $0xaaaaaa, R2
CMP $0xffffffffffa0, R3
CMPW $27745, R2
CMPW $0x60060, R2
ADDS $0xaaaaaa, R2, R3
CMN $0x1000000, R2
ADDW $0x12345678, R2, R3
RET
// logical covers the bitmask immediates of the logical family and the
// materialised fallback for the values a bitmask cannot carry.
TEXT ·logical(SB), NOSPLIT, $0-0
AND $0x3ff00000, R2, R3
BIC $0x22220000, R3, R4
ORR $0x3ff00000, R2
EOR $0x3ff00000, R2, R3
ANDS $0x3ff00000, R2
ORNW $0x3ff00000, R2
EONW $0x3ff00000, R2
BICSW $0x6006000060060, R5
TST $0x4900000049, R0
RET
+11
View File
@@ -43,6 +43,13 @@ const (
At // @ At // @
Hash // # Hash // #
Pipe // | Pipe // |
// Semicolon separates statements on one line (a Plan 9 statement
// terminator); Ampersand and Tilde are the expression operators & and ~
// of constant expressions. All three appear mostly inside macro bodies.
Semicolon // ;
Ampersand // &
Tilde // ~
) )
var kindNames = map[Kind]string{ var kindNames = map[Kind]string{
@@ -71,6 +78,10 @@ var kindNames = map[Kind]string{
At: "@", At: "@",
Hash: "#", Hash: "#",
Pipe: "|", Pipe: "|",
Semicolon: ";",
Ampersand: "&",
Tilde: "~",
} }
// String returns a human-readable name for the kind. // String returns a human-readable name for the kind.
+2
View File
@@ -32,6 +32,8 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/crypto_arm64.s", "../testdata/verify/crypto_arm64.s",
"../testdata/verify/integer_arm64.s", "../testdata/verify/integer_arm64.s",
"../testdata/verify/simd_arm64.s", "../testdata/verify/simd_arm64.s",
"../testdata/verify/widenimm_arm64.s",
"../testdata/verify/carryshift_arm64.s",
"../testdata/verify/system_arm64.s", "../testdata/verify/system_arm64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {
+4
View File
@@ -122,6 +122,10 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/crypto_amd64.s", "../testdata/verify/crypto_amd64.s",
"../testdata/verify/sse_amd64.s", "../testdata/verify/sse_amd64.s",
"../testdata/verify/avx_amd64.s", "../testdata/verify/avx_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/doubleshift_amd64.s", "../testdata/verify/doubleshift_amd64.s",
"../testdata/verify/ssestatic_amd64.s", "../testdata/verify/ssestatic_amd64.s",
} { } {
+2
View File
@@ -29,6 +29,8 @@ func TestGroundTruthLOONG64(t *testing.T) {
"../testdata/verify/branchu_loong64.s", "../testdata/verify/branchu_loong64.s",
"../testdata/verify/atomics_loong64.s", "../testdata/verify/atomics_loong64.s",
"../testdata/verify/vector_loong64.s", "../testdata/verify/vector_loong64.s",
"../testdata/verify/pcalign_loong64.s",
"../testdata/verify/l64forms_loong64.s",
"trampoline_loong64.s", "trampoline_loong64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {
+2
View File
@@ -33,6 +33,8 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/atomics_riscv64.s", "../testdata/verify/atomics_riscv64.s",
"../testdata/verify/vector_riscv64.s", "../testdata/verify/vector_riscv64.s",
"../testdata/verify/bitmanip_riscv64.s", "../testdata/verify/bitmanip_riscv64.s",
"../testdata/verify/pcalign_riscv64.s",
"../testdata/verify/branch_far_riscv64.s",
"trampoline_riscv64.s", "trampoline_riscv64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {