Compare commits

...
6 Commits
Author SHA1 Message Date
petrbalvin 97dfaa7526 docs: changelog for macro expansion and the corrected corpus audit
Test / test (push) Successful in 2m14s
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 66aa4dbc8b test(verify): register the campaign kernels in the ground-truth suites
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin dce5d31462 feat(amd64): LOCK and REP prefixes, literal data pseudo-ops and ADJSP
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9dc3987e02 feat(riscv64,loong64): PCALIGN, branch relaxation and operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9b238a525a feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin ad82aac663 feat(parser): macro expansion, conditionals and include splicing with -I
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
40 changed files with 6425 additions and 303 deletions
+15
View File
@@ -9,6 +9,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at
the point of use, `#undef` and the `#ifdef`/`#ifndef`/`#else`/
`#endif` family select branches, `#include` splices headers resolved
through the source directory and the new repeatable `-I` flag, `;`
separates statements, and constant expressions left in operands
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the
@@ -129,6 +139,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Fixed
- **The corpus audit attempts fewer files that no build would compile.**
Files named for Go ports gasm does not target (arm, 386, s390x, ...)
are reported as other-port and never attempted, the headline rate is
computed over attemptable files, and the audit searches the
toolchain's shipped headers (funcdata.h and friends) automatically.
- **riscv64 JALR silently jumped to the wrong register.** The trampoline
form `JALR X0, 0(X5)` read the memory operand's base as the destination,
encoding a jump to X0 with no diagnostic; the destination is the first
+569 -108
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"math/bits"
"strconv"
"strings"
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW":
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
"ADDS", "ADDSW", "SUBS", "SUBSW":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
if v >= 0 && v <= 0xFFF {
return 4
// Size exactly as the encoder will emit: a single imm12 word, the
// two-word ADDCON2 split, or a materialisation into REGTMP plus
// the register form. Anything else would desynchronise the label
// offsets of pass 1 from the bytes pass 2 lays down.
if v, ok := arm64ImmOperandValue(ops[0]); ok {
rn, rd := 0, 0
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
rd = n
}
if len(ops) == 3 {
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
rn = n
}
}
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
return 4 * len(ws)
}
}
if v >= -2048 && v < 0 {
return 4
}
// Larger immediates need MOV materialisation + op.
return 8
return 4
}
}
return 4
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Bitfield(mnem, enc.op, ops)
}
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
}
// EXTR.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
return encodeARM64Extr(mnem, enc.op, ops)
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below.
if spec, ok := a64SimdVTable[mnem]; ok {
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops)
}
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
}
// SIMD table lookup.
if mnem == "VTBL" {
return encodeARM64VTBL(ops)
if mnem == "VTBL" || mnem == "VTBX" {
return encodeARM64VTBL(mnem, ops)
}
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
}
op := ops[0]
// Branch to the program counter itself: JMP (PC) spins forever. The
// toolchain encodes it as an unconditional branch with a zero offset.
// Branch to the program counter: JMP (PC) spins forever, and a spelled
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
// units. The toolchain encodes both as a plain branch of that offset.
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
if link {
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
rel := op.Addr.Offset
if rel < -(1<<25) || rel >= (1<<25) {
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
}
return a64wordLE(a64Branch(0, 0)), nil
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, int32(rel))), nil
}
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
// the ADD/SUB-with-flags family an add/sub immediate.
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
mnem == "MVN" || mnem == "MVNW" ||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
var logical bool
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
"TST", "TSTW":
logical = true
}
if logical {
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
inverted := false
switch mnem {
case "BIC", "BICW", "BICS", "BICSW":
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
inverted = true
}
if inverted {
@@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
n, immr, imms, ok := a64LogicalImm(v, width)
if !ok {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
// Beyond the bitmask immediates the toolchain materialises
// the constant into REGTMP (R27) and uses the register form
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
// value, so the materialisation uses v before any inversion.
written := v
if inverted {
written = ^v
}
width := mnem
if strings.HasSuffix(mnem, "W") {
width = "MOVW"
} else {
width = "MOVD"
}
mw, merr := encodeARM64LoadImm(27, written, width)
var rn, rd int
switch len(ops) {
case 3:
rn = arm64RegNum(operandRegName(ops[1]))
rd = arm64RegNum(operandRegName(ops[2]))
default:
rd = arm64RegNum(operandRegName(ops[1]))
rn = rd
}
if isCmp {
rd = 31
}
if merr != nil || rn < 0 || rd < 0 {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
}
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
}
opc := (baseOp >> 29) & 7
sf := (baseOp >> 31) & 1
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
}
if !isExtend {
if amount < 0 || amount > 63 {
// ROR rides the shifted-register field only for the logical
// group; the toolchain reports "unsupported shift operator" for
// the arithmetic forms, whose shift=11 encoding is unallocated.
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
}
// The imm6 field is 5 bits and truncates at the 32-bit width.
limit := 63
if strings.HasSuffix(mnem, "W") {
limit = 31
}
if amount < 0 || amount > limit {
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
}
// SP-based ADD/SUB have no shifted-register encoding: the
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
switch len(ops) {
case 3:
// The carry family carries an immediate spelling in three operands
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
// operand, the same shape the two-operand form takes.
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
if v := arm64Imm64(ops[0]); v != 0 {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// OP Rm, Rn, Rd
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
// shifted-register group, the only forms whose register operand accepts the
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
func arm64LogicalShifted(mnem string) bool {
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
"TST", "TSTW", "MVN", "MVNW":
return true
}
return false
}
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
// data-processing instructions the toolchain accepts an immediate $0
// operand spelling for.
func arm64CarryOp(mnem string) bool {
switch mnem {
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
return true
}
return false
}
// arm64RegMod reports whether a register operand carries the shifted-register
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
// option (R0.UXTW, R3.SXTW<<2).
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
// arm64RegModifier resolves a modified register operand: the register number,
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
// option (UXTB=0..SXTX=7) with its shift amount.
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
// before the operator match.
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
name := operandRegName(op)
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
if before, after, ok0 := strings.Cut(name, "."); ok0 {
switch strings.ToUpper(strings.TrimSpace(after)) {
case "UXTB":
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
if rm < 0 {
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok || amount < 0 || amount > 4 {
return 0, 0, 0, false, 0, false
}
return rm, 0, extendOpt, true, amount, true
}
shiftKind = 0 // LSL
shift := strings.TrimSpace(op.Addr.Shift)
switch {
case strings.HasPrefix(shift, "<<"):
shiftKind = 0
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
default:
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok {
return 0, 0, 0, false, 0, false
}
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
// MUL mnemonic, not to these.
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// The widening three-operand forms (SMULL, UMNEGL, …) read the
// accumulate register as ZR, already preset in the table's base word.
if len(ops) == 3 {
switch mnem {
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
default:
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
}
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
// following the toolchain's immediate classification (asm7.go conclass and
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
// otherwise a constant materialisation into REGTMP (R27) followed by the
// register form.
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := arm64Imm64(ops[0])
v, ok := arm64ImmOperandValue(ops[0])
if !ok {
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
}
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd
if len(ops) == 3 {
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
sf := uint32(1) // 64-bit
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
sf = 0 // 32-bit
}
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
// CMP/CMN discard the destination.
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
rd = 31 // ZR
}
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
return a64WordsLE(ws...), nil
}
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
// their W forms. rn and rd are resolved register numbers (a comparison
// discards rd, so the caller passes 31).
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
w := strings.HasSuffix(mnem, "W")
sf := uint32(1) // 64-bit
d := v
if w {
sf = 0 // 32-bit
// The W forms classify the 32-bit value (asm7.go con32class).
d = int64(uint32(v))
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
op := uint32(0) // ADD
S := uint32(0)
if isSub {
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if isS {
S = 1
}
single := func(sh, imm12 uint32) []uint32 {
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
}
if v >= 0 && v <= 0xFFF {
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
// imm12: plain, then the one-shifted-by-12 form.
if d >= 0 && d <= 0xFFF {
return single(0, uint32(d)), nil
}
if v >= -2048 && v < 0 {
// Encode as the opposite operation with positive immediate.
opp := op ^ 1
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
return single(1, uint32(d>>12)), nil
}
// Try with shift by 12.
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
// split into two imm12 instructions, low half first (asm7.go case 48).
// The encoding is complete in itself: no REGTMP, no register form. The S
// forms must not break addition/subtraction, so the toolchain
// reclassifies them and falls through to the materialisation below.
dm := ^d
if w {
dm = ^d & 0xFFFFFFFF
}
// The imm12 field cannot carry the value; rejecting (rather than
// truncating) matches the toolchain, which reports the same shape.
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
return []uint32{
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
}, nil
}
// Constant into REGTMP (R27), then the register form. The first word
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
// the complement form, the bitmask ORR otherwise, and the full
// omovlconst sequence when no single word carries the value.
var seq []uint32
switch s := arm64Movcon(d); {
case s >= 0:
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
case arm64Movcon(dm) >= 0:
s := arm64Movcon(dm)
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
case isBitcon:
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
default:
seq = arm64MovLConst(d, sf)
}
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
for i := range seq[:len(seq)-1] {
seq[i] |= 27 // REGTMP
}
return seq, nil
}
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
// returned with the destination field clear so the caller can OR its own
// register in. movcon and movcon-of-complement must fail for d before this
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
func arm64MovLConst(d int64, sf uint32) []uint32 {
if sf == 0 {
// omovlconst AMOVW: both 16-bit halves, low first.
return []uint32{
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
}
}
dn := ^d
var immh [4]uint64
zero, neg := 0, 0
for i := range immh {
immh[i] = uint64(d>>(i*16)) & 0xFFFF
switch immh[i] {
case 0:
zero++
case 0xFFFF:
neg++
}
}
mw := func(opc uint32, val int64, chunk int) uint32 {
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
}
var os []uint32
switch {
case zero == 2:
// one MOVZ and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg == 2:
// one MOVN and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
// a single MOVK (constants from strength-reduced division).
if zero == 0 && neg == 0 {
for i := range 4 {
mask := uint64(0xFFFF) << (i * 16)
for period := 2; period <= 32; period *= 2 {
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
os = append(os, mw(3, d, i))
return os
}
}
}
}
switch {
case zero >= 1:
// one MOVZ and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg >= 1:
// one MOVN and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// one MOVZ and three MOVKs
os = append(os, mw(2, d, 0))
for i := 1; i < 4; i++ {
os = append(os, mw(3, d, i))
}
}
}
return os
}
// ---- MOV pseudo-instruction ----
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
}
}
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
var ws []uint32
first := true
for i := range 4 {
chunk := (d >> uint(i*16)) & 0xFFFF
if chunk == 0 {
continue
}
if first {
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
first = false
} else {
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
}
}
if len(ws) == 0 {
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
// for the first special 16-bit chunk, then MOVK per remaining one, with
// the bitmask-plus-fixup shortcut for strength-reduced constants).
ws := arm64MovLConst(d, sf)
for i := range ws {
ws[i] |= uint32(rd)
}
return a64WordsLE(ws...), nil
}
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
// with imms = lsb+width-1.
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
}
lsb := arm64Imm64(ops[0])
width := arm64Imm64(ops[2])
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
bits := int64(32) << (baseOp >> 31 & 1)
if lsb < 0 || lsb >= bits {
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
}
if width < 1 || width > bits || lsb+width > bits {
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
}
var immr, imms int64
switch mnem {
case "BFXIL", "BFXILW":
immr, imms = lsb, lsb+width-1
default: // BFI, SBFIZ, UBFIZ
immr, imms = (-lsb)%bits, width-1
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
// third field carries Rm or a 5-bit immediate in the same bits, at the
// toolchain's choice of register or immediate operand.
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(w), nil
}
switch mnem {
case "BRK", "SVC":
base := uint32(0xd4200000)
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
return a64wordLE(base | uint32(v)<<5), nil
case "DMB", "DSB", "ISB":
case "DMB", "DSB", "ISB", "CLREX":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if v < 0 || v > 15 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 127 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
}
return a64wordLE(0xd503201f | uint32(v)<<5), nil
case "BTI":
op := operandRegName(ops[0])
base, ok := map[string]uint32{"C": 0xd503245f}[op]
if !ok {
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
}
return a64wordLE(base), nil
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 0xFFFF {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
"DCPS3": 0xd4a00003}[mnem]
return a64wordLE(base | uint32(v)<<5), nil
case "DC":
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
return a64wordLE(base | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(base | uint32(rs)&31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
if !ok {
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
func specBit(i int) uint16 { return 1 << uint(i) }
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
return true
}
if !strings.HasPrefix(mnem, "VFCM") {
return false
}
s := strings.Join(strings.Fields(op.Raw), "")
s = strings.TrimPrefix(s, "$")
s = strings.Trim(s, "()")
return s == "0" || s == "0.0"
}
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
// zero immediate takes its compare-against-zero form instead, and the
// polynomial multiplies read the arrangement from their source operands
// alone, the result spelling (H8, Q1) riding no encoding bits.
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
// their compare-against-zero form instead, and the polynomial multiplies read
// the arrangement from their source operands alone, the result spelling
// (H8, Q1) riding no encoding bits.
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
if mnem == "VPMULL" || mnem == "VPMULL2" {
if len(ops) != 3 {
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
rd, _ := arm64VecOf(ops[2])
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
}
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
if arm64Imm64(ops[0]) != 0 {
if len(ops) == 3 && isImmOperand(ops[0]) {
base, ok := a64SimdVZero[mnem]
if !ok {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
if !arm64SimdZeroImm(mnem, ops[0]) {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
vn, ok1 := arm64VecOf(ops[1])
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
return nil, err
}
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
}
// VUADDLV spells its arrangement on the source alone; the rest take it
// on both.
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
// index register rides bits 19:16, the first table register bits 9:5, the
// destination bits 4:0 and the table length (registers minus one) bits
// 14:13. The table registers must be consecutive.
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) < 3 {
return nil, fmt.Errorf("VTBL expects index, table list and destination")
}
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
default:
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
}
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
base := uint32(0x0e000000)
if mnem == "VTBX" {
base |= 1 << 12
}
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
}
var immval int64
switch mnem {
case "VSHL":
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
if sh < 0 || sh >= esize {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
immval = esize + sh
default: // VUSHR, VSRI
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
if sh < 1 || sh > esize {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
}
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
sym *ast.Symbol // the parsed frame-relative reference (when mem)
}
aliases := map[string]alias{}
raws := map[string]string{}
for _, d := range f.Decls {
pre, ok := d.(*ast.Preproc)
if !ok {
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
continue
}
raws[name] = body
}
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
// transitively until nothing changes, bounded against cycles.
for range 8 {
changed := false
for name, body := range raws {
if next, ok := raws[body]; ok && next != body {
raws[name] = next
changed = true
}
}
if !changed {
break
}
}
for name, body := range raws {
isReg := func(s string) bool {
if arm64RegNum(s) >= 0 {
return true
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
return
}
// replace rewrites whole-word occurrences of the alias names in s.
replace := func(s string) string {
if s == "" {
return s
}
out := strings.Fields(s)
for i, w := range out {
if a, ok := aliases[w]; ok {
out[i] = a.raw
}
}
if len(out) == 0 {
return s
}
return strings.Join(out, " ")
}
// replaceToken rewrites an operand whose whole text is one alias use
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
// ending at a non-identifier character.
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
return s, false
}
// replaceScan rewrites alias uses inside a composite operand (a
// parenthesised memory operand or a bracketed register list): every
// identifier run of word and dot characters is matched against the alias
// names, everything else copies verbatim. The whitespace-split replace
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
// punctuation attached.
replaceScan := func(s string) string {
var b strings.Builder
for i := 0; i < len(s); {
if isAliasWordByte(s[i]) || s[i] == '.' {
j := i
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
j++
}
if nn, ok := replaceToken(s[i:j]); ok {
b.WriteString(nn)
} else {
b.WriteString(s[i:j])
}
i = j
continue
}
b.WriteByte(s[i])
i++
}
return b.String()
}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
op.Raw = a.raw
continue
}
op.Addr.Sym.Name = nn
op.Addr.Sym.Raw = nn
op.Raw = nn
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
// The span shape depends on what trailed the
// name: an element or arrangement selector
// (POLY.D[0], POLY.B16) rides in Shift and folds
// back onto the rewritten token; a shift
// operator stays in Shift while the span carries
// the bare register; a split list keeps its
// closing bracket, so the rewrite goes through
// the scan.
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
switch {
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
// Element or arrangement selectors and the
// closing bracket of a split list belong to
// the token text.
op.Raw = nn + sfx
op.Addr.Shift = ""
case op.Addr.Shift != "":
op.Raw = nn
default:
op.Raw = replaceScan(op.Raw)
}
continue
}
}
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
// [V0.B16, V1.B16] with aliased members.
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
op.Raw = replace(op.Raw)
op.Raw = replaceScan(op.Raw)
}
}
}
+292 -55
View File
@@ -308,6 +308,8 @@ const (
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
"AL": a64CondAL,
"NV": a64CondNV,
}
// ---- instruction format tags ----
@@ -335,46 +339,47 @@ var arm64CondMap = map[string]uint32{
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -487,6 +492,16 @@ func init() {
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// The widening multiplies: a 64-bit result riding the same layout, the
// three-operand forms reading the accumulate register as ZR.
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
// ---- move wide ----
// MOVZ/MOVN/MOVK
@@ -536,6 +551,15 @@ func init() {
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
@@ -716,6 +740,13 @@ func init() {
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
// the source width.
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
"REV16W": 0x5ac00400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
@@ -734,7 +765,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -785,6 +816,73 @@ func init() {
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// The remaining width and ordering spellings of the same shapes, and the
// CAS compare-and-swap family, word-verified against go tool asm.
lseMore := map[string]uint32{
"LDADDAB": 0x38a00000,
"LDADDAH": 0x78a00000,
"LDADDALB": 0x38e00000,
"LDADDALH": 0x78e00000,
"LDADDLB": 0x38600000,
"LDADDLD": 0xf8600000,
"LDADDLH": 0x78600000,
"LDADDLW": 0xb8600000,
"LDCLRAB": 0x38a01000,
"LDCLRAH": 0x78a01000,
"LDCLRALH": 0x78e01000,
"LDCLRB": 0x38201000,
"LDCLRD": 0xf8201000,
"LDCLRH": 0x78201000,
"LDCLRLB": 0x38601000,
"LDCLRLD": 0xf8601000,
"LDCLRLH": 0x78601000,
"LDCLRLW": 0xb8601000,
"LDCLRW": 0xb8201000,
"LDEORAB": 0x38a02000,
"LDEORAD": 0xf8a02000,
"LDEORAH": 0x78a02000,
"LDEORALB": 0x38e02000,
"LDEORALH": 0x78e02000,
"LDEORAW": 0xb8a02000,
"LDEORB": 0x38202000,
"LDEORD": 0xf8202000,
"LDEORH": 0x78202000,
"LDEORLB": 0x38602000,
"LDEORLD": 0xf8602000,
"LDEORLH": 0x78602000,
"LDEORLW": 0xb8602000,
"LDEORW": 0xb8202000,
"LDORAB": 0x38a03000,
"LDORAD": 0xf8a03000,
"LDORAH": 0x78a03000,
"LDORALH": 0x78e03000,
"LDORAW": 0xb8a03000,
"LDORB": 0x38203000,
"LDORD": 0xf8203000,
"LDORH": 0x78203000,
"LDORLB": 0x38603000,
"LDORLD": 0xf8603000,
"LDORLH": 0x78603000,
"LDORLW": 0xb8603000,
"LDORW": 0xb8203000,
"SWPAB": 0x38a08000,
"SWPAD": 0xf8a08000,
"SWPAH": 0x78a08000,
"SWPALH": 0x78e08000,
"SWPAW": 0xb8a08000,
"SWPB": 0x38208000,
"SWPH": 0x78208000,
"SWPLB": 0x38608000,
"SWPLD": 0xf8608000,
"SWPLH": 0x78608000,
"SWPLW": 0xb8608000,
"CASAD": 0xc8e07c00,
"CASALB": 0x08e0fc00,
"CASLW": 0x88a0fc00,
}
for m, op := range lseMore {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
@@ -794,7 +892,12 @@ func init() {
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
"NGC": 0xda000000, "NGCW": 0x5a000000,
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
"NEGSW": 0x6b000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
@@ -835,6 +938,12 @@ func init() {
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
var a64SimdQOnly = map[string]bool{
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
"VFABS": true, "VFNEG": true, "VFSQRT": true,
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
"VFMAXV": true, "VFMAXNMV": true,
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
@@ -918,19 +1044,96 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
// combination for them, so they stay out and the encoder rejects the shape.
var a64SimdVZero = map[string]uint32{
"VCMEQ": 0x0e209800,
"VCMGT": 0x0e208800,
"VCMGE": 0x2e208800,
"VCMLT": 0x0e20a800,
"VCMLE": 0x2e209800,
// FP compares against (0.0): the register forms above carry the U and op
// bits; the zero forms reshape them.
"VFCMEQ": 0x0ea0d800,
"VFCMGE": 0x2ea0c800,
"VFCMGT": 0x0ea0c800,
"VFCMLE": 0x2ea0d800,
"VFCMLT": 0x0ea0e800,
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
@@ -939,8 +1142,41 @@ var a64SimdVTable = map[string]a64SimdVSpec{
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
// MSR Rn, <sysreg>; the source register rides bits 4:0.
var a64MSRRegOps = map[string]uint32{
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
"ELR_EL1": 0xd5184020,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
+164 -13
View File
@@ -4,6 +4,7 @@
package asm
import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
}
}
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
// rejected instead of wrapping through int32.
func TestArm64AddSubImmRange(t *testing.T) {
for _, body := range []string{
"\tADD $0x100000000, R0, R1\n",
"\tSUB $-0x100000000, R0, R1\n",
"\tCMP $0x100000000, R0\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
// Comparisons never split, and the W forms classify the 32-bit value. Every
// word is go tool asm's own for the same source.
func TestArm64AddSubImmWide(t *testing.T) {
got := arm64Words(t, strings.Join([]string{
"\tADD $0xaaaaaa, R2, R3",
"\tSUB $0xaaaaaa, R2",
"\tADD $0x186a0, R2, R5",
"\tADD $0x1ffe00, R2, R3",
"\tADD $0x3fffffffc000, R5",
"\tADD $-100000, R2, R3",
"\tADD $-2048, R2, R3",
"\tCMP $0xaaaaaa, R2",
"\tCMP $0xffffffffffa0, R3",
"\tCMPW $27745, R2",
"\tCMPW $0x60060, R2",
"\tADDS $0xaaaaaa, R2, R3",
"\tADD $0x12345678, R2, R3",
"\tADDW $0x60060, R2",
"\tSUB $0xe7791f700, R3, R1",
"\tADDW $0x12345678, R2, R3",
"\tCMN $0x1000000, R2",
}, "\n")+"\n")
want := []uint32{
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
0x11018042, 0x11418042, // ADDW $0x60060: W split
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
// three operands, the ROR shift on the logical group (and its rejection for
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
// alias with an element selector. Words are go tool asm's own.
func TestArm64CarryShiftAlias(t *testing.T) {
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
want := []uint32{
0x9a1f0294, // ADC ZR, R20, R20
0x9a1f0284, // ADC ZR, R20, R4
0xfa1f008c, // SBCS ZR, R4, R12
0xfa0f008c, // SBCS R15, R4, R12
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
0x8ac18443, // AND R1 ROR 33, R2, R3
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
0xda0203e7, // SBC ZR, R7, R2
0x9b0eff77, // MSUB ZR, R27, R14, R23
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
}
}
// ROR on an arithmetic form is unallocated: the toolchain reports an
// unsupported shift operator.
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Error("ADD R1@>33: expected an error, got none")
}
}
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
// operand with an element selector and inside a split register list: the
// aliases resolve textually where the parser carries the selector apart from
// the name. Words are go tool asm's own.
func TestArm64VecAliasElement(t *testing.T) {
src := `#include "textflag.h"
#define POLY V15
#define ACC0 V8
#define ACC1 V9
TEXT ·f(SB), NOSPLIT, $0-0
VMOV R1, POLY.D[0]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [ACC0.B16]
VLD1.P (R0), [ACC0.B16, ACC1.B16]
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
want := []uint32{
0x4e081c2f, // INS V15.D[0], R1
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
0x4c407008, // VLD1 (R0), [V8.B16]
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
// once the value leaves every imm12 form: a constant sequence into REGTMP
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
// tool asm's own.
func TestArm64AddSubImmBeyond32(t *testing.T) {
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
want := []uint32{
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0x8b1b0001, // ADD R27, R0, R1
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
0xcb1b0001, // SUB R27, R0, R1
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0xeb1b001f, // CMP R27, R0
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
+55
View File
@@ -67,6 +67,9 @@ type spadjStep struct {
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
if err := checkAdjspBalance(t); err != nil {
return nil, nil, nil, nil, nil, err
}
fi := computeFrame(t)
chain := jumpChain(t)
resolve := func(name string) string {
@@ -203,6 +206,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
)
}
// frameBase is the SP delta the prologue leaves: 8 for the saved base
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
// statements' straight-line sum, so a mid-body step's value is the
// frame base plus what the body has opened so far.
frameBase, bodyDelta := 0, 0
if fi.useFP {
frameBase = 8 + fi.size
}
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
@@ -230,6 +241,16 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
ps[k].kind = RelCall
}
}
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
// The statement shifted SP mid-body: record the new running
// delta as the value in effect from just past the instruction.
v := s.Operands[0].Imm.Val
if s.Operands[0].Imm.Neg {
v = -v
}
bodyDelta += int(v)
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
}
patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...)
@@ -412,6 +433,40 @@ func hasCall(t *ast.Text) bool {
return false
}
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
// shifts SP away from the entry state and every RET must see the shifts
// closed. The assembler's own prologue and epilogue contribute matching
// deltas on both sides, so the statements' straight-line sum must be zero
// at each RET; branches do not reset the walk, which runs over the program
// list in source order. go tool asm reports an offender as "unbalanced
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
// $16/$-16 pair, per-RET rather than per-function).
func checkAdjspBalance(t *ast.Text) error {
delta := 0
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "ADJSP":
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
continue // reported during emission
}
v := in.Operands[0].Imm.Val
if in.Operands[0].Imm.Neg {
v = -v
}
delta += int(v)
case "RET":
if delta != 0 {
return fmt.Errorf("unbalanced PUSH/POP")
}
}
}
return nil
}
// guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form.
+129
View File
@@ -439,3 +439,132 @@ func TestSubSPEncodings(t *testing.T) {
}
}
}
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
// asserts the three behaviours the toolchain shows: each prefix statement is
// a standalone byte with a PC of its own (so a label placed on the LOCK
// points at the F0), the data pseudo-ops write their literal bytes inline,
// and END terminates nothing (the statements after it still belong to the
// function and carry no trace of it).
func TestAssemblePseudoStatements(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pseudo(SB), NOSPLIT, $0-0
pfx:
LOCK
CMPXCHGQ AX, (BX)
REP
MOVSQ
BYTE $0x0f
BYTE $0x1f
WORD $0x1234
END
BYTE $0x02
RET
`)
code, labels, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
want := []byte{
0xf0,
0x48, 0x0f, 0xb1, 0x03,
0xf3, 0x48, 0xa5,
0x0f, 0x1f, 0x34, 0x12,
0x02, 0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// The label sits on the LOCK byte, exactly where the toolchain's PC
// listing puts it.
if off := labels["pfx"]; off != 0 {
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
}
// The trailing BYTE lands where the layout says: after the 8 bytes of
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
// none.
if code[12] != 0x02 {
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
}
}
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
// ADJSP: the straight-line sum of the adjustments must be zero at each
// RET, branches in between counting for nothing (verified against go tool
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
// $16/$-16 pair with a JMP in between assembles).
func TestAssembleAdjspBalance(t *testing.T) {
// Balanced pair with a branch in between, bytes pinned from go tool asm.
fn := firstText(t, `
#include "textflag.h"
TEXT ·adjsp(SB), NOSPLIT, $0-0
ADJSP $16
JMP body
body:
ADJSP $-16
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
if hexBytes(code) != hexBytes(want) {
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// Unbalanced at the RET: the toolchain diagnoses, so must we.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·unbalanced(SB), NOSPLIT, $0-0
ADJSP $16
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// The check runs per RET: a closed pair before the first RET does not
// excuse an open adjustment before the second.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·tworet(SB), NOSPLIT, $0-0
ADJSP $8
ADJSP $-8
RET
mid:
ADJSP $8
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// A framed function: the assembler's own prologue and epilogue
// contribute matching deltas, so the pair in the body still balances,
// and the bytes match go tool asm end to end.
fn = firstText(t, `
#include "textflag.h"
TEXT ·framed(SB), $16-8
ADJSP $8
ADJSP $-8
RET
`)
code, _, err = Assemble(fn)
if err != nil {
t.Fatalf("Assemble framed: %v", err)
}
want = []byte{
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
0xC3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
+4 -1
View File
@@ -19,7 +19,10 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the
// SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
return true
}
if _, ok := noOperandTable[upper]; ok {
+77
View File
@@ -92,6 +92,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
// the SUBQ and ADDQ forms.
case "BYTE", "WORD", "LONG", "QUAD":
return e.encodeData(upper, ops)
case "END":
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
@@ -241,6 +251,73 @@ var prefetchVariant = map[string]int{
"PREFETCHT2": 3,
}
// dataWidth is the literal byte count of each data-emission pseudo-op.
var dataWidth = map[string]int{
"BYTE": 1,
"WORD": 2,
"LONG": 4,
"QUAD": 8,
}
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
// little-endian, with no opcode lookup. The value is truncated to the
// width rather than range-checked, exactly as go tool asm behaves (BYTE
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
// exactly one immediate is accepted: the toolchain rejects a list such as
// BYTE $1, $2, $3.
func (e *enc) encodeData(mnem string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s requires an integer immediate", mnem)
}
width := dataWidth[mnem]
out := make([]byte, width)
u := uint64(imm)
for i := range width {
out[i] = byte(u >> (8 * i))
}
e.out = append(e.out, out...)
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
// encoded body carries no trace of it, whatever operands follow the name
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
func (e *enc) encodeEnd(ops []Operand) error {
return nil
}
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
// picks (the same selection subSP and addSP make for the frame). go tool
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
// statement's effect on the SP balance is checked by the function-level
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
func (e *enc) encodeAdjsp(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("ADJSP requires an integer immediate")
}
switch v := int(imm); {
case v > 0:
e.out = append(e.out, subSP(v)...)
case v < 0:
e.out = append(e.out, addSP(-v)...)
default:
return fmt.Errorf("ADJSP $0 has no encoding")
}
return nil
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
+139
View File
@@ -925,3 +925,142 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
}
}
}
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
// prefix field merged into the following instruction, and it validates
// nothing about the pairing (LOCK before NOP assembles). The prefixed
// atomic and string shapes are the bytes the runtime's own kernels need.
func TestPrefixStatements(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"LOCK", "LOCK", nil, "f0"},
{"REP", "REP", nil, "f3"},
{"REPN", "REPN", nil, "f2"},
// LOCK; CMPXCHGQ AX, (BX)
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
// REP; MOVSQ
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
// REPN; MOVSB
{"REPN MOVSB", "MOVSB", nil, "a4"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The prefix statements take no operands, as the toolchain reports for
// LOCK AX.
if _, err := Encode("LOCK", AX); err == nil {
t.Error("LOCK AX assembled, want an error")
}
if _, err := Encode("REP", Imm(1)); err == nil {
t.Error("REP $1 assembled, want an error")
}
}
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked (go tool asm
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
func TestDataEmission(t *testing.T) {
cases := []struct {
name string
mnem string
imm Imm
want string
}{
{"BYTE", "BYTE", 0x0f, "0f"},
{"BYTE negative", "BYTE", -1, "ff"},
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
{"WORD", "WORD", 0x1234, "3412"},
{"WORD negative", "WORD", -1, "ffff"},
{"WORD truncated", "WORD", 0x12345, "4523"},
{"LONG", "LONG", 0x11223344, "44332211"},
{"LONG negative", "LONG", -1, "ffffffff"},
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
// register or a missing operand is no immediate at all.
if _, err := Encode("BYTE"); err == nil {
t.Error("BYTE with no operand assembled, want an error")
}
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
t.Error("BYTE $1, $2 assembled, want an error")
}
if _, err := Encode("WORD", AX); err == nil {
t.Error("WORD AX assembled, want an error")
}
}
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
// encodes to zero bytes and takes any operands without complaint (the
// toolchain accepts END $0 and END AX alike).
func TestEndIgnored(t *testing.T) {
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
code, err := Encode("END", ops...)
if err != nil {
t.Errorf("END: %v", err)
continue
}
if len(code) != 0 {
t.Errorf("END = %x, want no bytes", code)
}
}
}
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
// has no encoding (go tool asm refuses ADJSP $0 outright).
func TestAdjsp(t *testing.T) {
cases := []struct {
name string
imm Imm
want string
}{
{"imm8", 112, "4883ec70"},
{"imm8 negative", -112, "4883c470"},
{"imm32", 200, "4881ecc8000000"},
{"imm32 negative", -200, "4881c4c8000000"},
{"small", 8, "4883ec08"},
}
for _, c := range cases {
code, err := Encode("ADJSP", c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
}
}
if _, err := Encode("ADJSP", Imm(0)); err == nil {
t.Error("ADJSP $0 assembled, want an error")
}
if _, err := Encode("ADJSP"); err == nil {
t.Error("ADJSP with no operand assembled, want an error")
}
if _, err := Encode("ADJSP", AX); err == nil {
t.Error("ADJSP AX assembled, want an error")
}
}
+12
View File
@@ -65,6 +65,15 @@ var bitTestOp = map[string]int{
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
// fence names carry their opcode inside the 0F AE /digit group spelled out in
// full (E8/F0/F8), and PAUSE is F3 90.
//
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
// respectively), not as a prefix field merged into the next instruction: the
// statement that follows is encoded unaware of it, and nothing validates
// that the pairing is a legal one (LOCK before NOP assembles without
// complaint, each byte pinned against the toolchain). Because the bytes
// land in the stream before the following statement anyway, a LOCKed
// CMPXCHGQ encodes identically to a prefixed form.
var noOperandTable = map[string][]byte{
"CPUID": {0x0F, 0xA2},
"RDTSC": {0x0F, 0x31},
@@ -78,6 +87,9 @@ var noOperandTable = map[string][]byte{
"MFENCE": {0x0F, 0xAE, 0xF0},
"SFENCE": {0x0F, 0xAE, 0xF8},
"UNDEF": {0x0F, 0x0B},
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
}
// --- MOV --------------------------------------------------------------------
+392 -45
View File
@@ -46,30 +46,70 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := guardLen + len(prologue)
// The toolchain's parser counts N(PC) displacements over the source
// instructions at a uniform 4 bytes each, so a PC-relative branch
// resolves to the instruction N slots away in body order; the resolved
// target then participates in layout and loop-head padding like any
// branch target.
instrs := make([]*ast.Instr, 0, len(t.Body))
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += loong64InstrSize(s, fi)
if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" {
instrs = append(instrs, in)
}
}
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, which the first
// pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
parseIndex := make(map[*ast.Instr]int, len(instrs))
for i, in := range instrs {
parseIndex[in] = i
}
pcRelTarget := make(map[*ast.Instr]*ast.Instr)
for _, in := range instrs {
off, ok := loong64PCRelOffset(in)
if !ok {
continue
}
bodyLen = p - (guardLen + len(prologue))
tgt := parseIndex[in] + off
if tgt < 0 || tgt >= len(instrs) {
continue
}
pcRelTarget[in] = instrs[tgt]
}
// Pass 1: label offsets from the instruction sizes. PCALIGN contributes
// only its padding. On top of the explicit PCALIGNs, the toolchain pads
// every backward-branch target (loop head) to a 16-byte boundary, so the
// layout runs to a fixpoint over the alignment set.
loopAligns := map[string]bool{}
alignInstrs := map[*ast.Instr]bool{}
for {
offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
changed := false
for _, in := range instrs {
// A backward PC-relative target is the resolved instruction.
if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] {
alignInstrs[tgt] = true
changed = true
}
target, ok := loong64BranchTarget(in)
if !ok {
continue
}
tOff, ok := offsets[target]
if !ok || tOff >= pcs[in] || loopAligns[target] {
continue
}
loopAligns[target] = true
changed = true
}
if !changed {
break
}
}
// Final layout with the complete alignment set.
offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
bodyLen := bodyEnd - (guardLen + len(prologue))
pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget))
for in, tgt := range pcRelTarget {
pcRelPcs[in] = pcs[tgt]
}
var out []byte
if fi.needSplit {
@@ -84,7 +124,20 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
if !ok {
continue
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
// PCALIGN pads to the requested boundary with andi $0, $0, 0, the
// architecture's NOP, and encodes to nothing itself.
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
pad := loong64PCAlignPad(pc, in)
out = append(out, loong64PadBytes(pad)...)
pc += pad
continue
}
// Loop-head alignment padding precedes the instruction.
if pad := alignPad[in]; pad > 0 {
out = append(out, loong64PadBytes(pad)...)
pc += pad
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
@@ -115,6 +168,115 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
return out, offsets, relocs, lines, spadj, nil
}
// loong64PCRelOffset reports the N of a branch operand spelled N(PC): the
// displacement counted in source instructions from the branch itself.
func loong64PCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
branch := false
switch mnem {
case "JMP":
branch = len(instr.Operands) == 1
case "JAL", "CALL", "BL":
branch = len(instr.Operands) == 1 || len(instr.Operands) == 2
case "BFPT", "BFPF":
branch = len(instr.Operands) == 1
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
branch = len(instr.Operands) >= 2
}
if !branch {
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// loong64Layout walks the function body once and returns the label offsets,
// the loop-alignment padding due before each instruction (a pad of 0 needs
// nothing), the pc each instruction starts at (its padding included) and the
// first pc past the body. Explicit PCALIGN pads, the alignment pads for the
// labels in aligns and those for the instructions in alignInstrs (backward
// PC-relative targets) all contribute, mirroring the toolchain's layout
// pass.
func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) {
offsets := map[string]int{}
alignPad := map[*ast.Instr]int{}
pcs := map[*ast.Instr]int{}
pos := start
pendingAlign := false
var pendingNames []string
explicit := false
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
if aligns[s.Name.Text] {
pendingAlign = true
}
pendingNames = append(pendingNames, s.Name.Text)
// Provisional: a branch to the label lands here unless a loop
// alignment pad follows, in which case the label resolves to the
// padded instruction (the toolchain's labels bind to the branch
// target instruction, which the padding pass precedes).
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pos += loong64PCAlignPad(pos, s)
explicit = true
continue
}
if pendingAlign {
pendingAlign = false
if pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
}
if alignInstrs[s] && pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
if !explicit {
for _, n := range pendingNames {
offsets[n] = pos + alignPad[s]
}
}
pendingNames = nil
explicit = false
pcs[s] = pos + alignPad[s]
pos += alignPad[s] + loong64InstrSize(s, fi)
}
}
return offsets, alignPad, pcs, pos
}
// loong64BranchTarget reports the local label a branch-like instruction
// transfers to, the loop-head signal the toolchain derives from backward
// branch targets.
func loong64BranchTarget(instr *ast.Instr) (string, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var op *ast.Operand
switch {
case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF":
if len(ops) != 1 {
return "", false
}
op = ops[0]
case mnem == "TEQ" || mnem == "TNE":
return "", false
case len(ops) >= 2:
op = ops[len(ops)-1]
default:
return "", false
}
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go
@@ -177,21 +339,51 @@ func l64LabelOK(op *ast.Operand) (string, bool) {
return "", false
}
// l64SubToAdd rewrites the SUB family with an immediate first operand onto
// its ADD counterpart with the negated immediate: LoongArch has no
// subtract-immediate instructions, and the toolchain folds SUB $v into the
// ADD immediate form through the same optab matching (the $0 fold into 3R
// and the large-constant materialisations included). The negation is the
// second result; the operand is left untouched because the size pass
// normalises the same instruction.
func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch mnem {
case "SUB":
return "ADD", true
case "SUBW":
return "ADDW", true
case "SUBV", "SUBVU":
return "ADDV", true
}
}
return mnem, false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var neg bool
mnem, neg = l64SubToAdd(mnem, ops)
if mnem == "RET" {
return len(loong64Return(fi))
}
switch mnem {
case "TEQ", "TNE":
return 8 // bne/beq over the BREAK, then BREAK
case "PRELDX":
return 20 // the four-instruction constant materialisation + preldx
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0])
if neg {
v = -v
}
if v == 0 {
return 4 // folds into the 3R form (rk = R0)
}
@@ -229,11 +421,50 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
return 4
}
// loong64PCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func loong64PCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0
// (the architecture's NOP) for every full 4 bytes of pad.
func loong64PadBytes(pad int) []byte {
nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0))
out := make([]byte, 0, pad/4*len(nop))
for i := 0; i < pad/4; i++ {
out = append(out, nop...)
}
return out
}
// encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// The SUB family with an immediate first operand folds onto the ADD
// immediate form with the negated immediate; the negation happens on a
// copy of the operand, never on the shared syntax tree.
mnem, neg := l64SubToAdd(mnem, ops)
if neg {
c := *ops[0]
c.Imm.Val = -c.Imm.Val
ops2 := make([]*ast.Operand, len(ops))
ops2[0] = &c
copy(ops2[1:], ops[1:])
ops = ops2
}
// Pseudo-instructions and the branches first.
switch mnem {
case "RET":
@@ -249,10 +480,80 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "NEGW", "NEGV":
// The integer negation pseudo is a subtract from zero:
// NEGW src, dst → sub.w r0, src, dst.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := l64Reg(ops[0]), l64Reg(ops[1])
if src < 0 || dst < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sub := l64InstrTable["SUBW"].op
if mnem == "NEGV" {
sub = l64InstrTable["SUBV"].op
}
return l64wordLE(l64rrr(sub, src, 0, dst)), nil
case "TEQ", "TNE":
// The trap pseudo expands to two instructions: bne/beq rj, rd over
// the BREAK (offset 2 instruction units), then BREAK $code.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
code := int(immFromOperand(ops[0]))
rj, rd := 0, l64Reg(ops[len(ops)-1])
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
bop := l64branchTable["BNE"]
if mnem == "TNE" {
bop = l64branchTable["BEQ"]
}
return l64WordsLE(
l64irr16(bop, 2, rj, rd),
l64i15(l64InstrTable["BREAK"].op, code),
), nil
case "PRELDX":
// preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs
// (addrSeq, blockSize, blockNums, stride); the constant v built from
// it materialises in R30 across four instructions, then the preldx.
if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint")
}
rj := loong64RegNum(ops[0].Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
n := uint64(l64Imm64(ops[1]))
hint := int(l64Imm64(ops[2]))
addrSeq := (n >> 0) & 0x1
blkSize := (n >> 1) & 0x7ff
blkNums := (n >> 12) & 0x1ff
stride := (n >> 21) & 0xffff
v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 +
((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
ori = 0x00e << 22
preldx = 0x7058 << 15
)
return l64WordsLE(
l64ir(lu12iw, int(uint32(v>>12)), 30),
l64irr(ori, int(uint32(v)), 30, 30),
l64ir(lu32id, int(uint32(v>>32)), 30),
l64irr(lu52id, int(uint32(v>>52)), 30, 30),
l64rrr(preldx, 30, rj, hint),
), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
@@ -262,12 +563,12 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
if mnem == "JIRL" {
return encodeLOONG64Jirl(op, ops)
}
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// B/BL aliases reached only via JMP/JAL above.
@@ -533,11 +834,23 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
op := instr.Operands[0]
// PC-relative displacement: N(PC) resolves to the instruction N slots
// away in source order (the toolchain's parse-time count), and the field
// carries the final pc distance in instruction units.
if op.Addr.Sym == nil && op.Addr.Base == "PC" {
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset)
}
v := (targetPc - pc) >> 2
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base)
@@ -620,16 +933,28 @@ func l64offsetOperand(op *ast.Operand) (int32, bool) {
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[len(ops)-1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
var target string
var v int
lastOp := ops[len(ops)-1]
if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target = resolve(l64Label(lastOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
v := (targetOff - pc) >> 2
if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field.
@@ -690,33 +1015,55 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
isBF := mnem == "BFPT" || mnem == "BFPF"
if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
rj := 0 // BFPT/BFPF default to FCC0
if mnem != "BFPT" && mnem != "BFPF" {
var rj int
tgtOp := ops[len(ops)-1]
if isBF {
// BFPT/BFPF test an FCC condition register, defaulting to FCC0 when
// spelled without one.
rj = 0
if len(ops) == 2 {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
} else {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
var v int
if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target := resolve(l64Label(tgtOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
return nil, fmt.Errorf("branch %d too far (16-bit range)", v)
}
return l64wordLE(l64irr16(op, v, 0, rj)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
return nil, fmt.Errorf("branch %d too far (21-bit range)", v)
}
return l64wordLE(l64ir21(op, v, rj)), nil
}
+564 -45
View File
@@ -4,7 +4,9 @@
package asm
import (
"errors"
"fmt"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -31,32 +33,54 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
// Pass 1: collect instructions and compute label offsets assuming 4 bytes
// per instruction (or 8 for MOV $large-imm). No encoding yet.
// per instruction (or 8 for MOV $large-imm). No encoding yet. PCALIGN
// contributes only its padding, which is attached to the following
// instruction and emitted ahead of it. A relaxed branch carries the
// inverted condition and is followed by an inserted JMP rec (jmpTo set)
// that carries the original target.
type instrRec struct {
instr *ast.Instr
compressed bool
code []byte
pad int
relaxed bool
jmpTo string
}
var recs []instrRec
offsets := map[string]int{}
pos := guardLen + len(prologue)
pendingPad := 0
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
recs = append(recs, instrRec{instr: s})
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pendingPad += riscvPCAlignPad(pos, s)
pos += riscvPCAlignPad(pos, s)
continue
}
recs = append(recs, instrRec{instr: s, pad: pendingPad})
pendingPad = 0
pos += riscvInstrSize(s, fi)
}
}
// Pass 2: encode each instruction using Pass-1 offsets.
// Pass 2: encode each instruction using Pass-1 offsets. A branch or
// jump the offsets prove overlong encodes to a 4-byte placeholder: the
// relaxation pass rewrites it before the final encoding. pcRelPcs is
// unavailable this early, so the N(PC) forms take the same placeholder
// path.
pc := len(prologue)
for i := range recs {
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2
if err != nil {
branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil) // no relocs in Pass 2
if err != nil && !(branchLike && riscvIsRangeError(err)) {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
}
if err != nil {
code = make([]byte, 4)
}
recs[i].code = code
pc += len(code)
}
@@ -72,20 +96,100 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// Pass 4: recompute offsets with actual sizes. recs holds the
// instructions in emission order, so an index into it walks t.Body in
// lockstep (the same single pass Pass 1 uses) instead of rescanning the
// whole slice per statement.
// whole slice per statement. PCALIGN padding is recomputed here, since
// compression has shifted instruction sizes since Pass 1.
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
pendingPad = 0
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pad := riscvPCAlignPad(pos, s)
pendingPad += pad
pos += pad
continue
}
recs[ri].pad = pendingPad
pendingPad = 0
pos += len(recs[ri].code)
ri++
}
}
// Pass 4b: relax overlong conditional branches exactly as the toolchain
// does: invert the branch condition, point it at the instruction after an
// inserted JMP, let the JMP carry the original target, and re-layout until
// a pass inserts nothing. Inserted JMP recs share their branch's source
// line and trail it in emission order, so the body walk flushes them
// before every statement and at the end.
var pcRelPcs map[*ast.Instr]int
for {
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
pcs := make([]int, len(recs))
flushJmps := func() {
for ri < len(recs) && recs[ri].jmpTo != "" {
pcs[ri] = pos
pos += 4
ri++
}
}
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
flushJmps()
offsets[s.Name.Text] = pos
case *ast.Instr:
flushJmps()
if ri >= len(recs) {
continue
}
pcs[ri] = pos + recs[ri].pad
pos += recs[ri].pad + len(recs[ri].code)
ri++
}
}
flushJmps()
changed := false
for i := range recs {
r := &recs[i]
if r.relaxed || r.jmpTo != "" {
continue
}
mnem := strings.ToUpper(r.instr.Mnemonic.Text)
if !riscvIsCondBranch(mnem) || len(r.instr.Operands) == 0 {
continue
}
target := labelFromOperand(r.instr.Operands[len(r.instr.Operands)-1])
targetOff, ok := offsets[target]
if !ok {
continue
}
if delta := int32(targetOff - pcs[i]); delta < -4096 || delta >= 4096 {
r.relaxed = true
recs = slices.Insert(recs, i+1, instrRec{instr: r.instr, jmpTo: target})
changed = true
}
}
if !changed {
// Capture the final pcs for the N(PC) branch forms: their target
// is the instruction N source slots away, resolved by index.
pcRelPcs = map[*ast.Instr]int{}
for i := range recs {
if _, ok := riscvPCRelOffset(recs[i].instr); ok {
pcRelPcs[recs[i].instr] = pcs[i]
}
}
break
}
}
// Pass 5: re-encode branches with corrected offsets. Record relocations
// during this final pass (relocation offsets are relative to instruction
// start). The guard prefix precedes the prologue; its branches target
@@ -104,12 +208,40 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
preCount := len(relocs)
var lines []LineEntry
for _, r := range recs {
// PCALIGN padding precedes the instruction it was attached to.
if r.pad > 0 {
out = append(out, riscvPadBytes(r.pad)...)
pc += r.pad
}
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) {
out = append(out, r.code...)
pc += len(r.code)
} else {
code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs)
var code []byte
switch {
case r.jmpTo != "":
// The JMP a relaxation inserted: JAL X0 to the original target.
targetOff, ok := offsets[r.jmpTo]
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
return nil, nil, nil, nil, nil, err
}
word := riscvJType(0, offset)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.relaxed:
// The inverted half of a relaxed branch: it targets the inserted
// JMP, always the very next instruction (offset 4).
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
}
word := riscvBType(enc, rs1, rs2, 4)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.compressed && !isBranchLike(r.instr.Mnemonic.Text):
code = r.code
default:
var err error
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs)
if err != nil {
return nil, nil, nil, nil, nil, err
}
@@ -131,9 +263,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
}
out = append(out, code...)
pc += len(code)
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
relocs = append(relocs, guardReloc)
@@ -150,6 +282,8 @@ var riscvImmAlias = map[string]string{
"AND": "ANDI",
"OR": "ORI",
"XOR": "XORI",
"SLT": "SLTI",
"SLTU": "SLTIU",
"SLL": "SLLI",
"SRL": "SRLI",
"SRA": "SRAI",
@@ -178,6 +312,35 @@ func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
return mnem, false
}
// riscvPCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func riscvPCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// riscvPadBytes renders PCALIGN padding: 4-byte NOPs (addi $0, X0, X0) with a
// trailing 2-byte compressed NOP when the pad is 2 mod 4, exactly as the
// toolchain lays the bytes down.
func riscvPadBytes(pad int) []byte {
out := make([]byte, 0, pad)
for ; pad >= 4; pad -= 4 {
out = append(out, 0x13, 0x00, 0x00, 0x00)
}
if pad == 2 {
out = append(out, 0x01, 0x00)
}
return out
}
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
// Most instructions are 4 bytes; MOV with a large immediate and I-type
// arithmetic with a large immediate expand to several (possibly compressed)
@@ -224,6 +387,10 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
}
return riscvItypeImmediateSize(mnem, imm)
}
// BYTE lays down one raw byte per operand.
if mnem == "BYTE" {
return len(ops)
}
// The toolchain's synthesised instructions: some emit one word, others
// expand to a fixed sequence.
return riscvExtendedSize(mnem, ops)
@@ -318,12 +485,171 @@ func isBranchLike(mnem string) bool {
return false
}
// riscvIsCondBranch reports whether m is a conditional branch, the only
// instruction class branch relaxation rewrites.
func riscvIsCondBranch(mnem string) bool {
switch mnem {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
return true
}
return false
}
// riscvCSRNames maps the standard CSR mnemonics the assembler accepts onto
// their addresses.
var riscvCSRNames = map[string]int32{
"FFLAGS": 0x001,
"FRM": 0x002,
"FCSR": 0x003,
"VSTART": 0x008,
"VXSAT": 0x009,
"VXRM": 0x00A,
"VCSR": 0x00F,
"CYCLE": 0xC00,
"TIME": 0xC01,
"INSTRET": 0xC02,
"CYCLEH": 0xC80,
"TIMEH": 0xC81,
"INSTRETH": 0xC82,
"VL": 0xC20,
"VLENB": 0xC22,
}
// riscvCSRAddress resolves a CSR operand: an integer immediate or one of the
// standard CSR names.
func riscvCSRAddress(op *ast.Operand) (int32, bool) {
if isImmOperand(op) {
return immFromOperand(op), true
}
if op.Addr.Sym != nil {
if v, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
return v, true
}
}
return 0, false
}
// riscvPCRelOffset reports the N of a branch or jump operand spelled N(PC):
// the displacement counted in source instructions from the branch itself.
func riscvPCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
switch mnem {
case "JMP":
if len(instr.Operands) != 1 {
return 0, false
}
case "JAL":
if len(instr.Operands) != 1 && len(instr.Operands) != 2 {
return 0, false
}
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
if len(instr.Operands) < 2 {
return 0, false
}
default:
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// riscvPCRelTargetOff resolves the target displacement of a branch whose last
// operand is N(PC): the toolchain's parser counts the source instructions at
// a uniform 4 bytes, so the target is the instruction N slots away, and the
// displacement tracks that instruction's final pc. A nil pcRelPcs (the
// layout passes) yields a placeholder range error; the caller tolerates it
// for branch-like instructions.
func riscvPCRelTargetOff(instr *ast.Instr, pc int, pcRelPcs map[*ast.Instr]int) (int, bool, error) {
off, ok := riscvPCRelOffset(instr)
if !ok {
return 0, false, nil
}
if pcRelPcs == nil {
return 0, true, &riscvRangeError{"pc-relative placeholder"}
}
targetPc, ok := pcRelPcs[instr]
if !ok {
return 0, true, fmt.Errorf("PC-relative target %d out of range", off)
}
return targetPc, true, nil
}
// riscvInvertedBranchEnc returns the encoding of mnem's inverted condition
// for the given operands: InvertBranch's table applied at the encoding level.
// The register operands are already in position for the inverted form.
func riscvInvertedBranchEnc(mnem string, ops []*ast.Operand) (riscvEnc, int, int, bool) {
reg := func(i int) int { return regFromOperand(ops[i]) }
switch mnem {
case "BEQ": // → BNE rs1, rs2
return riscvEnc{0x63, 0x1, 0x00}, reg(0), reg(1), true
case "BNE": // → BEQ rs1, rs2
return riscvEnc{0x63, 0x0, 0x00}, reg(0), reg(1), true
case "BLT": // → BGE rs1, rs2
return riscvEnc{0x63, 0x5, 0x00}, reg(0), reg(1), true
case "BGE": // → BLT rs1, rs2
return riscvEnc{0x63, 0x4, 0x00}, reg(0), reg(1), true
case "BLTU": // → BGEU rs1, rs2
return riscvEnc{0x63, 0x7, 0x00}, reg(0), reg(1), true
case "BGEU": // → BLTU rs1, rs2
return riscvEnc{0x63, 0x6, 0x00}, reg(0), reg(1), true
case "BEQZ": // → BNEZ rs, X0
return riscvEnc{0x63, 0x1, 0x00}, reg(0), 0, true
case "BNEZ": // → BEQZ rs, X0
return riscvEnc{0x63, 0x0, 0x00}, reg(0), 0, true
case "BLTZ": // → BGEZ rs, X0
return riscvEnc{0x63, 0x5, 0x00}, reg(0), 0, true
case "BGEZ": // → BLTZ rs, X0
return riscvEnc{0x63, 0x4, 0x00}, reg(0), 0, true
case "BLEZ": // → BGTZ: blt X0, rs
return riscvEnc{0x63, 0x4, 0x00}, 0, reg(0), true
case "BGTZ": // → BLEZ: bge X0, rs
return riscvEnc{0x63, 0x5, 0x00}, 0, reg(0), true
case "BGT": // → BLE: bge rs2, rs1
return riscvEnc{0x63, 0x5, 0x00}, reg(1), reg(0), true
case "BLE": // → BGT: blt rs2, rs1
return riscvEnc{0x63, 0x4, 0x00}, reg(1), reg(0), true
case "BGTU": // → BLEU: bgeu rs2, rs1
return riscvEnc{0x63, 0x7, 0x00}, reg(1), reg(0), true
case "BLEU": // → BGTU: bltu rs2, rs1
return riscvEnc{0x63, 0x6, 0x00}, reg(1), reg(0), true
}
return riscvEnc{}, 0, 0, false
}
// riscvRangeError reports a branch or jump displacement beyond its
// architecture limit. The layout passes tolerate it (the relaxation pass
// rewrites overlong conditional branches before the final encoding); a range
// error reaching the final pass is a real failure.
type riscvRangeError struct{ msg string }
func (e *riscvRangeError) Error() string { return e.msg }
// riscvIsRangeError reports whether err is a displacement-range rejection.
func riscvIsRangeError(err error) bool {
var re *riscvRangeError
return errors.As(err, &re)
}
// riscvRoundModes maps the rounding-mode suffixes onto their funct7 codes.
var riscvRoundModes = map[string]uint32{
"RNE": 0,
"RTZ": 1,
"RDN": 2,
"RUP": 3,
"RMM": 4,
}
// riscvCheckBranchOffset rejects a B-type displacement outside its signed
// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an
// out-of-range offset would otherwise wrap to a wrong target.
func riscvCheckBranchOffset(target string, off int32) error {
if off < -4096 || off > 4094 {
return fmt.Errorf("branch to %q too far (13-bit range)", target)
return &riscvRangeError{fmt.Sprintf("branch to %q too far (13-bit range)", target)}
}
return nil
}
@@ -332,13 +658,13 @@ func riscvCheckBranchOffset(target string, off int32) error {
// 21-bit span [-1048576, 1048574].
func riscvCheckJumpOffset(target string, off int32) error {
if off < -1048576 || off > 1048574 {
return fmt.Errorf("jump to %q too far (21-bit range)", target)
return &riscvRangeError{fmt.Sprintf("jump to %q too far (21-bit range)", target)}
}
return nil
}
// encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := instr.Mnemonic.Text
ops := instr.Operands
var immNeg bool
@@ -351,6 +677,27 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil
case "WORD":
// WORD $w lays down a raw 32-bit little-endian word.
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
w := int64(immFromOperand(ops[0]))
if w < 0 || w > 0xFFFFFFFF {
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
}
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}, nil
case "BYTE":
// BYTE $b lays down one raw byte per operand.
var out []byte
for _, op := range ops {
b := int64(immFromOperand(op))
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "CALL":
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
// relocation. The Go assembler rejects CALL to a local branch label.
@@ -404,6 +751,17 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
targetOff, ok := offsets[target]
if !ok {
@@ -424,6 +782,18 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} else if len(ops) == 1 {
target = labelFromOperand(ops[0])
}
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff := off
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
@@ -456,10 +826,20 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
if rs < 0 {
return nil, fmt.Errorf("%s: invalid register", mnem)
}
target := labelFromOperand(ops[1])
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
targetOff := 0
target := ""
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
target = labelFromOperand(ops[1])
var ok bool
targetOff, ok = offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
}
var enc riscvEnc
rs1, rs2 := rs, 0
@@ -484,18 +864,25 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// System instructions with no operands.
case "FENCE", "ECALL", "EBREAK":
case "FENCE", "ECALL", "EBREAK", "FENCE.TSO", "PAUSE":
enc, ok := riscvInstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported system instruction %q", mnem)
}
// The bare FENCE expands to fence iorw, iorw: the predecessor and
// successor fields both carry 0xF in the I-type immediate
// (the toolchain's encodeFenceOperand TYPE_NONE default).
// (the toolchain's encodeFenceOperand TYPE_NONE default). FENCE.TSO
// carries the TSO fence mode with RW predecessor and successor.
imm := int32(0)
if mnem == "FENCE" {
imm = 0x0FF
}
if mnem == "FENCE.TSO" {
imm = 0x833
}
if mnem == "PAUSE" {
imm = 0x010
}
word = riscvIType(enc, 0, 0, imm)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
@@ -516,6 +903,29 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
// low three bits of the funct7 field.
if i := strings.IndexByte(mnem, '.'); i > 0 {
if base, ok := riscvCvtTable[mnem[:i]]; ok {
rm, ok := riscvRoundModes[mnem[i+1:]]
if !ok {
return nil, fmt.Errorf("unsupported rounding mode in %q", mnem)
}
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
base.funct7 = (base.funct7 &^ 7) | rm
word := riscvCvtType(base, rd, rs1)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
// R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last).
if fmaEnc, ok := riscvFmaTable[mnem]; ok {
if len(ops) != 4 {
@@ -532,29 +942,103 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last).
if csrEnc, ok := riscvCsrTable[mnem]; ok {
if len(ops) != 3 {
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last). The
// write-only pseudos (CSRS/CSRC/CSRW and their immediate forms) spell
// the source first, the CSR second, and read the destination as X0; the
// immediate or register variant follows the source operand's kind.
csrMnem := mnem
csrPseudo := false
csrRead := false
csrFix := int32(0)
switch mnem {
case "CSRS", "CSRW", "CSRC", "CSRSI", "CSRWI", "CSRCI":
csrMnem = map[string]string{
"CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC",
"CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI",
}[mnem]
csrPseudo = true
// CSRR csr, rd is CSRRS rd, csr, X0; the read pseudos RDCYCLE/RDTIME/
// RDINSTRET fix the CSR to cycle/time/instret.
case "CSRR":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
case "RDCYCLE", "RDTIME", "RDINSTRET":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
csrFix = map[string]int32{"RDCYCLE": 0xC00, "RDTIME": 0xC01, "RDINSTRET": 0xC02}[mnem]
}
if csrEnc, ok := riscvCsrTable[csrMnem]; ok {
if csrRead && len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
if csrPseudo && !csrRead && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if !csrPseudo && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
csr := immFromOperand(ops[0]) // CSR address (12-bit)
csrOp := ops[0]
srcOp := ops[0]
rdOp := ops[len(ops)-1]
switch {
case csrRead:
// CSRR csr, rd (or the fixed-CSR read pseudos with only rd).
case csrPseudo:
// src, csr.
if len(ops) > 1 {
csrOp, srcOp = ops[1], ops[0]
}
rdOp = nil
default:
// Either src, csr, rd or csr, src, rd: a CSR *name* in the
// second operand marks the toolchain's order.
srcOp = ops[1]
if op := ops[1]; op.Addr.Sym != nil {
if _, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
csrOp, srcOp = ops[1], ops[0]
}
}
}
csr, ok := riscvCSRAddress(csrOp)
if !ok && csrFix == 0 {
return nil, fmt.Errorf("%s: unknown CSR %q", mnem, csrOp.Raw)
}
if csrFix != 0 {
csr = csrFix
}
if csr < 0 || csr > 0xFFF {
return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr)
}
rd := regFromOperand(ops[2]) // destination register
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
rd := 0
if !csrPseudo {
rd = regFromOperand(rdOp) // destination register
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
}
if csrRead {
rd = regFromOperand(rdOp)
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
}
var src int
if csrEnc.imm {
// Immediate variant: ops[1] is a 5-bit unsigned immediate.
src = int(immFromOperand(ops[1]))
switch {
case csrRead:
// CSRR reads with rs1 = X0: src stays zero.
case isImmOperand(srcOp):
// Immediate variant: the source is a 5-bit unsigned immediate.
src = int(immFromOperand(srcOp))
if src < 0 || src > 31 {
return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem)
}
} else {
// Register variant: ops[1] is a register.
src = regFromOperand(ops[1])
case csrEnc.imm:
return nil, fmt.Errorf("%s expects an immediate source", mnem)
default:
// Register variant: the source is a register.
src = regFromOperand(srcOp)
if src < 0 {
return nil, fmt.Errorf("invalid source register in %s", mnem)
}
@@ -738,14 +1222,34 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvSType(enc, rs1, rs2, imm)
// Branches: rs1, rs2, label.
// Branches: rs1, rs2, label. BGT/BLE/BGTU/BLEU are the swapped-spelling
// forms of BLT/BGE/BLTU/BGEU (bgt rs1, rs2 is blt rs2, rs1).
case len(ops) == 3 && isBranchInstr(mnem):
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
target := labelFromOperand(ops[2])
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
switch mnem {
case "BGT":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, rs2, rs1 // blt rs2, rs1
case "BLE":
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, rs2, rs1 // bge rs2, rs1
case "BGTU":
enc, rs1, rs2 = riscvEnc{0x63, 0x6, 0x00}, rs2, rs1 // bltu rs2, rs1
case "BLEU":
enc, rs1, rs2 = riscvEnc{0x63, 0x7, 0x00}, rs2, rs1 // bgeu rs2, rs1
}
targetOff := 0
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
var ok bool
targetOff, ok = offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
}
offset := int32(targetOff - pc)
if rs1 < 0 || rs2 < 0 {
@@ -758,10 +1262,15 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm.
// U-type: rd, imm (or the toolchain testdata's INSTR $imm, rd).
case len(ops) == 2 && isUTypeInstr(mnem):
rd := regFromOperand(ops[0])
imm := immFromOperand(ops[1])
var rd int
var imm int32
if isImmOperand(ops[0]) {
imm, rd = immFromOperand(ops[0]), regFromOperand(ops[1])
} else {
rd, imm = regFromOperand(ops[0]), immFromOperand(ops[1])
}
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
@@ -1226,6 +1735,11 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := riscvCompressMnem(instr)
ops := instr.Operands
// The immediate aliases fold onto their I-type mnemonics before
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly
// as it compresses the spelling ADDI.
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
switch mnem {
case "LD", "MOV":
@@ -1289,6 +1803,9 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
case "ADDI":
rd, rs1, imm := extractITypeParams(instr)
if immNeg {
imm = -imm
}
if rd == -1 || rs1 == -1 {
return 0, false
}
@@ -2045,7 +2562,8 @@ func isRTypeInstr(m string) bool {
case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND",
"ADDW", "SUBW", "SLLW", "SRLW", "SRAW",
"MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU",
"MULW", "DIVW", "DIVUW", "REMW", "REMUW":
"MULW", "DIVW", "DIVUW", "REMW", "REMUW",
"CZEROEQZ", "CZERONEZ":
return true
}
return false
@@ -2085,7 +2603,7 @@ func isStoreInstr(m string) bool {
func isBranchInstr(m string) bool {
switch m {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU":
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU":
return true
}
return false
@@ -2111,7 +2629,8 @@ func isFPArithInstr(m string) bool {
switch m {
case "FADDS", "FSUBS", "FMULS", "FDIVS",
"FADDD", "FSUBD", "FMULD", "FDIVD",
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD":
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
"FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX":
return true
}
return false
+26 -4
View File
@@ -219,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01},
// Zicond conditional zeroing.
"CZEROEQZ": {0x33, 0x5, 0x07},
"CZERONEZ": {0x33, 0x7, 0x07},
// RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00},
@@ -247,13 +250,21 @@ var riscvInstrTable = map[string]riscvEnc{
"BGE": {0x63, 0x5, 0x00},
"BLTU": {0x63, 0x6, 0x00},
"BGEU": {0x63, 0x7, 0x00},
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
// with the register operands swapped.
"BGT": {0x63, 0x4, 0x00},
"BLE": {0x63, 0x5, 0x00},
"BGTU": {0x63, 0x6, 0x00},
"BLEU": {0x63, 0x7, 0x00},
// U-type.
"LUI": {0x37, 0x0, 0x00},
"AUIPC": {0x17, 0x0, 0x00},
// System.
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
"FENCE.TSO": {0x0F, 0x0, 0x00},
"PAUSE": {0x0F, 0x0, 0x00},
// JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00},
@@ -303,7 +314,14 @@ var riscvInstrTable = map[string]riscvEnc{
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
"FSGNJD": {0x53, 0x0, 0x11},
"FSGNJS": {0x53, 0x0, 0x10},
"FSGNJX": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x0, 0x15},
"FSGNJXS": {0x53, 0x0, 0x14},
"FSGNJND": {0x53, 0x1, 0x11},
"FSGNJNS": {0x53, 0x1, 0x10},
"FSGNJNX": {0x53, 0x1, 0x14},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
// The toolchain gives LR acquire ordering (aq = 1) and SC release
@@ -375,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
// The toolchain's W/D suffix spellings of the same moves.
"FMVXS": {0x70, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x53},
"FMVSX": {0x79, 0x0, 0x53},
}
// riscvCvtType encodes an FP conversion instruction.
+19 -6
View File
@@ -868,7 +868,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
@@ -905,9 +905,10 @@ func TestRISCVBranchJumpRange(t *testing.T) {
}
}
// TestRISCVBranchFarBody drives the range check through the full two-pass
// assembler: a forward branch over a body larger than the B-type span must
// error rather than wrap.
// TestRISCVBranchFarBody drives the relaxation pass through the full
// assembler: a forward branch over a body larger than the B-type span is
// rewritten as an inverted branch over an inserted JMP, the same layout the
// toolchain produces, instead of wrapping to a wrong target.
func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
@@ -916,8 +917,20 @@ func TestRISCVBranchFarBody(t *testing.T) {
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected a branch-out-of-range error, got none")
out, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
if !bytes.Equal(out[0:4], wantBranch) {
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
}
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
wantJmp := wordLE(riscvJType(0, 4404))
if !bytes.Equal(out[4:8], wantJmp) {
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
}
}
+30 -7
View File
@@ -37,7 +37,7 @@ import (
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -57,11 +57,13 @@ per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args())
return cmdAuditCorpus(fs.Args(), dirs)
}
archName := "amd64"
switch n := len(fs.Args()); {
@@ -395,8 +397,11 @@ func (t *corpusTally) fail(path, reason string) {
}
}
// cmdAuditCorpus implements audit-instructions --corpus.
func cmdAuditCorpus(args []string) error {
// cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
@@ -410,7 +415,25 @@ func cmdAuditCorpus(args []string) error {
}
root = filepath.Join(strings.TrimSpace(string(out)), "src")
}
stats, err := runCorpusAudit(root)
// The toolchain's shipped headers (funcdata.h and friends) define the
// macros GOROOT files include; a corpus audit measures those files, so
// the header directory joins the search path automatically. go_asm.h
// is compiler-generated per package and stays unresolvable on purpose.
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
seen := false
for _, d := range dirs {
if d == pkgInclude {
seen = true
}
}
if !seen {
dirs = append(dirs, pkgInclude)
}
}
}
stats, err := runCorpusAudit(root, dirs)
if err != nil {
return err
}
@@ -454,7 +477,7 @@ func otherPortFile(path string) bool {
return false
}
func runCorpusAudit(root string) (*corpusStats, error) {
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
return nil, err
@@ -479,7 +502,7 @@ func runCorpusAudit(root string) (*corpusStats, error) {
if err != nil {
return nil, err
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
+26 -11
View File
@@ -240,6 +240,16 @@ func readSource(path string) (string, error) {
return string(b), err
}
// includeDirs collects repeatable -I flags: the directories searched for
// #include files during macro expansion and include splicing.
type includeDirs []string
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
func (d *includeDirs) Set(v string) error {
*d = append(*d, v)
return nil
}
func cmdTokens(args []string) int {
fs := newCommand("tokens", "gasm tokens <file>", `
Print the lexical token stream of FILE: position, token kind and text, one
@@ -476,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
}
func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", `
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>", `
Assemble FILE without the Go toolchain: every TEXT function is encoded to
machine code and printed as a hex dump. Supported architectures: amd64
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
@@ -498,9 +508,11 @@ and the format version from go version).
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>")
return 2
}
// The format is validated before anything else, so a bogus value exits 2
@@ -526,7 +538,7 @@ and the format version from go version).
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
@@ -634,7 +646,7 @@ and the format version from go version).
// cmdDiff compares the machine code of two assembly files.
func cmdDiff(args []string) int {
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", `
set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
Compare the machine code produced by assembling two files.
Shows which functions differ and the byte-level differences.
Useful for verifying that two implementations produce identical code,
@@ -645,9 +657,11 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
`)
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
var dirs includeDirs
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
set.Parse(args)
if set.NArg() != 2 {
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>")
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
return 2
}
path1, path2 := set.Arg(0), set.Arg(1)
@@ -675,12 +689,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
}
// Assemble both files.
img1, err := assemblePath(path1, forced)
img1, err := assemblePath(path1, forced, dirs)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
return 1
}
img2, err := assemblePath(path2, forced)
img2, err := assemblePath(path2, forced, dirs)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
return 1
@@ -755,14 +769,15 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
}
}
// assemblePath reads, parses and assembles a file (used by cmdDiff). A
// non-Unknown forced architecture overrides the file-name suffix.
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) {
// assemblePath reads, preprocesses, parses and assembles a file (used by
// cmdDiff). A non-Unknown forced architecture overrides the file-name
// suffix.
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
src, err := readSource(path)
if err != nil {
return nil, err
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
+1 -1
View File
@@ -403,7 +403,7 @@ func TestRunCorpusAudit(t *testing.T) {
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
stats, err := runCorpusAudit(dir)
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
+109
View File
@@ -0,0 +1,109 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"path/filepath"
"strings"
"testing"
)
// writeTree writes a directory of files and returns its root.
func writeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
// in-file parameterised macro and an include resolved through -I, and checks
// the assembled bytes came from the expansion (the loop body counts six
// increments, two per expanded iteration).
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"inc/consts.h": "#define NITER 3\n",
"main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"consts.h\"\n" +
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
"\tXORQ AX, AX\n" +
"\tMOVQ $NITER, CX\n" +
"loop:\n" +
"\tSTEP(AX)\n" +
"\tDECQ CX\n" +
"\tJNZ loop\n" +
"\tMOVQ AX, ret+0(FP)\n" +
"\tRET\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// The macro expanded to two ADDQ $1 encodings in the static body; the
// iteration count lives in the runtime loop.
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
}
}
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
// including file's directory wins over the -I directories.
func TestAsmIncludeResolutionOrder(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"src/main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"vals.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ $VAL, AX\n" +
"\tRET\n",
"src/vals.h": "#define VAL 1\n",
"late/vals.h": "#define VAL 2\n",
"early/vals.h": "#define VAL 3\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// VAL came from src/vals.h, not from either -I directory: the image
// loads the immediate 1.
if !strings.Contains(stdout, "b8 01 00 00 00") {
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
}
}
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
// resolves nowhere on the assembly path.
func TestAsmMissingIncludeIsAnError(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
if code == 0 {
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
}
if !strings.Contains(stderr, `#include "nothere.h"`) {
t.Errorf("stderr does not name the failing include: %s", stderr)
}
}
+12 -3
View File
@@ -141,12 +141,13 @@ gasm lint kernel_amd64.s
## asm
```text
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>
Usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>
```
| Flag | Default | Effect |
|---|---|---|
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
| `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix |
| `-o` | empty | write the output to this file instead of a hex dump on stdout |
@@ -162,6 +163,13 @@ system toolchain; `goobj` emits the Go toolchain's own object format, which
installed: the object preamble is captured from `go tool asm` and the format
version from `go version`. `raw` and `elf` need no toolchain at all.
Assembly preprocessing matches the toolchain's: `#define` macros (object and
parameterised) expand at the point of use, `#undef`, `#ifdef`, `#ifndef`,
`#else` and `#endif` behave as in `go tool asm`, `;` separates statements,
and `#include "file"` splices the named file in, resolved against the source
directory and then each `-I` directory in order. `textflag.h` is the one
header that is not spliced: gasm consumes its flag names natively.
```sh
gasm asm hello_amd64.s
```
@@ -305,12 +313,13 @@ gasm debug --func add --cover hello_amd64.s
## diff
```text
Usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>
Usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>
```
| Flag | Default | Effect |
|---|---|---|
| `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
Functions are paired by exact name unless `--map` says otherwise, so
@@ -348,7 +357,7 @@ add: 16 bytes, args=24, frame=0 NOSPLIT
## audit-instructions
```text
Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]
Usage: gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]
```
Compare the gasm encoder for the given architecture (default amd64) against the
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-asm \- assemble Plan 9 assembly without the Go toolchain
.SH SYNOPSIS
.B gasm asm [\-\-format raw|elf|goobj] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
.B gasm asm [\-\-format raw|elf|goobj] [\-I dir] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
.SH DESCRIPTION
Assemble FILE without the Go toolchain: every TEXT function is encoded
to machine code and printed as a hex dump. Supported architectures:
@@ -47,6 +47,10 @@ functions link too.
.B \-\-format \fIraw|elf|goobj\fR
Output format; the default is raw.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-p \fIpkg\fR
Package path for --format goobj, qualifying the exported symbols.
.TP
+7 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
.SH SYNOPSIS
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [amd64|arm64|riscv64|loong64]
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-I dir] [amd64|arm64|riscv64|loong64]
.SH DESCRIPTION
Compare the gasm encoder for the given architecture (default amd64)
against
@@ -38,6 +38,12 @@ second.
.B \-\-corpus [\fIdir\fR]
Assemble a corpus of .s files and report pass rates and failure
reasons.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory. A corpus run whose files include
toolchain headers (such as GOROOT/pkg/include) needs it, the same -I a
toolchain comparison takes.
.SH EXIT STATUS
The mnemonic-diff mode reports through its output and exits 0; a failed
probe or an unknown architecture exits non-zero.
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-diff \- compare the machine code of two assembly files
.SH SYNOPSIS
.B gasm diff [\-GOARCH arch] <file1.s> <file2.s>
.B gasm diff [\-GOARCH arch] [\-I dir] <file1.s> <file2.s>
.SH DESCRIPTION
Compare the machine code produced by assembling two files. Shows which
functions differ and the byte-level differences. Useful for verifying
@@ -20,6 +20,10 @@ pairs two variants regardless of suffix.
Target architecture for both files: amd64, arm64, riscv64 or loong64;
overrides the file-name suffixes.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-\-map \fIspec\fR
Comma-separated old=new pairs to match functions with different names.
.SH EXIT STATUS
+44 -1
View File
@@ -119,7 +119,9 @@ func (l *Lexer) Next() token.Token {
// is a C-preprocessor line continuation (used by #define macros in the
// runtime .s files): splice the lines together by consuming both, so
// the whole macro becomes one logical line that the parser treats as an
// opaque preprocessor directive.
// opaque preprocessor directive. The backslash may also reach its
// newline across whitespace and a trailing comment ("…; \ // note\n"),
// which the toolchain's scanner skips the same way.
for {
c := l.cur()
if c == ' ' || c == '\t' || c == '\r' {
@@ -136,6 +138,16 @@ func (l *Lexer) Next() token.Token {
}
continue
}
if c == '\\' && l.continuationAhead() {
l.advance() // backslash, then the runes the scan saw
for !l.atEnd() && l.cur() != '\n' {
l.advance()
}
if !l.atEnd() {
l.advance() // the newline that closes the continuation
}
continue
}
break
}
@@ -185,6 +197,28 @@ func (l *Lexer) Next() token.Token {
}
}
// continuationAhead reports, without consuming anything, whether the
// backslash at the current position closes onto a newline through nothing
// but horizontal whitespace and one line comment. Positions after the
// backslash are inspected directly on the rune slice so a non-match leaves
// the scanner state untouched.
func (l *Lexer) continuationAhead() bool {
i := l.i + 1
for i < len(l.src) {
switch r := l.src[i]; {
case r == ' ' || r == '\t' || r == '\r':
i++
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
for i < len(l.src) && l.src[i] != '\n' {
i++
}
default:
return r == '\n'
}
}
return false
}
// lineComment consumes a // comment up to, but not including, the newline. A
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
// comment content, so it never enters the token text. Trimming only a \r
@@ -403,6 +437,15 @@ func (l *Lexer) punct(start token.Position) token.Token {
case '|':
l.advance()
return l.make(token.Pipe, start, "|")
case ';':
l.advance()
return l.make(token.Semicolon, start, ";")
case '&':
l.advance()
return l.make(token.Ampersand, start, "&")
case '~':
l.advance()
return l.make(token.Tilde, start, "~")
default:
// Unknown rune: emit it as Illegal and move on.
l.advance()
+146
View File
@@ -0,0 +1,146 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Constant-expression folding for operands. The toolchain's assembler
// evaluates arithmetic in every operand position, and macro-heavy GOROOT
// sources lean on it: parameterised bodies carry offsets like
// ((index*4)+0)(base), immediates like $(32-shift) and masks like
// $~63 or $(1<<0|1<<9). Substituting the parameters textually therefore
// leaves constant arithmetic behind, and the parser folds it here, keeping
// the operand AST identical to what the same literals written out would
// produce. Anything that is not a closed integer expression fails to fold
// and falls through to the ordinary operand paths.
package parser
import (
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// foldExpr evaluates the constant integer expression at the head of ts and
// returns its value together with the unconsumed tokens. ok is false when
// the tokens do not form an expression, which is the callers' signal to use
// the ordinary parsing paths.
func foldExpr(ts []token.Token) (val int64, rest []token.Token, ok bool) {
v, rest, ok := foldAdd(ts)
if !ok {
return 0, ts, false
}
return v, rest, true
}
// foldAdd parses addition-level expressions: +, - and | bind loosest, the
// Plan 9 convention that makes x<<1|3 read as (x<<1)|3.
func foldAdd(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldMul(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
kind := rest[0].Kind
if kind != token.Plus && kind != token.Minus && kind != token.Pipe {
return v, rest, true
}
w, r2, ok := foldMul(rest[1:])
if !ok {
return v, rest, true
}
switch kind {
case token.Plus:
v += w
case token.Minus:
v -= w
case token.Pipe:
v |= w
}
rest = r2
}
return v, rest, true
}
// foldMul parses multiplication-level expressions: *, / and the bit
// operators &, << and >>.
func foldMul(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldFactor(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
switch rest[0].Kind {
case token.Star:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v *= w
rest = r2
case token.Slash:
w, r2, ok := foldFactor(rest[1:])
if !ok || w == 0 {
return v, rest, true
}
v /= w
rest = r2
case token.Ampersand:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v &= w
rest = r2
case token.LShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v <<= uint(w)
rest = r2
case token.RShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v >>= uint(w)
rest = r2
default:
return v, rest, true
}
}
return v, rest, true
}
// foldFactor parses a number, a parenthesised expression, or a unary sign
// or complement.
func foldFactor(ts []token.Token) (int64, []token.Token, bool) {
if len(ts) == 0 {
return 0, ts, false
}
switch ts[0].Kind {
case token.Number:
v, ok := tryInt(ts[0].Text)
if !ok {
return 0, ts, false
}
return v, ts[1:], true
case token.LParen:
v, rest, ok := foldAdd(ts[1:])
if !ok || len(rest) == 0 || rest[0].Kind != token.RParen {
return 0, ts, false
}
return v, rest[1:], true
case token.Minus:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return -v, rest, true
case token.Plus:
return foldFactor(ts[1:])
case token.Tilde:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return ^v, rest, true
}
return 0, ts, false
}
+23
View File
@@ -408,6 +408,18 @@ func parseImmediate(g []token.Token) ast.Immediate {
return imm
}
}
// A constant expression introduced by '(' or '~'. Textual macro
// substitution leaves arithmetic such as $(32-shift) and $~63 behind,
// and the toolchain evaluates it in place; only shapes the ordinary
// paths below cannot read reach the folder, so every existing form
// keeps its exact parse.
if g[0].Kind == token.LParen || g[0].Kind == token.Tilde {
if v, rest, ok := foldExpr(g); ok && len(rest) == 0 {
imm.Val = v
imm.HasVal = true
return imm
}
}
i := 0
if g[i].Kind == token.Minus {
imm.Neg = true
@@ -459,6 +471,17 @@ func parseAddress(g []token.Token) ast.Address {
}
i := 0
// A parenthesised constant expression as the displacement: substituted
// macro bodies carry ((index*4)+0)(base) shapes. As with the signed
// number path below, the value is committed only when a base group
// follows.
if i < len(g) && g[i].Kind == token.LParen {
if v, rest, ok := foldExpr(g[i:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen {
addr.Offset = v
addr.HasOff = true
i = len(g) - len(rest)
}
}
// Optional leading displacement before a '(' base group. A sign pushes
// the parenthesis one token further out: -4(DX) has it at i+2.
if isSignedNumber(g, i) {
+475
View File
@@ -0,0 +1,475 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The preprocessor turns #define and #include directives into the token
// stream the parser really sees, the way the Go toolchain's assembler does:
// object and parameterised macros expand at the point of use, and an
// #include splices the named file's lines in place of the directive. The
// pass runs only on the assembly path (gasm asm, diff, the corpus audit),
// where the result is machine code; parsing for the linter, formatter and
// language server keeps the raw file so their view of #define lines, and
// therefore their macro-aware behaviour, is unchanged.
package parser
import (
"fmt"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Options controls the optional preprocessing applied before a file is
// parsed. The zero value reproduces Parse exactly.
type Options struct {
// IncludeDirs lists the -I directories searched for #include files,
// in order, after the including file's own directory.
IncludeDirs []string
// Expand enables macro expansion, include splicing and the
// statement-separator reading of ';' that the expanded bodies rely on.
Expand bool
}
// ParseWithOptions parses src like Parse, optionally preprocessing it first.
// The returned file is usable even when errors is non-empty.
func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) {
tokens := lexer.Tokenize(src)
var lines [][]token.Token
var errs []error
if opts.Expand {
pp := &preproc{opts: opts, macros: map[string]*macroDef{}}
lines = pp.fileLines(path, tokens, token.Position{})
errs = pp.errs
} else {
lines = splitLines(tokens)
}
p := &state{path: path}
p.parse(lines)
return p.file, append(errs, p.errs...)
}
// maxExpansionDepth bounds recursive macro expansion; the toolchain's
// assembler gives up after 100 nested invocations without producing a token.
const maxExpansionDepth = 100
// textflagHeader names the one header gasm does not splice: its flag macros
// (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser,
// encoders and linter, and expanding them to their numeric constants would
// leave every consumer blind to them.
const textflagHeader = "textflag.h"
// macroDef is one #define. A nil args slice is an object macro; a non-nil
// (possibly empty) one is parameterised, the C distinction between
// "#define A(x)" and "#define A (x)".
type macroDef struct {
name string
args []string
body []token.Token
}
// preproc carries the state of one expansion pass: the live macro table, the
// chain of files currently being read, for cycle detection, and the
// conditional-inclusion stack of #ifdef regions.
type preproc struct {
opts Options
macros map[string]*macroDef
errs []error
stack []string // absolute paths of files being read, innermost last
ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth
}
// enabled reports whether the position being read is inside a live
// conditional branch. Directives inside a disabled branch contribute
// nothing, and its content lines are dropped, exactly as the toolchain's
// input stack does.
func (pp *preproc) enabled() bool {
return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1]
}
func (pp *preproc) errorf(pos token.Position, format string, args ...any) {
pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
}
// fileLines tokenizes and preprocesses one file into logical lines.
// Directive lines are kept (the parser records them for the tooling);
// #include lines are replaced by the included file's lines. includePos is
// the position of the #include that pulled this file in, zero for the
// top-level file, and only serves cycle diagnostics.
func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token {
abs, err := filepath.Abs(path)
if err != nil {
abs = filepath.Clean(path)
}
if slices.Contains(pp.stack, abs) {
if includePos.IsValid() {
pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path))
}
return nil
}
pp.stack = append(pp.stack, abs)
var out [][]token.Token
for _, line := range splitLines(tokens) {
if len(line) == 0 {
out = append(out, line)
continue
}
if line[0].Kind == token.Hash {
out = append(out, pp.directive(line, filepath.Dir(path))...)
continue
}
if !pp.enabled() {
continue
}
out = append(out, splitOnSemicolons(pp.expandTokens(line))...)
}
pp.stack = pp.stack[:len(pp.stack)-1]
if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 {
// The stack is per-input, shared across includes, so only the
// top-level file's end can decide the input was left unclosed.
pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef")
}
return out
}
// directive processes one '#' line and returns the lines to keep in the
// stream: every directive line is kept as-is for the parser (which records
// it), except #include, which is replaced by the spliced content.
// Conditionals are tracked on every line; every other directive is inert
// inside a disabled branch.
func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token {
if len(line) < 2 || line[1].Kind != token.Ident {
return [][]token.Token{line}
}
switch line[1].Text {
case "ifdef", "ifndef":
pp.ifdef(line, line[1].Text == "ifndef")
case "else":
pp.elseBranch(line)
case "endif":
pp.endif(line)
case "define":
if pp.enabled() {
pp.define(line)
}
case "undef":
if pp.enabled() {
pp.undef(line)
}
case "include":
if pp.enabled() {
return pp.include(line, dir)
}
default:
// #line and unknown directives are recorded but not interpreted:
// conservative support keeps the parser's view intact and files
// using them fail on their content, not silently.
}
return [][]token.Token{line}
}
// ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth
// onto the conditional stack. A branch opened inside a disabled region is
// itself disabled, however the name resolves.
func (pp *preproc) ifdef(line []token.Token, inverted bool) {
truth := false
if len(line) >= 3 && line[2].Kind == token.Ident {
_, defined := pp.macros[line[2].Text]
truth = defined != inverted
} else {
pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text)
}
if !pp.enabled() {
truth = false
}
pp.ifdefStack = append(pp.ifdefStack, truth)
}
// elseBranch flips the innermost conditional's truth, but only when the
// region enclosing it is itself live: the toolchain keeps outer overrides.
func (pp *preproc) elseBranch(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #else")
return
}
if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] {
pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1]
}
}
// endif closes the innermost conditional.
func (pp *preproc) endif(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #endif")
return
}
pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1]
}
// define parses "#define NAME[(formals)] body" into the macro table. The
// body runs to the end of the logical line (the lexer has already spliced
// backslash continuations) and stops at a comment, which never expands.
func (pp *preproc) define(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
name := line[2]
args := []string(nil)
body := line[3:]
// The definition is parameterised only when '(' follows the name
// directly; the toolchain separates "#define A(x)" from
// "#define A (x)" by adjacency, and so does the column check here.
if len(body) > 0 && body[0].Kind == token.LParen &&
body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) {
args = []string{}
i := 1
for i < len(body) && body[i].Kind != token.RParen {
if body[i].Kind == token.Ident {
args = append(args, body[i].Text)
}
i++
}
if i < len(body) {
body = body[i+1:]
} else {
body = nil
}
}
if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 {
body = body[:i]
}
if _, exists := pp.macros[name.Text]; exists {
// The toolchain refuses redefinition, so a file the oracle accepts
// never redefines; failing here keeps that contract visible.
pp.errorf(name.Pos, "redefinition of macro %s", name.Text)
}
pp.macros[name.Text] = &macroDef{name: name.Text, args: args, body: pp.bodyWithBreaks(body)}
}
// bodyWithBreaks records the statement boundaries the continuations carry.
// The lexer splices backslash-continued lines into one logical line, but the
// toolchain keeps the newline as a token in the stored body, which is how a
// multi-instruction body without semicolons (the arm64 style) still splits
// into statements on expansion. A line change inside the logical line is
// exactly a continuation, so the boundary is restored from the positions.
func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token {
out := make([]token.Token, 0, len(body))
for i, t := range body {
if i > 0 && t.Pos.Line != body[i-1].Pos.Line {
out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos})
}
out = append(out, t)
}
return out
}
// undef handles "#undef NAME", which the toolchain honours and requires to
// name a defined macro.
func (pp *preproc) undef(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
if _, ok := pp.macros[line[2].Text]; !ok {
pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text)
return
}
delete(pp.macros, line[2].Text)
}
// include resolves and splices "#include \"file\"". A header that cannot be
// read keeps the directive line in the stream, with a diagnostic.
func (pp *preproc) include(line []token.Token, dir string) [][]token.Token {
if len(line) < 3 || line[2].Kind != token.String {
return [][]token.Token{line}
}
header := line[2]
name, err := strconv.Unquote(header.Text)
if err != nil {
pp.errorf(header.Pos, "unquoting include file name: %v", err)
return [][]token.Token{line}
}
if filepath.Base(name) == textflagHeader {
// Flag macros are handled natively (see textflagHeader); the
// directive stays so tools still see the include.
return [][]token.Token{line}
}
resolved, ok := pp.resolve(name, dir)
if !ok {
searched := append([]string{dir}, pp.opts.IncludeDirs...)
pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", "))
return [][]token.Token{line}
}
src, err := os.ReadFile(resolved)
if err != nil {
pp.errorf(header.Pos, "#include %q: %v", name, err)
return [][]token.Token{line}
}
return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos)
}
// resolve looks an include name up the way the toolchain does: as written
// (relative to the working directory), then relative to the including
// file's directory, then in each -I directory in order.
func (pp *preproc) resolve(name, dir string) (string, bool) {
candidates := []string{name}
if !filepath.IsAbs(name) {
candidates = append(candidates, filepath.Join(dir, name))
for _, d := range pp.opts.IncludeDirs {
candidates = append(candidates, filepath.Join(d, name))
}
}
for _, c := range candidates {
if st, err := os.Stat(c); err == nil && !st.IsDir() {
return c, true
}
}
return "", false
}
// expandTokens expands every macro invocation in a token sequence,
// recursively, with a depth guard. A body is spliced into the sequence in
// place and rescanned, the way the toolchain's input stack re-reads pushed
// tokens: an object macro may name a parameterised one, and the argument
// list of the expansion may then come from the tokens that follow.
func (pp *preproc) expandTokens(in []token.Token) []token.Token {
s := in
i := 0
consecutive := 0
for i < len(s) {
t := s[i]
if t.Kind != token.Ident {
i++
consecutive = 0
continue
}
def := pp.macros[t.Text]
if def == nil {
i++
consecutive = 0
continue
}
// The guard mirrors the toolchain's: 100 nested invocations in a
// row without a plain token between them means recursion.
consecutive++
if consecutive > maxExpansionDepth {
pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth)
return nil
}
if def.args == nil {
s = append(s[:i], append(restamp(def.body, t.Pos), s[i+1:]...)...)
continue
}
// A parameterised macro invoked without its parentheses stands
// unexpanded, naming itself, as in the toolchain.
if i+1 >= len(s) || s[i+1].Kind != token.LParen {
i++
consecutive = 0
continue
}
args, next := pp.collectArgs(s, i+1, t)
if args == nil {
return nil
}
// A zero-argument macro may be invoked as NAME().
if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 {
args = nil
}
if len(args) != len(def.args) {
pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args))
i = next
consecutive = 0
continue
}
sub := make([]token.Token, 0, len(def.body))
for _, bt := range def.body {
if bt.Kind == token.Ident {
if k := slices.Index(def.args, bt.Text); k >= 0 {
sub = append(sub, restamp(args[k], t.Pos)...)
continue
}
}
sub = append(sub, bt)
}
s = append(s[:i], append(sub, s[next:]...)...)
}
return s
}
// collectArgs reads the actual argument tokens of an invocation; the opening
// parenthesis is at start. Commas separate arguments except inside nested
// parentheses. A nil result means the list was unterminated, which is a
// diagnostic.
func (pp *preproc) collectArgs(in []token.Token, start int, name token.Token) ([][]token.Token, int) {
var args [][]token.Token
var cur []token.Token
nesting := 0
for i := start + 1; i < len(in); i++ {
t := in[i]
switch t.Kind {
case token.LParen:
nesting++
cur = append(cur, t)
case token.RParen:
if nesting == 0 {
return append(args, cur), i + 1
}
nesting--
cur = append(cur, t)
case token.Comma:
if nesting == 0 {
args = append(args, cur)
cur = nil
continue
}
cur = append(cur, t)
case token.Comment:
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, i
default:
cur = append(cur, t)
}
}
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, len(in)
}
// restamp copies body tokens to the invocation's position, so diagnostics
// and the line table point where the macro was used, as the toolchain's
// input stack does.
func restamp(body []token.Token, pos token.Position) []token.Token {
out := make([]token.Token, len(body))
for i, t := range body {
t.Pos, t.End = pos, pos
out[i] = t
}
return out
}
// splitOnSemicolons breaks a token sequence at ';' statement separators and
// at the Newline markers that record continuation boundaries inside macro
// bodies, producing the logical lines the parser expects. The separators
// carry no meaning beyond the break, so the pieces are exactly what the same
// statements on separate lines would produce.
func splitOnSemicolons(ts []token.Token) [][]token.Token {
var out [][]token.Token
start := 0
for i, t := range ts {
if t.Kind == token.Semicolon || t.Kind == token.Newline {
if i > start {
out = append(out, ts[start:i])
}
start = i + 1
}
}
if start < len(ts) {
out = append(out, ts[start:])
}
return out
}
+552
View File
@@ -0,0 +1,552 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// expand parses src with preprocessing enabled and returns the first TEXT's
// body instructions as "MNEMONIC operand|operand" strings, the shape the
// expansion assertions below compare against. Runs of spaces are
// collapsed: Raw renders a token group as its tokens joined with single
// spaces, so "$(32-7)" arrives as "$ ( 32 - 7 )" and the comparison must
// not depend on that spelling.
func expand(t *testing.T, src string) (*ast.File, []string) {
t.Helper()
f, errs := ParseWithOptions("t_amd64.s", src, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
ts := texts(f)
if len(ts) == 0 {
t.Fatalf("no TEXT in:\n%s", src)
}
var got []string
for _, s := range ts[0].Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
var ops []string
for _, op := range in.Operands {
ops = append(ops, op.Raw)
}
line := in.Mnemonic.Text + " " + strings.Join(ops, ", ")
got = append(got, strings.ReplaceAll(line, " ", ""))
}
return f, got
}
func wantLines(t *testing.T, got []string, want ...string) {
t.Helper()
strip := func(lines []string) string {
var out []string
for _, l := range lines {
out = append(out, strings.ReplaceAll(l, " ", ""))
}
return strings.Join(out, "\n")
}
if strip(got) != strip(want) {
t.Errorf("expanded body:\n %s\nwant:\n %s", strings.Join(got, "\n "), strings.Join(want, "\n "))
}
}
func TestObjectMacroExpandsAtUse(t *testing.T) {
_, got := expand(t, `
#define REGTMP CX
#define TWICE ADDQ CX, AX; ADDQ CX, AX
TEXT ·f(SB), NOSPLIT, $0
MOVQ 8(SP), REGTMP
TWICE
RET
`)
wantLines(t, got,
"MOVQ 8(SP), CX",
"ADDQ CX, AX",
"ADDQ CX, AX",
"RET",
)
}
func TestParameterisedMacroSubstitutesArguments(t *testing.T) {
f, errs := ParseWithOptions("t_amd64.s", `
#define ROUND1(a, index, const, shift) \
ADDQ $const, a; \
MOVW (index*4)(SP), a; \
RORQ $(32-shift), a
TEXT ·f(SB), NOSPLIT, $0
ROUND1(AX, 3, 0xd76aa478, 7)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
add := body[0].(*ast.Instr)
if add.Mnemonic.Text != "ADDQ" || !add.Operands[0].Imm.HasVal ||
add.Operands[0].Imm.Val != 0xd76aa478 || add.Operands[1].Addr.Sym == nil ||
add.Operands[1].Addr.Sym.Name != "AX" {
t.Errorf("ADDQ operands substituted wrong: %+v %+v", add.Operands[0].Imm, add.Operands[1].Addr)
}
mov := body[1].(*ast.Instr)
if addr := mov.Operands[0].Addr; !addr.HasOff || addr.Offset != 12 {
t.Errorf("MOVW offset = %+v, want 12 from 3*4", addr)
}
ror := body[2].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25 from (32-7)", ror.Operands[0].Imm)
}
}
func TestMacroArgumentsKeepCommasInParens(t *testing.T) {
// An argument may itself be an unparenthesised expression: the tokens
// substitute verbatim and the parser folds the result, as the
// toolchain's parser does.
f, errs := ParseWithOptions("t_amd64.s", `
#define LOAD(dst, off) MOVQ off(SP), dst
TEXT ·f(SB), NOSPLIT, $0
LOAD(AX, 1*8)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
addr := in.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 {
t.Errorf("offset = %+v, want 8", addr)
}
if sym := in.Operands[1].Addr.Sym; sym == nil || sym.Name != "AX" {
t.Errorf("destination = %+v, want AX", in.Operands[1].Addr)
}
}
func TestNestedMacroInvocations(t *testing.T) {
// An object macro naming a parameterised one, and a parameterised body
// invoking another parameterised macro: the toolchain's input stack
// rescans substituted tokens, and so does expansion here.
_, got := expand(t, `
#define DOUBLE(x) ADDQ x, x
#define TWICE2 DOUBLE
#define FOUR(a, b) DOUBLE(a); DOUBLE(b)
TEXT ·f(SB), NOSPLIT, $0
TWICE2(AX)
FOUR(AX, CX)
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestMultiLineBodySplitsWithoutSemicolons(t *testing.T) {
// The arm64 style: backslash-continued lines with no semicolons. The
// continuation newline is a statement boundary, as in the toolchain.
_, got := expand(t, `
#define PAIR \
ADDQ AX, AX \
MOVQ AX, CX
TEXT ·f(SB), NOSPLIT, $0
PAIR
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"MOVQ AX, CX",
"RET",
)
}
func TestZeroArgumentMacro(t *testing.T) {
_, got := expand(t, `
#define BARRIER()
TEXT ·f(SB), NOSPLIT, $0
BARRIER()
RET
`)
wantLines(t, got, "RET")
}
func TestParameterisedWithoutParensStandsAsName(t *testing.T) {
// A parameterised macro invoked without its parentheses names itself,
// which the parser then reports as an unknown instruction rather than
// silently expanding nothing.
f, errs := ParseWithOptions("t_amd64.s", `
#define M(x) ADDQ x, x
TEXT ·f(SB), NOSPLIT, $0
M
RET
`, Options{Expand: true})
if len(errs) != 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
if len(fn.Body) == 0 {
t.Fatal("body empty")
}
in, ok := fn.Body[0].(*ast.Instr)
if !ok || in.Mnemonic.Text != "M" {
t.Fatalf("bare parameterised macro did not stand as its name: %+v", fn.Body[0])
}
}
func TestDefinitionScoping(t *testing.T) {
// A definition applies from its point onward: the use before the
// #define stays untouched.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
SPECIAL
#define SPECIAL ADDQ AX, AX
SPECIAL
RET
`)
wantLines(t, got,
"SPECIAL",
"ADDQ AX, AX",
"RET",
)
}
func TestUndefRemovesMacro(t *testing.T) {
_, got := expand(t, `
#define TEMP AX
TEXT ·f(SB), NOSPLIT, $0
TEMP
#undef TEMP
TEMP
RET
`)
wantLines(t, got,
"AX",
"TEMP",
"RET",
)
}
func TestUndefUndefinedMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#undef NOSUCH\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "undefined macro NOSUCH") {
t.Fatalf("#undef of an undefined macro: got %v, want an error naming it", errs)
}
}
func TestRedefinitionIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A X\n#define A Y\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "redefinition of macro A") {
t.Fatalf("redefinition: got %v, want an error", errs)
}
}
func TestRecursiveMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A B\n#define B A\nTEXT ·f(SB), NOSPLIT, $0\n\tA\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "recursive macro invocation") {
t.Fatalf("recursion: got %v, want a recursive-macro error, not a hang", errs)
}
}
func TestWrongArgumentCountIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define M(a, b) ADDQ a, b\nTEXT ·f(SB), NOSPLIT, $0\n\tM(AX)\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "wrong arg count for macro M") {
t.Fatalf("arg count: got %v, want an error", errs)
}
}
func TestConditionalsSelectOneBranch(t *testing.T) {
_, got := expand(t, `
#define MODE2
TEXT ·f(SB), NOSPLIT, $0
#ifdef MODE2
ADDQ AX, AX
#else
SUBQ AX, AX
#endif
#ifndef MODE2
SUBQ CX, CX
#else
ADDQ CX, CX
#endif
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestConditionalsHideDefinitionsAndIncludes(t *testing.T) {
// A definition inside a disabled branch must not exist, and an
// unresolvable include there must not be followed.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
#ifdef NOTDEFINED
#define HIDEN ADDQ AX, AX
#include "nowhere.h"
#endif
HIDEN
RET
`)
wantLines(t, got, "HIDEN", "RET")
}
func TestUnclosedConditionalIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#ifdef X\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unclosed #ifdef") {
t.Fatalf("unclosed conditional: got %v, want an error", errs)
}
}
func TestUnmatchedConditionalDelimitersAreErrors(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#endif\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #endif") {
t.Fatalf("unmatched #endif: got %v, want an error", errs)
}
_, errs = ParseWithOptions("t_amd64.s", "#else\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #else") {
t.Fatalf("unmatched #else: got %v, want an error", errs)
}
}
// includeTree writes a directory of include files and returns its path.
func includeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
func TestIncludeSplicesAndDefinesAreShared(t *testing.T) {
dir := includeTree(t, map[string]string{
"consts.h": "#define KONST $42\n",
})
f, errs := ParseWithOptions("t_amd64.s", `
#include "consts.h"
TEXT ·f(SB), NOSPLIT, $0
MOVQ KONST, AX
RET
`, Options{Expand: true, IncludeDirs: []string{dir}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if in.Mnemonic.Text != "MOVQ" || strings.ReplaceAll(in.Operands[0].Raw, " ", "") != "$42" {
t.Fatalf("include splicing failed: %+v", in)
}
}
func TestIncludeResolutionOrder(t *testing.T) {
// The including file's directory wins over the -I list, and the -I list
// is searched in order.
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"inc/which.h": "#define WHO ONE\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
// The header exists in the including file's directory and in two -I
// directories; the source-directory copy must win.
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
found := false
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO ONE") {
found = true
}
}
if !found {
t.Error("the including file's directory did not win include resolution")
}
}
func TestIncludeSearchesIncludeDirsInOrder(t *testing.T) {
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO THREE") {
t.Error("the second -I directory was searched before the first")
}
}
}
func TestIncludeCycleIsDetected(t *testing.T) {
src := includeTree(t, map[string]string{
"a.s": "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"b.s": "#include \"a.s\"\n",
})
_, errs := ParseWithOptions(filepath.Join(src, "a.s"), "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "include cycle") {
t.Fatalf("include cycle: got %v, want a cycle diagnostic, not a hang", errs)
}
}
func TestUnresolvableIncludeIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true, IncludeDirs: []string{t.TempDir()}})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), `#include "nothere.h"`) {
t.Fatalf("missing include: got %v, want a clear diagnostic", errs)
}
}
func TestTextflagHeaderIsNeverSpliced(t *testing.T) {
// textflag.h resolves nowhere here, yet the file must parse: the flag
// names are consumed natively and the include stays in the tree.
f, errs := ParseWithOptions("t_amd64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
hasInclude := false
for _, d := range f.Decls {
if _, ok := d.(*ast.Include); ok {
hasInclude = true
}
}
if !hasInclude {
t.Error("textflag.h include was dropped from the tree")
}
}
func TestSemicolonSplitsRawLinesToo(t *testing.T) {
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
BYTE $0x0f; BYTE $0x1f
RET
`)
wantLines(t, got, "BYTE $0x0f", "BYTE $0x1f", "RET")
}
func TestParseUnchangedWithoutExpand(t *testing.T) {
// Without Expand the preprocessor must not exist: a macro invocation
// stays an unexpanded instruction line and ';' keeps the old parse.
f, errs := Parse("t_amd64.s", `
#define TWICE ADDQ AX, AX
TEXT ·f(SB), NOSPLIT, $0
TWICE
BYTE $0x0f; BYTE $0x1f
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
var mnemonics []string
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
mnemonics = append(mnemonics, in.Mnemonic.Text)
}
}
if strings.Join(mnemonics, " ") != "TWICE BYTE RET" {
t.Errorf("non-expanding parse changed: %v", mnemonics)
}
}
func TestConstantExpressionFolding(t *testing.T) {
// The shapes substituted macro bodies leave behind: parenthesised
// arithmetic in immediates and displacements, tilde complements. The
// assertions read the semantic fields; Raw keeps the operand's tokens
// in the canonicalised rendering, not the folded values.
f, errs := ParseWithOptions("t_amd64.s", `
TEXT ·f(SB), NOSPLIT, $0
RORQ $(32-7), AX
ANDQ $~63, AX
MOVQ ((2*4)+0)(SP), AX
MOVQ $((1<<3)|(1<<1)), AX
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
ror := body[0].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25", ror.Operands[0].Imm)
}
and := body[1].(*ast.Instr)
if !and.Operands[0].Imm.HasVal || and.Operands[0].Imm.Val != -64 {
t.Errorf("ANDQ immediate = %+v, want -64", and.Operands[0].Imm)
}
mov := body[2].(*ast.Instr)
addr := mov.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 || addr.Base != "SP" {
t.Errorf("MOVQ address = %+v, want 8(SP)", addr)
}
mov2 := body[3].(*ast.Instr)
if !mov2.Operands[0].Imm.HasVal || mov2.Operands[0].Imm.Val != 10 {
t.Errorf("MOVQ immediate = %+v, want 10", mov2.Operands[0].Imm)
}
}
func TestConstantExpressionFoldsWithoutExpand(t *testing.T) {
// Folding is a parser capability, not a preprocessing one: a
// hand-written $(32-7) folds the same way with expansion off.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tRORQ $(32-7), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if !in.Operands[0].Imm.HasVal || in.Operands[0].Imm.Val != 25 {
t.Errorf("Imm = %+v, want 25", in.Operands[0].Imm)
}
}
func TestNotAnExpressionFallsBack(t *testing.T) {
// Symbol immediates and floats must keep their ordinary parse.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ $1.5, AX\n\tMOVQ $·sym(SB), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
mov1 := fn.Body[0].(*ast.Instr)
if mov1.Operands[0].Imm.HasVal || mov1.Operands[0].Imm.Float != "1.5" {
t.Errorf("float immediate parsed as %+v", mov1.Operands[0].Imm)
}
mov2 := fn.Body[1].(*ast.Instr)
if mov2.Operands[0].Imm.Sym == nil {
t.Errorf("symbol immediate parsed as %+v", mov2.Operands[0].Imm)
}
}
File diff suppressed because it is too large Load Diff
+48
View File
@@ -0,0 +1,48 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Carry arithmetic, logical shifts, register aliases with element selectors
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
#include "textflag.h"
#define acc0 V8
#define acc1 V9
#define const0 R15
#define POLY V15
// carry pins the ADC/SBC family: the $0 spellings in two and three
// operands, and the register-carry forms.
TEXT ·carry(SB), NOSPLIT, $0-0
ADC $0, R20
ADC $0, R20, R4
SBCS $0, R4
SBCS $0, R4, R12
SBCS R15, R4, R12
SBC $0, R1
ADCSW $0, R2, R3
RET
// shift pins the shifted-register forms including ROR, which only the
// logical family accepts.
TEXT ·shift(SB), NOSPLIT, $0-0
ANDW R9@>7, R19, R26
AND R1@>33, R2, R3
ADD R1<<11, R2, R3
SUB R1->33, R2
ORR R5<<2, R6, R7
RET
// vecalias pins the vector aliases with element selectors and the
// structure loads with aliased members.
TEXT ·vecalias(SB), NOSPLIT, $0-0
MOVD $0xC2, R1
VMOV R1, POLY.D[0]
VMOV R0, POLY.D[1]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [acc0.B16]
VLD1.P (R0), [acc0.B16, acc1.B16]
VST1 [acc0.B16, acc1.B16], (R1)
VST1.P [acc0.B16, acc1.B16], 32(R1)
RET
+51
View File
@@ -0,0 +1,51 @@
// The subtract-immediate fold, the TEQ/TNE trap pseudos, PRELDX, the FP
// condition branches and the N(PC) branch spellings, against the toolchain.
#include "textflag.h"
// func SubFold(x int64) int64
TEXT ·SubFold(SB), NOSPLIT, $0-16
MOVV x+0(FP), R8
SUBV $0, R8
SUBV $4, R9, R10
SUBV $4096, R11
SUBV $-4, R12
SUB $1, R13
SUBVU $4, R14
SUBV $1048576, R15
MOVV R8, ret+8(FP)
RET
// func Traps(x int64) int64
TEXT ·Traps(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
TEQ $4, R4, R5
TEQ $4, R4
TNE $6, R5, R6
MOVV R4, ret+8(FP)
RET
// func Prefetch(x int64) int64
TEXT ·Prefetch(SB), NOSPLIT, $0-16
MOVV x+0(FP), R7
PRELDX 0(R7), $0x80001021, $0
PRELDX -1(R7), $0x1021, $2
MOVV R7, ret+8(FP)
RET
// func BranchForms(x int64) int64
TEXT ·BranchForms(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
l1:
BFPT l1
BFPT FCC3, l1
BFPF l1
JMP -4(PC)
JAL 1(PC)
JAL (R4)
loop:
ADDV $1, R4
BEQ R4, R5, loop
BNE R4, l1
RET
+33
View File
@@ -0,0 +1,33 @@
// PCALIGN padding on loong64: andi $0, $0, 0 (the architecture's NOP), plus
// the automatic loop-head alignment to a 16-byte boundary.
#include "textflag.h"
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $16
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $32
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func LoopAlign(x int64) int64
TEXT ·LoopAlign(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
MOVV $10, R5
loop:
BEQ R4, R5, done
ADDV $1, R4
JMP loop
done:
MOVV R4, ret+8(FP)
RET
+35
View File
@@ -0,0 +1,35 @@
// PCALIGN padding on riscv64: 4-byte NOPs with a 2-byte compressed NOP when
// the pad is 2 mod 4, exactly as the toolchain lays the bytes down.
#include "textflag.h"
// func Pad8(x int64) int64
TEXT ·Pad8(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $16
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $32
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func PadAfterOdd(x int64) int64
TEXT ·PadAfterOdd(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
+103
View File
@@ -0,0 +1,103 @@
// Instruction prefixes: LOCK, REP and REPN. go tool asm encodes each
// statement as a standalone one-byte instruction with a PC of its own (F0,
// F3 and F2 respectively); the statement that follows is encoded unaware of
// it, and nothing validates the pairing. The shapes are the runtime's
// atomic read-modify-write family and the string moves, every result folded
// back.
#include "textflag.h"
// func cas64(ptr *uint64, old, new uint64) bool
TEXT ·cas64(SB), NOSPLIT, $0-25
MOVQ ptr+0(FP), BX
MOVQ old+8(FP), AX
MOVQ new+16(FP), CX
LOCK
CMPXCHGQ CX, 0(BX)
SETEQ ret+24(FP)
RET
// func casloop(addr *uint64, v uint64) uint64
// The runtime's Or64 shape: a LOCK inside a branch loop, the backward jump
// measuring over the prefix statement's own byte.
TEXT ·casloop(SB), NOSPLIT, $0-24
MOVQ addr+0(FP), BX
MOVQ v+8(FP), CX
loop:
MOVQ CX, DX
MOVQ (BX), AX
ORQ AX, DX
LOCK
CMPXCHGQ DX, (BX)
JNZ loop
MOVQ AX, ret+16(FP)
RET
// func xadd64(p *uint64, v uint64) uint64
TEXT ·xadd64(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
LOCK
XADDQ BX, (AX)
MOVQ AX, ret+16(FP)
RET
// func xaddw(p *uint16, v uint16) uint16
TEXT ·xaddw(SB), NOSPLIT, $0-12
MOVQ p+0(FP), AX
MOVW v+8(FP), BX
LOCK
XADDW BX, (AX)
MOVW AX, ret+8(FP)
RET
// func lockarith(p *uint64)
TEXT ·lockarith(SB), NOSPLIT, $0-8
MOVQ p+0(FP), AX
LOCK
ORQ CX, (AX)
LOCK
ANDL CX, (AX)
LOCK
INCQ (AX)
LOCK
DECQ (AX)
LOCK
ORB BX, (AX)
RET
// func repstring(dst, src *byte, n int)
// The memmove shapes: forward copy by quadwords, backward tails.
TEXT ·repstring(SB), NOSPLIT, $0-24
MOVQ dst+0(FP), DI
MOVQ src+8(FP), SI
REP
MOVSQ
REP
MOVSB
REPN
MOVSB
REP
STOSQ
REP
STOSB
RET
// func pfxlabel()
// Labels pinned on prefix statements' own bytes: pfx: sits on the LOCK,
// mid: on the REPN.
TEXT ·pfxlabel(SB), NOSPLIT, $0-0
pfx:
LOCK
XCHGL BX, (AX)
JMP done
mid:
REPN
MOVSB
done:
REP
STOSB
RET
+62
View File
@@ -0,0 +1,62 @@
// Literal data emission: BYTE, WORD, LONG and QUAD write the immediate
// into the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked; END is
// accepted and ignored, contributing no bytes and ending nothing. The
// shapes mirror the runtime's hand-laid markers
// (crypto/internal/boring/sig/sig_amd64.s) and its syscall stubs
// (runtime/sys_linux_amd64.s).
#include "textflag.h"
// func marker()
// A boring/crypto-style marker: a hand-laid forward branch whose skip
// distance is patched at runtime. One BYTE per statement, as the
// runtime's own file spells it: the semicolon-separated one-liner the
// sys_linux_amd64.s stub uses does not survive gasm fmt, which drops the
// statement separators.
TEXT ·marker(SB), NOSPLIT, $0-0
BYTE $0xEB
BYTE $0x1D
BYTE $0xF4
BYTE $0x48
BYTE $0xF4
BYTE $0x4B
BYTE $0xC3
RET
// func stub()
// The sys_linux_amd64.s stub bytes: the sign-extended
// "48 c7 c0 0f 00 00 00" form of MOVQ $rt_sigreturn, AX.
TEXT ·stub(SB), NOSPLIT, $0-0
BYTE $0x48
BYTE $0xc7
BYTE $0xc0
BYTE $0x0f
BYTE $0x00
BYTE $0x00
BYTE $0x00
RET
// func words()
// The wider literals, and an END that ends nothing: the WORD after it
// still lands in this function.
TEXT ·words(SB), NOSPLIT, $0-0
WORD $0x1234
WORD $-1
LONG $0x11223344
LONG $-1
QUAD $0x1122334455667788
QUAD $-2
END
WORD $0xBEEF
RET
// func trunc()
// Truncation, not a range check: each literal keeps its low bytes, exactly
// as go tool asm emits them.
TEXT ·trunc(SB), NOSPLIT, $0-0
BYTE $0x1FF
WORD $0x12345
LONG $0x123456789
QUAD $-2
RET
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Wide-immediate arithmetic: every classification band of the ADD/SUB
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
// their materialised fallback. Byte-for-byte against go tool asm.
#include "textflag.h"
// imm12 covers the plain and shifted-by-12 imm12 forms.
TEXT ·imm12(SB), NOSPLIT, $0-0
ADD $1, R2, R3
ADD $0x000aaa, R2, R3
ADD $0xaaa000, R2
SUB $0x000aaa, R2, R3
SUB $0xaaa000, R2
ADDW $40960, R0
CMP $40960, R0
CMPW $40960, R0
RET
// split pins the ADDCON2 band: two imm12 instructions, low half first.
TEXT ·split(SB), NOSPLIT, $0-0
ADD $0xaaaaaa, R2, R3
SUB $0xaaaaaa, R2
ADD $0x186a0, R2, R5
SUB $0x186a0, R2, R3
ADDW $0x60060, R2
RET
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
// MOVN for the complement form, the bitmask ORR otherwise.
TEXT ·regtmp(SB), NOSPLIT, $0-0
ADD $0x1ffe00, R2, R3
ADD $0x3fffffffc000, R5
ADD $-2048, R2, R3
ADD $-100000, R2, R3
CMP $0x1000000, R2
CMP $0x100000000, R0
SUB $-0x100000000, R0, R1
RET
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
// compare forms that never split.
TEXT ·movseq(SB), NOSPLIT, $0-0
ADD $0x12345678, R2, R3
SUB $0xe7791f700, R3, R1
CMP $0xaaaaaa, R2
CMP $0xffffffffffa0, R3
CMPW $27745, R2
CMPW $0x60060, R2
ADDS $0xaaaaaa, R2, R3
CMN $0x1000000, R2
ADDW $0x12345678, R2, R3
RET
// logical covers the bitmask immediates of the logical family and the
// materialised fallback for the values a bitmask cannot carry.
TEXT ·logical(SB), NOSPLIT, $0-0
AND $0x3ff00000, R2, R3
BIC $0x22220000, R3, R4
ORR $0x3ff00000, R2
EOR $0x3ff00000, R2, R3
ANDS $0x3ff00000, R2
ORNW $0x3ff00000, R2
EONW $0x3ff00000, R2
BICSW $0x6006000060060, R5
TST $0x4900000049, R0
RET
+11
View File
@@ -43,6 +43,13 @@ const (
At // @
Hash // #
Pipe // |
// Semicolon separates statements on one line (a Plan 9 statement
// terminator); Ampersand and Tilde are the expression operators & and ~
// of constant expressions. All three appear mostly inside macro bodies.
Semicolon // ;
Ampersand // &
Tilde // ~
)
var kindNames = map[Kind]string{
@@ -71,6 +78,10 @@ var kindNames = map[Kind]string{
At: "@",
Hash: "#",
Pipe: "|",
Semicolon: ";",
Ampersand: "&",
Tilde: "~",
}
// String returns a human-readable name for the kind.
+2
View File
@@ -32,6 +32,8 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/crypto_arm64.s",
"../testdata/verify/integer_arm64.s",
"../testdata/verify/simd_arm64.s",
"../testdata/verify/widenimm_arm64.s",
"../testdata/verify/carryshift_arm64.s",
"../testdata/verify/system_arm64.s",
} {
t.Run(path, func(t *testing.T) {
+4
View File
@@ -122,6 +122,10 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/crypto_amd64.s",
"../testdata/verify/sse_amd64.s",
"../testdata/verify/avx_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/doubleshift_amd64.s",
"../testdata/verify/ssestatic_amd64.s",
} {
+2
View File
@@ -29,6 +29,8 @@ func TestGroundTruthLOONG64(t *testing.T) {
"../testdata/verify/branchu_loong64.s",
"../testdata/verify/atomics_loong64.s",
"../testdata/verify/vector_loong64.s",
"../testdata/verify/pcalign_loong64.s",
"../testdata/verify/l64forms_loong64.s",
"trampoline_loong64.s",
} {
t.Run(path, func(t *testing.T) {
+2
View File
@@ -33,6 +33,8 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/atomics_riscv64.s",
"../testdata/verify/vector_riscv64.s",
"../testdata/verify/bitmanip_riscv64.s",
"../testdata/verify/pcalign_riscv64.s",
"../testdata/verify/branch_far_riscv64.s",
"trampoline_riscv64.s",
} {
t.Run(path, func(t *testing.T) {