Compare commits
6
Commits
0629f5e2df
...
97dfaa7526
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
97dfaa7526 | ||
|
|
66aa4dbc8b | ||
|
|
dce5d31462 | ||
|
|
9dc3987e02 | ||
|
|
9b238a525a | ||
|
|
ad82aac663 |
@@ -9,6 +9,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Added
|
||||
|
||||
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
|
||||
`gasm audit-instructions` now preprocess assembly the way the
|
||||
toolchain does: object and parameterised `#define` macros expand at
|
||||
the point of use, `#undef` and the `#ifdef`/`#ifndef`/`#else`/
|
||||
`#endif` family select branches, `#include` splices headers resolved
|
||||
through the source directory and the new repeatable `-I` flag, `;`
|
||||
separates statements, and constant expressions left in operands
|
||||
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
|
||||
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
|
||||
language server keep reading the raw file.
|
||||
- **The GOROOT instruction wave, part 1.** The encoder now covers the
|
||||
instruction families GOROOT's real code uses that gasm lacked,
|
||||
byte-verified against `go tool asm`: on amd64 the carry ALU, the
|
||||
@@ -129,6 +139,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Fixed
|
||||
|
||||
- **The corpus audit attempts fewer files that no build would compile.**
|
||||
Files named for Go ports gasm does not target (arm, 386, s390x, ...)
|
||||
are reported as other-port and never attempted, the headline rate is
|
||||
computed over attemptable files, and the audit searches the
|
||||
toolchain's shipped headers (funcdata.h and friends) automatically.
|
||||
- **riscv64 JALR silently jumped to the wrong register.** The trampoline
|
||||
form `JALR X0, 0(X5)` read the memory operand's base as the destination,
|
||||
encoding a jump to X0 with no diagnostic; the destination is the first
|
||||
|
||||
+566
-105
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/bits"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
||||
"FMOVS", "FMOVD":
|
||||
return arm64MovSize(mnem, ops, fi)
|
||||
case "ADD", "ADDW", "SUB", "SUBW":
|
||||
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
|
||||
"ADDS", "ADDSW", "SUBS", "SUBSW":
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
v := arm64Imm64(ops[0])
|
||||
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return 4
|
||||
// Size exactly as the encoder will emit: a single imm12 word, the
|
||||
// two-word ADDCON2 split, or a materialisation into REGTMP plus
|
||||
// the register form. Anything else would desynchronise the label
|
||||
// offsets of pass 1 from the bytes pass 2 lays down.
|
||||
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
||||
rn, rd := 0, 0
|
||||
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
|
||||
rd = n
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
return 4
|
||||
if len(ops) == 3 {
|
||||
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
|
||||
rn = n
|
||||
}
|
||||
// Larger immediates need MOV materialisation + op.
|
||||
return 8
|
||||
}
|
||||
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
|
||||
return 4 * len(ws)
|
||||
}
|
||||
}
|
||||
return 4
|
||||
}
|
||||
}
|
||||
return 4
|
||||
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64Bitfield(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
|
||||
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// EXTR.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
||||
return encodeARM64Extr(mnem, enc.op, ops)
|
||||
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||
// this check precedes the plain SIMD3 path below.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok {
|
||||
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
||||
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
||||
// an empty register-form spec.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
||||
return encodeARM64SimdV(mnem, spec, ops)
|
||||
}
|
||||
|
||||
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
}
|
||||
|
||||
// SIMD table lookup.
|
||||
if mnem == "VTBL" {
|
||||
return encodeARM64VTBL(ops)
|
||||
if mnem == "VTBL" || mnem == "VTBX" {
|
||||
return encodeARM64VTBL(mnem, ops)
|
||||
}
|
||||
|
||||
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
||||
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
}
|
||||
op := ops[0]
|
||||
|
||||
// Branch to the program counter itself: JMP (PC) spins forever. The
|
||||
// toolchain encodes it as an unconditional branch with a zero offset.
|
||||
// Branch to the program counter: JMP (PC) spins forever, and a spelled
|
||||
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
|
||||
// units. The toolchain encodes both as a plain branch of that offset.
|
||||
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
||||
if link {
|
||||
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
|
||||
rel := op.Addr.Offset
|
||||
if rel < -(1<<25) || rel >= (1<<25) {
|
||||
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
|
||||
}
|
||||
return a64wordLE(a64Branch(0, 0)), nil
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
||||
}
|
||||
|
||||
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
||||
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
|
||||
// the ADD/SUB-with-flags family an add/sub immediate.
|
||||
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
|
||||
mnem == "MVN" || mnem == "MVNW" ||
|
||||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
|
||||
|
||||
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
||||
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
||||
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
var logical bool
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
||||
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
|
||||
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
||||
"TST", "TSTW":
|
||||
logical = true
|
||||
}
|
||||
if logical {
|
||||
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
inverted := false
|
||||
switch mnem {
|
||||
case "BIC", "BICW", "BICS", "BICSW":
|
||||
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
||||
inverted = true
|
||||
}
|
||||
if inverted {
|
||||
@@ -700,8 +727,38 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
n, immr, imms, ok := a64LogicalImm(v, width)
|
||||
if !ok {
|
||||
// Beyond the bitmask immediates the toolchain materialises
|
||||
// the constant into REGTMP (R27) and uses the register form
|
||||
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
|
||||
// value, so the materialisation uses v before any inversion.
|
||||
written := v
|
||||
if inverted {
|
||||
written = ^v
|
||||
}
|
||||
width := mnem
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
width = "MOVW"
|
||||
} else {
|
||||
width = "MOVD"
|
||||
}
|
||||
mw, merr := encodeARM64LoadImm(27, written, width)
|
||||
var rn, rd int
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
rn = arm64RegNum(operandRegName(ops[1]))
|
||||
rd = arm64RegNum(operandRegName(ops[2]))
|
||||
default:
|
||||
rd = arm64RegNum(operandRegName(ops[1]))
|
||||
rn = rd
|
||||
}
|
||||
if isCmp {
|
||||
rd = 31
|
||||
}
|
||||
if merr != nil || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
||||
}
|
||||
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
|
||||
}
|
||||
opc := (baseOp >> 29) & 7
|
||||
sf := (baseOp >> 31) & 1
|
||||
var rn, rd int
|
||||
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
||||
}
|
||||
if !isExtend {
|
||||
if amount < 0 || amount > 63 {
|
||||
// ROR rides the shifted-register field only for the logical
|
||||
// group; the toolchain reports "unsupported shift operator" for
|
||||
// the arithmetic forms, whose shift=11 encoding is unallocated.
|
||||
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
|
||||
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
|
||||
}
|
||||
// The imm6 field is 5 bits and truncates at the 32-bit width.
|
||||
limit := 63
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
limit = 31
|
||||
}
|
||||
if amount < 0 || amount > limit {
|
||||
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
||||
}
|
||||
// SP-based ADD/SUB have no shifted-register encoding: the
|
||||
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
// The carry family carries an immediate spelling in three operands
|
||||
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
|
||||
// operand, the same shape the two-operand form takes.
|
||||
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
|
||||
if v := arm64Imm64(ops[0]); v != 0 {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
// OP Rm, Rn, Rd
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
|
||||
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
|
||||
// shifted-register group, the only forms whose register operand accepts the
|
||||
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
|
||||
func arm64LogicalShifted(mnem string) bool {
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
|
||||
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
|
||||
"TST", "TSTW", "MVN", "MVNW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
|
||||
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
|
||||
// data-processing instructions the toolchain accepts an immediate $0
|
||||
// operand spelling for.
|
||||
func arm64CarryOp(mnem string) bool {
|
||||
switch mnem {
|
||||
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64RegMod reports whether a register operand carries the shifted-register
|
||||
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
||||
// option (R0.UXTW, R3.SXTW<<2).
|
||||
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
|
||||
|
||||
// arm64RegModifier resolves a modified register operand: the register number,
|
||||
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
||||
// option (UXTB=0..SXTX=7) with its shift amount.
|
||||
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
|
||||
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
|
||||
// before the operator match.
|
||||
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
||||
name := operandRegName(op)
|
||||
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
||||
switch strings.ToUpper(strings.TrimSpace(after)) {
|
||||
case "UXTB":
|
||||
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
if rm < 0 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok || amount < 0 || amount > 4 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
return rm, 0, extendOpt, true, amount, true
|
||||
}
|
||||
shiftKind = 0 // LSL
|
||||
shift := strings.TrimSpace(op.Addr.Shift)
|
||||
switch {
|
||||
case strings.HasPrefix(shift, "<<"):
|
||||
shiftKind = 0
|
||||
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
default:
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
|
||||
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
||||
// MUL mnemonic, not to these.
|
||||
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
// The widening three-operand forms (SMULL, UMNEGL, …) read the
|
||||
// accumulate register as ZR, already preset in the table's base word.
|
||||
if len(ops) == 3 {
|
||||
switch mnem {
|
||||
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
|
||||
default:
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
|
||||
}
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rm < 0 || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
if len(ops) != 4 {
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
|
||||
// ---- ADD/SUB immediate ----
|
||||
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
|
||||
// following the toolchain's immediate classification (asm7.go conclass and
|
||||
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
|
||||
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
|
||||
// otherwise a constant materialisation into REGTMP (R27) followed by the
|
||||
// register form.
|
||||
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 2 && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
v, ok := arm64ImmOperandValue(ops[0])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
||||
}
|
||||
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
||||
rn := rd
|
||||
if len(ops) == 3 {
|
||||
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
|
||||
sf = 0 // 32-bit
|
||||
}
|
||||
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
|
||||
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
|
||||
// CMP/CMN discard the destination.
|
||||
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
||||
rd = 31 // ZR
|
||||
}
|
||||
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", mnem, err)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
|
||||
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
|
||||
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
|
||||
// their W forms. rn and rd are resolved register numbers (a comparison
|
||||
// discards rd, so the caller passes 31).
|
||||
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
|
||||
w := strings.HasSuffix(mnem, "W")
|
||||
sf := uint32(1) // 64-bit
|
||||
d := v
|
||||
if w {
|
||||
sf = 0 // 32-bit
|
||||
// The W forms classify the 32-bit value (asm7.go con32class).
|
||||
d = int64(uint32(v))
|
||||
}
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
op := uint32(0) // ADD
|
||||
S := uint32(0)
|
||||
if isSub {
|
||||
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if isS {
|
||||
S = 1
|
||||
}
|
||||
single := func(sh, imm12 uint32) []uint32 {
|
||||
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
|
||||
}
|
||||
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
|
||||
// imm12: plain, then the one-shifted-by-12 form.
|
||||
if d >= 0 && d <= 0xFFF {
|
||||
return single(0, uint32(d)), nil
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
// Encode as the opposite operation with positive immediate.
|
||||
opp := op ^ 1
|
||||
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
|
||||
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
|
||||
return single(1, uint32(d>>12)), nil
|
||||
}
|
||||
// Try with shift by 12.
|
||||
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
|
||||
|
||||
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
|
||||
// split into two imm12 instructions, low half first (asm7.go case 48).
|
||||
// The encoding is complete in itself: no REGTMP, no register form. The S
|
||||
// forms must not break addition/subtraction, so the toolchain
|
||||
// reclassifies them and falls through to the materialisation below.
|
||||
dm := ^d
|
||||
if w {
|
||||
dm = ^d & 0xFFFFFFFF
|
||||
}
|
||||
// The imm12 field cannot carry the value; rejecting (rather than
|
||||
// truncating) matches the toolchain, which reports the same shape.
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
|
||||
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
|
||||
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
|
||||
return []uint32{
|
||||
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
|
||||
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Constant into REGTMP (R27), then the register form. The first word
|
||||
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
|
||||
// the complement form, the bitmask ORR otherwise, and the full
|
||||
// omovlconst sequence when no single word carries the value.
|
||||
var seq []uint32
|
||||
switch s := arm64Movcon(d); {
|
||||
case s >= 0:
|
||||
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
|
||||
case arm64Movcon(dm) >= 0:
|
||||
s := arm64Movcon(dm)
|
||||
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
|
||||
case isBitcon:
|
||||
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
|
||||
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
|
||||
default:
|
||||
seq = arm64MovLConst(d, sf)
|
||||
}
|
||||
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
|
||||
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
|
||||
for i := range seq[:len(seq)-1] {
|
||||
seq[i] |= 27 // REGTMP
|
||||
}
|
||||
return seq, nil
|
||||
}
|
||||
|
||||
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
|
||||
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
|
||||
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
|
||||
// returned with the destination field clear so the caller can OR its own
|
||||
// register in. movcon and movcon-of-complement must fail for d before this
|
||||
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
|
||||
func arm64MovLConst(d int64, sf uint32) []uint32 {
|
||||
if sf == 0 {
|
||||
// omovlconst AMOVW: both 16-bit halves, low first.
|
||||
return []uint32{
|
||||
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
|
||||
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
|
||||
}
|
||||
}
|
||||
dn := ^d
|
||||
var immh [4]uint64
|
||||
zero, neg := 0, 0
|
||||
for i := range immh {
|
||||
immh[i] = uint64(d>>(i*16)) & 0xFFFF
|
||||
switch immh[i] {
|
||||
case 0:
|
||||
zero++
|
||||
case 0xFFFF:
|
||||
neg++
|
||||
}
|
||||
}
|
||||
mw := func(opc uint32, val int64, chunk int) uint32 {
|
||||
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
|
||||
}
|
||||
var os []uint32
|
||||
switch {
|
||||
case zero == 2:
|
||||
// one MOVZ and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg == 2:
|
||||
// one MOVN and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
|
||||
// a single MOVK (constants from strength-reduced division).
|
||||
if zero == 0 && neg == 0 {
|
||||
for i := range 4 {
|
||||
mask := uint64(0xFFFF) << (i * 16)
|
||||
for period := 2; period <= 32; period *= 2 {
|
||||
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
|
||||
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
|
||||
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
|
||||
os = append(os, mw(3, d, i))
|
||||
return os
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case zero >= 1:
|
||||
// one MOVZ and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg >= 1:
|
||||
// one MOVN and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// one MOVZ and three MOVKs
|
||||
os = append(os, mw(2, d, 0))
|
||||
for i := 1; i < 4; i++ {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
}
|
||||
return os
|
||||
}
|
||||
|
||||
// ---- MOV pseudo-instruction ----
|
||||
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
|
||||
var ws []uint32
|
||||
first := true
|
||||
for i := range 4 {
|
||||
chunk := (d >> uint(i*16)) & 0xFFFF
|
||||
if chunk == 0 {
|
||||
continue
|
||||
}
|
||||
if first {
|
||||
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
|
||||
first = false
|
||||
} else {
|
||||
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
|
||||
}
|
||||
}
|
||||
if len(ws) == 0 {
|
||||
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
|
||||
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
||||
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
|
||||
// for the first special 16-bit chunk, then MOVK per remaining one, with
|
||||
// the bitmask-plus-fixup shortcut for strength-reduced constants).
|
||||
ws := arm64MovLConst(d, sf)
|
||||
for i := range ws {
|
||||
ws[i] |= uint32(rd)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
|
||||
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
|
||||
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
|
||||
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
|
||||
// with imms = lsb+width-1.
|
||||
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
||||
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
||||
}
|
||||
lsb := arm64Imm64(ops[0])
|
||||
width := arm64Imm64(ops[2])
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[3]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
bits := int64(32) << (baseOp >> 31 & 1)
|
||||
if lsb < 0 || lsb >= bits {
|
||||
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
||||
}
|
||||
if width < 1 || width > bits || lsb+width > bits {
|
||||
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
||||
}
|
||||
var immr, imms int64
|
||||
switch mnem {
|
||||
case "BFXIL", "BFXILW":
|
||||
immr, imms = lsb, lsb+width-1
|
||||
default: // BFI, SBFIZ, UBFIZ
|
||||
immr, imms = (-lsb)%bits, width-1
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
||||
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
||||
// toolchain's choice of register or immediate operand.
|
||||
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
||||
// PRFM (Rn), $imm|<op>
|
||||
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
// Operand-less returns and pointer-authentication hints.
|
||||
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
||||
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
|
||||
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
|
||||
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
||||
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("%s expects no operand", mnem)
|
||||
}
|
||||
return a64wordLE(w), nil
|
||||
}
|
||||
switch mnem {
|
||||
case "BRK", "SVC":
|
||||
base := uint32(0xd4200000)
|
||||
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DMB", "DSB", "ISB":
|
||||
case "DMB", "DSB", "ISB", "CLREX":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if v < 0 || v > 15 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<8), nil
|
||||
case "HINT":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 127 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
|
||||
}
|
||||
return a64wordLE(0xd503201f | uint32(v)<<5), nil
|
||||
case "BTI":
|
||||
op := operandRegName(ops[0])
|
||||
base, ok := map[string]uint32{"C": 0xd503245f}[op]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
|
||||
}
|
||||
return a64wordLE(base), nil
|
||||
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 0xFFFF {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
|
||||
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
|
||||
"DCPS3": 0xd4a00003}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DC":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("DC expects <op>, Rn")
|
||||
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
}
|
||||
return a64wordLE(base | uint32(rd)&31), nil
|
||||
case "MSR":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
||||
}
|
||||
if !isImmOperand(ops[0]) {
|
||||
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
|
||||
base, ok := a64MSRRegOps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("MSR: invalid source register")
|
||||
}
|
||||
return a64wordLE(base | uint32(rs)&31), nil
|
||||
}
|
||||
base, ok := a64MSROps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
|
||||
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
||||
func specBit(i int) uint16 { return 1 << uint(i) }
|
||||
|
||||
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
|
||||
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
|
||||
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
|
||||
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
|
||||
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
|
||||
return true
|
||||
}
|
||||
if !strings.HasPrefix(mnem, "VFCM") {
|
||||
return false
|
||||
}
|
||||
s := strings.Join(strings.Fields(op.Raw), "")
|
||||
s = strings.TrimPrefix(s, "$")
|
||||
s = strings.Trim(s, "()")
|
||||
return s == "0" || s == "0.0"
|
||||
}
|
||||
|
||||
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
|
||||
// zero immediate takes its compare-against-zero form instead, and the
|
||||
// polynomial multiplies read the arrangement from their source operands
|
||||
// alone, the result spelling (H8, Q1) riding no encoding bits.
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
|
||||
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
|
||||
// their compare-against-zero form instead, and the polynomial multiplies read
|
||||
// the arrangement from their source operands alone, the result spelling
|
||||
// (H8, Q1) riding no encoding bits.
|
||||
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
||||
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
||||
if len(ops) != 3 {
|
||||
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
rd, _ := arm64VecOf(ops[2])
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
||||
}
|
||||
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
if arm64Imm64(ops[0]) != 0 {
|
||||
if len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
base, ok := a64SimdVZero[mnem]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
if !arm64SimdZeroImm(mnem, ops[0]) {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
vn, ok1 := arm64VecOf(ops[1])
|
||||
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
|
||||
allowed := uint16(0x7f)
|
||||
if strings.HasPrefix(mnem, "VFCM") {
|
||||
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if spec.fixed {
|
||||
arrBits = 0
|
||||
}
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
}
|
||||
|
||||
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
}
|
||||
// VUADDLV spells its arrangement on the source alone; the rest take it
|
||||
// on both.
|
||||
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
||||
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
|
||||
// index register rides bits 19:16, the first table register bits 9:5, the
|
||||
// destination bits 4:0 and the table length (registers minus one) bits
|
||||
// 14:13. The table registers must be consecutive.
|
||||
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) < 3 {
|
||||
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
||||
}
|
||||
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
default:
|
||||
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
||||
}
|
||||
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
base := uint32(0x0e000000)
|
||||
if mnem == "VTBX" {
|
||||
base |= 1 << 12
|
||||
}
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
|
||||
}
|
||||
var immval int64
|
||||
switch mnem {
|
||||
case "VSHL":
|
||||
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
|
||||
if sh < 0 || sh >= esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
||||
}
|
||||
immval = esize + sh
|
||||
default: // VUSHR, VSRI
|
||||
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
|
||||
if sh < 1 || sh > esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
||||
}
|
||||
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
||||
}
|
||||
aliases := map[string]alias{}
|
||||
raws := map[string]string{}
|
||||
for _, d := range f.Decls {
|
||||
pre, ok := d.(*ast.Preproc)
|
||||
if !ok {
|
||||
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
||||
continue
|
||||
}
|
||||
raws[name] = body
|
||||
}
|
||||
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
|
||||
// transitively until nothing changes, bounded against cycles.
|
||||
for range 8 {
|
||||
changed := false
|
||||
for name, body := range raws {
|
||||
if next, ok := raws[body]; ok && next != body {
|
||||
raws[name] = next
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
}
|
||||
for name, body := range raws {
|
||||
isReg := func(s string) bool {
|
||||
if arm64RegNum(s) >= 0 {
|
||||
return true
|
||||
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return
|
||||
}
|
||||
|
||||
// replace rewrites whole-word occurrences of the alias names in s.
|
||||
replace := func(s string) string {
|
||||
if s == "" {
|
||||
return s
|
||||
}
|
||||
out := strings.Fields(s)
|
||||
for i, w := range out {
|
||||
if a, ok := aliases[w]; ok {
|
||||
out[i] = a.raw
|
||||
}
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return s
|
||||
}
|
||||
return strings.Join(out, " ")
|
||||
}
|
||||
// replaceToken rewrites an operand whose whole text is one alias use
|
||||
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
||||
// ending at a non-identifier character.
|
||||
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return s, false
|
||||
}
|
||||
|
||||
// replaceScan rewrites alias uses inside a composite operand (a
|
||||
// parenthesised memory operand or a bracketed register list): every
|
||||
// identifier run of word and dot characters is matched against the alias
|
||||
// names, everything else copies verbatim. The whitespace-split replace
|
||||
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
|
||||
// punctuation attached.
|
||||
replaceScan := func(s string) string {
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); {
|
||||
if isAliasWordByte(s[i]) || s[i] == '.' {
|
||||
j := i
|
||||
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
|
||||
j++
|
||||
}
|
||||
if nn, ok := replaceToken(s[i:j]); ok {
|
||||
b.WriteString(nn)
|
||||
} else {
|
||||
b.WriteString(s[i:j])
|
||||
}
|
||||
i = j
|
||||
continue
|
||||
}
|
||||
b.WriteByte(s[i])
|
||||
i++
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
op.Raw = a.raw
|
||||
continue
|
||||
}
|
||||
op.Addr.Sym.Name = nn
|
||||
op.Addr.Sym.Raw = nn
|
||||
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
|
||||
// The span shape depends on what trailed the
|
||||
// name: an element or arrangement selector
|
||||
// (POLY.D[0], POLY.B16) rides in Shift and folds
|
||||
// back onto the rewritten token; a shift
|
||||
// operator stays in Shift while the span carries
|
||||
// the bare register; a split list keeps its
|
||||
// closing bracket, so the rewrite goes through
|
||||
// the scan.
|
||||
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
switch {
|
||||
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
|
||||
// Element or arrangement selectors and the
|
||||
// closing bracket of a split list belong to
|
||||
// the token text.
|
||||
op.Raw = nn + sfx
|
||||
op.Addr.Shift = ""
|
||||
case op.Addr.Shift != "":
|
||||
op.Raw = nn
|
||||
default:
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
continue
|
||||
}
|
||||
}
|
||||
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
// [V0.B16, V1.B16] with aliased members.
|
||||
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
||||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
||||
op.Raw = replace(op.Raw)
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+238
-1
@@ -308,6 +308,8 @@ const (
|
||||
a64CondLT = 0xb
|
||||
a64CondGT = 0xc
|
||||
a64CondLE = 0xd
|
||||
a64CondAL = 0xe
|
||||
a64CondNV = 0xf
|
||||
)
|
||||
|
||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
|
||||
"LT": a64CondLT,
|
||||
"GT": a64CondGT,
|
||||
"LE": a64CondLE,
|
||||
"AL": a64CondAL,
|
||||
"NV": a64CondNV,
|
||||
}
|
||||
|
||||
// ---- instruction format tags ----
|
||||
@@ -343,6 +347,7 @@ const (
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
@@ -487,6 +492,16 @@ func init() {
|
||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||
// The widening multiplies: a 64-bit result riding the same layout, the
|
||||
// three-operand forms reading the accumulate register as ZR.
|
||||
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
||||
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
||||
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
||||
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
||||
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
||||
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
||||
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
||||
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
@@ -536,6 +551,15 @@ func init() {
|
||||
// ---- bitfield ----
|
||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
||||
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
||||
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
||||
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
||||
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
@@ -716,6 +740,13 @@ func init() {
|
||||
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
||||
// the source width.
|
||||
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
||||
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
||||
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
||||
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
||||
"REV16W": 0x5ac00400,
|
||||
}
|
||||
for m, op := range dp1 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||
@@ -734,7 +765,7 @@ func init() {
|
||||
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||
|
||||
// ---- system operations ----
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||
}
|
||||
|
||||
@@ -785,6 +816,73 @@ func init() {
|
||||
for m, op := range lse {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
// The remaining width and ordering spellings of the same shapes, and the
|
||||
// CAS compare-and-swap family, word-verified against go tool asm.
|
||||
lseMore := map[string]uint32{
|
||||
"LDADDAB": 0x38a00000,
|
||||
"LDADDAH": 0x78a00000,
|
||||
"LDADDALB": 0x38e00000,
|
||||
"LDADDALH": 0x78e00000,
|
||||
"LDADDLB": 0x38600000,
|
||||
"LDADDLD": 0xf8600000,
|
||||
"LDADDLH": 0x78600000,
|
||||
"LDADDLW": 0xb8600000,
|
||||
"LDCLRAB": 0x38a01000,
|
||||
"LDCLRAH": 0x78a01000,
|
||||
"LDCLRALH": 0x78e01000,
|
||||
"LDCLRB": 0x38201000,
|
||||
"LDCLRD": 0xf8201000,
|
||||
"LDCLRH": 0x78201000,
|
||||
"LDCLRLB": 0x38601000,
|
||||
"LDCLRLD": 0xf8601000,
|
||||
"LDCLRLH": 0x78601000,
|
||||
"LDCLRLW": 0xb8601000,
|
||||
"LDCLRW": 0xb8201000,
|
||||
"LDEORAB": 0x38a02000,
|
||||
"LDEORAD": 0xf8a02000,
|
||||
"LDEORAH": 0x78a02000,
|
||||
"LDEORALB": 0x38e02000,
|
||||
"LDEORALH": 0x78e02000,
|
||||
"LDEORAW": 0xb8a02000,
|
||||
"LDEORB": 0x38202000,
|
||||
"LDEORD": 0xf8202000,
|
||||
"LDEORH": 0x78202000,
|
||||
"LDEORLB": 0x38602000,
|
||||
"LDEORLD": 0xf8602000,
|
||||
"LDEORLH": 0x78602000,
|
||||
"LDEORLW": 0xb8602000,
|
||||
"LDEORW": 0xb8202000,
|
||||
"LDORAB": 0x38a03000,
|
||||
"LDORAD": 0xf8a03000,
|
||||
"LDORAH": 0x78a03000,
|
||||
"LDORALH": 0x78e03000,
|
||||
"LDORAW": 0xb8a03000,
|
||||
"LDORB": 0x38203000,
|
||||
"LDORD": 0xf8203000,
|
||||
"LDORH": 0x78203000,
|
||||
"LDORLB": 0x38603000,
|
||||
"LDORLD": 0xf8603000,
|
||||
"LDORLH": 0x78603000,
|
||||
"LDORLW": 0xb8603000,
|
||||
"LDORW": 0xb8203000,
|
||||
"SWPAB": 0x38a08000,
|
||||
"SWPAD": 0xf8a08000,
|
||||
"SWPAH": 0x78a08000,
|
||||
"SWPALH": 0x78e08000,
|
||||
"SWPAW": 0xb8a08000,
|
||||
"SWPB": 0x38208000,
|
||||
"SWPH": 0x78208000,
|
||||
"SWPLB": 0x38608000,
|
||||
"SWPLD": 0xf8608000,
|
||||
"SWPLH": 0x78608000,
|
||||
"SWPLW": 0xb8608000,
|
||||
"CASAD": 0xc8e07c00,
|
||||
"CASALB": 0x08e0fc00,
|
||||
"CASLW": 0x88a0fc00,
|
||||
}
|
||||
for m, op := range lseMore {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
|
||||
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||
@@ -794,6 +892,11 @@ func init() {
|
||||
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
||||
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
||||
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
||||
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
||||
"NEGSW": 0x6b000000,
|
||||
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||
}
|
||||
@@ -835,6 +938,12 @@ func init() {
|
||||
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
||||
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||
// the across-vector reductions.
|
||||
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||
|
||||
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||
// and pairwise compares among them. Word-verified against go tool asm.
|
||||
var a64SimdQOnly = map[string]bool{
|
||||
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
||||
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
||||
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
||||
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
||||
"VFMAXV": true, "VFMAXNMV": true,
|
||||
}
|
||||
|
||||
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||
@@ -928,19 +1054,129 @@ var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
"VZIP1": {0x0e003800, 0x7f, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||
"VCMGT": {0x0e203400, 0x7f, false},
|
||||
"VCMHI": {0x2e203400, 0x7f, false},
|
||||
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// FP arithmetic shares the same arrangement restriction.
|
||||
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||
// tool asm.
|
||||
"VBIC": {0x0e601c00, 0x7f, false},
|
||||
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||
"VBSL": {0x6e601c00, 0x7f, false},
|
||||
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VMLA": {0x4ea09400, 0x7f, false},
|
||||
"VMLS": {0x6ea09400, 0x7f, false},
|
||||
"VORN": {0x4ee01c00, 0x7f, false},
|
||||
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||
"VUMAX": {0x2e206400, 0x7f, false},
|
||||
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||
"VUZP1": {0x0e001800, 0x7f, false},
|
||||
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
}
|
||||
|
||||
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
||||
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
||||
// combination for them, so they stay out and the encoder rejects the shape.
|
||||
var a64SimdVZero = map[string]uint32{
|
||||
"VCMEQ": 0x0e209800,
|
||||
"VCMGT": 0x0e208800,
|
||||
"VCMGE": 0x2e208800,
|
||||
"VCMLT": 0x0e20a800,
|
||||
"VCMLE": 0x2e209800,
|
||||
// FP compares against (0.0): the register forms above carry the U and op
|
||||
// bits; the zero forms reshape them.
|
||||
"VFCMEQ": 0x0ea0d800,
|
||||
"VFCMGE": 0x2ea0c800,
|
||||
"VFCMGT": 0x0ea0c800,
|
||||
"VFCMLE": 0x2ea0d800,
|
||||
"VFCMLT": 0x0ea0e800,
|
||||
}
|
||||
|
||||
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
||||
// the register pair spelling ORR Vd, Vn, Vm.
|
||||
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||
"VREV64": {0x0e200800, 0x3f, false},
|
||||
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
// Two-register data-processing across one arrangement.
|
||||
"VABS": {0x0e20b800, 0x7f, false},
|
||||
"VNEG": {0x2e20b800, 0x7f, false},
|
||||
"VCLS": {0x0e204800, 0x7f, false},
|
||||
"VCLZ": {0x2e204800, 0x7f, false},
|
||||
"VCNT": {0x0e205800, 0x7f, false},
|
||||
"VNOT": {0x2e205800, 0x7f, false},
|
||||
"VSQABS": {0x0e207800, 0x7f, false},
|
||||
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||
"VRBIT": {0x6e605800, 0x7f, false},
|
||||
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||
// Across-vector reductions: the operand arrangement rides as usual and
|
||||
// the destination stays a bare V register.
|
||||
"VADDV": {0x0e31b800, 0x3f, false},
|
||||
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||
}
|
||||
|
||||
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
|
||||
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||
var a64MSRRegOps = map[string]uint32{
|
||||
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||
"ELR_EL1": 0xd5184020,
|
||||
}
|
||||
|
||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||
|
||||
+161
-10
@@ -4,6 +4,7 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
|
||||
// rejected instead of wrapping through int32.
|
||||
func TestArm64AddSubImmRange(t *testing.T) {
|
||||
for _, body := range []string{
|
||||
"\tADD $0x100000000, R0, R1\n",
|
||||
"\tSUB $-0x100000000, R0, R1\n",
|
||||
"\tCMP $0x100000000, R0\n",
|
||||
} {
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
|
||||
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
|
||||
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
|
||||
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
|
||||
// Comparisons never split, and the W forms classify the 32-bit value. Every
|
||||
// word is go tool asm's own for the same source.
|
||||
func TestArm64AddSubImmWide(t *testing.T) {
|
||||
got := arm64Words(t, strings.Join([]string{
|
||||
"\tADD $0xaaaaaa, R2, R3",
|
||||
"\tSUB $0xaaaaaa, R2",
|
||||
"\tADD $0x186a0, R2, R5",
|
||||
"\tADD $0x1ffe00, R2, R3",
|
||||
"\tADD $0x3fffffffc000, R5",
|
||||
"\tADD $-100000, R2, R3",
|
||||
"\tADD $-2048, R2, R3",
|
||||
"\tCMP $0xaaaaaa, R2",
|
||||
"\tCMP $0xffffffffffa0, R3",
|
||||
"\tCMPW $27745, R2",
|
||||
"\tCMPW $0x60060, R2",
|
||||
"\tADDS $0xaaaaaa, R2, R3",
|
||||
"\tADD $0x12345678, R2, R3",
|
||||
"\tADDW $0x60060, R2",
|
||||
"\tSUB $0xe7791f700, R3, R1",
|
||||
"\tADDW $0x12345678, R2, R3",
|
||||
"\tCMN $0x1000000, R2",
|
||||
}, "\n")+"\n")
|
||||
want := []uint32{
|
||||
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
|
||||
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
|
||||
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
|
||||
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
|
||||
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
|
||||
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
|
||||
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
|
||||
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
|
||||
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
|
||||
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
|
||||
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
|
||||
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
|
||||
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
|
||||
0x11018042, 0x11418042, // ADDW $0x60060: W split
|
||||
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
|
||||
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
|
||||
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
|
||||
// three operands, the ROR shift on the logical group (and its rejection for
|
||||
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
|
||||
// alias with an element selector. Words are go tool asm's own.
|
||||
func TestArm64CarryShiftAlias(t *testing.T) {
|
||||
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
|
||||
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
|
||||
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
|
||||
want := []uint32{
|
||||
0x9a1f0294, // ADC ZR, R20, R20
|
||||
0x9a1f0284, // ADC ZR, R20, R4
|
||||
0xfa1f008c, // SBCS ZR, R4, R12
|
||||
0xfa0f008c, // SBCS R15, R4, R12
|
||||
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
|
||||
0x8ac18443, // AND R1 ROR 33, R2, R3
|
||||
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
|
||||
0xda0203e7, // SBC ZR, R7, R2
|
||||
0x9b0eff77, // MSUB ZR, R27, R14, R23
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
|
||||
// ROR on an arithmetic form is unallocated: the toolchain reports an
|
||||
// unsupported shift operator.
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if _, err := AssembleFileARM64(f); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", body)
|
||||
t.Error("ADD R1@>33: expected an error, got none")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
|
||||
// operand with an element selector and inside a split register list: the
|
||||
// aliases resolve textually where the parser carries the selector apart from
|
||||
// the name. Words are go tool asm's own.
|
||||
func TestArm64VecAliasElement(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
#define POLY V15
|
||||
#define ACC0 V8
|
||||
#define ACC1 V9
|
||||
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
VMOV R1, POLY.D[0]
|
||||
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||
VLD1 (R0), [ACC0.B16]
|
||||
VLD1.P (R0), [ACC0.B16, ACC1.B16]
|
||||
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
got := leWords(img.Code)
|
||||
want := []uint32{
|
||||
0x4e081c2f, // INS V15.D[0], R1
|
||||
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
|
||||
0x4c407008, // VLD1 (R0), [V8.B16]
|
||||
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
|
||||
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
|
||||
// once the value leaves every imm12 form: a constant sequence into REGTMP
|
||||
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
|
||||
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
|
||||
// tool asm's own.
|
||||
func TestArm64AddSubImmBeyond32(t *testing.T) {
|
||||
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
|
||||
want := []uint32{
|
||||
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||
0x8b1b0001, // ADD R27, R0, R1
|
||||
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
|
||||
0xcb1b0001, // SUB R27, R0, R1
|
||||
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||
0xeb1b001f, // CMP R27, R0
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -67,6 +67,9 @@ type spadjStep struct {
|
||||
// patch sites (for the file-level layout to resolve), the label table and the
|
||||
// stack-adjustment boundaries.
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||
if err := checkAdjspBalance(t); err != nil {
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
fi := computeFrame(t)
|
||||
chain := jumpChain(t)
|
||||
resolve := func(name string) string {
|
||||
@@ -203,6 +206,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
||||
)
|
||||
}
|
||||
// frameBase is the SP delta the prologue leaves: 8 for the saved base
|
||||
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
|
||||
// statements' straight-line sum, so a mid-body step's value is the
|
||||
// frame base plus what the body has opened so far.
|
||||
frameBase, bodyDelta := 0, 0
|
||||
if fi.useFP {
|
||||
frameBase = 8 + fi.size
|
||||
}
|
||||
pos := guardLen + len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
s, ok := stmt.(*ast.Instr)
|
||||
@@ -230,6 +241,16 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
ps[k].kind = RelCall
|
||||
}
|
||||
}
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
|
||||
// The statement shifted SP mid-body: record the new running
|
||||
// delta as the value in effect from just past the instruction.
|
||||
v := s.Operands[0].Imm.Val
|
||||
if s.Operands[0].Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
bodyDelta += int(v)
|
||||
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
|
||||
}
|
||||
patches = append(patches, ps...)
|
||||
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||
out = append(out, code...)
|
||||
@@ -412,6 +433,40 @@ func hasCall(t *ast.Text) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
|
||||
// shifts SP away from the entry state and every RET must see the shifts
|
||||
// closed. The assembler's own prologue and epilogue contribute matching
|
||||
// deltas on both sides, so the statements' straight-line sum must be zero
|
||||
// at each RET; branches do not reset the walk, which runs over the program
|
||||
// list in source order. go tool asm reports an offender as "unbalanced
|
||||
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
|
||||
// $16/$-16 pair, per-RET rather than per-function).
|
||||
func checkAdjspBalance(t *ast.Text) error {
|
||||
delta := 0
|
||||
for _, stmt := range t.Body {
|
||||
in, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "ADJSP":
|
||||
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
|
||||
continue // reported during emission
|
||||
}
|
||||
v := in.Operands[0].Imm.Val
|
||||
if in.Operands[0].Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
delta += int(v)
|
||||
case "RET":
|
||||
if delta != 0 {
|
||||
return fmt.Errorf("unbalanced PUSH/POP")
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// guardLen returns the byte length of the stack-split guard prefix. The
|
||||
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
||||
// short form and 6 in the long form.
|
||||
|
||||
@@ -439,3 +439,132 @@ func TestSubSPEncodings(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
|
||||
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
|
||||
// asserts the three behaviours the toolchain shows: each prefix statement is
|
||||
// a standalone byte with a PC of its own (so a label placed on the LOCK
|
||||
// points at the F0), the data pseudo-ops write their literal bytes inline,
|
||||
// and END terminates nothing (the statements after it still belong to the
|
||||
// function and carry no trace of it).
|
||||
func TestAssemblePseudoStatements(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·pseudo(SB), NOSPLIT, $0-0
|
||||
pfx:
|
||||
LOCK
|
||||
CMPXCHGQ AX, (BX)
|
||||
REP
|
||||
MOVSQ
|
||||
BYTE $0x0f
|
||||
BYTE $0x1f
|
||||
WORD $0x1234
|
||||
END
|
||||
BYTE $0x02
|
||||
RET
|
||||
`)
|
||||
code, labels, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
|
||||
want := []byte{
|
||||
0xf0,
|
||||
0x48, 0x0f, 0xb1, 0x03,
|
||||
0xf3, 0x48, 0xa5,
|
||||
0x0f, 0x1f, 0x34, 0x12,
|
||||
0x02, 0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
// The label sits on the LOCK byte, exactly where the toolchain's PC
|
||||
// listing puts it.
|
||||
if off := labels["pfx"]; off != 0 {
|
||||
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
|
||||
}
|
||||
// The trailing BYTE lands where the layout says: after the 8 bytes of
|
||||
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
|
||||
// none.
|
||||
if code[12] != 0x02 {
|
||||
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
|
||||
// ADJSP: the straight-line sum of the adjustments must be zero at each
|
||||
// RET, branches in between counting for nothing (verified against go tool
|
||||
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
|
||||
// $16/$-16 pair with a JMP in between assembles).
|
||||
func TestAssembleAdjspBalance(t *testing.T) {
|
||||
// Balanced pair with a branch in between, bytes pinned from go tool asm.
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·adjsp(SB), NOSPLIT, $0-0
|
||||
ADJSP $16
|
||||
JMP body
|
||||
body:
|
||||
ADJSP $-16
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
|
||||
// Unbalanced at the RET: the toolchain diagnoses, so must we.
|
||||
_, _, err = Assemble(firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·unbalanced(SB), NOSPLIT, $0-0
|
||||
ADJSP $16
|
||||
RET
|
||||
`))
|
||||
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||
}
|
||||
|
||||
// The check runs per RET: a closed pair before the first RET does not
|
||||
// excuse an open adjustment before the second.
|
||||
_, _, err = Assemble(firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·tworet(SB), NOSPLIT, $0-0
|
||||
ADJSP $8
|
||||
ADJSP $-8
|
||||
RET
|
||||
mid:
|
||||
ADJSP $8
|
||||
RET
|
||||
`))
|
||||
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||
}
|
||||
|
||||
// A framed function: the assembler's own prologue and epilogue
|
||||
// contribute matching deltas, so the pair in the body still balances,
|
||||
// and the bytes match go tool asm end to end.
|
||||
fn = firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·framed(SB), $16-8
|
||||
ADJSP $8
|
||||
ADJSP $-8
|
||||
RET
|
||||
`)
|
||||
code, _, err = Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble framed: %v", err)
|
||||
}
|
||||
want = []byte{
|
||||
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
|
||||
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
|
||||
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
|
||||
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
|
||||
0xC3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
+4
-1
@@ -19,7 +19,10 @@ func Encodable(mnemonic string) bool {
|
||||
// Fixed-name instructions (no size suffix).
|
||||
switch upper {
|
||||
case "RET", "NOP", "CALL", "JMP",
|
||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
|
||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and the
|
||||
// SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
|
||||
return true
|
||||
}
|
||||
if _, ok := noOperandTable[upper]; ok {
|
||||
|
||||
@@ -92,6 +92,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||
case "SHA256RNDS2":
|
||||
return e.encodeSha256rnds2(ops)
|
||||
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
|
||||
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
|
||||
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
|
||||
// the SUBQ and ADDQ forms.
|
||||
case "BYTE", "WORD", "LONG", "QUAD":
|
||||
return e.encodeData(upper, ops)
|
||||
case "END":
|
||||
return e.encodeEnd(ops)
|
||||
case "ADJSP":
|
||||
return e.encodeAdjsp(ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
@@ -241,6 +251,73 @@ var prefetchVariant = map[string]int{
|
||||
"PREFETCHT2": 3,
|
||||
}
|
||||
|
||||
// dataWidth is the literal byte count of each data-emission pseudo-op.
|
||||
var dataWidth = map[string]int{
|
||||
"BYTE": 1,
|
||||
"WORD": 2,
|
||||
"LONG": 4,
|
||||
"QUAD": 8,
|
||||
}
|
||||
|
||||
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
|
||||
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
|
||||
// little-endian, with no opcode lookup. The value is truncated to the
|
||||
// width rather than range-checked, exactly as go tool asm behaves (BYTE
|
||||
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
|
||||
// exactly one immediate is accepted: the toolchain rejects a list such as
|
||||
// BYTE $1, $2, $3.
|
||||
func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s requires an integer immediate", mnem)
|
||||
}
|
||||
width := dataWidth[mnem]
|
||||
out := make([]byte, width)
|
||||
u := uint64(imm)
|
||||
for i := range width {
|
||||
out[i] = byte(u >> (8 * i))
|
||||
}
|
||||
e.out = append(e.out, out...)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||
// the statements after an END still belong to the same function and the
|
||||
// encoded body carries no trace of it, whatever operands follow the name
|
||||
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
|
||||
func (e *enc) encodeEnd(ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
|
||||
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
|
||||
// picks (the same selection subSP and addSP make for the frame). go tool
|
||||
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
|
||||
// statement's effect on the SP balance is checked by the function-level
|
||||
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
|
||||
func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("ADJSP requires an integer immediate")
|
||||
}
|
||||
switch v := int(imm); {
|
||||
case v > 0:
|
||||
e.out = append(e.out, subSP(v)...)
|
||||
case v < 0:
|
||||
e.out = append(e.out, addSP(-v)...)
|
||||
default:
|
||||
return fmt.Errorf("ADJSP $0 has no encoding")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
|
||||
@@ -925,3 +925,142 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
|
||||
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
|
||||
// prefix field merged into the following instruction, and it validates
|
||||
// nothing about the pairing (LOCK before NOP assembles). The prefixed
|
||||
// atomic and string shapes are the bytes the runtime's own kernels need.
|
||||
func TestPrefixStatements(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"LOCK", "LOCK", nil, "f0"},
|
||||
{"REP", "REP", nil, "f3"},
|
||||
{"REPN", "REPN", nil, "f2"},
|
||||
// LOCK; CMPXCHGQ AX, (BX)
|
||||
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||
// REP; MOVSQ
|
||||
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
|
||||
// REPN; MOVSB
|
||||
{"REPN MOVSB", "MOVSB", nil, "a4"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
// The prefix statements take no operands, as the toolchain reports for
|
||||
// LOCK AX.
|
||||
if _, err := Encode("LOCK", AX); err == nil {
|
||||
t.Error("LOCK AX assembled, want an error")
|
||||
}
|
||||
if _, err := Encode("REP", Imm(1)); err == nil {
|
||||
t.Error("REP $1 assembled, want an error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
|
||||
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
|
||||
// lookup, truncated to the width rather than range-checked (go tool asm
|
||||
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
|
||||
func TestDataEmission(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
imm Imm
|
||||
want string
|
||||
}{
|
||||
{"BYTE", "BYTE", 0x0f, "0f"},
|
||||
{"BYTE negative", "BYTE", -1, "ff"},
|
||||
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
|
||||
{"WORD", "WORD", 0x1234, "3412"},
|
||||
{"WORD negative", "WORD", -1, "ffff"},
|
||||
{"WORD truncated", "WORD", 0x12345, "4523"},
|
||||
{"LONG", "LONG", 0x11223344, "44332211"},
|
||||
{"LONG negative", "LONG", -1, "ffffffff"},
|
||||
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
|
||||
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.imm)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
|
||||
// register or a missing operand is no immediate at all.
|
||||
if _, err := Encode("BYTE"); err == nil {
|
||||
t.Error("BYTE with no operand assembled, want an error")
|
||||
}
|
||||
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
|
||||
t.Error("BYTE $1, $2 assembled, want an error")
|
||||
}
|
||||
if _, err := Encode("WORD", AX); err == nil {
|
||||
t.Error("WORD AX assembled, want an error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
|
||||
// encodes to zero bytes and takes any operands without complaint (the
|
||||
// toolchain accepts END $0 and END AX alike).
|
||||
func TestEndIgnored(t *testing.T) {
|
||||
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
|
||||
code, err := Encode("END", ops...)
|
||||
if err != nil {
|
||||
t.Errorf("END: %v", err)
|
||||
continue
|
||||
}
|
||||
if len(code) != 0 {
|
||||
t.Errorf("END = %x, want no bytes", code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
|
||||
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
|
||||
// has no encoding (go tool asm refuses ADJSP $0 outright).
|
||||
func TestAdjsp(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
imm Imm
|
||||
want string
|
||||
}{
|
||||
{"imm8", 112, "4883ec70"},
|
||||
{"imm8 negative", -112, "4883c470"},
|
||||
{"imm32", 200, "4881ecc8000000"},
|
||||
{"imm32 negative", -200, "4881c4c8000000"},
|
||||
{"small", 8, "4883ec08"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode("ADJSP", c.imm)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
|
||||
}
|
||||
}
|
||||
if _, err := Encode("ADJSP", Imm(0)); err == nil {
|
||||
t.Error("ADJSP $0 assembled, want an error")
|
||||
}
|
||||
if _, err := Encode("ADJSP"); err == nil {
|
||||
t.Error("ADJSP with no operand assembled, want an error")
|
||||
}
|
||||
if _, err := Encode("ADJSP", AX); err == nil {
|
||||
t.Error("ADJSP AX assembled, want an error")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -65,6 +65,15 @@ var bitTestOp = map[string]int{
|
||||
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||
//
|
||||
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
|
||||
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
|
||||
// respectively), not as a prefix field merged into the next instruction: the
|
||||
// statement that follows is encoded unaware of it, and nothing validates
|
||||
// that the pairing is a legal one (LOCK before NOP assembles without
|
||||
// complaint, each byte pinned against the toolchain). Because the bytes
|
||||
// land in the stream before the following statement anyway, a LOCKed
|
||||
// CMPXCHGQ encodes identically to a prefixed form.
|
||||
var noOperandTable = map[string][]byte{
|
||||
"CPUID": {0x0F, 0xA2},
|
||||
"RDTSC": {0x0F, 0x31},
|
||||
@@ -78,6 +87,9 @@ var noOperandTable = map[string][]byte{
|
||||
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||
"UNDEF": {0x0F, 0x0B},
|
||||
"LOCK": {0xF0},
|
||||
"REP": {0xF3},
|
||||
"REPN": {0xF2},
|
||||
}
|
||||
|
||||
// --- MOV --------------------------------------------------------------------
|
||||
|
||||
+387
-40
@@ -46,30 +46,70 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
|
||||
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
|
||||
}
|
||||
|
||||
// Pass 1: label offsets from the instruction sizes.
|
||||
offsets := map[string]int{}
|
||||
pos := guardLen + len(prologue)
|
||||
// The toolchain's parser counts N(PC) displacements over the source
|
||||
// instructions at a uniform 4 bytes each, so a PC-relative branch
|
||||
// resolves to the instruction N slots away in body order; the resolved
|
||||
// target then participates in layout and loop-head padding like any
|
||||
// branch target.
|
||||
instrs := make([]*ast.Instr, 0, len(t.Body))
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
pos += loong64InstrSize(s, fi)
|
||||
if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" {
|
||||
instrs = append(instrs, in)
|
||||
}
|
||||
}
|
||||
parseIndex := make(map[*ast.Instr]int, len(instrs))
|
||||
for i, in := range instrs {
|
||||
parseIndex[in] = i
|
||||
}
|
||||
pcRelTarget := make(map[*ast.Instr]*ast.Instr)
|
||||
for _, in := range instrs {
|
||||
off, ok := loong64PCRelOffset(in)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
tgt := parseIndex[in] + off
|
||||
if tgt < 0 || tgt >= len(instrs) {
|
||||
continue
|
||||
}
|
||||
pcRelTarget[in] = instrs[tgt]
|
||||
}
|
||||
|
||||
// Pass 2: encode. The guard prefix precedes the prologue; its branches
|
||||
// target the morestack block at the end of the function, which the first
|
||||
// pass has sized.
|
||||
bodyLen := 0
|
||||
{
|
||||
p := guardLen + len(prologue)
|
||||
for _, stmt := range t.Body {
|
||||
if in, ok := stmt.(*ast.Instr); ok {
|
||||
p += loong64InstrSize(in, fi)
|
||||
// Pass 1: label offsets from the instruction sizes. PCALIGN contributes
|
||||
// only its padding. On top of the explicit PCALIGNs, the toolchain pads
|
||||
// every backward-branch target (loop head) to a 16-byte boundary, so the
|
||||
// layout runs to a fixpoint over the alignment set.
|
||||
loopAligns := map[string]bool{}
|
||||
alignInstrs := map[*ast.Instr]bool{}
|
||||
for {
|
||||
offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
|
||||
changed := false
|
||||
for _, in := range instrs {
|
||||
// A backward PC-relative target is the resolved instruction.
|
||||
if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] {
|
||||
alignInstrs[tgt] = true
|
||||
changed = true
|
||||
}
|
||||
target, ok := loong64BranchTarget(in)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
tOff, ok := offsets[target]
|
||||
if !ok || tOff >= pcs[in] || loopAligns[target] {
|
||||
continue
|
||||
}
|
||||
loopAligns[target] = true
|
||||
changed = true
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
}
|
||||
bodyLen = p - (guardLen + len(prologue))
|
||||
// Final layout with the complete alignment set.
|
||||
offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
|
||||
bodyLen := bodyEnd - (guardLen + len(prologue))
|
||||
pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget))
|
||||
for in, tgt := range pcRelTarget {
|
||||
pcRelPcs[in] = pcs[tgt]
|
||||
}
|
||||
var out []byte
|
||||
if fi.needSplit {
|
||||
@@ -84,7 +124,20 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
|
||||
// PCALIGN pads to the requested boundary with andi $0, $0, 0, the
|
||||
// architecture's NOP, and encodes to nothing itself.
|
||||
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
|
||||
pad := loong64PCAlignPad(pc, in)
|
||||
out = append(out, loong64PadBytes(pad)...)
|
||||
pc += pad
|
||||
continue
|
||||
}
|
||||
// Loop-head alignment padding precedes the instruction.
|
||||
if pad := alignPad[in]; pad > 0 {
|
||||
out = append(out, loong64PadBytes(pad)...)
|
||||
pc += pad
|
||||
}
|
||||
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
|
||||
}
|
||||
@@ -115,6 +168,115 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
|
||||
return out, offsets, relocs, lines, spadj, nil
|
||||
}
|
||||
|
||||
// loong64PCRelOffset reports the N of a branch operand spelled N(PC): the
|
||||
// displacement counted in source instructions from the branch itself.
|
||||
func loong64PCRelOffset(instr *ast.Instr) (int, bool) {
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
branch := false
|
||||
switch mnem {
|
||||
case "JMP":
|
||||
branch = len(instr.Operands) == 1
|
||||
case "JAL", "CALL", "BL":
|
||||
branch = len(instr.Operands) == 1 || len(instr.Operands) == 2
|
||||
case "BFPT", "BFPF":
|
||||
branch = len(instr.Operands) == 1
|
||||
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
|
||||
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
|
||||
branch = len(instr.Operands) >= 2
|
||||
}
|
||||
if !branch {
|
||||
return 0, false
|
||||
}
|
||||
op := instr.Operands[len(instr.Operands)-1]
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
|
||||
return int(op.Addr.Offset), true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// loong64Layout walks the function body once and returns the label offsets,
|
||||
// the loop-alignment padding due before each instruction (a pad of 0 needs
|
||||
// nothing), the pc each instruction starts at (its padding included) and the
|
||||
// first pc past the body. Explicit PCALIGN pads, the alignment pads for the
|
||||
// labels in aligns and those for the instructions in alignInstrs (backward
|
||||
// PC-relative targets) all contribute, mirroring the toolchain's layout
|
||||
// pass.
|
||||
func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) {
|
||||
offsets := map[string]int{}
|
||||
alignPad := map[*ast.Instr]int{}
|
||||
pcs := map[*ast.Instr]int{}
|
||||
pos := start
|
||||
pendingAlign := false
|
||||
var pendingNames []string
|
||||
explicit := false
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
if aligns[s.Name.Text] {
|
||||
pendingAlign = true
|
||||
}
|
||||
pendingNames = append(pendingNames, s.Name.Text)
|
||||
// Provisional: a branch to the label lands here unless a loop
|
||||
// alignment pad follows, in which case the label resolves to the
|
||||
// padded instruction (the toolchain's labels bind to the branch
|
||||
// target instruction, which the padding pass precedes).
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
|
||||
pos += loong64PCAlignPad(pos, s)
|
||||
explicit = true
|
||||
continue
|
||||
}
|
||||
if pendingAlign {
|
||||
pendingAlign = false
|
||||
if pos&15 != 0 {
|
||||
alignPad[s] = 16 - pos&15
|
||||
}
|
||||
}
|
||||
if alignInstrs[s] && pos&15 != 0 {
|
||||
alignPad[s] = 16 - pos&15
|
||||
}
|
||||
if !explicit {
|
||||
for _, n := range pendingNames {
|
||||
offsets[n] = pos + alignPad[s]
|
||||
}
|
||||
}
|
||||
pendingNames = nil
|
||||
explicit = false
|
||||
pcs[s] = pos + alignPad[s]
|
||||
pos += alignPad[s] + loong64InstrSize(s, fi)
|
||||
}
|
||||
}
|
||||
return offsets, alignPad, pcs, pos
|
||||
}
|
||||
|
||||
// loong64BranchTarget reports the local label a branch-like instruction
|
||||
// transfers to, the loop-head signal the toolchain derives from backward
|
||||
// branch targets.
|
||||
func loong64BranchTarget(instr *ast.Instr) (string, bool) {
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
ops := instr.Operands
|
||||
var op *ast.Operand
|
||||
switch {
|
||||
case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF":
|
||||
if len(ops) != 1 {
|
||||
return "", false
|
||||
}
|
||||
op = ops[0]
|
||||
case mnem == "TEQ" || mnem == "TNE":
|
||||
return "", false
|
||||
case len(ops) >= 2:
|
||||
op = ops[len(ops)-1]
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
||||
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
||||
return op.Addr.Sym.Name, true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
|
||||
// branch-chasing pass: a label whose first instruction is an unconditional
|
||||
// local jump redirects its own jumpers to the ultimate target. The Go
|
||||
@@ -177,21 +339,51 @@ func l64LabelOK(op *ast.Operand) (string, bool) {
|
||||
return "", false
|
||||
}
|
||||
|
||||
// l64SubToAdd rewrites the SUB family with an immediate first operand onto
|
||||
// its ADD counterpart with the negated immediate: LoongArch has no
|
||||
// subtract-immediate instructions, and the toolchain folds SUB $v into the
|
||||
// ADD immediate form through the same optab matching (the $0 fold into 3R
|
||||
// and the large-constant materialisations included). The negation is the
|
||||
// second result; the operand is left untouched because the size pass
|
||||
// normalises the same instruction.
|
||||
func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
switch mnem {
|
||||
case "SUB":
|
||||
return "ADD", true
|
||||
case "SUBW":
|
||||
return "ADDW", true
|
||||
case "SUBV", "SUBVU":
|
||||
return "ADDV", true
|
||||
}
|
||||
}
|
||||
return mnem, false
|
||||
}
|
||||
|
||||
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
|
||||
// most, more for the multi-instruction expansions.
|
||||
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
ops := instr.Operands
|
||||
var neg bool
|
||||
mnem, neg = l64SubToAdd(mnem, ops)
|
||||
|
||||
if mnem == "RET" {
|
||||
return len(loong64Return(fi))
|
||||
}
|
||||
switch mnem {
|
||||
case "TEQ", "TNE":
|
||||
return 8 // bne/beq over the BREAK, then BREAK
|
||||
case "PRELDX":
|
||||
return 20 // the four-instruction constant materialisation + preldx
|
||||
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
|
||||
return loong64MovSize(mnem, ops, fi)
|
||||
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
v := l64Imm64(ops[0])
|
||||
if neg {
|
||||
v = -v
|
||||
}
|
||||
if v == 0 {
|
||||
return 4 // folds into the 3R form (rk = R0)
|
||||
}
|
||||
@@ -229,11 +421,50 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
|
||||
return 4
|
||||
}
|
||||
|
||||
// loong64PCAlignPad returns the padding PCALIGN inserts before the next
|
||||
// instruction so that it starts at the requested boundary relative to the
|
||||
// function start. The boundary must be a power of two between 8 and 2048, as
|
||||
// the toolchain requires; anything else pads nothing.
|
||||
func loong64PCAlignPad(pos int, instr *ast.Instr) int {
|
||||
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
|
||||
return 0
|
||||
}
|
||||
align := int(immFromOperand(instr.Operands[0]))
|
||||
if align < 8 || align > 2048 || align&(align-1) != 0 {
|
||||
return 0
|
||||
}
|
||||
return (align - pos%align) % align
|
||||
}
|
||||
|
||||
// loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0
|
||||
// (the architecture's NOP) for every full 4 bytes of pad.
|
||||
func loong64PadBytes(pad int) []byte {
|
||||
nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0))
|
||||
out := make([]byte, 0, pad/4*len(nop))
|
||||
for i := 0; i < pad/4; i++ {
|
||||
out = append(out, nop...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// encodeLOONG64Instr encodes a single LoongArch instruction.
|
||||
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
|
||||
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
ops := instr.Operands
|
||||
|
||||
// The SUB family with an immediate first operand folds onto the ADD
|
||||
// immediate form with the negated immediate; the negation happens on a
|
||||
// copy of the operand, never on the shared syntax tree.
|
||||
mnem, neg := l64SubToAdd(mnem, ops)
|
||||
if neg {
|
||||
c := *ops[0]
|
||||
c.Imm.Val = -c.Imm.Val
|
||||
ops2 := make([]*ast.Operand, len(ops))
|
||||
ops2[0] = &c
|
||||
copy(ops2[1:], ops[1:])
|
||||
ops = ops2
|
||||
}
|
||||
|
||||
// Pseudo-instructions and the branches first.
|
||||
switch mnem {
|
||||
case "RET":
|
||||
@@ -249,10 +480,80 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
||||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
|
||||
case "NEGW", "NEGV":
|
||||
// The integer negation pseudo is a subtract from zero:
|
||||
// NEGW src, dst → sub.w r0, src, dst.
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
src, dst := l64Reg(ops[0]), l64Reg(ops[1])
|
||||
if src < 0 || dst < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand")
|
||||
}
|
||||
sub := l64InstrTable["SUBW"].op
|
||||
if mnem == "NEGV" {
|
||||
sub = l64InstrTable["SUBV"].op
|
||||
}
|
||||
return l64wordLE(l64rrr(sub, src, 0, dst)), nil
|
||||
case "TEQ", "TNE":
|
||||
// The trap pseudo expands to two instructions: bne/beq rj, rd over
|
||||
// the BREAK (offset 2 instruction units), then BREAK $code.
|
||||
if len(ops) != 2 && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
code := int(immFromOperand(ops[0]))
|
||||
rj, rd := 0, l64Reg(ops[len(ops)-1])
|
||||
if len(ops) == 3 {
|
||||
rj = l64Reg(ops[1])
|
||||
}
|
||||
if rj < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand")
|
||||
}
|
||||
bop := l64branchTable["BNE"]
|
||||
if mnem == "TNE" {
|
||||
bop = l64branchTable["BEQ"]
|
||||
}
|
||||
return l64WordsLE(
|
||||
l64irr16(bop, 2, rj, rd),
|
||||
l64i15(l64InstrTable["BREAK"].op, code),
|
||||
), nil
|
||||
case "PRELDX":
|
||||
// preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs
|
||||
// (addrSeq, blockSize, blockNums, stride); the constant v built from
|
||||
// it materialises in R30 across four instructions, then the preldx.
|
||||
if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) {
|
||||
return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint")
|
||||
}
|
||||
rj := loong64RegNum(ops[0].Addr.Base)
|
||||
if rj < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand")
|
||||
}
|
||||
n := uint64(l64Imm64(ops[1]))
|
||||
hint := int(l64Imm64(ops[2]))
|
||||
addrSeq := (n >> 0) & 0x1
|
||||
blkSize := (n >> 1) & 0x7ff
|
||||
blkNums := (n >> 12) & 0x1ff
|
||||
stride := (n >> 21) & 0xffff
|
||||
v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 +
|
||||
((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44
|
||||
const (
|
||||
lu12iw = 0x0a << 25
|
||||
lu32id = 0x0b << 25
|
||||
lu52id = 0x00c << 22
|
||||
ori = 0x00e << 22
|
||||
preldx = 0x7058 << 15
|
||||
)
|
||||
return l64WordsLE(
|
||||
l64ir(lu12iw, int(uint32(v>>12)), 30),
|
||||
l64irr(ori, int(uint32(v)), 30, 30),
|
||||
l64ir(lu32id, int(uint32(v>>32)), 30),
|
||||
l64irr(lu52id, int(uint32(v>>52)), 30, 30),
|
||||
l64rrr(preldx, 30, rj, hint),
|
||||
), nil
|
||||
case "JMP", "B":
|
||||
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
|
||||
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
|
||||
case "JAL", "CALL", "BL":
|
||||
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
|
||||
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
|
||||
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
|
||||
return encodeLOONG64Mov(instr, mnem, fi, relocs)
|
||||
}
|
||||
@@ -262,12 +563,12 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
||||
if mnem == "JIRL" {
|
||||
return encodeLOONG64Jirl(op, ops)
|
||||
}
|
||||
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
|
||||
return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
|
||||
}
|
||||
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
|
||||
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
|
||||
if op, ok := l64branch21Table[mnem]; ok {
|
||||
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
|
||||
return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
|
||||
}
|
||||
// B/BL aliases reached only via JMP/JAL above.
|
||||
|
||||
@@ -533,11 +834,23 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
||||
//
|
||||
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
|
||||
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
|
||||
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
|
||||
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
|
||||
if len(instr.Operands) != 1 {
|
||||
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
|
||||
}
|
||||
op := instr.Operands[0]
|
||||
// PC-relative displacement: N(PC) resolves to the instruction N slots
|
||||
// away in source order (the toolchain's parse-time count), and the field
|
||||
// carries the final pc distance in instruction units.
|
||||
if op.Addr.Sym == nil && op.Addr.Base == "PC" {
|
||||
targetPc, ok := pcRelPcs[instr]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset)
|
||||
}
|
||||
v := (targetPc - pc) >> 2
|
||||
opc := l64jumpTable[mnem]
|
||||
return l64wordLE(l64bbl(opc, v)), nil
|
||||
}
|
||||
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
|
||||
// Indirect: (rj) → jirl.
|
||||
rj := loong64RegNum(op.Addr.Base)
|
||||
@@ -620,16 +933,28 @@ func l64offsetOperand(op *ast.Operand) (int32, bool) {
|
||||
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
|
||||
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
|
||||
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
|
||||
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
||||
func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
|
||||
if len(ops) != 2 && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
target := resolve(l64Label(ops[len(ops)-1]))
|
||||
var target string
|
||||
var v int
|
||||
lastOp := ops[len(ops)-1]
|
||||
if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" {
|
||||
// N(PC) resolves to the instruction N slots away in source order.
|
||||
targetPc, ok := pcRelPcs[instr]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset)
|
||||
}
|
||||
v = (targetPc - pc) >> 2
|
||||
} else {
|
||||
target = resolve(l64Label(lastOp))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
v := (targetOff - pc) >> 2
|
||||
v = (targetOff - pc) >> 2
|
||||
}
|
||||
if len(ops) == 2 {
|
||||
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
|
||||
// BGEZ-family aliases encoded with rj in the rj field.
|
||||
@@ -690,33 +1015,55 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
|
||||
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
|
||||
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
|
||||
// and a 16-bit offset, are handled separately.
|
||||
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
||||
if len(ops) != 2 {
|
||||
func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
|
||||
isBF := mnem == "BFPT" || mnem == "BFPF"
|
||||
if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
target := resolve(l64Label(ops[1]))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
v := (targetOff - pc) >> 2
|
||||
rj := 0 // BFPT/BFPF default to FCC0
|
||||
if mnem != "BFPT" && mnem != "BFPF" {
|
||||
var rj int
|
||||
tgtOp := ops[len(ops)-1]
|
||||
if isBF {
|
||||
// BFPT/BFPF test an FCC condition register, defaulting to FCC0 when
|
||||
// spelled without one.
|
||||
rj = 0
|
||||
if len(ops) == 2 {
|
||||
rj = l64Reg(ops[0])
|
||||
if rj < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand")
|
||||
}
|
||||
}
|
||||
} else {
|
||||
rj = l64Reg(ops[0])
|
||||
if rj < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand")
|
||||
}
|
||||
}
|
||||
var v int
|
||||
if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" {
|
||||
// N(PC) resolves to the instruction N slots away in source order.
|
||||
targetPc, ok := pcRelPcs[instr]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset)
|
||||
}
|
||||
v = (targetPc - pc) >> 2
|
||||
} else {
|
||||
target := resolve(l64Label(tgtOp))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
v = (targetOff - pc) >> 2
|
||||
}
|
||||
if mnem == "BGTZ" || mnem == "BLEZ" {
|
||||
// The toolchain swaps the register into the rd field and keeps the
|
||||
// 16-bit offset form.
|
||||
if (v<<16)>>16 != v {
|
||||
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
|
||||
return nil, fmt.Errorf("branch %d too far (16-bit range)", v)
|
||||
}
|
||||
return l64wordLE(l64irr16(op, v, 0, rj)), nil
|
||||
}
|
||||
if (v<<11)>>11 != v {
|
||||
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
||||
return nil, fmt.Errorf("branch %d too far (21-bit range)", v)
|
||||
}
|
||||
return l64wordLE(l64ir21(op, v, rj)), nil
|
||||
}
|
||||
|
||||
+557
-38
@@ -4,7 +4,9 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
@@ -31,32 +33,54 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
}
|
||||
|
||||
// Pass 1: collect instructions and compute label offsets assuming 4 bytes
|
||||
// per instruction (or 8 for MOV $large-imm). No encoding yet.
|
||||
// per instruction (or 8 for MOV $large-imm). No encoding yet. PCALIGN
|
||||
// contributes only its padding, which is attached to the following
|
||||
// instruction and emitted ahead of it. A relaxed branch carries the
|
||||
// inverted condition and is followed by an inserted JMP rec (jmpTo set)
|
||||
// that carries the original target.
|
||||
type instrRec struct {
|
||||
instr *ast.Instr
|
||||
compressed bool
|
||||
code []byte
|
||||
pad int
|
||||
relaxed bool
|
||||
jmpTo string
|
||||
}
|
||||
var recs []instrRec
|
||||
offsets := map[string]int{}
|
||||
pos := guardLen + len(prologue)
|
||||
pendingPad := 0
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
recs = append(recs, instrRec{instr: s})
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
|
||||
pendingPad += riscvPCAlignPad(pos, s)
|
||||
pos += riscvPCAlignPad(pos, s)
|
||||
continue
|
||||
}
|
||||
recs = append(recs, instrRec{instr: s, pad: pendingPad})
|
||||
pendingPad = 0
|
||||
pos += riscvInstrSize(s, fi)
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: encode each instruction using Pass-1 offsets.
|
||||
// Pass 2: encode each instruction using Pass-1 offsets. A branch or
|
||||
// jump the offsets prove overlong encodes to a 4-byte placeholder: the
|
||||
// relaxation pass rewrites it before the final encoding. pcRelPcs is
|
||||
// unavailable this early, so the N(PC) forms take the same placeholder
|
||||
// path.
|
||||
pc := len(prologue)
|
||||
for i := range recs {
|
||||
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2
|
||||
if err != nil {
|
||||
branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
|
||||
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil) // no relocs in Pass 2
|
||||
if err != nil && !(branchLike && riscvIsRangeError(err)) {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
|
||||
}
|
||||
if err != nil {
|
||||
code = make([]byte, 4)
|
||||
}
|
||||
recs[i].code = code
|
||||
pc += len(code)
|
||||
}
|
||||
@@ -72,20 +96,100 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
// Pass 4: recompute offsets with actual sizes. recs holds the
|
||||
// instructions in emission order, so an index into it walks t.Body in
|
||||
// lockstep (the same single pass Pass 1 uses) instead of rescanning the
|
||||
// whole slice per statement.
|
||||
// whole slice per statement. PCALIGN padding is recomputed here, since
|
||||
// compression has shifted instruction sizes since Pass 1.
|
||||
offsets = map[string]int{}
|
||||
pos = guardLen + len(prologue)
|
||||
ri := 0
|
||||
pendingPad = 0
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
|
||||
pad := riscvPCAlignPad(pos, s)
|
||||
pendingPad += pad
|
||||
pos += pad
|
||||
continue
|
||||
}
|
||||
recs[ri].pad = pendingPad
|
||||
pendingPad = 0
|
||||
pos += len(recs[ri].code)
|
||||
ri++
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 4b: relax overlong conditional branches exactly as the toolchain
|
||||
// does: invert the branch condition, point it at the instruction after an
|
||||
// inserted JMP, let the JMP carry the original target, and re-layout until
|
||||
// a pass inserts nothing. Inserted JMP recs share their branch's source
|
||||
// line and trail it in emission order, so the body walk flushes them
|
||||
// before every statement and at the end.
|
||||
var pcRelPcs map[*ast.Instr]int
|
||||
for {
|
||||
offsets = map[string]int{}
|
||||
pos = guardLen + len(prologue)
|
||||
ri := 0
|
||||
pcs := make([]int, len(recs))
|
||||
flushJmps := func() {
|
||||
for ri < len(recs) && recs[ri].jmpTo != "" {
|
||||
pcs[ri] = pos
|
||||
pos += 4
|
||||
ri++
|
||||
}
|
||||
}
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
flushJmps()
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
flushJmps()
|
||||
if ri >= len(recs) {
|
||||
continue
|
||||
}
|
||||
pcs[ri] = pos + recs[ri].pad
|
||||
pos += recs[ri].pad + len(recs[ri].code)
|
||||
ri++
|
||||
}
|
||||
}
|
||||
flushJmps()
|
||||
|
||||
changed := false
|
||||
for i := range recs {
|
||||
r := &recs[i]
|
||||
if r.relaxed || r.jmpTo != "" {
|
||||
continue
|
||||
}
|
||||
mnem := strings.ToUpper(r.instr.Mnemonic.Text)
|
||||
if !riscvIsCondBranch(mnem) || len(r.instr.Operands) == 0 {
|
||||
continue
|
||||
}
|
||||
target := labelFromOperand(r.instr.Operands[len(r.instr.Operands)-1])
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if delta := int32(targetOff - pcs[i]); delta < -4096 || delta >= 4096 {
|
||||
r.relaxed = true
|
||||
recs = slices.Insert(recs, i+1, instrRec{instr: r.instr, jmpTo: target})
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
// Capture the final pcs for the N(PC) branch forms: their target
|
||||
// is the instruction N source slots away, resolved by index.
|
||||
pcRelPcs = map[*ast.Instr]int{}
|
||||
for i := range recs {
|
||||
if _, ok := riscvPCRelOffset(recs[i].instr); ok {
|
||||
pcRelPcs[recs[i].instr] = pcs[i]
|
||||
}
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 5: re-encode branches with corrected offsets. Record relocations
|
||||
// during this final pass (relocation offsets are relative to instruction
|
||||
// start). The guard prefix precedes the prologue; its branches target
|
||||
@@ -104,12 +208,40 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
preCount := len(relocs)
|
||||
var lines []LineEntry
|
||||
for _, r := range recs {
|
||||
// PCALIGN padding precedes the instruction it was attached to.
|
||||
if r.pad > 0 {
|
||||
out = append(out, riscvPadBytes(r.pad)...)
|
||||
pc += r.pad
|
||||
}
|
||||
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
|
||||
if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) {
|
||||
out = append(out, r.code...)
|
||||
pc += len(r.code)
|
||||
} else {
|
||||
code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs)
|
||||
var code []byte
|
||||
switch {
|
||||
case r.jmpTo != "":
|
||||
// The JMP a relaxation inserted: JAL X0 to the original target.
|
||||
targetOff, ok := offsets[r.jmpTo]
|
||||
if !ok {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
|
||||
}
|
||||
offset := int32(targetOff - pc)
|
||||
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
word := riscvJType(0, offset)
|
||||
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
|
||||
case r.relaxed:
|
||||
// The inverted half of a relaxed branch: it targets the inserted
|
||||
// JMP, always the very next instruction (offset 4).
|
||||
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
|
||||
if !ok {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
|
||||
}
|
||||
word := riscvBType(enc, rs1, rs2, 4)
|
||||
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
|
||||
case r.compressed && !isBranchLike(r.instr.Mnemonic.Text):
|
||||
code = r.code
|
||||
default:
|
||||
var err error
|
||||
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
@@ -131,10 +263,10 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
|
||||
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
|
||||
}
|
||||
}
|
||||
out = append(out, code...)
|
||||
pc += len(code)
|
||||
}
|
||||
}
|
||||
if fi.needSplit {
|
||||
relocs = append(relocs, guardReloc)
|
||||
}
|
||||
@@ -150,6 +282,8 @@ var riscvImmAlias = map[string]string{
|
||||
"AND": "ANDI",
|
||||
"OR": "ORI",
|
||||
"XOR": "XORI",
|
||||
"SLT": "SLTI",
|
||||
"SLTU": "SLTIU",
|
||||
"SLL": "SLLI",
|
||||
"SRL": "SRLI",
|
||||
"SRA": "SRAI",
|
||||
@@ -178,6 +312,35 @@ func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
|
||||
return mnem, false
|
||||
}
|
||||
|
||||
// riscvPCAlignPad returns the padding PCALIGN inserts before the next
|
||||
// instruction so that it starts at the requested boundary relative to the
|
||||
// function start. The boundary must be a power of two between 8 and 2048, as
|
||||
// the toolchain requires; anything else pads nothing.
|
||||
func riscvPCAlignPad(pos int, instr *ast.Instr) int {
|
||||
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
|
||||
return 0
|
||||
}
|
||||
align := int(immFromOperand(instr.Operands[0]))
|
||||
if align < 8 || align > 2048 || align&(align-1) != 0 {
|
||||
return 0
|
||||
}
|
||||
return (align - pos%align) % align
|
||||
}
|
||||
|
||||
// riscvPadBytes renders PCALIGN padding: 4-byte NOPs (addi $0, X0, X0) with a
|
||||
// trailing 2-byte compressed NOP when the pad is 2 mod 4, exactly as the
|
||||
// toolchain lays the bytes down.
|
||||
func riscvPadBytes(pad int) []byte {
|
||||
out := make([]byte, 0, pad)
|
||||
for ; pad >= 4; pad -= 4 {
|
||||
out = append(out, 0x13, 0x00, 0x00, 0x00)
|
||||
}
|
||||
if pad == 2 {
|
||||
out = append(out, 0x01, 0x00)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
|
||||
// Most instructions are 4 bytes; MOV with a large immediate and I-type
|
||||
// arithmetic with a large immediate expand to several (possibly compressed)
|
||||
@@ -224,6 +387,10 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
|
||||
}
|
||||
return riscvItypeImmediateSize(mnem, imm)
|
||||
}
|
||||
// BYTE lays down one raw byte per operand.
|
||||
if mnem == "BYTE" {
|
||||
return len(ops)
|
||||
}
|
||||
// The toolchain's synthesised instructions: some emit one word, others
|
||||
// expand to a fixed sequence.
|
||||
return riscvExtendedSize(mnem, ops)
|
||||
@@ -318,12 +485,171 @@ func isBranchLike(mnem string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// riscvIsCondBranch reports whether m is a conditional branch, the only
|
||||
// instruction class branch relaxation rewrites.
|
||||
func riscvIsCondBranch(mnem string) bool {
|
||||
switch mnem {
|
||||
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU",
|
||||
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// riscvCSRNames maps the standard CSR mnemonics the assembler accepts onto
|
||||
// their addresses.
|
||||
var riscvCSRNames = map[string]int32{
|
||||
"FFLAGS": 0x001,
|
||||
"FRM": 0x002,
|
||||
"FCSR": 0x003,
|
||||
"VSTART": 0x008,
|
||||
"VXSAT": 0x009,
|
||||
"VXRM": 0x00A,
|
||||
"VCSR": 0x00F,
|
||||
"CYCLE": 0xC00,
|
||||
"TIME": 0xC01,
|
||||
"INSTRET": 0xC02,
|
||||
"CYCLEH": 0xC80,
|
||||
"TIMEH": 0xC81,
|
||||
"INSTRETH": 0xC82,
|
||||
"VL": 0xC20,
|
||||
"VLENB": 0xC22,
|
||||
}
|
||||
|
||||
// riscvCSRAddress resolves a CSR operand: an integer immediate or one of the
|
||||
// standard CSR names.
|
||||
func riscvCSRAddress(op *ast.Operand) (int32, bool) {
|
||||
if isImmOperand(op) {
|
||||
return immFromOperand(op), true
|
||||
}
|
||||
if op.Addr.Sym != nil {
|
||||
if v, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// riscvPCRelOffset reports the N of a branch or jump operand spelled N(PC):
|
||||
// the displacement counted in source instructions from the branch itself.
|
||||
func riscvPCRelOffset(instr *ast.Instr) (int, bool) {
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
switch mnem {
|
||||
case "JMP":
|
||||
if len(instr.Operands) != 1 {
|
||||
return 0, false
|
||||
}
|
||||
case "JAL":
|
||||
if len(instr.Operands) != 1 && len(instr.Operands) != 2 {
|
||||
return 0, false
|
||||
}
|
||||
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
|
||||
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
|
||||
if len(instr.Operands) < 2 {
|
||||
return 0, false
|
||||
}
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
op := instr.Operands[len(instr.Operands)-1]
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
|
||||
return int(op.Addr.Offset), true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// riscvPCRelTargetOff resolves the target displacement of a branch whose last
|
||||
// operand is N(PC): the toolchain's parser counts the source instructions at
|
||||
// a uniform 4 bytes, so the target is the instruction N slots away, and the
|
||||
// displacement tracks that instruction's final pc. A nil pcRelPcs (the
|
||||
// layout passes) yields a placeholder range error; the caller tolerates it
|
||||
// for branch-like instructions.
|
||||
func riscvPCRelTargetOff(instr *ast.Instr, pc int, pcRelPcs map[*ast.Instr]int) (int, bool, error) {
|
||||
off, ok := riscvPCRelOffset(instr)
|
||||
if !ok {
|
||||
return 0, false, nil
|
||||
}
|
||||
if pcRelPcs == nil {
|
||||
return 0, true, &riscvRangeError{"pc-relative placeholder"}
|
||||
}
|
||||
targetPc, ok := pcRelPcs[instr]
|
||||
if !ok {
|
||||
return 0, true, fmt.Errorf("PC-relative target %d out of range", off)
|
||||
}
|
||||
return targetPc, true, nil
|
||||
}
|
||||
|
||||
// riscvInvertedBranchEnc returns the encoding of mnem's inverted condition
|
||||
// for the given operands: InvertBranch's table applied at the encoding level.
|
||||
// The register operands are already in position for the inverted form.
|
||||
func riscvInvertedBranchEnc(mnem string, ops []*ast.Operand) (riscvEnc, int, int, bool) {
|
||||
reg := func(i int) int { return regFromOperand(ops[i]) }
|
||||
switch mnem {
|
||||
case "BEQ": // → BNE rs1, rs2
|
||||
return riscvEnc{0x63, 0x1, 0x00}, reg(0), reg(1), true
|
||||
case "BNE": // → BEQ rs1, rs2
|
||||
return riscvEnc{0x63, 0x0, 0x00}, reg(0), reg(1), true
|
||||
case "BLT": // → BGE rs1, rs2
|
||||
return riscvEnc{0x63, 0x5, 0x00}, reg(0), reg(1), true
|
||||
case "BGE": // → BLT rs1, rs2
|
||||
return riscvEnc{0x63, 0x4, 0x00}, reg(0), reg(1), true
|
||||
case "BLTU": // → BGEU rs1, rs2
|
||||
return riscvEnc{0x63, 0x7, 0x00}, reg(0), reg(1), true
|
||||
case "BGEU": // → BLTU rs1, rs2
|
||||
return riscvEnc{0x63, 0x6, 0x00}, reg(0), reg(1), true
|
||||
case "BEQZ": // → BNEZ rs, X0
|
||||
return riscvEnc{0x63, 0x1, 0x00}, reg(0), 0, true
|
||||
case "BNEZ": // → BEQZ rs, X0
|
||||
return riscvEnc{0x63, 0x0, 0x00}, reg(0), 0, true
|
||||
case "BLTZ": // → BGEZ rs, X0
|
||||
return riscvEnc{0x63, 0x5, 0x00}, reg(0), 0, true
|
||||
case "BGEZ": // → BLTZ rs, X0
|
||||
return riscvEnc{0x63, 0x4, 0x00}, reg(0), 0, true
|
||||
case "BLEZ": // → BGTZ: blt X0, rs
|
||||
return riscvEnc{0x63, 0x4, 0x00}, 0, reg(0), true
|
||||
case "BGTZ": // → BLEZ: bge X0, rs
|
||||
return riscvEnc{0x63, 0x5, 0x00}, 0, reg(0), true
|
||||
case "BGT": // → BLE: bge rs2, rs1
|
||||
return riscvEnc{0x63, 0x5, 0x00}, reg(1), reg(0), true
|
||||
case "BLE": // → BGT: blt rs2, rs1
|
||||
return riscvEnc{0x63, 0x4, 0x00}, reg(1), reg(0), true
|
||||
case "BGTU": // → BLEU: bgeu rs2, rs1
|
||||
return riscvEnc{0x63, 0x7, 0x00}, reg(1), reg(0), true
|
||||
case "BLEU": // → BGTU: bltu rs2, rs1
|
||||
return riscvEnc{0x63, 0x6, 0x00}, reg(1), reg(0), true
|
||||
}
|
||||
return riscvEnc{}, 0, 0, false
|
||||
}
|
||||
|
||||
// riscvRangeError reports a branch or jump displacement beyond its
|
||||
// architecture limit. The layout passes tolerate it (the relaxation pass
|
||||
// rewrites overlong conditional branches before the final encoding); a range
|
||||
// error reaching the final pass is a real failure.
|
||||
type riscvRangeError struct{ msg string }
|
||||
|
||||
func (e *riscvRangeError) Error() string { return e.msg }
|
||||
|
||||
// riscvIsRangeError reports whether err is a displacement-range rejection.
|
||||
func riscvIsRangeError(err error) bool {
|
||||
var re *riscvRangeError
|
||||
return errors.As(err, &re)
|
||||
}
|
||||
|
||||
// riscvRoundModes maps the rounding-mode suffixes onto their funct7 codes.
|
||||
var riscvRoundModes = map[string]uint32{
|
||||
"RNE": 0,
|
||||
"RTZ": 1,
|
||||
"RDN": 2,
|
||||
"RUP": 3,
|
||||
"RMM": 4,
|
||||
}
|
||||
|
||||
// riscvCheckBranchOffset rejects a B-type displacement outside its signed
|
||||
// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an
|
||||
// out-of-range offset would otherwise wrap to a wrong target.
|
||||
func riscvCheckBranchOffset(target string, off int32) error {
|
||||
if off < -4096 || off > 4094 {
|
||||
return fmt.Errorf("branch to %q too far (13-bit range)", target)
|
||||
return &riscvRangeError{fmt.Sprintf("branch to %q too far (13-bit range)", target)}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -332,13 +658,13 @@ func riscvCheckBranchOffset(target string, off int32) error {
|
||||
// 21-bit span [-1048576, 1048574].
|
||||
func riscvCheckJumpOffset(target string, off int32) error {
|
||||
if off < -1048576 || off > 1048574 {
|
||||
return fmt.Errorf("jump to %q too far (21-bit range)", target)
|
||||
return &riscvRangeError{fmt.Sprintf("jump to %q too far (21-bit range)", target)}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeRISCVInstr encodes a single RISC-V instruction.
|
||||
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
|
||||
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
|
||||
mnem := instr.Mnemonic.Text
|
||||
ops := instr.Operands
|
||||
var immNeg bool
|
||||
@@ -351,6 +677,27 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
// RET = epilogue (restore LR and close the frame when present) +
|
||||
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
|
||||
return riscvReturn(fi), nil
|
||||
case "WORD":
|
||||
// WORD $w lays down a raw 32-bit little-endian word.
|
||||
if len(ops) != 1 {
|
||||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
w := int64(immFromOperand(ops[0]))
|
||||
if w < 0 || w > 0xFFFFFFFF {
|
||||
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
|
||||
}
|
||||
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}, nil
|
||||
case "BYTE":
|
||||
// BYTE $b lays down one raw byte per operand.
|
||||
var out []byte
|
||||
for _, op := range ops {
|
||||
b := int64(immFromOperand(op))
|
||||
if b < 0 || b > 0xFF {
|
||||
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
|
||||
}
|
||||
out = append(out, byte(b))
|
||||
}
|
||||
return out, nil
|
||||
case "CALL":
|
||||
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
|
||||
// relocation. The Go assembler rejects CALL to a local branch label.
|
||||
@@ -404,6 +751,17 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
offset := int32(off - pc)
|
||||
if err := riscvCheckJumpOffset("", offset); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
word = riscvJType(0, offset)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
}
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
@@ -424,6 +782,18 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
} else if len(ops) == 1 {
|
||||
target = labelFromOperand(ops[0])
|
||||
}
|
||||
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
targetOff := off
|
||||
offset := int32(targetOff - pc)
|
||||
if err := riscvCheckJumpOffset(target, offset); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
word = riscvJType(rd, offset)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
@@ -456,11 +826,21 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid register", mnem)
|
||||
}
|
||||
target := labelFromOperand(ops[1])
|
||||
targetOff, ok := offsets[target]
|
||||
targetOff := 0
|
||||
target := ""
|
||||
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
targetOff = off
|
||||
} else {
|
||||
target = labelFromOperand(ops[1])
|
||||
var ok bool
|
||||
targetOff, ok = offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
}
|
||||
var enc riscvEnc
|
||||
rs1, rs2 := rs, 0
|
||||
switch mnem {
|
||||
@@ -484,18 +864,25 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
|
||||
// System instructions with no operands.
|
||||
case "FENCE", "ECALL", "EBREAK":
|
||||
case "FENCE", "ECALL", "EBREAK", "FENCE.TSO", "PAUSE":
|
||||
enc, ok := riscvInstrTable[mnem]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unsupported system instruction %q", mnem)
|
||||
}
|
||||
// The bare FENCE expands to fence iorw, iorw: the predecessor and
|
||||
// successor fields both carry 0xF in the I-type immediate
|
||||
// (the toolchain's encodeFenceOperand TYPE_NONE default).
|
||||
// (the toolchain's encodeFenceOperand TYPE_NONE default). FENCE.TSO
|
||||
// carries the TSO fence mode with RW predecessor and successor.
|
||||
imm := int32(0)
|
||||
if mnem == "FENCE" {
|
||||
imm = 0x0FF
|
||||
}
|
||||
if mnem == "FENCE.TSO" {
|
||||
imm = 0x833
|
||||
}
|
||||
if mnem == "PAUSE" {
|
||||
imm = 0x010
|
||||
}
|
||||
word = riscvIType(enc, 0, 0, imm)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
@@ -516,6 +903,29 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
|
||||
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
|
||||
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
|
||||
// low three bits of the funct7 field.
|
||||
if i := strings.IndexByte(mnem, '.'); i > 0 {
|
||||
if base, ok := riscvCvtTable[mnem[:i]]; ok {
|
||||
rm, ok := riscvRoundModes[mnem[i+1:]]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unsupported rounding mode in %q", mnem)
|
||||
}
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rs1 := regFromOperand(ops[0])
|
||||
rd := regFromOperand(ops[1])
|
||||
if rd < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||
}
|
||||
base.funct7 = (base.funct7 &^ 7) | rm
|
||||
word := riscvCvtType(base, rd, rs1)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
}
|
||||
|
||||
// R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last).
|
||||
if fmaEnc, ok := riscvFmaTable[mnem]; ok {
|
||||
if len(ops) != 4 {
|
||||
@@ -532,29 +942,103 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
|
||||
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last).
|
||||
if csrEnc, ok := riscvCsrTable[mnem]; ok {
|
||||
if len(ops) != 3 {
|
||||
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last). The
|
||||
// write-only pseudos (CSRS/CSRC/CSRW and their immediate forms) spell
|
||||
// the source first, the CSR second, and read the destination as X0; the
|
||||
// immediate or register variant follows the source operand's kind.
|
||||
csrMnem := mnem
|
||||
csrPseudo := false
|
||||
csrRead := false
|
||||
csrFix := int32(0)
|
||||
switch mnem {
|
||||
case "CSRS", "CSRW", "CSRC", "CSRSI", "CSRWI", "CSRCI":
|
||||
csrMnem = map[string]string{
|
||||
"CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC",
|
||||
"CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI",
|
||||
}[mnem]
|
||||
csrPseudo = true
|
||||
// CSRR csr, rd is CSRRS rd, csr, X0; the read pseudos RDCYCLE/RDTIME/
|
||||
// RDINSTRET fix the CSR to cycle/time/instret.
|
||||
case "CSRR":
|
||||
csrMnem = "CSRRS"
|
||||
csrPseudo = true
|
||||
csrRead = true
|
||||
case "RDCYCLE", "RDTIME", "RDINSTRET":
|
||||
csrMnem = "CSRRS"
|
||||
csrPseudo = true
|
||||
csrRead = true
|
||||
csrFix = map[string]int32{"RDCYCLE": 0xC00, "RDTIME": 0xC01, "RDINSTRET": 0xC02}[mnem]
|
||||
}
|
||||
if csrEnc, ok := riscvCsrTable[csrMnem]; ok {
|
||||
if csrRead && len(ops) != 1 && len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
if csrPseudo && !csrRead && len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
if !csrPseudo && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
csr := immFromOperand(ops[0]) // CSR address (12-bit)
|
||||
csrOp := ops[0]
|
||||
srcOp := ops[0]
|
||||
rdOp := ops[len(ops)-1]
|
||||
switch {
|
||||
case csrRead:
|
||||
// CSRR csr, rd (or the fixed-CSR read pseudos with only rd).
|
||||
case csrPseudo:
|
||||
// src, csr.
|
||||
if len(ops) > 1 {
|
||||
csrOp, srcOp = ops[1], ops[0]
|
||||
}
|
||||
rdOp = nil
|
||||
default:
|
||||
// Either src, csr, rd or csr, src, rd: a CSR *name* in the
|
||||
// second operand marks the toolchain's order.
|
||||
srcOp = ops[1]
|
||||
if op := ops[1]; op.Addr.Sym != nil {
|
||||
if _, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
|
||||
csrOp, srcOp = ops[1], ops[0]
|
||||
}
|
||||
}
|
||||
}
|
||||
csr, ok := riscvCSRAddress(csrOp)
|
||||
if !ok && csrFix == 0 {
|
||||
return nil, fmt.Errorf("%s: unknown CSR %q", mnem, csrOp.Raw)
|
||||
}
|
||||
if csrFix != 0 {
|
||||
csr = csrFix
|
||||
}
|
||||
if csr < 0 || csr > 0xFFF {
|
||||
return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr)
|
||||
}
|
||||
rd := regFromOperand(ops[2]) // destination register
|
||||
rd := 0
|
||||
if !csrPseudo {
|
||||
rd = regFromOperand(rdOp) // destination register
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("invalid destination register in %s", mnem)
|
||||
}
|
||||
}
|
||||
if csrRead {
|
||||
rd = regFromOperand(rdOp)
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("invalid destination register in %s", mnem)
|
||||
}
|
||||
}
|
||||
var src int
|
||||
if csrEnc.imm {
|
||||
// Immediate variant: ops[1] is a 5-bit unsigned immediate.
|
||||
src = int(immFromOperand(ops[1]))
|
||||
switch {
|
||||
case csrRead:
|
||||
// CSRR reads with rs1 = X0: src stays zero.
|
||||
case isImmOperand(srcOp):
|
||||
// Immediate variant: the source is a 5-bit unsigned immediate.
|
||||
src = int(immFromOperand(srcOp))
|
||||
if src < 0 || src > 31 {
|
||||
return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem)
|
||||
}
|
||||
} else {
|
||||
// Register variant: ops[1] is a register.
|
||||
src = regFromOperand(ops[1])
|
||||
case csrEnc.imm:
|
||||
return nil, fmt.Errorf("%s expects an immediate source", mnem)
|
||||
default:
|
||||
// Register variant: the source is a register.
|
||||
src = regFromOperand(srcOp)
|
||||
if src < 0 {
|
||||
return nil, fmt.Errorf("invalid source register in %s", mnem)
|
||||
}
|
||||
@@ -738,15 +1222,35 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
}
|
||||
word = riscvSType(enc, rs1, rs2, imm)
|
||||
|
||||
// Branches: rs1, rs2, label.
|
||||
// Branches: rs1, rs2, label. BGT/BLE/BGTU/BLEU are the swapped-spelling
|
||||
// forms of BLT/BGE/BLTU/BGEU (bgt rs1, rs2 is blt rs2, rs1).
|
||||
case len(ops) == 3 && isBranchInstr(mnem):
|
||||
rs1 := regFromOperand(ops[0])
|
||||
rs2 := regFromOperand(ops[1])
|
||||
target := labelFromOperand(ops[2])
|
||||
targetOff, ok := offsets[target]
|
||||
switch mnem {
|
||||
case "BGT":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, rs2, rs1 // blt rs2, rs1
|
||||
case "BLE":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, rs2, rs1 // bge rs2, rs1
|
||||
case "BGTU":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x6, 0x00}, rs2, rs1 // bltu rs2, rs1
|
||||
case "BLEU":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x7, 0x00}, rs2, rs1 // bgeu rs2, rs1
|
||||
}
|
||||
targetOff := 0
|
||||
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
targetOff = off
|
||||
} else {
|
||||
var ok bool
|
||||
targetOff, ok = offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
}
|
||||
offset := int32(targetOff - pc)
|
||||
if rs1 < 0 || rs2 < 0 {
|
||||
return nil, fmt.Errorf("invalid register in %s", mnem)
|
||||
@@ -758,10 +1262,15 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
|
||||
word = riscvBType(enc, rs1, rs2, offset)
|
||||
|
||||
// U-type: rd, imm.
|
||||
// U-type: rd, imm (or the toolchain testdata's INSTR $imm, rd).
|
||||
case len(ops) == 2 && isUTypeInstr(mnem):
|
||||
rd := regFromOperand(ops[0])
|
||||
imm := immFromOperand(ops[1])
|
||||
var rd int
|
||||
var imm int32
|
||||
if isImmOperand(ops[0]) {
|
||||
imm, rd = immFromOperand(ops[0]), regFromOperand(ops[1])
|
||||
} else {
|
||||
rd, imm = regFromOperand(ops[0]), immFromOperand(ops[1])
|
||||
}
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register in %s", mnem)
|
||||
}
|
||||
@@ -1226,6 +1735,11 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
|
||||
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
|
||||
mnem := riscvCompressMnem(instr)
|
||||
ops := instr.Operands
|
||||
// The immediate aliases fold onto their I-type mnemonics before
|
||||
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly
|
||||
// as it compresses the spelling ADDI.
|
||||
var immNeg bool
|
||||
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
|
||||
|
||||
switch mnem {
|
||||
case "LD", "MOV":
|
||||
@@ -1289,6 +1803,9 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
|
||||
|
||||
case "ADDI":
|
||||
rd, rs1, imm := extractITypeParams(instr)
|
||||
if immNeg {
|
||||
imm = -imm
|
||||
}
|
||||
if rd == -1 || rs1 == -1 {
|
||||
return 0, false
|
||||
}
|
||||
@@ -2045,7 +2562,8 @@ func isRTypeInstr(m string) bool {
|
||||
case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND",
|
||||
"ADDW", "SUBW", "SLLW", "SRLW", "SRAW",
|
||||
"MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU",
|
||||
"MULW", "DIVW", "DIVUW", "REMW", "REMUW":
|
||||
"MULW", "DIVW", "DIVUW", "REMW", "REMUW",
|
||||
"CZEROEQZ", "CZERONEZ":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
@@ -2085,7 +2603,7 @@ func isStoreInstr(m string) bool {
|
||||
|
||||
func isBranchInstr(m string) bool {
|
||||
switch m {
|
||||
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU":
|
||||
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
@@ -2111,7 +2629,8 @@ func isFPArithInstr(m string) bool {
|
||||
switch m {
|
||||
case "FADDS", "FSUBS", "FMULS", "FDIVS",
|
||||
"FADDD", "FSUBD", "FMULD", "FDIVD",
|
||||
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD":
|
||||
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
|
||||
"FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
|
||||
@@ -219,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"DIVUW": {0x3B, 0x5, 0x01},
|
||||
"REMW": {0x3B, 0x6, 0x01},
|
||||
"REMUW": {0x3B, 0x7, 0x01},
|
||||
// Zicond conditional zeroing.
|
||||
"CZEROEQZ": {0x33, 0x5, 0x07},
|
||||
"CZERONEZ": {0x33, 0x7, 0x07},
|
||||
// RV64I, I-type arithmetic.
|
||||
"ADDI": {0x13, 0x0, 0x00},
|
||||
"ADDIW": {0x1B, 0x0, 0x00},
|
||||
@@ -247,6 +250,12 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"BGE": {0x63, 0x5, 0x00},
|
||||
"BLTU": {0x63, 0x6, 0x00},
|
||||
"BGEU": {0x63, 0x7, 0x00},
|
||||
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
|
||||
// with the register operands swapped.
|
||||
"BGT": {0x63, 0x4, 0x00},
|
||||
"BLE": {0x63, 0x5, 0x00},
|
||||
"BGTU": {0x63, 0x6, 0x00},
|
||||
"BLEU": {0x63, 0x7, 0x00},
|
||||
// U-type.
|
||||
"LUI": {0x37, 0x0, 0x00},
|
||||
"AUIPC": {0x17, 0x0, 0x00},
|
||||
@@ -254,6 +263,8 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"ECALL": {0x73, 0x0, 0x00},
|
||||
"EBREAK": {0x73, 0x0, 0x00},
|
||||
"FENCE": {0x0F, 0x0, 0x00},
|
||||
"FENCE.TSO": {0x0F, 0x0, 0x00},
|
||||
"PAUSE": {0x0F, 0x0, 0x00},
|
||||
// JALR, indirect jump/call (I-type).
|
||||
"JALR": {0x67, 0x0, 0x00},
|
||||
|
||||
@@ -304,6 +315,13 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
// FP sign injection (double): rs2 carries the sign source.
|
||||
"FSGNJD": {0x53, 0x0, 0x11},
|
||||
"FSGNJS": {0x53, 0x0, 0x10},
|
||||
"FSGNJX": {0x53, 0x0, 0x14},
|
||||
"FSGNJXD": {0x53, 0x0, 0x15},
|
||||
"FSGNJXS": {0x53, 0x0, 0x14},
|
||||
"FSGNJND": {0x53, 0x1, 0x11},
|
||||
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||
"FSGNJNX": {0x53, 0x1, 0x14},
|
||||
|
||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||
@@ -375,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
// The toolchain's W/D suffix spellings of the same moves.
|
||||
"FMVXS": {0x70, 0x0, 0x53},
|
||||
"FMVFS": {0x78, 0x0, 0x53},
|
||||
"FMVSX": {0x79, 0x0, 0x53},
|
||||
}
|
||||
|
||||
// riscvCvtType encodes an FP conversion instruction.
|
||||
|
||||
@@ -868,7 +868,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
|
||||
t.Helper()
|
||||
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
||||
instr := fn.Body[0].(*ast.Instr)
|
||||
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
|
||||
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil)
|
||||
}
|
||||
|
||||
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
||||
@@ -905,9 +905,10 @@ func TestRISCVBranchJumpRange(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRISCVBranchFarBody drives the range check through the full two-pass
|
||||
// assembler: a forward branch over a body larger than the B-type span must
|
||||
// error rather than wrap.
|
||||
// TestRISCVBranchFarBody drives the relaxation pass through the full
|
||||
// assembler: a forward branch over a body larger than the B-type span is
|
||||
// rewritten as an inverted branch over an inserted JMP, the same layout the
|
||||
// toolchain produces, instead of wrapping to a wrong target.
|
||||
func TestRISCVBranchFarBody(t *testing.T) {
|
||||
var sb strings.Builder
|
||||
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
||||
@@ -916,8 +917,20 @@ func TestRISCVBranchFarBody(t *testing.T) {
|
||||
}
|
||||
sb.WriteString("done:\n\tRET\n")
|
||||
fn := firstTextRISCV(t, sb.String())
|
||||
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||
t.Error("expected a branch-out-of-range error, got none")
|
||||
out, _, _, _, _, err := assembleRISCV(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
|
||||
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
|
||||
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
|
||||
if !bytes.Equal(out[0:4], wantBranch) {
|
||||
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
|
||||
}
|
||||
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
|
||||
wantJmp := wordLE(riscvJType(0, 4404))
|
||||
if !bytes.Equal(out[4:8], wantJmp) {
|
||||
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+30
-7
@@ -37,7 +37,7 @@ import (
|
||||
// construction and are excluded from the diff; the other architectures list
|
||||
// their conditional branches outright.
|
||||
func cmdAuditInstructions(args []string) error {
|
||||
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
|
||||
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
|
||||
Compare the gasm encoder for the given architecture (default amd64) against
|
||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
||||
@@ -57,11 +57,13 @@ per-architecture pass rates and the most common failure reasons, which drive
|
||||
the encodability backlog by frequency rather than by table order.
|
||||
`)
|
||||
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
||||
var dirs includeDirs
|
||||
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
if *corpus {
|
||||
return cmdAuditCorpus(fs.Args())
|
||||
return cmdAuditCorpus(fs.Args(), dirs)
|
||||
}
|
||||
archName := "amd64"
|
||||
switch n := len(fs.Args()); {
|
||||
@@ -395,8 +397,11 @@ func (t *corpusTally) fail(path, reason string) {
|
||||
}
|
||||
}
|
||||
|
||||
// cmdAuditCorpus implements audit-instructions --corpus.
|
||||
func cmdAuditCorpus(args []string) error {
|
||||
// cmdAuditCorpus implements audit-instructions --corpus. The include
|
||||
// directories carry #include resolution over a corpus whose files refer to
|
||||
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
|
||||
// needs.
|
||||
func cmdAuditCorpus(args []string, dirs includeDirs) error {
|
||||
if len(args) > 1 {
|
||||
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
||||
}
|
||||
@@ -410,7 +415,25 @@ func cmdAuditCorpus(args []string) error {
|
||||
}
|
||||
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
||||
}
|
||||
stats, err := runCorpusAudit(root)
|
||||
// The toolchain's shipped headers (funcdata.h and friends) define the
|
||||
// macros GOROOT files include; a corpus audit measures those files, so
|
||||
// the header directory joins the search path automatically. go_asm.h
|
||||
// is compiler-generated per package and stays unresolvable on purpose.
|
||||
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
|
||||
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
|
||||
seen := false
|
||||
for _, d := range dirs {
|
||||
if d == pkgInclude {
|
||||
seen = true
|
||||
}
|
||||
}
|
||||
if !seen {
|
||||
dirs = append(dirs, pkgInclude)
|
||||
}
|
||||
}
|
||||
}
|
||||
stats, err := runCorpusAudit(root, dirs)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -454,7 +477,7 @@ func otherPortFile(path string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func runCorpusAudit(root string) (*corpusStats, error) {
|
||||
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
|
||||
files, err := asmFiles(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -479,7 +502,7 @@ func runCorpusAudit(root string) (*corpusStats, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
|
||||
|
||||
var wanted []int // indexes into targets
|
||||
if a := arch.FromFilename(path); a != arch.Unknown {
|
||||
|
||||
+26
-11
@@ -240,6 +240,16 @@ func readSource(path string) (string, error) {
|
||||
return string(b), err
|
||||
}
|
||||
|
||||
// includeDirs collects repeatable -I flags: the directories searched for
|
||||
// #include files during macro expansion and include splicing.
|
||||
type includeDirs []string
|
||||
|
||||
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
|
||||
func (d *includeDirs) Set(v string) error {
|
||||
*d = append(*d, v)
|
||||
return nil
|
||||
}
|
||||
|
||||
func cmdTokens(args []string) int {
|
||||
fs := newCommand("tokens", "gasm tokens <file>", `
|
||||
Print the lexical token stream of FILE: position, token kind and text, one
|
||||
@@ -476,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
||||
}
|
||||
|
||||
func cmdAsm(args []string) int {
|
||||
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", `
|
||||
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>", `
|
||||
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
||||
machine code and printed as a hex dump. Supported architectures: amd64
|
||||
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
||||
@@ -498,9 +508,11 @@ and the format version from go version).
|
||||
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
||||
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
||||
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
|
||||
var dirs includeDirs
|
||||
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>")
|
||||
return 2
|
||||
}
|
||||
// The format is validated before anything else, so a bogus value exits 2
|
||||
@@ -526,7 +538,7 @@ and the format version from go version).
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
@@ -634,7 +646,7 @@ and the format version from go version).
|
||||
|
||||
// cmdDiff compares the machine code of two assembly files.
|
||||
func cmdDiff(args []string) int {
|
||||
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", `
|
||||
set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
|
||||
Compare the machine code produced by assembling two files.
|
||||
Shows which functions differ and the byte-level differences.
|
||||
Useful for verifying that two implementations produce identical code,
|
||||
@@ -645,9 +657,11 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
||||
`)
|
||||
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
||||
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
|
||||
var dirs includeDirs
|
||||
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||
set.Parse(args)
|
||||
if set.NArg() != 2 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>")
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
|
||||
return 2
|
||||
}
|
||||
path1, path2 := set.Arg(0), set.Arg(1)
|
||||
@@ -675,12 +689,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
||||
}
|
||||
|
||||
// Assemble both files.
|
||||
img1, err := assemblePath(path1, forced)
|
||||
img1, err := assemblePath(path1, forced, dirs)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
||||
return 1
|
||||
}
|
||||
img2, err := assemblePath(path2, forced)
|
||||
img2, err := assemblePath(path2, forced, dirs)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
||||
return 1
|
||||
@@ -755,14 +769,15 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// assemblePath reads, parses and assembles a file (used by cmdDiff). A
|
||||
// non-Unknown forced architecture overrides the file-name suffix.
|
||||
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) {
|
||||
// assemblePath reads, preprocesses, parses and assembles a file (used by
|
||||
// cmdDiff). A non-Unknown forced architecture overrides the file-name
|
||||
// suffix.
|
||||
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
|
||||
@@ -403,7 +403,7 @@ func TestRunCorpusAudit(t *testing.T) {
|
||||
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
||||
|
||||
stats, err := runCorpusAudit(dir)
|
||||
stats, err := runCorpusAudit(dir, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("runCorpusAudit: %v", err)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// writeTree writes a directory of files and returns its root.
|
||||
func writeTree(t *testing.T, files map[string]string) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
for name, content := range files {
|
||||
path := filepath.Join(dir, name)
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
|
||||
// in-file parameterised macro and an include resolved through -I, and checks
|
||||
// the assembled bytes came from the expansion (the loop body counts six
|
||||
// increments, two per expanded iteration).
|
||||
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
dir := writeTree(t, map[string]string{
|
||||
"inc/consts.h": "#define NITER 3\n",
|
||||
"main_amd64.s": "#include \"textflag.h\"\n" +
|
||||
"#include \"consts.h\"\n" +
|
||||
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
|
||||
"\tXORQ AX, AX\n" +
|
||||
"\tMOVQ $NITER, CX\n" +
|
||||
"loop:\n" +
|
||||
"\tSTEP(AX)\n" +
|
||||
"\tDECQ CX\n" +
|
||||
"\tJNZ loop\n" +
|
||||
"\tMOVQ AX, ret+0(FP)\n" +
|
||||
"\tRET\n",
|
||||
})
|
||||
stdout, stderr, code := capture(func() int {
|
||||
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
|
||||
})
|
||||
if code != 0 {
|
||||
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||
}
|
||||
// The macro expanded to two ADDQ $1 encodings in the static body; the
|
||||
// iteration count lives in the runtime loop.
|
||||
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
|
||||
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
|
||||
// including file's directory wins over the -I directories.
|
||||
func TestAsmIncludeResolutionOrder(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
dir := writeTree(t, map[string]string{
|
||||
"src/main_amd64.s": "#include \"textflag.h\"\n" +
|
||||
"#include \"vals.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ $VAL, AX\n" +
|
||||
"\tRET\n",
|
||||
"src/vals.h": "#define VAL 1\n",
|
||||
"late/vals.h": "#define VAL 2\n",
|
||||
"early/vals.h": "#define VAL 3\n",
|
||||
})
|
||||
stdout, stderr, code := capture(func() int {
|
||||
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
|
||||
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
|
||||
})
|
||||
if code != 0 {
|
||||
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||
}
|
||||
// VAL came from src/vals.h, not from either -I directory: the image
|
||||
// loads the immediate 1.
|
||||
if !strings.Contains(stdout, "b8 01 00 00 00") {
|
||||
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
|
||||
// resolves nowhere on the assembly path.
|
||||
func TestAsmMissingIncludeIsAnError(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("runs the assembler end to end")
|
||||
}
|
||||
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
|
||||
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
|
||||
if code == 0 {
|
||||
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
|
||||
}
|
||||
if !strings.Contains(stderr, `#include "nothere.h"`) {
|
||||
t.Errorf("stderr does not name the failing include: %s", stderr)
|
||||
}
|
||||
}
|
||||
+12
-3
@@ -141,12 +141,13 @@ gasm lint kernel_amd64.s
|
||||
## asm
|
||||
|
||||
```text
|
||||
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>
|
||||
Usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>
|
||||
```
|
||||
|
||||
| Flag | Default | Effect |
|
||||
|---|---|---|
|
||||
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
|
||||
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
|
||||
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
|
||||
| `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix |
|
||||
| `-o` | empty | write the output to this file instead of a hex dump on stdout |
|
||||
@@ -162,6 +163,13 @@ system toolchain; `goobj` emits the Go toolchain's own object format, which
|
||||
installed: the object preamble is captured from `go tool asm` and the format
|
||||
version from `go version`. `raw` and `elf` need no toolchain at all.
|
||||
|
||||
Assembly preprocessing matches the toolchain's: `#define` macros (object and
|
||||
parameterised) expand at the point of use, `#undef`, `#ifdef`, `#ifndef`,
|
||||
`#else` and `#endif` behave as in `go tool asm`, `;` separates statements,
|
||||
and `#include "file"` splices the named file in, resolved against the source
|
||||
directory and then each `-I` directory in order. `textflag.h` is the one
|
||||
header that is not spliced: gasm consumes its flag names natively.
|
||||
|
||||
```sh
|
||||
gasm asm hello_amd64.s
|
||||
```
|
||||
@@ -305,12 +313,13 @@ gasm debug --func add --cover hello_amd64.s
|
||||
## diff
|
||||
|
||||
```text
|
||||
Usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>
|
||||
Usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>
|
||||
```
|
||||
|
||||
| Flag | Default | Effect |
|
||||
|---|---|---|
|
||||
| `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes |
|
||||
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
|
||||
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
|
||||
|
||||
Functions are paired by exact name unless `--map` says otherwise, so
|
||||
@@ -348,7 +357,7 @@ add: 16 bytes, args=24, frame=0 NOSPLIT
|
||||
## audit-instructions
|
||||
|
||||
```text
|
||||
Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]
|
||||
Usage: gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]
|
||||
```
|
||||
|
||||
Compare the gasm encoder for the given architecture (default amd64) against the
|
||||
|
||||
+5
-1
@@ -2,7 +2,7 @@
|
||||
.SH NAME
|
||||
gasm-asm \- assemble Plan 9 assembly without the Go toolchain
|
||||
.SH SYNOPSIS
|
||||
.B gasm asm [\-\-format raw|elf|goobj] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
|
||||
.B gasm asm [\-\-format raw|elf|goobj] [\-I dir] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
|
||||
.SH DESCRIPTION
|
||||
Assemble FILE without the Go toolchain: every TEXT function is encoded
|
||||
to machine code and printed as a hex dump. Supported architectures:
|
||||
@@ -47,6 +47,10 @@ functions link too.
|
||||
.B \-\-format \fIraw|elf|goobj\fR
|
||||
Output format; the default is raw.
|
||||
.TP
|
||||
.B \-I \fIdir\fR
|
||||
Directory to search for #include files; may be repeated, searched in
|
||||
order after the source directory.
|
||||
.TP
|
||||
.B \-p \fIpkg\fR
|
||||
Package path for --format goobj, qualifying the exported symbols.
|
||||
.TP
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
.SH NAME
|
||||
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
|
||||
.SH SYNOPSIS
|
||||
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [amd64|arm64|riscv64|loong64]
|
||||
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-I dir] [amd64|arm64|riscv64|loong64]
|
||||
.SH DESCRIPTION
|
||||
Compare the gasm encoder for the given architecture (default amd64)
|
||||
against
|
||||
@@ -38,6 +38,12 @@ second.
|
||||
.B \-\-corpus [\fIdir\fR]
|
||||
Assemble a corpus of .s files and report pass rates and failure
|
||||
reasons.
|
||||
.TP
|
||||
.B \-I \fIdir\fR
|
||||
Directory to search for #include files; may be repeated, searched in
|
||||
order after the source directory. A corpus run whose files include
|
||||
toolchain headers (such as GOROOT/pkg/include) needs it, the same -I a
|
||||
toolchain comparison takes.
|
||||
.SH EXIT STATUS
|
||||
The mnemonic-diff mode reports through its output and exits 0; a failed
|
||||
probe or an unknown architecture exits non-zero.
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
.SH NAME
|
||||
gasm-diff \- compare the machine code of two assembly files
|
||||
.SH SYNOPSIS
|
||||
.B gasm diff [\-GOARCH arch] <file1.s> <file2.s>
|
||||
.B gasm diff [\-GOARCH arch] [\-I dir] <file1.s> <file2.s>
|
||||
.SH DESCRIPTION
|
||||
Compare the machine code produced by assembling two files. Shows which
|
||||
functions differ and the byte-level differences. Useful for verifying
|
||||
@@ -20,6 +20,10 @@ pairs two variants regardless of suffix.
|
||||
Target architecture for both files: amd64, arm64, riscv64 or loong64;
|
||||
overrides the file-name suffixes.
|
||||
.TP
|
||||
.B \-I \fIdir\fR
|
||||
Directory to search for #include files; may be repeated, searched in
|
||||
order after the source directory.
|
||||
.TP
|
||||
.B \-\-map \fIspec\fR
|
||||
Comma-separated old=new pairs to match functions with different names.
|
||||
.SH EXIT STATUS
|
||||
|
||||
+44
-1
@@ -119,7 +119,9 @@ func (l *Lexer) Next() token.Token {
|
||||
// is a C-preprocessor line continuation (used by #define macros in the
|
||||
// runtime .s files): splice the lines together by consuming both, so
|
||||
// the whole macro becomes one logical line that the parser treats as an
|
||||
// opaque preprocessor directive.
|
||||
// opaque preprocessor directive. The backslash may also reach its
|
||||
// newline across whitespace and a trailing comment ("…; \ // note\n"),
|
||||
// which the toolchain's scanner skips the same way.
|
||||
for {
|
||||
c := l.cur()
|
||||
if c == ' ' || c == '\t' || c == '\r' {
|
||||
@@ -136,6 +138,16 @@ func (l *Lexer) Next() token.Token {
|
||||
}
|
||||
continue
|
||||
}
|
||||
if c == '\\' && l.continuationAhead() {
|
||||
l.advance() // backslash, then the runes the scan saw
|
||||
for !l.atEnd() && l.cur() != '\n' {
|
||||
l.advance()
|
||||
}
|
||||
if !l.atEnd() {
|
||||
l.advance() // the newline that closes the continuation
|
||||
}
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
@@ -185,6 +197,28 @@ func (l *Lexer) Next() token.Token {
|
||||
}
|
||||
}
|
||||
|
||||
// continuationAhead reports, without consuming anything, whether the
|
||||
// backslash at the current position closes onto a newline through nothing
|
||||
// but horizontal whitespace and one line comment. Positions after the
|
||||
// backslash are inspected directly on the rune slice so a non-match leaves
|
||||
// the scanner state untouched.
|
||||
func (l *Lexer) continuationAhead() bool {
|
||||
i := l.i + 1
|
||||
for i < len(l.src) {
|
||||
switch r := l.src[i]; {
|
||||
case r == ' ' || r == '\t' || r == '\r':
|
||||
i++
|
||||
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
|
||||
for i < len(l.src) && l.src[i] != '\n' {
|
||||
i++
|
||||
}
|
||||
default:
|
||||
return r == '\n'
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// lineComment consumes a // comment up to, but not including, the newline. A
|
||||
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
|
||||
// comment content, so it never enters the token text. Trimming only a \r
|
||||
@@ -403,6 +437,15 @@ func (l *Lexer) punct(start token.Position) token.Token {
|
||||
case '|':
|
||||
l.advance()
|
||||
return l.make(token.Pipe, start, "|")
|
||||
case ';':
|
||||
l.advance()
|
||||
return l.make(token.Semicolon, start, ";")
|
||||
case '&':
|
||||
l.advance()
|
||||
return l.make(token.Ampersand, start, "&")
|
||||
case '~':
|
||||
l.advance()
|
||||
return l.make(token.Tilde, start, "~")
|
||||
default:
|
||||
// Unknown rune: emit it as Illegal and move on.
|
||||
l.advance()
|
||||
|
||||
+146
@@ -0,0 +1,146 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Constant-expression folding for operands. The toolchain's assembler
|
||||
// evaluates arithmetic in every operand position, and macro-heavy GOROOT
|
||||
// sources lean on it: parameterised bodies carry offsets like
|
||||
// ((index*4)+0)(base), immediates like $(32-shift) and masks like
|
||||
// $~63 or $(1<<0|1<<9). Substituting the parameters textually therefore
|
||||
// leaves constant arithmetic behind, and the parser folds it here, keeping
|
||||
// the operand AST identical to what the same literals written out would
|
||||
// produce. Anything that is not a closed integer expression fails to fold
|
||||
// and falls through to the ordinary operand paths.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// foldExpr evaluates the constant integer expression at the head of ts and
|
||||
// returns its value together with the unconsumed tokens. ok is false when
|
||||
// the tokens do not form an expression, which is the callers' signal to use
|
||||
// the ordinary parsing paths.
|
||||
func foldExpr(ts []token.Token) (val int64, rest []token.Token, ok bool) {
|
||||
v, rest, ok := foldAdd(ts)
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
return v, rest, true
|
||||
}
|
||||
|
||||
// foldAdd parses addition-level expressions: +, - and | bind loosest, the
|
||||
// Plan 9 convention that makes x<<1|3 read as (x<<1)|3.
|
||||
func foldAdd(ts []token.Token) (int64, []token.Token, bool) {
|
||||
v, rest, ok := foldMul(ts)
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
for len(rest) > 0 {
|
||||
kind := rest[0].Kind
|
||||
if kind != token.Plus && kind != token.Minus && kind != token.Pipe {
|
||||
return v, rest, true
|
||||
}
|
||||
w, r2, ok := foldMul(rest[1:])
|
||||
if !ok {
|
||||
return v, rest, true
|
||||
}
|
||||
switch kind {
|
||||
case token.Plus:
|
||||
v += w
|
||||
case token.Minus:
|
||||
v -= w
|
||||
case token.Pipe:
|
||||
v |= w
|
||||
}
|
||||
rest = r2
|
||||
}
|
||||
return v, rest, true
|
||||
}
|
||||
|
||||
// foldMul parses multiplication-level expressions: *, / and the bit
|
||||
// operators &, << and >>.
|
||||
func foldMul(ts []token.Token) (int64, []token.Token, bool) {
|
||||
v, rest, ok := foldFactor(ts)
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
for len(rest) > 0 {
|
||||
switch rest[0].Kind {
|
||||
case token.Star:
|
||||
w, r2, ok := foldFactor(rest[1:])
|
||||
if !ok {
|
||||
return v, rest, true
|
||||
}
|
||||
v *= w
|
||||
rest = r2
|
||||
case token.Slash:
|
||||
w, r2, ok := foldFactor(rest[1:])
|
||||
if !ok || w == 0 {
|
||||
return v, rest, true
|
||||
}
|
||||
v /= w
|
||||
rest = r2
|
||||
case token.Ampersand:
|
||||
w, r2, ok := foldFactor(rest[1:])
|
||||
if !ok {
|
||||
return v, rest, true
|
||||
}
|
||||
v &= w
|
||||
rest = r2
|
||||
case token.LShift:
|
||||
w, r2, ok := foldFactor(rest[1:])
|
||||
if !ok || w < 0 || w >= 64 {
|
||||
return v, rest, true
|
||||
}
|
||||
v <<= uint(w)
|
||||
rest = r2
|
||||
case token.RShift:
|
||||
w, r2, ok := foldFactor(rest[1:])
|
||||
if !ok || w < 0 || w >= 64 {
|
||||
return v, rest, true
|
||||
}
|
||||
v >>= uint(w)
|
||||
rest = r2
|
||||
default:
|
||||
return v, rest, true
|
||||
}
|
||||
}
|
||||
return v, rest, true
|
||||
}
|
||||
|
||||
// foldFactor parses a number, a parenthesised expression, or a unary sign
|
||||
// or complement.
|
||||
func foldFactor(ts []token.Token) (int64, []token.Token, bool) {
|
||||
if len(ts) == 0 {
|
||||
return 0, ts, false
|
||||
}
|
||||
switch ts[0].Kind {
|
||||
case token.Number:
|
||||
v, ok := tryInt(ts[0].Text)
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
return v, ts[1:], true
|
||||
case token.LParen:
|
||||
v, rest, ok := foldAdd(ts[1:])
|
||||
if !ok || len(rest) == 0 || rest[0].Kind != token.RParen {
|
||||
return 0, ts, false
|
||||
}
|
||||
return v, rest[1:], true
|
||||
case token.Minus:
|
||||
v, rest, ok := foldFactor(ts[1:])
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
return -v, rest, true
|
||||
case token.Plus:
|
||||
return foldFactor(ts[1:])
|
||||
case token.Tilde:
|
||||
v, rest, ok := foldFactor(ts[1:])
|
||||
if !ok {
|
||||
return 0, ts, false
|
||||
}
|
||||
return ^v, rest, true
|
||||
}
|
||||
return 0, ts, false
|
||||
}
|
||||
@@ -408,6 +408,18 @@ func parseImmediate(g []token.Token) ast.Immediate {
|
||||
return imm
|
||||
}
|
||||
}
|
||||
// A constant expression introduced by '(' or '~'. Textual macro
|
||||
// substitution leaves arithmetic such as $(32-shift) and $~63 behind,
|
||||
// and the toolchain evaluates it in place; only shapes the ordinary
|
||||
// paths below cannot read reach the folder, so every existing form
|
||||
// keeps its exact parse.
|
||||
if g[0].Kind == token.LParen || g[0].Kind == token.Tilde {
|
||||
if v, rest, ok := foldExpr(g); ok && len(rest) == 0 {
|
||||
imm.Val = v
|
||||
imm.HasVal = true
|
||||
return imm
|
||||
}
|
||||
}
|
||||
i := 0
|
||||
if g[i].Kind == token.Minus {
|
||||
imm.Neg = true
|
||||
@@ -459,6 +471,17 @@ func parseAddress(g []token.Token) ast.Address {
|
||||
}
|
||||
|
||||
i := 0
|
||||
// A parenthesised constant expression as the displacement: substituted
|
||||
// macro bodies carry ((index*4)+0)(base) shapes. As with the signed
|
||||
// number path below, the value is committed only when a base group
|
||||
// follows.
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
if v, rest, ok := foldExpr(g[i:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen {
|
||||
addr.Offset = v
|
||||
addr.HasOff = true
|
||||
i = len(g) - len(rest)
|
||||
}
|
||||
}
|
||||
// Optional leading displacement before a '(' base group. A sign pushes
|
||||
// the parenthesis one token further out: -4(DX) has it at i+2.
|
||||
if isSignedNumber(g, i) {
|
||||
|
||||
@@ -0,0 +1,475 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// The preprocessor turns #define and #include directives into the token
|
||||
// stream the parser really sees, the way the Go toolchain's assembler does:
|
||||
// object and parameterised macros expand at the point of use, and an
|
||||
// #include splices the named file's lines in place of the directive. The
|
||||
// pass runs only on the assembly path (gasm asm, diff, the corpus audit),
|
||||
// where the result is machine code; parsing for the linter, formatter and
|
||||
// language server keeps the raw file so their view of #define lines, and
|
||||
// therefore their macro-aware behaviour, is unchanged.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unicode/utf8"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Options controls the optional preprocessing applied before a file is
|
||||
// parsed. The zero value reproduces Parse exactly.
|
||||
type Options struct {
|
||||
// IncludeDirs lists the -I directories searched for #include files,
|
||||
// in order, after the including file's own directory.
|
||||
IncludeDirs []string
|
||||
// Expand enables macro expansion, include splicing and the
|
||||
// statement-separator reading of ';' that the expanded bodies rely on.
|
||||
Expand bool
|
||||
}
|
||||
|
||||
// ParseWithOptions parses src like Parse, optionally preprocessing it first.
|
||||
// The returned file is usable even when errors is non-empty.
|
||||
func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) {
|
||||
tokens := lexer.Tokenize(src)
|
||||
var lines [][]token.Token
|
||||
var errs []error
|
||||
if opts.Expand {
|
||||
pp := &preproc{opts: opts, macros: map[string]*macroDef{}}
|
||||
lines = pp.fileLines(path, tokens, token.Position{})
|
||||
errs = pp.errs
|
||||
} else {
|
||||
lines = splitLines(tokens)
|
||||
}
|
||||
p := &state{path: path}
|
||||
p.parse(lines)
|
||||
return p.file, append(errs, p.errs...)
|
||||
}
|
||||
|
||||
// maxExpansionDepth bounds recursive macro expansion; the toolchain's
|
||||
// assembler gives up after 100 nested invocations without producing a token.
|
||||
const maxExpansionDepth = 100
|
||||
|
||||
// textflagHeader names the one header gasm does not splice: its flag macros
|
||||
// (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser,
|
||||
// encoders and linter, and expanding them to their numeric constants would
|
||||
// leave every consumer blind to them.
|
||||
const textflagHeader = "textflag.h"
|
||||
|
||||
// macroDef is one #define. A nil args slice is an object macro; a non-nil
|
||||
// (possibly empty) one is parameterised, the C distinction between
|
||||
// "#define A(x)" and "#define A (x)".
|
||||
type macroDef struct {
|
||||
name string
|
||||
args []string
|
||||
body []token.Token
|
||||
}
|
||||
|
||||
// preproc carries the state of one expansion pass: the live macro table, the
|
||||
// chain of files currently being read, for cycle detection, and the
|
||||
// conditional-inclusion stack of #ifdef regions.
|
||||
type preproc struct {
|
||||
opts Options
|
||||
macros map[string]*macroDef
|
||||
errs []error
|
||||
stack []string // absolute paths of files being read, innermost last
|
||||
ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth
|
||||
}
|
||||
|
||||
// enabled reports whether the position being read is inside a live
|
||||
// conditional branch. Directives inside a disabled branch contribute
|
||||
// nothing, and its content lines are dropped, exactly as the toolchain's
|
||||
// input stack does.
|
||||
func (pp *preproc) enabled() bool {
|
||||
return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1]
|
||||
}
|
||||
|
||||
func (pp *preproc) errorf(pos token.Position, format string, args ...any) {
|
||||
pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
|
||||
}
|
||||
|
||||
// fileLines tokenizes and preprocesses one file into logical lines.
|
||||
// Directive lines are kept (the parser records them for the tooling);
|
||||
// #include lines are replaced by the included file's lines. includePos is
|
||||
// the position of the #include that pulled this file in, zero for the
|
||||
// top-level file, and only serves cycle diagnostics.
|
||||
func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token {
|
||||
abs, err := filepath.Abs(path)
|
||||
if err != nil {
|
||||
abs = filepath.Clean(path)
|
||||
}
|
||||
if slices.Contains(pp.stack, abs) {
|
||||
if includePos.IsValid() {
|
||||
pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
pp.stack = append(pp.stack, abs)
|
||||
|
||||
var out [][]token.Token
|
||||
for _, line := range splitLines(tokens) {
|
||||
if len(line) == 0 {
|
||||
out = append(out, line)
|
||||
continue
|
||||
}
|
||||
if line[0].Kind == token.Hash {
|
||||
out = append(out, pp.directive(line, filepath.Dir(path))...)
|
||||
continue
|
||||
}
|
||||
if !pp.enabled() {
|
||||
continue
|
||||
}
|
||||
out = append(out, splitOnSemicolons(pp.expandTokens(line))...)
|
||||
}
|
||||
pp.stack = pp.stack[:len(pp.stack)-1]
|
||||
if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 {
|
||||
// The stack is per-input, shared across includes, so only the
|
||||
// top-level file's end can decide the input was left unclosed.
|
||||
pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef")
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// directive processes one '#' line and returns the lines to keep in the
|
||||
// stream: every directive line is kept as-is for the parser (which records
|
||||
// it), except #include, which is replaced by the spliced content.
|
||||
// Conditionals are tracked on every line; every other directive is inert
|
||||
// inside a disabled branch.
|
||||
func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token {
|
||||
if len(line) < 2 || line[1].Kind != token.Ident {
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
switch line[1].Text {
|
||||
case "ifdef", "ifndef":
|
||||
pp.ifdef(line, line[1].Text == "ifndef")
|
||||
case "else":
|
||||
pp.elseBranch(line)
|
||||
case "endif":
|
||||
pp.endif(line)
|
||||
case "define":
|
||||
if pp.enabled() {
|
||||
pp.define(line)
|
||||
}
|
||||
case "undef":
|
||||
if pp.enabled() {
|
||||
pp.undef(line)
|
||||
}
|
||||
case "include":
|
||||
if pp.enabled() {
|
||||
return pp.include(line, dir)
|
||||
}
|
||||
default:
|
||||
// #line and unknown directives are recorded but not interpreted:
|
||||
// conservative support keeps the parser's view intact and files
|
||||
// using them fail on their content, not silently.
|
||||
}
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
|
||||
// ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth
|
||||
// onto the conditional stack. A branch opened inside a disabled region is
|
||||
// itself disabled, however the name resolves.
|
||||
func (pp *preproc) ifdef(line []token.Token, inverted bool) {
|
||||
truth := false
|
||||
if len(line) >= 3 && line[2].Kind == token.Ident {
|
||||
_, defined := pp.macros[line[2].Text]
|
||||
truth = defined != inverted
|
||||
} else {
|
||||
pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text)
|
||||
}
|
||||
if !pp.enabled() {
|
||||
truth = false
|
||||
}
|
||||
pp.ifdefStack = append(pp.ifdefStack, truth)
|
||||
}
|
||||
|
||||
// elseBranch flips the innermost conditional's truth, but only when the
|
||||
// region enclosing it is itself live: the toolchain keeps outer overrides.
|
||||
func (pp *preproc) elseBranch(line []token.Token) {
|
||||
if len(pp.ifdefStack) == 0 {
|
||||
pp.errorf(line[0].Pos, "unmatched #else")
|
||||
return
|
||||
}
|
||||
if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] {
|
||||
pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1]
|
||||
}
|
||||
}
|
||||
|
||||
// endif closes the innermost conditional.
|
||||
func (pp *preproc) endif(line []token.Token) {
|
||||
if len(pp.ifdefStack) == 0 {
|
||||
pp.errorf(line[0].Pos, "unmatched #endif")
|
||||
return
|
||||
}
|
||||
pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1]
|
||||
}
|
||||
|
||||
// define parses "#define NAME[(formals)] body" into the macro table. The
|
||||
// body runs to the end of the logical line (the lexer has already spliced
|
||||
// backslash continuations) and stops at a comment, which never expands.
|
||||
func (pp *preproc) define(line []token.Token) {
|
||||
if len(line) < 3 || line[2].Kind != token.Ident {
|
||||
return
|
||||
}
|
||||
name := line[2]
|
||||
args := []string(nil)
|
||||
body := line[3:]
|
||||
// The definition is parameterised only when '(' follows the name
|
||||
// directly; the toolchain separates "#define A(x)" from
|
||||
// "#define A (x)" by adjacency, and so does the column check here.
|
||||
if len(body) > 0 && body[0].Kind == token.LParen &&
|
||||
body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) {
|
||||
args = []string{}
|
||||
i := 1
|
||||
for i < len(body) && body[i].Kind != token.RParen {
|
||||
if body[i].Kind == token.Ident {
|
||||
args = append(args, body[i].Text)
|
||||
}
|
||||
i++
|
||||
}
|
||||
if i < len(body) {
|
||||
body = body[i+1:]
|
||||
} else {
|
||||
body = nil
|
||||
}
|
||||
}
|
||||
if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 {
|
||||
body = body[:i]
|
||||
}
|
||||
if _, exists := pp.macros[name.Text]; exists {
|
||||
// The toolchain refuses redefinition, so a file the oracle accepts
|
||||
// never redefines; failing here keeps that contract visible.
|
||||
pp.errorf(name.Pos, "redefinition of macro %s", name.Text)
|
||||
}
|
||||
pp.macros[name.Text] = ¯oDef{name: name.Text, args: args, body: pp.bodyWithBreaks(body)}
|
||||
|
||||
}
|
||||
|
||||
// bodyWithBreaks records the statement boundaries the continuations carry.
|
||||
// The lexer splices backslash-continued lines into one logical line, but the
|
||||
// toolchain keeps the newline as a token in the stored body, which is how a
|
||||
// multi-instruction body without semicolons (the arm64 style) still splits
|
||||
// into statements on expansion. A line change inside the logical line is
|
||||
// exactly a continuation, so the boundary is restored from the positions.
|
||||
func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token {
|
||||
out := make([]token.Token, 0, len(body))
|
||||
for i, t := range body {
|
||||
if i > 0 && t.Pos.Line != body[i-1].Pos.Line {
|
||||
out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos})
|
||||
}
|
||||
out = append(out, t)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// undef handles "#undef NAME", which the toolchain honours and requires to
|
||||
// name a defined macro.
|
||||
func (pp *preproc) undef(line []token.Token) {
|
||||
if len(line) < 3 || line[2].Kind != token.Ident {
|
||||
return
|
||||
}
|
||||
if _, ok := pp.macros[line[2].Text]; !ok {
|
||||
pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text)
|
||||
return
|
||||
}
|
||||
delete(pp.macros, line[2].Text)
|
||||
}
|
||||
|
||||
// include resolves and splices "#include \"file\"". A header that cannot be
|
||||
// read keeps the directive line in the stream, with a diagnostic.
|
||||
func (pp *preproc) include(line []token.Token, dir string) [][]token.Token {
|
||||
if len(line) < 3 || line[2].Kind != token.String {
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
header := line[2]
|
||||
name, err := strconv.Unquote(header.Text)
|
||||
if err != nil {
|
||||
pp.errorf(header.Pos, "unquoting include file name: %v", err)
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
if filepath.Base(name) == textflagHeader {
|
||||
// Flag macros are handled natively (see textflagHeader); the
|
||||
// directive stays so tools still see the include.
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
resolved, ok := pp.resolve(name, dir)
|
||||
if !ok {
|
||||
searched := append([]string{dir}, pp.opts.IncludeDirs...)
|
||||
pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", "))
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
src, err := os.ReadFile(resolved)
|
||||
if err != nil {
|
||||
pp.errorf(header.Pos, "#include %q: %v", name, err)
|
||||
return [][]token.Token{line}
|
||||
}
|
||||
return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos)
|
||||
}
|
||||
|
||||
// resolve looks an include name up the way the toolchain does: as written
|
||||
// (relative to the working directory), then relative to the including
|
||||
// file's directory, then in each -I directory in order.
|
||||
func (pp *preproc) resolve(name, dir string) (string, bool) {
|
||||
candidates := []string{name}
|
||||
if !filepath.IsAbs(name) {
|
||||
candidates = append(candidates, filepath.Join(dir, name))
|
||||
for _, d := range pp.opts.IncludeDirs {
|
||||
candidates = append(candidates, filepath.Join(d, name))
|
||||
}
|
||||
}
|
||||
for _, c := range candidates {
|
||||
if st, err := os.Stat(c); err == nil && !st.IsDir() {
|
||||
return c, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// expandTokens expands every macro invocation in a token sequence,
|
||||
// recursively, with a depth guard. A body is spliced into the sequence in
|
||||
// place and rescanned, the way the toolchain's input stack re-reads pushed
|
||||
// tokens: an object macro may name a parameterised one, and the argument
|
||||
// list of the expansion may then come from the tokens that follow.
|
||||
func (pp *preproc) expandTokens(in []token.Token) []token.Token {
|
||||
s := in
|
||||
i := 0
|
||||
consecutive := 0
|
||||
for i < len(s) {
|
||||
t := s[i]
|
||||
if t.Kind != token.Ident {
|
||||
i++
|
||||
consecutive = 0
|
||||
continue
|
||||
}
|
||||
def := pp.macros[t.Text]
|
||||
if def == nil {
|
||||
i++
|
||||
consecutive = 0
|
||||
continue
|
||||
}
|
||||
// The guard mirrors the toolchain's: 100 nested invocations in a
|
||||
// row without a plain token between them means recursion.
|
||||
consecutive++
|
||||
if consecutive > maxExpansionDepth {
|
||||
pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth)
|
||||
return nil
|
||||
}
|
||||
if def.args == nil {
|
||||
s = append(s[:i], append(restamp(def.body, t.Pos), s[i+1:]...)...)
|
||||
continue
|
||||
}
|
||||
// A parameterised macro invoked without its parentheses stands
|
||||
// unexpanded, naming itself, as in the toolchain.
|
||||
if i+1 >= len(s) || s[i+1].Kind != token.LParen {
|
||||
i++
|
||||
consecutive = 0
|
||||
continue
|
||||
}
|
||||
args, next := pp.collectArgs(s, i+1, t)
|
||||
if args == nil {
|
||||
return nil
|
||||
}
|
||||
// A zero-argument macro may be invoked as NAME().
|
||||
if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 {
|
||||
args = nil
|
||||
}
|
||||
if len(args) != len(def.args) {
|
||||
pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args))
|
||||
i = next
|
||||
consecutive = 0
|
||||
continue
|
||||
}
|
||||
sub := make([]token.Token, 0, len(def.body))
|
||||
for _, bt := range def.body {
|
||||
if bt.Kind == token.Ident {
|
||||
if k := slices.Index(def.args, bt.Text); k >= 0 {
|
||||
sub = append(sub, restamp(args[k], t.Pos)...)
|
||||
continue
|
||||
}
|
||||
}
|
||||
sub = append(sub, bt)
|
||||
}
|
||||
s = append(s[:i], append(sub, s[next:]...)...)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// collectArgs reads the actual argument tokens of an invocation; the opening
|
||||
// parenthesis is at start. Commas separate arguments except inside nested
|
||||
// parentheses. A nil result means the list was unterminated, which is a
|
||||
// diagnostic.
|
||||
func (pp *preproc) collectArgs(in []token.Token, start int, name token.Token) ([][]token.Token, int) {
|
||||
var args [][]token.Token
|
||||
var cur []token.Token
|
||||
nesting := 0
|
||||
for i := start + 1; i < len(in); i++ {
|
||||
t := in[i]
|
||||
switch t.Kind {
|
||||
case token.LParen:
|
||||
nesting++
|
||||
cur = append(cur, t)
|
||||
case token.RParen:
|
||||
if nesting == 0 {
|
||||
return append(args, cur), i + 1
|
||||
}
|
||||
nesting--
|
||||
cur = append(cur, t)
|
||||
case token.Comma:
|
||||
if nesting == 0 {
|
||||
args = append(args, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
case token.Comment:
|
||||
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
|
||||
return nil, i
|
||||
default:
|
||||
cur = append(cur, t)
|
||||
}
|
||||
}
|
||||
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
|
||||
return nil, len(in)
|
||||
}
|
||||
|
||||
// restamp copies body tokens to the invocation's position, so diagnostics
|
||||
// and the line table point where the macro was used, as the toolchain's
|
||||
// input stack does.
|
||||
func restamp(body []token.Token, pos token.Position) []token.Token {
|
||||
out := make([]token.Token, len(body))
|
||||
for i, t := range body {
|
||||
t.Pos, t.End = pos, pos
|
||||
out[i] = t
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// splitOnSemicolons breaks a token sequence at ';' statement separators and
|
||||
// at the Newline markers that record continuation boundaries inside macro
|
||||
// bodies, producing the logical lines the parser expects. The separators
|
||||
// carry no meaning beyond the break, so the pieces are exactly what the same
|
||||
// statements on separate lines would produce.
|
||||
func splitOnSemicolons(ts []token.Token) [][]token.Token {
|
||||
var out [][]token.Token
|
||||
start := 0
|
||||
for i, t := range ts {
|
||||
if t.Kind == token.Semicolon || t.Kind == token.Newline {
|
||||
if i > start {
|
||||
out = append(out, ts[start:i])
|
||||
}
|
||||
start = i + 1
|
||||
}
|
||||
}
|
||||
if start < len(ts) {
|
||||
out = append(out, ts[start:])
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,552 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// expand parses src with preprocessing enabled and returns the first TEXT's
|
||||
// body instructions as "MNEMONIC operand|operand" strings, the shape the
|
||||
// expansion assertions below compare against. Runs of spaces are
|
||||
// collapsed: Raw renders a token group as its tokens joined with single
|
||||
// spaces, so "$(32-7)" arrives as "$ ( 32 - 7 )" and the comparison must
|
||||
// not depend on that spelling.
|
||||
func expand(t *testing.T, src string) (*ast.File, []string) {
|
||||
t.Helper()
|
||||
f, errs := ParseWithOptions("t_amd64.s", src, Options{Expand: true})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
ts := texts(f)
|
||||
if len(ts) == 0 {
|
||||
t.Fatalf("no TEXT in:\n%s", src)
|
||||
}
|
||||
var got []string
|
||||
for _, s := range ts[0].Body {
|
||||
in, ok := s.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
var ops []string
|
||||
for _, op := range in.Operands {
|
||||
ops = append(ops, op.Raw)
|
||||
}
|
||||
line := in.Mnemonic.Text + " " + strings.Join(ops, ", ")
|
||||
got = append(got, strings.ReplaceAll(line, " ", ""))
|
||||
}
|
||||
return f, got
|
||||
}
|
||||
|
||||
func wantLines(t *testing.T, got []string, want ...string) {
|
||||
t.Helper()
|
||||
strip := func(lines []string) string {
|
||||
var out []string
|
||||
for _, l := range lines {
|
||||
out = append(out, strings.ReplaceAll(l, " ", ""))
|
||||
}
|
||||
return strings.Join(out, "\n")
|
||||
}
|
||||
if strip(got) != strip(want) {
|
||||
t.Errorf("expanded body:\n %s\nwant:\n %s", strings.Join(got, "\n "), strings.Join(want, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
func TestObjectMacroExpandsAtUse(t *testing.T) {
|
||||
_, got := expand(t, `
|
||||
#define REGTMP CX
|
||||
#define TWICE ADDQ CX, AX; ADDQ CX, AX
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
MOVQ 8(SP), REGTMP
|
||||
TWICE
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"MOVQ 8(SP), CX",
|
||||
"ADDQ CX, AX",
|
||||
"ADDQ CX, AX",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestParameterisedMacroSubstitutesArguments(t *testing.T) {
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
#define ROUND1(a, index, const, shift) \
|
||||
ADDQ $const, a; \
|
||||
MOVW (index*4)(SP), a; \
|
||||
RORQ $(32-shift), a
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
ROUND1(AX, 3, 0xd76aa478, 7)
|
||||
RET
|
||||
`, Options{Expand: true})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
body := texts(f)[0].Body
|
||||
add := body[0].(*ast.Instr)
|
||||
if add.Mnemonic.Text != "ADDQ" || !add.Operands[0].Imm.HasVal ||
|
||||
add.Operands[0].Imm.Val != 0xd76aa478 || add.Operands[1].Addr.Sym == nil ||
|
||||
add.Operands[1].Addr.Sym.Name != "AX" {
|
||||
t.Errorf("ADDQ operands substituted wrong: %+v %+v", add.Operands[0].Imm, add.Operands[1].Addr)
|
||||
}
|
||||
mov := body[1].(*ast.Instr)
|
||||
if addr := mov.Operands[0].Addr; !addr.HasOff || addr.Offset != 12 {
|
||||
t.Errorf("MOVW offset = %+v, want 12 from 3*4", addr)
|
||||
}
|
||||
ror := body[2].(*ast.Instr)
|
||||
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
|
||||
t.Errorf("RORQ immediate = %+v, want 25 from (32-7)", ror.Operands[0].Imm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMacroArgumentsKeepCommasInParens(t *testing.T) {
|
||||
// An argument may itself be an unparenthesised expression: the tokens
|
||||
// substitute verbatim and the parser folds the result, as the
|
||||
// toolchain's parser does.
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
#define LOAD(dst, off) MOVQ off(SP), dst
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
LOAD(AX, 1*8)
|
||||
RET
|
||||
`, Options{Expand: true})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
in := texts(f)[0].Body[0].(*ast.Instr)
|
||||
addr := in.Operands[0].Addr
|
||||
if !addr.HasOff || addr.Offset != 8 {
|
||||
t.Errorf("offset = %+v, want 8", addr)
|
||||
}
|
||||
if sym := in.Operands[1].Addr.Sym; sym == nil || sym.Name != "AX" {
|
||||
t.Errorf("destination = %+v, want AX", in.Operands[1].Addr)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNestedMacroInvocations(t *testing.T) {
|
||||
// An object macro naming a parameterised one, and a parameterised body
|
||||
// invoking another parameterised macro: the toolchain's input stack
|
||||
// rescans substituted tokens, and so does expansion here.
|
||||
_, got := expand(t, `
|
||||
#define DOUBLE(x) ADDQ x, x
|
||||
#define TWICE2 DOUBLE
|
||||
#define FOUR(a, b) DOUBLE(a); DOUBLE(b)
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
TWICE2(AX)
|
||||
FOUR(AX, CX)
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"ADDQ AX, AX",
|
||||
"ADDQ AX, AX",
|
||||
"ADDQ CX, CX",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestMultiLineBodySplitsWithoutSemicolons(t *testing.T) {
|
||||
// The arm64 style: backslash-continued lines with no semicolons. The
|
||||
// continuation newline is a statement boundary, as in the toolchain.
|
||||
_, got := expand(t, `
|
||||
#define PAIR \
|
||||
ADDQ AX, AX \
|
||||
MOVQ AX, CX
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
PAIR
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"ADDQ AX, AX",
|
||||
"MOVQ AX, CX",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestZeroArgumentMacro(t *testing.T) {
|
||||
_, got := expand(t, `
|
||||
#define BARRIER()
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
BARRIER()
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got, "RET")
|
||||
}
|
||||
|
||||
func TestParameterisedWithoutParensStandsAsName(t *testing.T) {
|
||||
// A parameterised macro invoked without its parentheses names itself,
|
||||
// which the parser then reports as an unknown instruction rather than
|
||||
// silently expanding nothing.
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
#define M(x) ADDQ x, x
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
M
|
||||
RET
|
||||
`, Options{Expand: true})
|
||||
if len(errs) != 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fn := texts(f)[0]
|
||||
if len(fn.Body) == 0 {
|
||||
t.Fatal("body empty")
|
||||
}
|
||||
in, ok := fn.Body[0].(*ast.Instr)
|
||||
if !ok || in.Mnemonic.Text != "M" {
|
||||
t.Fatalf("bare parameterised macro did not stand as its name: %+v", fn.Body[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionScoping(t *testing.T) {
|
||||
// A definition applies from its point onward: the use before the
|
||||
// #define stays untouched.
|
||||
_, got := expand(t, `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
SPECIAL
|
||||
#define SPECIAL ADDQ AX, AX
|
||||
SPECIAL
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"SPECIAL",
|
||||
"ADDQ AX, AX",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestUndefRemovesMacro(t *testing.T) {
|
||||
_, got := expand(t, `
|
||||
#define TEMP AX
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
TEMP
|
||||
#undef TEMP
|
||||
TEMP
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"AX",
|
||||
"TEMP",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestUndefUndefinedMacroIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#undef NOSUCH\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "undefined macro NOSUCH") {
|
||||
t.Fatalf("#undef of an undefined macro: got %v, want an error naming it", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRedefinitionIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#define A X\n#define A Y\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "redefinition of macro A") {
|
||||
t.Fatalf("redefinition: got %v, want an error", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecursiveMacroIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#define A B\n#define B A\nTEXT ·f(SB), NOSPLIT, $0\n\tA\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "recursive macro invocation") {
|
||||
t.Fatalf("recursion: got %v, want a recursive-macro error, not a hang", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrongArgumentCountIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#define M(a, b) ADDQ a, b\nTEXT ·f(SB), NOSPLIT, $0\n\tM(AX)\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "wrong arg count for macro M") {
|
||||
t.Fatalf("arg count: got %v, want an error", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConditionalsSelectOneBranch(t *testing.T) {
|
||||
_, got := expand(t, `
|
||||
#define MODE2
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
#ifdef MODE2
|
||||
ADDQ AX, AX
|
||||
#else
|
||||
SUBQ AX, AX
|
||||
#endif
|
||||
#ifndef MODE2
|
||||
SUBQ CX, CX
|
||||
#else
|
||||
ADDQ CX, CX
|
||||
#endif
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got,
|
||||
"ADDQ AX, AX",
|
||||
"ADDQ CX, CX",
|
||||
"RET",
|
||||
)
|
||||
}
|
||||
|
||||
func TestConditionalsHideDefinitionsAndIncludes(t *testing.T) {
|
||||
// A definition inside a disabled branch must not exist, and an
|
||||
// unresolvable include there must not be followed.
|
||||
_, got := expand(t, `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
#ifdef NOTDEFINED
|
||||
#define HIDEN ADDQ AX, AX
|
||||
#include "nowhere.h"
|
||||
#endif
|
||||
HIDEN
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got, "HIDEN", "RET")
|
||||
}
|
||||
|
||||
func TestUnclosedConditionalIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#ifdef X\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unclosed #ifdef") {
|
||||
t.Fatalf("unclosed conditional: got %v, want an error", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmatchedConditionalDelimitersAreErrors(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#endif\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #endif") {
|
||||
t.Fatalf("unmatched #endif: got %v, want an error", errs)
|
||||
}
|
||||
_, errs = ParseWithOptions("t_amd64.s", "#else\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #else") {
|
||||
t.Fatalf("unmatched #else: got %v, want an error", errs)
|
||||
}
|
||||
}
|
||||
|
||||
// includeTree writes a directory of include files and returns its path.
|
||||
func includeTree(t *testing.T, files map[string]string) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
for name, content := range files {
|
||||
path := filepath.Join(dir, name)
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
func TestIncludeSplicesAndDefinesAreShared(t *testing.T) {
|
||||
dir := includeTree(t, map[string]string{
|
||||
"consts.h": "#define KONST $42\n",
|
||||
})
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
#include "consts.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
MOVQ KONST, AX
|
||||
RET
|
||||
`, Options{Expand: true, IncludeDirs: []string{dir}})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
in := texts(f)[0].Body[0].(*ast.Instr)
|
||||
if in.Mnemonic.Text != "MOVQ" || strings.ReplaceAll(in.Operands[0].Raw, " ", "") != "$42" {
|
||||
t.Fatalf("include splicing failed: %+v", in)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIncludeResolutionOrder(t *testing.T) {
|
||||
// The including file's directory wins over the -I list, and the -I list
|
||||
// is searched in order.
|
||||
src := includeTree(t, map[string]string{
|
||||
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"inc/which.h": "#define WHO ONE\n",
|
||||
"first/which.h": "#define WHO TWO\n",
|
||||
"second/which.h": "#define WHO THREE\n",
|
||||
})
|
||||
main := filepath.Join(src, "inc", "main.s")
|
||||
body, err := os.ReadFile(main)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The header exists in the including file's directory and in two -I
|
||||
// directories; the source-directory copy must win.
|
||||
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
|
||||
filepath.Join(src, "first"), filepath.Join(src, "second"),
|
||||
}})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
found := false
|
||||
for _, d := range f.Decls {
|
||||
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO ONE") {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("the including file's directory did not win include resolution")
|
||||
}
|
||||
}
|
||||
|
||||
func TestIncludeSearchesIncludeDirsInOrder(t *testing.T) {
|
||||
src := includeTree(t, map[string]string{
|
||||
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"first/which.h": "#define WHO TWO\n",
|
||||
"second/which.h": "#define WHO THREE\n",
|
||||
})
|
||||
main := filepath.Join(src, "inc", "main.s")
|
||||
body, err := os.ReadFile(main)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
|
||||
filepath.Join(src, "first"), filepath.Join(src, "second"),
|
||||
}})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO THREE") {
|
||||
t.Error("the second -I directory was searched before the first")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestIncludeCycleIsDetected(t *testing.T) {
|
||||
src := includeTree(t, map[string]string{
|
||||
"a.s": "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"b.s": "#include \"a.s\"\n",
|
||||
})
|
||||
_, errs := ParseWithOptions(filepath.Join(src, "a.s"), "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
|
||||
Options{Expand: true})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "include cycle") {
|
||||
t.Fatalf("include cycle: got %v, want a cycle diagnostic, not a hang", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnresolvableIncludeIsAnError(t *testing.T) {
|
||||
_, errs := ParseWithOptions("t_amd64.s", "#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
|
||||
Options{Expand: true, IncludeDirs: []string{t.TempDir()}})
|
||||
if len(errs) == 0 || !strings.Contains(errs[0].Error(), `#include "nothere.h"`) {
|
||||
t.Fatalf("missing include: got %v, want a clear diagnostic", errs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextflagHeaderIsNeverSpliced(t *testing.T) {
|
||||
// textflag.h resolves nowhere here, yet the file must parse: the flag
|
||||
// names are consumed natively and the include stays in the tree.
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
`, Options{Expand: true})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
hasInclude := false
|
||||
for _, d := range f.Decls {
|
||||
if _, ok := d.(*ast.Include); ok {
|
||||
hasInclude = true
|
||||
}
|
||||
}
|
||||
if !hasInclude {
|
||||
t.Error("textflag.h include was dropped from the tree")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSemicolonSplitsRawLinesToo(t *testing.T) {
|
||||
_, got := expand(t, `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
BYTE $0x0f; BYTE $0x1f
|
||||
RET
|
||||
`)
|
||||
wantLines(t, got, "BYTE $0x0f", "BYTE $0x1f", "RET")
|
||||
}
|
||||
|
||||
func TestParseUnchangedWithoutExpand(t *testing.T) {
|
||||
// Without Expand the preprocessor must not exist: a macro invocation
|
||||
// stays an unexpanded instruction line and ';' keeps the old parse.
|
||||
f, errs := Parse("t_amd64.s", `
|
||||
#define TWICE ADDQ AX, AX
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
TWICE
|
||||
BYTE $0x0f; BYTE $0x1f
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fn := texts(f)[0]
|
||||
var mnemonics []string
|
||||
for _, s := range fn.Body {
|
||||
if in, ok := s.(*ast.Instr); ok {
|
||||
mnemonics = append(mnemonics, in.Mnemonic.Text)
|
||||
}
|
||||
}
|
||||
if strings.Join(mnemonics, " ") != "TWICE BYTE RET" {
|
||||
t.Errorf("non-expanding parse changed: %v", mnemonics)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConstantExpressionFolding(t *testing.T) {
|
||||
// The shapes substituted macro bodies leave behind: parenthesised
|
||||
// arithmetic in immediates and displacements, tilde complements. The
|
||||
// assertions read the semantic fields; Raw keeps the operand's tokens
|
||||
// in the canonicalised rendering, not the folded values.
|
||||
f, errs := ParseWithOptions("t_amd64.s", `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RORQ $(32-7), AX
|
||||
ANDQ $~63, AX
|
||||
MOVQ ((2*4)+0)(SP), AX
|
||||
MOVQ $((1<<3)|(1<<1)), AX
|
||||
RET
|
||||
`, Options{Expand: true})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
body := texts(f)[0].Body
|
||||
ror := body[0].(*ast.Instr)
|
||||
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
|
||||
t.Errorf("RORQ immediate = %+v, want 25", ror.Operands[0].Imm)
|
||||
}
|
||||
and := body[1].(*ast.Instr)
|
||||
if !and.Operands[0].Imm.HasVal || and.Operands[0].Imm.Val != -64 {
|
||||
t.Errorf("ANDQ immediate = %+v, want -64", and.Operands[0].Imm)
|
||||
}
|
||||
mov := body[2].(*ast.Instr)
|
||||
addr := mov.Operands[0].Addr
|
||||
if !addr.HasOff || addr.Offset != 8 || addr.Base != "SP" {
|
||||
t.Errorf("MOVQ address = %+v, want 8(SP)", addr)
|
||||
}
|
||||
mov2 := body[3].(*ast.Instr)
|
||||
if !mov2.Operands[0].Imm.HasVal || mov2.Operands[0].Imm.Val != 10 {
|
||||
t.Errorf("MOVQ immediate = %+v, want 10", mov2.Operands[0].Imm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConstantExpressionFoldsWithoutExpand(t *testing.T) {
|
||||
// Folding is a parser capability, not a preprocessing one: a
|
||||
// hand-written $(32-7) folds the same way with expansion off.
|
||||
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tRORQ $(32-7), AX\n\tRET\n", Options{})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
in := texts(f)[0].Body[0].(*ast.Instr)
|
||||
if !in.Operands[0].Imm.HasVal || in.Operands[0].Imm.Val != 25 {
|
||||
t.Errorf("Imm = %+v, want 25", in.Operands[0].Imm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNotAnExpressionFallsBack(t *testing.T) {
|
||||
// Symbol immediates and floats must keep their ordinary parse.
|
||||
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ $1.5, AX\n\tMOVQ $·sym(SB), AX\n\tRET\n", Options{})
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
fn := texts(f)[0]
|
||||
mov1 := fn.Body[0].(*ast.Instr)
|
||||
if mov1.Operands[0].Imm.HasVal || mov1.Operands[0].Imm.Float != "1.5" {
|
||||
t.Errorf("float immediate parsed as %+v", mov1.Operands[0].Imm)
|
||||
}
|
||||
mov2 := fn.Body[1].(*ast.Instr)
|
||||
if mov2.Operands[0].Imm.Sym == nil {
|
||||
t.Errorf("symbol immediate parsed as %+v", mov2.Operands[0].Imm)
|
||||
}
|
||||
}
|
||||
Vendored
+2110
File diff suppressed because it is too large
Load Diff
Vendored
+48
@@ -0,0 +1,48 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Carry arithmetic, logical shifts, register aliases with element selectors
|
||||
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
|
||||
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
#define acc0 V8
|
||||
#define acc1 V9
|
||||
#define const0 R15
|
||||
#define POLY V15
|
||||
|
||||
// carry pins the ADC/SBC family: the $0 spellings in two and three
|
||||
// operands, and the register-carry forms.
|
||||
TEXT ·carry(SB), NOSPLIT, $0-0
|
||||
ADC $0, R20
|
||||
ADC $0, R20, R4
|
||||
SBCS $0, R4
|
||||
SBCS $0, R4, R12
|
||||
SBCS R15, R4, R12
|
||||
SBC $0, R1
|
||||
ADCSW $0, R2, R3
|
||||
RET
|
||||
|
||||
// shift pins the shifted-register forms including ROR, which only the
|
||||
// logical family accepts.
|
||||
TEXT ·shift(SB), NOSPLIT, $0-0
|
||||
ANDW R9@>7, R19, R26
|
||||
AND R1@>33, R2, R3
|
||||
ADD R1<<11, R2, R3
|
||||
SUB R1->33, R2
|
||||
ORR R5<<2, R6, R7
|
||||
RET
|
||||
|
||||
// vecalias pins the vector aliases with element selectors and the
|
||||
// structure loads with aliased members.
|
||||
TEXT ·vecalias(SB), NOSPLIT, $0-0
|
||||
MOVD $0xC2, R1
|
||||
VMOV R1, POLY.D[0]
|
||||
VMOV R0, POLY.D[1]
|
||||
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||
VLD1 (R0), [acc0.B16]
|
||||
VLD1.P (R0), [acc0.B16, acc1.B16]
|
||||
VST1 [acc0.B16, acc1.B16], (R1)
|
||||
VST1.P [acc0.B16, acc1.B16], 32(R1)
|
||||
RET
|
||||
Vendored
+51
@@ -0,0 +1,51 @@
|
||||
// The subtract-immediate fold, the TEQ/TNE trap pseudos, PRELDX, the FP
|
||||
// condition branches and the N(PC) branch spellings, against the toolchain.
|
||||
#include "textflag.h"
|
||||
|
||||
// func SubFold(x int64) int64
|
||||
TEXT ·SubFold(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R8
|
||||
SUBV $0, R8
|
||||
SUBV $4, R9, R10
|
||||
SUBV $4096, R11
|
||||
SUBV $-4, R12
|
||||
SUB $1, R13
|
||||
SUBVU $4, R14
|
||||
SUBV $1048576, R15
|
||||
MOVV R8, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func Traps(x int64) int64
|
||||
TEXT ·Traps(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R4
|
||||
TEQ $4, R4, R5
|
||||
TEQ $4, R4
|
||||
TNE $6, R5, R6
|
||||
MOVV R4, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func Prefetch(x int64) int64
|
||||
TEXT ·Prefetch(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R7
|
||||
PRELDX 0(R7), $0x80001021, $0
|
||||
PRELDX -1(R7), $0x1021, $2
|
||||
MOVV R7, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func BranchForms(x int64) int64
|
||||
TEXT ·BranchForms(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R4
|
||||
|
||||
l1:
|
||||
BFPT l1
|
||||
BFPT FCC3, l1
|
||||
BFPF l1
|
||||
JMP -4(PC)
|
||||
JAL 1(PC)
|
||||
JAL (R4)
|
||||
|
||||
loop:
|
||||
ADDV $1, R4
|
||||
BEQ R4, R5, loop
|
||||
BNE R4, l1
|
||||
RET
|
||||
Vendored
+33
@@ -0,0 +1,33 @@
|
||||
// PCALIGN padding on loong64: andi $0, $0, 0 (the architecture's NOP), plus
|
||||
// the automatic loop-head alignment to a 16-byte boundary.
|
||||
#include "textflag.h"
|
||||
|
||||
// func Pad16(x int64) int64
|
||||
TEXT ·Pad16(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R4
|
||||
PCALIGN $16
|
||||
ADDV $1, R4
|
||||
MOVV R4, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func Pad32(x int64) int64
|
||||
TEXT ·Pad32(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R4
|
||||
PCALIGN $32
|
||||
ADDV $1, R4
|
||||
MOVV R4, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func LoopAlign(x int64) int64
|
||||
TEXT ·LoopAlign(SB), NOSPLIT, $0-16
|
||||
MOVV x+0(FP), R4
|
||||
MOVV $10, R5
|
||||
|
||||
loop:
|
||||
BEQ R4, R5, done
|
||||
ADDV $1, R4
|
||||
JMP loop
|
||||
|
||||
done:
|
||||
MOVV R4, ret+8(FP)
|
||||
RET
|
||||
Vendored
+35
@@ -0,0 +1,35 @@
|
||||
// PCALIGN padding on riscv64: 4-byte NOPs with a 2-byte compressed NOP when
|
||||
// the pad is 2 mod 4, exactly as the toolchain lays the bytes down.
|
||||
#include "textflag.h"
|
||||
|
||||
// func Pad8(x int64) int64
|
||||
TEXT ·Pad8(SB), NOSPLIT, $0-16
|
||||
MOV x+0(FP), X5
|
||||
PCALIGN $8
|
||||
ADD $1, X5
|
||||
MOV X5, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func Pad16(x int64) int64
|
||||
TEXT ·Pad16(SB), NOSPLIT, $0-16
|
||||
MOV x+0(FP), X5
|
||||
PCALIGN $16
|
||||
ADD $1, X5
|
||||
MOV X5, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func Pad32(x int64) int64
|
||||
TEXT ·Pad32(SB), NOSPLIT, $0-16
|
||||
MOV x+0(FP), X5
|
||||
PCALIGN $32
|
||||
ADD $1, X5
|
||||
MOV X5, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func PadAfterOdd(x int64) int64
|
||||
TEXT ·PadAfterOdd(SB), NOSPLIT, $0-16
|
||||
MOV x+0(FP), X5
|
||||
PCALIGN $8
|
||||
ADD $1, X5
|
||||
MOV X5, ret+8(FP)
|
||||
RET
|
||||
Vendored
+103
@@ -0,0 +1,103 @@
|
||||
// Instruction prefixes: LOCK, REP and REPN. go tool asm encodes each
|
||||
// statement as a standalone one-byte instruction with a PC of its own (F0,
|
||||
// F3 and F2 respectively); the statement that follows is encoded unaware of
|
||||
// it, and nothing validates the pairing. The shapes are the runtime's
|
||||
// atomic read-modify-write family and the string moves, every result folded
|
||||
// back.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func cas64(ptr *uint64, old, new uint64) bool
|
||||
TEXT ·cas64(SB), NOSPLIT, $0-25
|
||||
MOVQ ptr+0(FP), BX
|
||||
MOVQ old+8(FP), AX
|
||||
MOVQ new+16(FP), CX
|
||||
LOCK
|
||||
CMPXCHGQ CX, 0(BX)
|
||||
SETEQ ret+24(FP)
|
||||
RET
|
||||
|
||||
// func casloop(addr *uint64, v uint64) uint64
|
||||
// The runtime's Or64 shape: a LOCK inside a branch loop, the backward jump
|
||||
// measuring over the prefix statement's own byte.
|
||||
TEXT ·casloop(SB), NOSPLIT, $0-24
|
||||
MOVQ addr+0(FP), BX
|
||||
MOVQ v+8(FP), CX
|
||||
|
||||
loop:
|
||||
MOVQ CX, DX
|
||||
MOVQ (BX), AX
|
||||
ORQ AX, DX
|
||||
LOCK
|
||||
CMPXCHGQ DX, (BX)
|
||||
JNZ loop
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func xadd64(p *uint64, v uint64) uint64
|
||||
TEXT ·xadd64(SB), NOSPLIT, $0-24
|
||||
MOVQ p+0(FP), AX
|
||||
MOVQ v+8(FP), BX
|
||||
LOCK
|
||||
XADDQ BX, (AX)
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func xaddw(p *uint16, v uint16) uint16
|
||||
TEXT ·xaddw(SB), NOSPLIT, $0-12
|
||||
MOVQ p+0(FP), AX
|
||||
MOVW v+8(FP), BX
|
||||
LOCK
|
||||
XADDW BX, (AX)
|
||||
MOVW AX, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func lockarith(p *uint64)
|
||||
TEXT ·lockarith(SB), NOSPLIT, $0-8
|
||||
MOVQ p+0(FP), AX
|
||||
LOCK
|
||||
ORQ CX, (AX)
|
||||
LOCK
|
||||
ANDL CX, (AX)
|
||||
LOCK
|
||||
INCQ (AX)
|
||||
LOCK
|
||||
DECQ (AX)
|
||||
LOCK
|
||||
ORB BX, (AX)
|
||||
RET
|
||||
|
||||
// func repstring(dst, src *byte, n int)
|
||||
// The memmove shapes: forward copy by quadwords, backward tails.
|
||||
TEXT ·repstring(SB), NOSPLIT, $0-24
|
||||
MOVQ dst+0(FP), DI
|
||||
MOVQ src+8(FP), SI
|
||||
REP
|
||||
MOVSQ
|
||||
REP
|
||||
MOVSB
|
||||
REPN
|
||||
MOVSB
|
||||
REP
|
||||
STOSQ
|
||||
REP
|
||||
STOSB
|
||||
RET
|
||||
|
||||
// func pfxlabel()
|
||||
// Labels pinned on prefix statements' own bytes: pfx: sits on the LOCK,
|
||||
// mid: on the REPN.
|
||||
TEXT ·pfxlabel(SB), NOSPLIT, $0-0
|
||||
pfx:
|
||||
LOCK
|
||||
XCHGL BX, (AX)
|
||||
JMP done
|
||||
|
||||
mid:
|
||||
REPN
|
||||
MOVSB
|
||||
|
||||
done:
|
||||
REP
|
||||
STOSB
|
||||
RET
|
||||
Vendored
+62
@@ -0,0 +1,62 @@
|
||||
// Literal data emission: BYTE, WORD, LONG and QUAD write the immediate
|
||||
// into the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
|
||||
// lookup, truncated to the width rather than range-checked; END is
|
||||
// accepted and ignored, contributing no bytes and ending nothing. The
|
||||
// shapes mirror the runtime's hand-laid markers
|
||||
// (crypto/internal/boring/sig/sig_amd64.s) and its syscall stubs
|
||||
// (runtime/sys_linux_amd64.s).
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func marker()
|
||||
// A boring/crypto-style marker: a hand-laid forward branch whose skip
|
||||
// distance is patched at runtime. One BYTE per statement, as the
|
||||
// runtime's own file spells it: the semicolon-separated one-liner the
|
||||
// sys_linux_amd64.s stub uses does not survive gasm fmt, which drops the
|
||||
// statement separators.
|
||||
TEXT ·marker(SB), NOSPLIT, $0-0
|
||||
BYTE $0xEB
|
||||
BYTE $0x1D
|
||||
BYTE $0xF4
|
||||
BYTE $0x48
|
||||
BYTE $0xF4
|
||||
BYTE $0x4B
|
||||
BYTE $0xC3
|
||||
RET
|
||||
|
||||
// func stub()
|
||||
// The sys_linux_amd64.s stub bytes: the sign-extended
|
||||
// "48 c7 c0 0f 00 00 00" form of MOVQ $rt_sigreturn, AX.
|
||||
TEXT ·stub(SB), NOSPLIT, $0-0
|
||||
BYTE $0x48
|
||||
BYTE $0xc7
|
||||
BYTE $0xc0
|
||||
BYTE $0x0f
|
||||
BYTE $0x00
|
||||
BYTE $0x00
|
||||
BYTE $0x00
|
||||
RET
|
||||
|
||||
// func words()
|
||||
// The wider literals, and an END that ends nothing: the WORD after it
|
||||
// still lands in this function.
|
||||
TEXT ·words(SB), NOSPLIT, $0-0
|
||||
WORD $0x1234
|
||||
WORD $-1
|
||||
LONG $0x11223344
|
||||
LONG $-1
|
||||
QUAD $0x1122334455667788
|
||||
QUAD $-2
|
||||
END
|
||||
WORD $0xBEEF
|
||||
RET
|
||||
|
||||
// func trunc()
|
||||
// Truncation, not a range check: each literal keeps its low bytes, exactly
|
||||
// as go tool asm emits them.
|
||||
TEXT ·trunc(SB), NOSPLIT, $0-0
|
||||
BYTE $0x1FF
|
||||
WORD $0x12345
|
||||
LONG $0x123456789
|
||||
QUAD $-2
|
||||
RET
|
||||
Vendored
+70
@@ -0,0 +1,70 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Wide-immediate arithmetic: every classification band of the ADD/SUB
|
||||
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
|
||||
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
|
||||
// their materialised fallback. Byte-for-byte against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// imm12 covers the plain and shifted-by-12 imm12 forms.
|
||||
TEXT ·imm12(SB), NOSPLIT, $0-0
|
||||
ADD $1, R2, R3
|
||||
ADD $0x000aaa, R2, R3
|
||||
ADD $0xaaa000, R2
|
||||
SUB $0x000aaa, R2, R3
|
||||
SUB $0xaaa000, R2
|
||||
ADDW $40960, R0
|
||||
CMP $40960, R0
|
||||
CMPW $40960, R0
|
||||
RET
|
||||
|
||||
// split pins the ADDCON2 band: two imm12 instructions, low half first.
|
||||
TEXT ·split(SB), NOSPLIT, $0-0
|
||||
ADD $0xaaaaaa, R2, R3
|
||||
SUB $0xaaaaaa, R2
|
||||
ADD $0x186a0, R2, R5
|
||||
SUB $0x186a0, R2, R3
|
||||
ADDW $0x60060, R2
|
||||
RET
|
||||
|
||||
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
|
||||
// MOVN for the complement form, the bitmask ORR otherwise.
|
||||
TEXT ·regtmp(SB), NOSPLIT, $0-0
|
||||
ADD $0x1ffe00, R2, R3
|
||||
ADD $0x3fffffffc000, R5
|
||||
ADD $-2048, R2, R3
|
||||
ADD $-100000, R2, R3
|
||||
CMP $0x1000000, R2
|
||||
CMP $0x100000000, R0
|
||||
SUB $-0x100000000, R0, R1
|
||||
RET
|
||||
|
||||
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
|
||||
// compare forms that never split.
|
||||
TEXT ·movseq(SB), NOSPLIT, $0-0
|
||||
ADD $0x12345678, R2, R3
|
||||
SUB $0xe7791f700, R3, R1
|
||||
CMP $0xaaaaaa, R2
|
||||
CMP $0xffffffffffa0, R3
|
||||
CMPW $27745, R2
|
||||
CMPW $0x60060, R2
|
||||
ADDS $0xaaaaaa, R2, R3
|
||||
CMN $0x1000000, R2
|
||||
ADDW $0x12345678, R2, R3
|
||||
RET
|
||||
|
||||
// logical covers the bitmask immediates of the logical family and the
|
||||
// materialised fallback for the values a bitmask cannot carry.
|
||||
TEXT ·logical(SB), NOSPLIT, $0-0
|
||||
AND $0x3ff00000, R2, R3
|
||||
BIC $0x22220000, R3, R4
|
||||
ORR $0x3ff00000, R2
|
||||
EOR $0x3ff00000, R2, R3
|
||||
ANDS $0x3ff00000, R2
|
||||
ORNW $0x3ff00000, R2
|
||||
EONW $0x3ff00000, R2
|
||||
BICSW $0x6006000060060, R5
|
||||
TST $0x4900000049, R0
|
||||
RET
|
||||
@@ -43,6 +43,13 @@ const (
|
||||
At // @
|
||||
Hash // #
|
||||
Pipe // |
|
||||
|
||||
// Semicolon separates statements on one line (a Plan 9 statement
|
||||
// terminator); Ampersand and Tilde are the expression operators & and ~
|
||||
// of constant expressions. All three appear mostly inside macro bodies.
|
||||
Semicolon // ;
|
||||
Ampersand // &
|
||||
Tilde // ~
|
||||
)
|
||||
|
||||
var kindNames = map[Kind]string{
|
||||
@@ -71,6 +78,10 @@ var kindNames = map[Kind]string{
|
||||
At: "@",
|
||||
Hash: "#",
|
||||
Pipe: "|",
|
||||
|
||||
Semicolon: ";",
|
||||
Ampersand: "&",
|
||||
Tilde: "~",
|
||||
}
|
||||
|
||||
// String returns a human-readable name for the kind.
|
||||
|
||||
@@ -32,6 +32,8 @@ func TestGroundTruthARM64(t *testing.T) {
|
||||
"../testdata/verify/crypto_arm64.s",
|
||||
"../testdata/verify/integer_arm64.s",
|
||||
"../testdata/verify/simd_arm64.s",
|
||||
"../testdata/verify/widenimm_arm64.s",
|
||||
"../testdata/verify/carryshift_arm64.s",
|
||||
"../testdata/verify/system_arm64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
|
||||
@@ -122,6 +122,10 @@ func TestGroundTruthAMD64(t *testing.T) {
|
||||
"../testdata/verify/crypto_amd64.s",
|
||||
"../testdata/verify/sse_amd64.s",
|
||||
"../testdata/verify/avx_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/doubleshift_amd64.s",
|
||||
"../testdata/verify/ssestatic_amd64.s",
|
||||
} {
|
||||
|
||||
@@ -29,6 +29,8 @@ func TestGroundTruthLOONG64(t *testing.T) {
|
||||
"../testdata/verify/branchu_loong64.s",
|
||||
"../testdata/verify/atomics_loong64.s",
|
||||
"../testdata/verify/vector_loong64.s",
|
||||
"../testdata/verify/pcalign_loong64.s",
|
||||
"../testdata/verify/l64forms_loong64.s",
|
||||
"trampoline_loong64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
|
||||
@@ -33,6 +33,8 @@ func TestGroundTruthRISCV(t *testing.T) {
|
||||
"../testdata/verify/atomics_riscv64.s",
|
||||
"../testdata/verify/vector_riscv64.s",
|
||||
"../testdata/verify/bitmanip_riscv64.s",
|
||||
"../testdata/verify/pcalign_riscv64.s",
|
||||
"../testdata/verify/branch_far_riscv64.s",
|
||||
"trampoline_riscv64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user