From 0629f5e2df7b6a59d4f5c9ecf0595b7a03f2a610 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 11:49:05 +0200 Subject: [PATCH] feat(arm64): assemble PCALIGN padding and BYTE literal bytes Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 1093 ++++++++++++++++++++++++++++++++++++++--- asm/arm64_encode.go | 72 +++ 2 files changed, 1108 insertions(+), 57 deletions(-) diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 899ecb7..5686b5a 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -56,7 +56,11 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: - pos += arm64InstrSize(s, fi) + if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { + pos += arm64PCAlignPad(pos, s) + } else { + pos += arm64InstrSize(s, fi, pos) + } } } @@ -68,7 +72,7 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ p := guardLen + len(prologue) for _, stmt := range t.Body { if in, ok := stmt.(*ast.Instr); ok { - p += arm64InstrSize(in, fi) + p += arm64InstrSize(in, fi, p) } } bodyLen = p - (guardLen + len(prologue)) @@ -86,6 +90,21 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ if !ok { continue } + if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" { + pad := arm64PCAlignPad(pc, in) + for i := 0; i < pad/4; i++ { + out = append(out, a64wordLE(a64NOP)...) + pc += 4 + } + continue + } + if strings.ToUpper(in.Mnemonic.Text) == "BYTE" { + for _, op := range in.Operands { + out = append(out, byte(arm64Imm64(op))) + pc++ + } + continue + } code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits) if err != nil { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) @@ -178,25 +197,81 @@ func arm64LabelOK(op *ast.Operand) (string, bool) { return "", false } +// arm64WritebackSuffix splits a mnemonic carrying the toolchain's post-index +// (.P) or pre-index (.W) suffix, as in MOVD.P or LDP.W. It reports the base +// mnemonic, the suffix letter and whether a suffix was present. +func arm64WritebackSuffix(mnem string) (base, wb string, ok bool) { + if before, ok0 := strings.CutSuffix(mnem, ".P"); ok0 { + return before, "P", true + } + if before, ok0 := strings.CutSuffix(mnem, ".W"); ok0 { + return before, "W", true + } + return mnem, "", false +} + +// arm64PCAlignPad returns the padding PCALIGN inserts before the next +// instruction so that it starts at the requested boundary relative to the +// function start. The boundary must be a power of two between 8 and 2048, +// as the toolchain requires. +func arm64PCAlignPad(pos int, instr *ast.Instr) int { + if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) { + return 0 + } + align := int(arm64Imm64(instr.Operands[0])) + if align < 8 || align > 2048 || align&(align-1) != 0 { + return 0 + } + return (align - pos%align) % align +} + +// isARM64MovMnemonic reports whether m is one of the MOV-family spellings the +// arm64 encoder treats as the MOV pseudo-instruction. +func isARM64MovMnemonic(m string) bool { + switch m { + case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", + "FMOVS", "FMOVD": + return true + } + return false +} + // arm64InstrSize returns the encoded size of an instruction: 4 bytes for // most, more for the multi-instruction expansions. -func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { +func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands if mnem == "RET" { return len(arm64Return(fi)) } + if mnem == "PCALIGN" { + return arm64PCAlignPad(pos, instr) + } + if mnem == "BYTE" { + return len(ops) + } switch mnem { case "VMOVS", "VMOVD", "VMOVQ": // ADRP + ADD + wide load against a pooled literal. return 12 } + // Writeback (.P/.W) forms are always a single instruction. + if base, _, ok := arm64WritebackSuffix(mnem); ok { + if isARM64MovMnemonic(base) || a64InstrTable[base].format == a64FPair { + return 4 + } + } + // The funcdata pseudo-statements contribute no bytes. + switch mnem { + case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": + return 0 + } switch mnem { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return arm64MovSize(mnem, ops, fi) - case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW": + case "ADD", "ADDW", "SUB", "SUBW": if len(ops) >= 2 && isImmOperand(ops[0]) { v := arm64Imm64(ops[0]) // Small immediate (0..4095 or -2048..-1) fits in one instruction. @@ -243,7 +318,29 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve) case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": - return encodeARM64Mov(instr, mnem, fi, relocs) + return encodeARM64Mov(instr, mnem, "", fi, relocs) + } + + // Post-index (.P) and pre-index (.W) writeback forms: the MOV family and + // the load/store pair family carry the suffix on the mnemonic itself. + // (The SIMD VLD1.P/VST1.P/VLD1R.P/VLD4R.P spellings also end in .P, but + // for them the suffix is part of the mnemonic and the table routes them.) + if base, wb, ok := arm64WritebackSuffix(mnem); ok { + switch { + case isARM64MovMnemonic(base): + return encodeARM64Mov(instr, base, wb, fi, relocs) + case a64InstrTable[base].format == a64FPair: + return encodeARM64Pair(base, a64InstrTable[base].op, ops, fi, wb, relocs) + } + } + + // The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS, + // GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine + // code: the toolchain emits zero instruction bytes for them, and so does + // the encoder here. + switch mnem { + case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": + return nil, nil } // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). @@ -258,7 +355,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 // ADD/SUB immediate. if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" || - mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { + mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || + mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" { if len(ops) >= 2 && isImmOperand(ops[0]) { return encodeARM64AddSubImm(mnem, ops) } @@ -351,9 +449,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64AcqRel(mnem, enc.op, ops) } - // Load/store pairs (LDP, STP, LDPW, STPW, FLDPD, FSTPD). + // Load/store pairs (LDP, STP, LDPW, STPW, FLDPD, FSTPD). The .P/.W + // writeback forms are routed earlier, straight from the mnemonic. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair { - return encodeARM64Pair(mnem, enc.op, ops, fi) + return encodeARM64Pair(mnem, enc.op, ops, fi, "", relocs) } // Compare-and-branch and test-and-branch to a label. @@ -460,6 +559,15 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri } op := ops[0] + // Branch to the program counter itself: JMP (PC) spins forever. The + // toolchain encodes it as an unconditional branch with a zero offset. + if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { + if link { + return nil, fmt.Errorf("%s: branch to PC is not a call", mnem) + } + return a64wordLE(a64Branch(0, 0)), nil + } + // Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The // toolchain's spelling carries no offset and no index; anything else // is reported rather than silently dropped. @@ -532,14 +640,17 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } - target := resolve(arm64Label(ops[0])) - targetOff, ok := offsets[target] - if !ok { - return nil, fmt.Errorf("undefined label %q", target) + rel, pcRel := arm64PCRelOffset(ops[0]) + if !pcRel { + target := resolve(arm64Label(ops[0])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel = (targetOff - pc) >> 2 } - rel := (targetOff - pc) >> 2 if rel < -(1<<18) || rel >= (1<<18) { - return nil, fmt.Errorf("branch to %q too far (19-bit range)", target) + return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel) } // The condition code is in the low 4 bits of baseOp. cond := baseOp & 0xF @@ -552,10 +663,122 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in // For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd). // For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR). // For NEG: NEG Rm, Rd (Rn=ZR). +// The first operand may carry the toolchain's modifier shapes: a shifted +// register (R0<<2, R1>>3) or an extend modifier (R0.UXTW, R3.SXTW<<2). +// Logical instructions (AND/ANDS/BIC/…) also accept a bitmask immediate, and +// the ADD/SUB-with-flags family an add/sub immediate. func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW" + // Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating + // bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the + // complement of the written value. + if len(ops) >= 2 && len(ops) <= 3 && isImmOperand(ops[0]) { + var logical bool + switch mnem { + case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW", + "BIC", "BICW", "BICS", "BICSW", "TST", "TSTW": + logical = true + } + if logical { + v, ok := arm64ImmOperandValue(ops[0]) + if !ok { + return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw) + } + inverted := false + switch mnem { + case "BIC", "BICW", "BICS", "BICSW": + inverted = true + } + if inverted { + v = ^v + } + width := 64 + if strings.HasSuffix(mnem, "W") { + width = 32 + } + n, immr, imms, ok := a64LogicalImm(v, width) + if !ok { + return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " ")) + } + opc := (baseOp >> 29) & 7 + sf := (baseOp >> 31) & 1 + var rn, rd int + switch len(ops) { + case 3: + rn = arm64RegNum(operandRegName(ops[1])) + rd = arm64RegNum(operandRegName(ops[2])) + default: + rd = arm64RegNum(operandRegName(ops[1])) + rn = rd + } + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(sf<<31 | opc<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | + uint32(rn)<<5 | uint32(rd)), nil + } + } + + // Shifted-register and extend-modifier first operand: OP Rm<= 2 && arm64RegMod(ops[0]) { + rm, shiftBits, extendOpt, isExtend, amount, ok := arm64RegModifier(ops[0]) + if !ok { + return nil, fmt.Errorf("%s: invalid register modifier %q", mnem, ops[0].Raw) + } + isAddSub := strings.HasPrefix(mnem, "ADD") || strings.HasPrefix(mnem, "SUB") || + isCmp || isNeg || mnem == "ADC" || mnem == "ADCS" || mnem == "SBC" || mnem == "SBCS" || + mnem == "ADCW" || mnem == "ADCSW" || mnem == "SBCW" || mnem == "SBCSW" + if isExtend && !isAddSub { + return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem) + } + if !isExtend { + if amount < 0 || amount > 63 { + return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount) + } + // SP-based ADD/SUB have no shifted-register encoding: the + // toolchain canonicalises LSL #n to the extend form (UXTX, or + // UXTW in the 32-bit forms) and rejects a right shift. + spInvolved := false + for _, op := range ops[1:] { + if n := operandRegName(op); n == "SP" || n == "RSP" { + spInvolved = true + } + } + if spInvolved && strings.HasPrefix(mnem, "ADD") || spInvolved && strings.HasPrefix(mnem, "SUB") { + if shiftBits != 0 { + return nil, fmt.Errorf("%s: right shift not encodable against SP", mnem) + } + opt := uint32(3) // UXTX + if strings.HasSuffix(mnem, "W") { + opt = 2 // UXTW + } + baseOp |= 1<<21 | opt<<13 | uint32(amount)<<10 + } else { + baseOp |= shiftBits<<22 | uint32(amount)<<10 + } + } else { + baseOp |= 1<<21 | extendOpt<<13 | uint32(amount)<<10 + } + rd := arm64RegNum(operandRegName(ops[len(ops)-1])) + rn := rd + if len(ops) == 3 { + rn = arm64RegNum(operandRegName(ops[1])) + } + if isCmp { + rd = 31 + } + if isNeg && len(ops) == 2 { + rn = 31 + } + if rm < 0 || rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil + } + switch len(ops) { case 3: // OP Rm, Rn, Rd @@ -610,6 +833,104 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } +// arm64RegMod reports whether a register operand carries the shifted-register +// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend +// option (R0.UXTW, R3.SXTW<<2). +func arm64RegMod(op *ast.Operand) bool { + if op.Addr.Shift != "" { + return true + } + name := operandRegName(op) + if _, after, ok := strings.Cut(name, "."); ok { + return strings.IndexByte(after, '[') < 0 // element selectors are not extend modifiers + } + return false +} + +// arm64RegModifier resolves a modified register operand: the register number, +// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend +// option (UXTB=0..SXTX=7) with its shift amount. +func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) { + name := operandRegName(op) + if before, after, ok0 := strings.Cut(name, "."); ok0 { + switch strings.ToUpper(strings.TrimSpace(after)) { + case "UXTB": + extendOpt = 0 + case "UXTH": + extendOpt = 1 + case "UXTW", "UXTW32": + extendOpt = 2 + case "UXTX": + extendOpt = 3 + case "SXTB": + extendOpt = 4 + case "SXTH": + extendOpt = 5 + case "SXTW": + extendOpt = 6 + case "SXTX": + extendOpt = 7 + default: + return 0, 0, 0, false, 0, false + } + extend = true + rm = arm64RegNum(strings.TrimSpace(before)) + if rm < 0 { + return 0, 0, 0, false, 0, false + } + amount, ok = arm64ShiftAmount(op.Addr.Shift) + if !ok || amount < 0 || amount > 4 { + return 0, 0, 0, false, 0, false + } + return rm, 0, extendOpt, true, amount, true + } + shiftKind = 0 // LSL + shift := strings.TrimSpace(op.Addr.Shift) + switch { + case strings.HasPrefix(shift, "<<"): + shiftKind = 0 + case strings.HasPrefix(shift, ">>"): + shiftKind = 1 // LSR + case strings.HasPrefix(shift, "->"): + shiftKind = 2 // ASR + case strings.HasPrefix(shift, "@>"): + shiftKind = 3 // ROR + default: + return 0, 0, 0, false, 0, false + } + amount, ok = arm64ShiftAmount(op.Addr.Shift) + if !ok { + return 0, 0, 0, false, 0, false + } + rm = arm64RegNum(name) + if rm < 0 { + return 0, 0, 0, false, 0, false + } + return rm, shiftKind, 0, false, amount, true +} + +// arm64ShiftAmount extracts the integer after the shift operator in a +// shift suffix (<<, >>, ->, @>). +func arm64ShiftAmount(shift string) (int, bool) { + s := strings.TrimSpace(shift) + s = strings.TrimPrefix(s, "<<") + s = strings.TrimPrefix(s, ">>") + s = strings.TrimPrefix(s, "->") + s = strings.TrimPrefix(s, "@>") + s = strings.TrimSpace(s) + if s == "" { + if shift == "" { + return 0, true + } + return 0, false + } + v, err := strconv.Atoi(s) + if err != nil { + return 0, false + } + return v, true +} + // encodeARM64Shift encodes LSL/LSR/ASR/ROR in both widths. The operand order // is source first, destination last: OP $sh|Rm, Rn, Rd or OP $sh|Rm, Rd. // With an immediate the shift is the SBFM/UBFM (ROR: EXTR) alias, with a @@ -709,16 +1030,16 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" - isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" + isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW" + isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || + mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" sf := uint32(1) // 64-bit - if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" { + if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" { sf = 0 // 32-bit } - if mnem == "CMP" || mnem == "CMPW" { - rd = 31 // ZR - } - if mnem == "CMN" || mnem == "CMNW" { + // CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings + // keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n). + if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { rd = 31 // ZR } @@ -761,13 +1082,41 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { // MOVx $sym(SB), rd address of a static symbol (ADRP+ADD) // MOVx sym(SB), rd load from a static symbol (ADRP+LDR) // MOVx rd, sym(SB) store to a static symbol (ADRP+STR) -func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) { +// +// With the writeback suffix (MOVD.P, MOVD.W, …) the memory form becomes a +// post-index or pre-index access whose offset is the base writeback amount. +// Storing a $0 immediate stores ZR; any other immediate is rejected, matching +// the toolchain. +func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src, dst := ops[0], ops[1] + if wb != "" { + switch { + case isMemOperand(src) && !isMemOperand(dst): + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s: invalid destination register", mnem) + } + return encodeARM64MemOp(mnem, src, rd, true, fi, wb) + case isMemOperand(dst) && !isMemOperand(src): + rs := arm64RegNum(operandRegName(src)) + if rs < 0 { + if !isImmOperand(src) || arm64Imm64(src) != 0 { + return nil, fmt.Errorf("%s: invalid source register", mnem) + } + // Storing a constant zero stores the zero register. + rs = 31 + } + return encodeARM64MemOp(mnem, dst, rs, false, fi, wb) + default: + return nil, fmt.Errorf("%s: writeback form needs a register and a memory operand", mnem) + } + } + // Immediate → register (including $sym(SB)). if isImmOperand(src) && !isMemOperand(src) { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { @@ -800,20 +1149,41 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[] return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs) } + // System-register moves: MOVD NZCV, R0 reads (MRS) and MOVD R0, NZCV + // writes (MSR) the flag and FP status registers. + if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "" && src.Addr.Base == "" { + if base, ok := a64MRSOps[src.Addr.Sym.Name]; ok { + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s %s: invalid destination register", mnem, src.Addr.Sym.Name) + } + return a64wordLE(base | uint32(rd)&31), nil + } + } + if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "" && dst.Addr.Base == "" { + if base, ok := a64MSRRegOps[dst.Addr.Sym.Name]; ok { + rs := arm64RegNum(operandRegName(src)) + if rs < 0 { + return nil, fmt.Errorf("%s %s: invalid source register", mnem, dst.Addr.Sym.Name) + } + return a64wordLE(base | uint32(rs)&31), nil + } + } + // Memory load/store with offset. - if isMemOperand(src) && !isMemOperand(dst) { + if arm64IsMemOperand(src) && !arm64IsMemOperand(dst) { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s: invalid destination register", mnem) } - return encodeARM64MemOp(mnem, src, rd, true, fi) + return encodeARM64MemOp(mnem, src, rd, true, fi, "") } - if !isMemOperand(src) && isMemOperand(dst) { + if !arm64IsMemOperand(src) && arm64IsMemOperand(dst) { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s: invalid source register", mnem) } - return encodeARM64MemOp(mnem, dst, rs, false, fi) + return encodeARM64MemOp(mnem, dst, rs, false, fi, "") } // Register → register. @@ -831,6 +1201,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // ADRP + ADD } + if isMemOperand(dst) { + // Only the $0 (ZR store) immediate reaches memory, in one word. + return 4 + } // Size the immediate exactly as the encoder will emit it: multi-chunk // values expand to up to four words and the W forms truncate first. // Anything else would desynchronise the label offsets of pass 1 from @@ -1081,7 +1455,12 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { } // encodeARM64MemOp encodes a memory load or store with offset. -func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm64FrameInfo) ([]byte, error) { +// encodeARM64MemOp encodes a MOV-family load/store. With wb set ("P" post +// or "W" pre) the offset is the signed writeback amount applied to the base +// register after (post) or before (pre) the access; the offset must fit the +// unscaled 9-bit field and the base must be a real register, since a pseudo +// frame base cannot be written back. +func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm64FrameInfo, wb string) ([]byte, error) { rn, off := arm64MemWithFrame(mem, fi) if rn < 0 { return nil, fmt.Errorf("invalid memory operand") @@ -1100,6 +1479,23 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 } else { opc = storeOpc } + + if wb != "" { + if mem.Addr.Sym != nil && mem.Addr.Sym.Pseudo != "" { + return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem) + } + if off < -256 || off > 255 { + return nil, fmt.Errorf("%s: writeback offset %d out of range (-256..255)", mnem, off) + } + w := a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg) + if wb == "P" { + w |= 1 << 10 // post-index + } else { + w |= 3 << 10 // pre-index + } + return a64wordLE(w), nil + } + // Scaled unsigned offset first, then the unscaled ±255 form. if off >= 0 && off%scale == 0 && off/scale < 4096 { return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil @@ -1212,6 +1608,34 @@ func arm64Imm64(op *ast.Operand) int64 { return 0 } +// arm64ImmOperandValue returns the immediate an operand stands for, falling +// back to a raw evaluation for the spellings the parser leaves unevaluated: +// the one's-complement form $~n and parenthesised constant expressions. +// The second result reports whether a value could be recovered. +func arm64ImmOperandValue(op *ast.Operand) (int64, bool) { + if op.Imm.HasVal { + v := op.Imm.Val + if op.Imm.Neg { + v = -v + } + return v, true + } + s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) + inverted := false + if i := strings.IndexByte(s, '~'); i >= 0 { + inverted = true + s = s[i+1:] + } + v, ok := arm64EvalExpr(s) + if !ok { + return 0, false + } + if inverted { + v = ^v + } + return v, true +} + // arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo- // registers via the frame mapping. The offset stays 64-bit: the AST carries // int64 displacements and truncating here would wrap offsets beyond 2^31 @@ -1221,9 +1645,207 @@ func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int64) { base, pseudo := arm64ResolvePseudo(op.Addr.Sym, fi) return base, int64(pseudo) } + if rn, off, ok := arm64ExprMem(op); ok { + return rn, off + } return arm64RegNum(op.Addr.Base), op.Addr.Offset } +// arm64IsMemOperand is the arm64-side memory test: the shared syntactic test +// plus the parenthesised-expression form (8*1)(RSP), which the parser leaves +// unstructured (empty base) because the offset is not a plain integer. +func arm64IsMemOperand(op *ast.Operand) bool { + if isMemOperand(op) { + return true + } + _, _, ok := arm64ExprMem(op) + return ok +} + +// arm64ExprMem recovers a base register and an evaluated offset from a +// parenthesised-expression memory operand such as (8*22)(RSP) or (0*8)(R0). +// It reports ok=false for anything else. +func arm64ExprMem(op *ast.Operand) (rn int, off int64, ok bool) { + if op.Addr.Base != "" || op.Addr.Sym != nil { + return 0, 0, false + } + s := strings.Join(strings.Fields(op.Raw), " ") + if !strings.HasSuffix(s, ")") { + return 0, 0, false + } + // Split the trailing "( REG )" from the leading "( EXPR )". + inner := strings.LastIndex(s, "(") + if inner <= 0 { + return 0, 0, false + } + regPart := strings.TrimSpace(s[inner+1 : len(s)-1]) + head := strings.TrimSpace(s[:inner]) + if !strings.HasPrefix(head, "(") || !strings.HasSuffix(head, ")") { + return 0, 0, false + } + expr := strings.TrimSpace(head[1 : len(head)-1]) + v, ok := arm64EvalExpr(expr) + if !ok { + return 0, 0, false + } + rn = arm64RegNum(regPart) + if rn < 0 { + return 0, 0, false + } + return rn, v, true +} + +// arm64EvalExpr evaluates the Plan 9 constant arithmetic the assembler +// accepts inside memory operands: integers with unary minus and the + - * << +// >> & | ^ operators. Operator precedence follows the Plan 9 convention +// (shifts bind tighter than +, * tighter than shifts); expressions it cannot +// fully reduce report ok=false. +func arm64EvalExpr(s string) (int64, bool) { + type parser struct { + toks []string + pos int + } + var scan func(string) []string + scan = func(s string) []string { + var out []string + for s = strings.TrimSpace(s); s != ""; s = strings.TrimSpace(s) { + switch { + case s[0] == '(' || s[0] == ')': + out = append(out, s[:1]) + s = s[1:] + case s[0] >= '0' && s[0] <= '9': + i := 0 + for i < len(s) && ((s[i] >= '0' && s[i] <= '9') || s[i] == 'x' || s[i] == 'X' || + s[i] >= 'a' && s[i] <= 'f' || s[i] >= 'A' && s[i] <= 'F') { + i++ + } + out = append(out, s[:i]) + s = s[i:] + case strings.HasPrefix(s, "<<"), strings.HasPrefix(s, ">>"): + out = append(out, s[:2]) + s = s[2:] + case strings.IndexByte("+-*&|^~", s[0]) >= 0: + out = append(out, s[:1]) + s = s[1:] + default: + return nil + } + } + return out + } + toks := scan(s) + if toks == nil { + return 0, false + } + p := &parser{toks: toks} + + var primary func() (int64, bool) + var expr func() (int64, bool) + primary = func() (int64, bool) { + if p.pos >= len(p.toks) { + return 0, false + } + t := p.toks[p.pos] + switch { + case t == "-": + p.pos++ + v, ok := primary() + return -v, ok + case t == "+": + p.pos++ + return primary() + case t == "~": + p.pos++ + v, ok := primary() + return ^v, ok + case t == "(": + p.pos++ + v, ok := expr() + if !ok || p.pos >= len(p.toks) || p.toks[p.pos] != ")" { + return 0, false + } + p.pos++ + return v, true + } + if t[0] < '0' || t[0] > '9' { + return 0, false + } + v, err := strconv.ParseInt(strings.TrimPrefix(strings.TrimPrefix(t, "0X"), "0x"), 0, 64) + if err != nil { + return 0, false + } + p.pos++ + return v, true + } + var binop func(minLevel int) (int64, bool) + level := func(op string) int { + switch op { + case "|", "^": + return 1 + case "&": + return 2 + case "<<", ">>": + return 3 + case "*": + return 4 + case "+", "-": + return 5 + } + return 0 + } + var apply func(v int64, op string, w int64) (int64, bool) + apply = func(v int64, op string, w int64) (int64, bool) { + switch op { + case "+": + return v + w, true + case "-": + return v - w, true + case "*": + return v * w, true + case "<<": + return v << uint(w), true + case ">>": + return v >> uint(w), true + case "&": + return v & w, true + case "|": + return v | w, true + case "^": + return v ^ w, true + } + return 0, false + } + binop = func(minLevel int) (int64, bool) { + v, ok := primary() + if !ok { + return 0, false + } + for p.pos < len(p.toks) { + op := p.toks[p.pos] + lv := level(op) + if lv == 0 || lv < minLevel { + break + } + p.pos++ + w, ok := binop(lv + 1) + if !ok { + return 0, false + } + v, ok = apply(v, op, w) + if !ok { + return 0, false + } + } + return v, true + } + expr = func() (int64, bool) { return binop(1) } + v, ok := binop(1) + if !ok || p.pos != len(p.toks) { + return 0, false + } + return v, true +} + // arm64Label returns the label name of an operand. func arm64Label(op *ast.Operand) string { if op.Addr.Sym != nil { @@ -1235,14 +1857,18 @@ func arm64Label(op *ast.Operand) string { // ---- FP instruction encoding ---- // encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd). -// FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. +// FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. The two-operand spelling +// (FMULD F3, F5: multiply into the second operand) folds Rn into Rd. func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { - if len(ops) != 3 { - return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + if len(ops) != 3 && len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) - rn := arm64RegNum(operandRegName(ops[1])) - rd := arm64RegNum(operandRegName(ops[2])) + rd := arm64RegNum(operandRegName(ops[len(ops)-1])) + rn := rd + if len(ops) == 3 { + rn = arm64RegNum(operandRegName(ops[1])) + } if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } @@ -1692,6 +2318,20 @@ func encodeARM64CondCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, // encodeARM64Branch19 encodes CBZ/CBNZ: (Rt, label), // word = base | imm19<<5 | Rt with imm19 = (target - pc) >> 2. +// arm64PCRelOffset recognises the toolchain's forward branch spelling +// n(PC): the number counts INSTRUCTIONS from the branch itself. It reports +// ok for operands spelled that way and leaves everything else alone. +func arm64PCRelOffset(op *ast.Operand) (int, bool) { + if op.Addr.Base != "PC" && !(op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { + return 0, false + } + off := op.Addr.Offset + if !op.Addr.HasOff { + off = 0 + } + return int(off), true +} + func encodeARM64Branch19(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) @@ -1700,14 +2340,17 @@ func encodeARM64Branch19(mnem string, baseOp uint32, ops []*ast.Operand, pc int, if rt < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - target := resolve(arm64Label(ops[1])) - targetOff, ok := offsets[target] - if !ok { - return nil, fmt.Errorf("undefined label %q", target) + rel, pcRel := arm64PCRelOffset(ops[1]) + if !pcRel { + target := resolve(arm64Label(ops[1])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel = (targetOff - pc) >> 2 } - rel := (targetOff - pc) >> 2 if rel < -(1<<18) || rel >= (1<<18) { - return nil, fmt.Errorf("branch to %q too far (19-bit range)", target) + return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel) } return a64wordLE(baseOp | uint32(rel)&0x7FFFF<<5 | uint32(rt)), nil } @@ -1727,14 +2370,17 @@ func encodeARM64TestBranch(mnem string, baseOp uint32, ops []*ast.Operand, pc in if rt < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - target := resolve(arm64Label(ops[2])) - targetOff, ok := offsets[target] - if !ok { - return nil, fmt.Errorf("undefined label %q", target) + rel, pcRel := arm64PCRelOffset(ops[2]) + if !pcRel { + target := resolve(arm64Label(ops[2])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel = (targetOff - pc) >> 2 } - rel := (targetOff - pc) >> 2 if rel < -(1<<13) || rel >= (1<<13) { - return nil, fmt.Errorf("branch to %q too far (14-bit range)", target) + return nil, fmt.Errorf("%s: branch offset %d out of 14-bit range", mnem, rel) } return a64wordLE(baseOp | uint32(bit>>5)<<31 | uint32(bit&31)<<19 | uint32(rel)&0x3FFF<<5 | uint32(rt)), nil } @@ -1743,7 +2389,11 @@ func encodeARM64TestBranch(mnem string, baseOp uint32, ops []*ast.Operand, pc in // (mem, (Rt1, Rt2)), stores (Rt1, Rt2), mem; the scaled immediate rides // imm7 at bits 21:15 and must fit -64..63 after division by the access // size (8 bytes for the D forms, 4 for the W forms). -func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64FrameInfo) ([]byte, error) { +// encodeARM64Pair encodes the load/store pair family. wb selects the +// addressing mode: "" the signed-offset form, "P" post-index, "W" pre-index; +// in the writeback forms the immediate is the amount added to the base +// register around the access. +func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64FrameInfo, wb string, relocs *[]Reloc) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } @@ -1756,23 +2406,53 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra if !load { memOp, pairOp = ops[1], ops[0] } - if !isMemOperand(memOp) { + if !arm64IsMemOperand(memOp) { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } rt1, rt2, ok := arm64PairOf(pairOp) if !ok { return nil, fmt.Errorf("%s expects a register pair (Rt1, Rt2)", mnem) } + + // Pair access against a static symbol: ADRP R27, sym; ADD R27, R27, #lo; + // LDP/STP (R27), (Rt1, Rt2), with the R_ADDRARM64 pair riding the first + // two words, exactly like the toolchain lays it out. + if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" { + if wb != "" { + return nil, fmt.Errorf("%s: writeback is not supported on a symbol operand", mnem) + } + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset}, + Reloc{Off: 4, After: 4, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 27), // ADRP R27, 0 + a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27 + baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (Rt1, Rt2) + ), nil + } + rn, off := arm64MemWithFrame(memOp, fi) if rn < 0 { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } + if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" { + return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem) + } if off%scale != 0 || off < -64*scale || off > 63*scale { return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale) } imm7 := off / scale - // The base carries the opc, V and L halves; only the scaled immediate - // and the three registers are filled in here. + // The signed-offset form carries bits 24:23 = 10; post-index drops bit 24 + // and pre-index sets both, with the base carrying the opc, V and L halves. + switch wb { + case "P": + baseOp = baseOp&^(1<<24) | 1<<23 + case "W": + baseOp |= 1 << 23 + } return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil } @@ -2307,15 +2987,41 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { return nil, fmt.Errorf("%s expects an element operand Vn.[i]", mnem) } if !ok1 || !src.hasIdx { - // General register into a vector element. - f, ok := a64ElemField(dst.arr, dst.idx) - if !ok { - return nil, fmt.Errorf("%s: invalid element operand", mnem) - } + // General register into a vector. An arranged destination without a + // lane index duplicates the register across every lane (DUP Vd.T, + // Rn); an indexed one is an INS into that single lane. rs := arm64RegNum(operandRegName(ops[0])) if rs < 0 { return nil, fmt.Errorf("%s: source must be a general register", mnem) } + if !dst.hasIdx { + var imm5, q uint32 + switch dst.arr { + case "B8": + imm5, q = 1, 0 + case "B16": + imm5, q = 1, 1<<30 + case "H4": + imm5, q = 2, 0 + case "H8": + imm5, q = 2, 1<<30 + case "S2": + imm5, q = 4, 0 + case "S4": + imm5, q = 4, 1<<30 + case "D1": + imm5, q = 8, 0 + case "D2": + imm5, q = 8, 1<<30 + default: + return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) + } + return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil + } + f, ok := a64ElemField(dst.arr, dst.idx) + if !ok { + return nil, fmt.Errorf("%s: invalid element operand", mnem) + } return a64wordLE(0x4e001c00 | f<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil } sf, ok := a64ElemField(src.arr, src.idx) @@ -2360,13 +3066,27 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { // // VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn) // VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn) +// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane) // VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3] // -// The post-index forms set the post bit and Rm = 11111; the increment is -// implied by the register list, and a spelled offset rides along the way the -// toolchain's own encodings ignore it. +// The post-index forms set the post bit and Rm = 11111. A spelled offset +// rides along (the encoding ignores it; the toolchain only checks that it +// matches the access size), and a multi-register post-index list takes no +// offset at all, the increment following from the list. func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) { load := strings.HasPrefix(mnem, "VLD") + + // One-lane forms spell a single Vt.T[i] operand, not a bracketed list. + laneIdx := 1 + if !load { + laneIdx = 0 + } + if len(ops) > laneIdx { + if v, ok := arm64VecOf(ops[laneIdx]); ok && v.hasIdx { + return encodeARM64VLDSTLane(mnem, post, ops, load, laneIdx, v) + } + } + listStart, memAt := 0, 1 if load { // Every load spells the memory operand first. @@ -2405,7 +3125,11 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err if !ok { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) } - return a64wordLE(base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)), nil + w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg) + if post != 0 { + w |= 1<<23 | 0x1f<<16 + } + return a64wordLE(w), nil } if len(vs) < 1 || len(vs) > 4 { @@ -2435,6 +3159,50 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil } +// encodeARM64VLDSTLane encodes the one-lane structure forms: +// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and +// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the +// toolchain's table and is rejected. +func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem) + } + memIdx := laneIdx ^ 1 + rn, _ := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{}) + if rn < 0 { + return nil, fmt.Errorf("%s: invalid memory operand", mnem) + } + if !load && post == 0 { + return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem) + } + w := uint32(0x0d400000) + switch strings.ToUpper(v.arr) { + case "B": + // Index at bits 12:10 (the size field doubles as the low index bits). + w |= uint32(v.idx) << 10 + case "H": + // Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11. + w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30 + case "S": + // Index<0> at bit 12, index<1> at bit 30. + w |= 4<<13 | uint32(v.idx&1)<<12 | uint32(v.idx>>1&1)<<30 + case "D": + // Index<0> at bit 30, fixed size field 01. + w |= 4<<13 | 1<<10 | uint32(v.idx&1)<<30 + default: + return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr) + } + // The base carries bit 22 (L) set; a store clears it. The post-index + // forms add bit 23 and Rm = 11111. + if !load { + w &^= 1 << 22 + } + if post != 0 { + w |= 1<<23 | 0x1f<<16 + } + return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil +} + // a64ArrSizeQ maps an arrangement to its size code (bits 11:10) and 128-bit // flag for the structure load/store words. func a64ArrSizeQ(arr string) (size, q uint32, ok bool) { @@ -2619,6 +3387,12 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order } // immediates; the object-file emitters record R_ADDRARM64 relocations for // the linker. func AssembleFileARM64(f *ast.File) (*Image, error) { + // Resolve the file's own simple #define aliases (RARG0 → R0, NR → R9, + // TEB_error → 0x68, B0 → V0) the way the toolchain's preprocessor does + // textually. Parameterised macros and multi-line bodies are beyond + // token substitution and stay untouched. + arm64ResolveAliases(f) + dataSyms, err := collectData(f) if err != nil { return nil, err @@ -2713,3 +3487,208 @@ func AssembleFileARM64(f *ast.File) (*Image, error) { markExternals(img, dataSyms) return img, nil } + +// arm64ResolveAliases applies the file's own simple #define aliases to every +// instruction operand, the way the toolchain's preprocessor substitutes them +// textually. Only single-line, non-parameterised bodies whose value is a +// register name or an integer constant are resolved: anything else +// (parameterised macros, multi-instruction bodies, header-supplied names) +// stays as written and surfaces as a normal operand error. +func arm64ResolveAliases(f *ast.File) { + type alias struct { + raw string // replacement text + reg bool // the body is a register name + value int64 // the body as an integer (when !reg) + isInt bool // the body parsed as an integer + mem bool // the body is a frame-relative memory reference + sym *ast.Symbol // the parsed frame-relative reference (when mem) + } + aliases := map[string]alias{} + for _, d := range f.Decls { + pre, ok := d.(*ast.Preproc) + if !ok { + continue + } + fields := strings.Fields(pre.Raw) + if len(fields) < 3 || fields[0] != "define" { + continue + } + name, body := fields[1], strings.Join(fields[2:], " ") + // A parameterised macro spells its parameter list right after the + // name; a multi-instruction body needs statement expansion. + if strings.ContainsAny(name, "(") || body == "" || + strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") { + continue + } + isReg := func(s string) bool { + if arm64RegNum(s) >= 0 { + return true + } + if v, ok := a64VecReg(s); ok && !v.hasIdx { + return true + } + return false + } + if isReg(body) { + aliases[name] = alias{raw: body, reg: true} + continue + } + if sym, ok := arm64FrameAliasBody(body); ok { + aliases[name] = alias{raw: body, mem: true, sym: sym} + continue + } + v, err := strconv.ParseInt(body, 0, 64) + if err != nil { + continue + } + aliases[name] = alias{raw: body, value: v, isInt: true} + } + if len(aliases) == 0 { + return + } + + // replace rewrites whole-word occurrences of the alias names in s. + replace := func(s string) string { + if s == "" { + return s + } + out := strings.Fields(s) + for i, w := range out { + if a, ok := aliases[w]; ok { + out[i] = a.raw + } + } + if len(out) == 0 { + return s + } + return strings.Join(out, " ") + } + // replaceToken rewrites an operand whose whole text is one alias use + // possibly followed by syntax (POLY.D[0]): the alias must be a prefix + // ending at a non-identifier character. + replaceToken := func(s string) (string, bool) { + for name, a := range aliases { + if s == name { + return a.raw, true + } + if strings.HasPrefix(s, name) { + rest := s[len(name):] + if rest != "" && !isAliasWordByte(rest[0]) { + return a.raw + rest, true + } + } + } + return s, false + } + + for _, d := range f.Decls { + t, ok := d.(*ast.Text) + if !ok { + continue + } + for _, stmt := range t.Body { + in, ok := stmt.(*ast.Instr) + if !ok { + continue + } + for _, op := range in.Operands { + // Immediate aliases: $CLOCK_REALTIME → $0. The parser may + // leave the unevaluable name in Raw alone or carry it as an + // unevaluated symbol immediate; both shapes resolve here. + if op.Kind == ast.OpImmediate && !op.Imm.HasVal { + name := "" + if op.Imm.Sym != nil && op.Imm.Sym.Pseudo == "" { + name = op.Imm.Sym.Name + } else if op.Addr.Sym == nil && op.Addr.Base == "" { + name = strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) + } + if a, ok := aliases[name]; ok && a.isInt { + op.Imm.Val, op.Imm.HasVal = a.value, true + op.Imm.Sym = nil + op.Addr = ast.Address{} + op.Raw = "$" + a.raw + continue + } + } + // Memory operand whose displacement is an alias: + // TEB_error(R18_PLATFORM) with TEB_error → 0x68. + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base != "" { + if a, ok := aliases[op.Addr.Sym.Name]; ok && a.isInt { + op.Addr.Offset, op.Addr.HasOff = a.value, true + op.Addr.Sym = nil + op.Raw = a.raw + "(" + op.Addr.Base + ")" + continue + } + } + // Register alias as a memory base. + if op.Addr.Base != "" { + if a, ok := aliases[op.Addr.Base]; ok && a.reg { + op.Addr.Base = a.raw + } + } + // Bare and suffixed symbol tokens: registers, vector + // registers with arrangement or lane, branch labels. + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" { + if nn, changed := replaceToken(op.Addr.Sym.Name); changed { + // A frame-relative body (ret+24(FP)) rebuilds the + // operand as a full memory reference. + if a, ok := aliases[op.Addr.Sym.Name]; ok && a.mem { + op.Addr.Sym, op.Addr.Base, op.Addr.Shift = a.sym, "", "" + op.Raw = a.raw + continue + } + op.Addr.Sym.Name = nn + op.Addr.Sym.Raw = nn + op.Raw = nn + continue + } + } + // Bracketed groups and lists travel in Raw: (RARG0, R1) and + // [V0.B16, V1.B16] with aliased members. + if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") || + strings.HasPrefix(strings.TrimSpace(op.Raw), "[") { + op.Raw = replace(op.Raw) + } + } + } + } +} + +// isAliasWordByte reports whether b can appear inside an identifier, so a +// substitution ending here would have merged two tokens. +func isAliasWordByte(b byte) bool { + return b == '_' || b >= '0' && b <= '9' || b >= 'a' && b <= 'z' || b >= 'A' && b <= 'Z' +} + +// arm64FrameAliasBody parses an alias body of the shape NAME, NAME+off or +// NAME+off(PSEUDO) with PSEUDO one of FP/SP: the frame-relative memory +// references the runtime headers alias wholesale (LOCAL_RETVALID +// → ret+24(FP)). ok is false for anything else. +func arm64FrameAliasBody(body string) (*ast.Symbol, bool) { + s := strings.Join(strings.Fields(body), "") + i := strings.LastIndexByte(s, '(') + if i < 0 || !strings.HasSuffix(s, ")") { + return nil, false + } + pseudo := s[i+1 : len(s)-1] + if pseudo != "FP" && pseudo != "SP" { + return nil, false + } + head := s[:i] + name, offStr := head, "" + if j := strings.LastIndexByte(head, '+'); j >= 0 { + name, offStr = head[:j], head[j+1:] + } + if name == "" { + return nil, false + } + off, hasOff := int64(0), false + if offStr != "" { + v, err := strconv.ParseInt(offStr, 0, 64) + if err != nil { + return nil, false + } + off, hasOff = v, true + } + return &ast.Symbol{Name: name, Pseudo: pseudo, Offset: off, HasOff: hasOff, Raw: s}, true +} diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 6c18727..762cd13 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -29,6 +29,7 @@ package asm import ( "maps" + "math/bits" "strconv" "strings" @@ -159,6 +160,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 { return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd } +// ---- logical immediate ---- + +// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the +// given lane width (32 or 64): it returns the N, immr and imms fields of the +// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's +// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest +// repeating element, find the run of ones and its rotation. ok is false when +// v is not expressible (all zeros, all ones, or not a single cyclic run). +func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) { + u := uint64(v) + if width == 32 { + u &= 0xFFFFFFFF + } + size := uint64(width) + mask := ^uint64(0) + if size < 64 { + mask = uint64(1)< 2 { + half := size / 2 + hm := uint64(1)<>half&hm { + size = half + u &= hm + } else { + break + } + } + ones := bits.OnesCount64(u) + // Find the right-rotation that lays the ones out contiguously at the + // bottom of the element; the hardware applies the inverse rotation. + em := uint64(1)<>r | u<<(int(size)-r) + if size < 64 { + rotated &= em + } + if rotated == expected { + rot = r + break + } + } + if rot < 0 { + return 0, 0, 0, false + } + if size == 64 { + n = 1 + } + immr = uint32((int(size) - rot) % int(size)) + imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1) + return n, immr, imms, true +} + // ---- load/store (unsigned immediate, scaled) ---- // a64LSU encodes a load/store register (unsigned immediate, scaled): @@ -778,7 +840,9 @@ func init() { a64InstrTable["VST1"] = a64Enc{format: a64FVLDST} a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1} } // a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word, @@ -904,6 +968,14 @@ var a64MRSOps = map[string]uint32{ "ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000, "CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040, "DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480, + "NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420, +} + +// a64MSRRegOps maps the system register names GOROOT writes through the +// MSR (register) form, spelled in Go assembly as MOVD Rn, or +// MSR Rn, ; the source register rides bits 4:0. +var a64MSRRegOps = map[string]uint32{ + "NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420, } // a64MSROps maps the system register names GOROOT writes to their fixed