diff --git a/arch/arm64.go b/arch/arm64.go index 9e73d7f..1eaf1dc 100644 --- a/arch/arm64.go +++ b/arch/arm64.go @@ -145,7 +145,7 @@ func arm64Curated() []Instr { for _, op := range []string{ "LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW", "LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW", - "LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL", + "LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", } { t = append(t, i(op, "Atomic memory operation")) } diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index d2a0fdd..8efcfaa 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -189,7 +189,7 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { return arm64MovSize(mnem, ops, fi) case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW": if len(ops) >= 2 && isImmOperand(ops[0]) { - v := immFromOperand(ops[0]) + v := arm64Imm64(ops[0]) // Small immediate (0..4095 or -2048..-1) fits in one instruction. if v >= 0 && v <= 0xFFF { return 4 @@ -221,7 +221,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 if len(ops) != 1 { return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } - return a64wordLE(uint32(immFromOperand(ops[0]))), nil + w := arm64Imm64(ops[0]) + if w < 0 || w > 0xFFFFFFFF { + return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w) + } + return a64wordLE(uint32(w)), nil case "B", "JMP": return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve) case "BL", "CALL": @@ -249,14 +253,19 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 } } + // Shifts: immediate forms alias SBFM/UBFM/EXTR, register forms are the + // two-source LSLV/LSRV/ASRV/RORV. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShift { + return encodeARM64Shift(mnem, enc.op, ops) + } + + // Multiply-accumulate: MADD/MSUB Rm, Ra, Rn, Rd. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPR4 { + return encodeARM64MAddSub(mnem, enc.op, ops) + } + // Register-register data processing. - // ASR/LSL/LSR/ROR with immediate operands use bitfield encoding (SBFM/UBFM). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR { - isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" || - mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW" - if isShift && len(ops) >= 2 && isImmOperand(ops[0]) { - return encodeARM64Bitfield(mnem, enc.op, ops) - } return encodeARM64DPSR(mnem, enc.op, ops) } @@ -478,6 +487,88 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } +// encodeARM64Shift encodes LSL/LSR/ASR/ROR in both widths. The operand order +// is source first, destination last: OP $sh|Rm, Rn, Rd or OP $sh|Rm, Rd. +// With an immediate the shift is the SBFM/UBFM (ROR: EXTR) alias, with a +// register it is the data-processing (2 source) LSLV/LSRV/ASRV/RORV; the +// two-source opcode rides the same 0xd6<<21 field as SDIV/UDIV, with +// LSLV=0b001000, LSRV=0b001001, ASRV=0b001010, RORV=0b001011 at bits 15:10. +func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[len(ops)-1])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + + if isImmOperand(ops[0]) { + width := uint32(64) + if strings.HasSuffix(mnem, "W") { + width = 32 + } + sh := arm64Imm64(ops[0]) + if sh < 0 || uint32(sh) >= width { + return nil, fmt.Errorf("%s: shift amount %d out of range for %d-bit form", mnem, sh, width) + } + switch mnem { + case "LSL", "LSLW": + // UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh + immr := (width - uint32(sh)) % width + return a64wordLE(baseOp | immr<<16 | (width-1-uint32(sh))<<10 | uint32(rn)<<5 | uint32(rd)), nil + case "LSR", "LSRW": + // UBFM Rd, Rn, #sh, #(W-1) + return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil + case "ASR", "ASRW": + // SBFM Rd, Rn, #sh, #(W-1) + return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil + default: + // ROR, RORW: EXTR Rd, Rn, Rn, #sh (Rm = Rn, imms = sh). + return a64wordLE(baseOp | uint32(rn)<<16 | uint32(sh)<<10 | uint32(rn)<<5 | uint32(rd)), nil + } + } + + rm := arm64RegNum(operandRegName(ops[0])) + if rm < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + op2 := uint32(8) // LSLV + switch mnem { + case "LSR", "LSRW": + op2 = 9 // LSRV + case "ASR", "ASRW": + op2 = 10 // ASRV + case "ROR", "RORW": + op2 = 11 // RORV + } + sf := uint32(1) + if strings.HasSuffix(mnem, "W") { + sf = 0 + } + return a64wordLE(sf<<31 | 0xd6<<21 | op2<<10 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64MAddSub encodes MADD/MSUB/MADDW/MSUBW. The toolchain's operand +// order is Rm, Ra, Rn, Rd (its optab case 15 comment says exactly that), so +// the accumulate register is the SECOND operand: base | Rm<<16 | Ra<<10 | +// Rn<<5 | Rd. The optab has no shorter row for these mnemonics, so all four +// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the +// MUL mnemonic, not to these. +func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops)) + } + rm := arm64RegNum(operandRegName(ops[0])) + ra := arm64RegNum(operandRegName(ops[1])) + rn := arm64RegNum(operandRegName(ops[2])) + rd := arm64RegNum(operandRegName(ops[3])) + if rm < 0 || rn < 0 || ra < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(ra)<<10 | uint32(rn)<<5 | uint32(rd)), nil +} + // ---- ADD/SUB immediate ---- // encodeARM64AddSubImm encodes an ADD/SUB immediate instruction. @@ -485,7 +576,7 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } - v := immFromOperand(ops[0]) + v := arm64Imm64(ops[0]) rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { @@ -529,6 +620,8 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 { return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil } + // The imm12 field cannot carry the value; rejecting (rather than + // truncating) matches the toolchain, which reports the same shape. return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v) } @@ -615,14 +708,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // ADRP + ADD } - v := arm64Imm64(src) - if v == 0 { + // Size the immediate exactly as the encoder will emit it: multi-chunk + // values expand to up to four words and the W forms truncate first. + // Anything else would desynchronise the label offsets of pass 1 from + // the bytes pass 2 lays down, corrupting every later branch. + b, err := encodeARM64LoadImm(31, arm64Imm64(src), mnem) + if err != nil { return 4 } - if arm64Movcon(v) >= 0 || arm64Movcon(^v) >= 0 { - return 4 - } - return 8 // MOVZ + MOVK + return len(b) case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": return 8 // ADRP + LDR case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": @@ -638,7 +732,7 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if !ok { lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access } - scale := int32(1) << uint(lt.size) + scale := int64(1) << uint(lt.size) if off >= 0 && off%scale == 0 && off/scale < 4096 { return 4 } @@ -655,28 +749,28 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { } // encodeARM64LoadImm loads an immediate into a register, matching the -// toolchain's MOVZ/MOVN/MOVK sequence. +// toolchain's MOVZ/MOVN/MOVK sequence. W forms truncate to 32 bits first and +// every classification (movcon, complement, chunk count) runs on the truncated +// value, so a 32-bit immediate never reaches the 64-bit halves: MOVW $-1 +// truncates to 0xFFFFFFFF, whose complement is a single zero chunk, and encodes +// as MOVN W, #0. func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { d := v - // For 32-bit MOVW, zero-extend. + sf := uint32(1) // 64-bit if mnem == "MOVW" || mnem == "MOVWU" { d = int64(uint32(v)) + sf = 0 } if d == 0 { // ORR Rd, ZR, ZR (MOV $0, Rd) op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit - if mnem == "MOVW" || mnem == "MOVWU" { + if sf == 0 { op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit } return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil } - sf := uint32(1) // 64-bit - if mnem == "MOVW" || mnem == "MOVWU" { - sf = 0 - } - // The Go toolchain classifies immediates: // - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values // - Negative values: MOVN first, then bitmask @@ -691,15 +785,21 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { } } - // Try MOVZ (single non-zero 16-bit chunk). + // Try MOVZ (single non-zero 16-bit chunk) and MOVN (single non-0xFFFF + // chunk of the complement). The W forms must look inside the 32-bit + // window only, so the complement is masked to the operand width; d is + // already truncated and needs no mask. + width := uint64(0xFFFFFFFF) + if sf == 1 { + width = 0xFFFFFFFFFFFFFFFF + } s := arm64Movcon(d) if s >= 0 { return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil } - // Try MOVN (single non-0xFFFF 16-bit chunk of ^d). - sn := arm64Movcon(^d) + sn := arm64Movcon(^d & int64(width)) if sn >= 0 { - return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil + return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32(((^d)>>uint(sn))&0xFFFF), uint32(rd))), nil } // For values outside the bitmask-first range that are not movcon: try bitmask. @@ -710,7 +810,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { } } - // Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk. + // Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk. var ws []uint32 first := true for i := range 4 { @@ -805,7 +905,7 @@ func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) { // Integer → integer: ORR Rd, ZR, Rs. // FP → FP: FMOV Fd, Fn (FP data processing). // FP ↔ GP: FMOV general (FPCVTI encoding). -// Go Plan 9 syntax: MOV dst, src (first operand = destination). +// Go Plan 9 syntax is source first, destination last: MOV src, dst. func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { rs := arm64RegNum(operandRegName(src)) rd := arm64RegNum(operandRegName(dst)) @@ -826,8 +926,7 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { } // GP ↔ FP: FMOV general (FPCVTI encoding). - // Go syntax: FMOV FPdst, GPsrc or FMOV GPdst, FPsrc. - // First operand = destination, second = source. + // Go syntax: FMOV GPsrc, FPdst or FMOV FPsrc, GPdst, source first. if sc == arm64ClsFP && dc == arm64ClsGR { // FP → GP: FMOV Wd/Xd, Sn/Dn. opcode bits[20:16]=6. sf, typ := uint32(0), uint32(0) @@ -867,7 +966,7 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 lt = a64LoadTable["MOVD"] } - scale := int32(1) << uint(lt.size) + scale := int64(1) << uint(lt.size) storeOpc := a64StoreOpc(lt) var opc int if load { @@ -880,30 +979,31 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil } if off >= -256 && off <= 255 { - return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil + return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil } // Large offset: materialise the base in REGTMP (R27) the way the - // toolchain does and access what remains. + // toolchain does and access what remains. The ADD offsets from the + // operand's own base register, [SP] and [Rn] alike. addImm, addShift, access, ok := arm64SplitOffset(off, scale) if !ok { return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) } return a64WordsLE( - a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm< 0xF { + return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv) + } + return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil } // encodeARM64FPSel encodes a FP conditional select. @@ -1233,8 +1339,23 @@ func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e // ---- Atomics encoding ---- +// arm64ExclMem resolves the memory operand of an exclusive or atomic +// instruction. These encodings have no immediate field: the toolchain +// rejects `LDXR 8(R1), R2` as an illegal combination, so a non-zero offset is +// reported rather than silently dropped (which would read the wrong address). +func arm64ExclMem(mnem string, op *ast.Operand) (int, error) { + rn, off := arm64MemWithFrame(op, arm64FrameInfo{}) + if rn < 0 { + return 0, fmt.Errorf("invalid memory operand in %s", mnem) + } + if off != 0 { + return 0, fmt.Errorf("%s: offset %d not supported, exclusive and atomic accesses take a plain (Rn) operand", mnem, off) + } + return rn, nil +} + // encodeARM64Excl encodes an exclusive load/store instruction. -// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg or reg, mem) +// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg) // STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status) func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { // LDXR/STXR have different operand forms. @@ -1244,9 +1365,12 @@ func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } - rn, _ := arm64MemWithFrame(ops[0], arm64FrameInfo{}) + rn, err := arm64ExclMem(mnem, ops[0]) + if err != nil { + return nil, err + } rt := arm64RegNum(operandRegName(ops[1])) - if rn < 0 || rt < 0 { + if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil @@ -1256,9 +1380,15 @@ func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs := arm64RegNum(operandRegName(ops[0])) - rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{}) + if rs < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + rn, err := arm64ExclMem(mnem, ops[1]) + if err != nil { + return nil, err + } rt := arm64RegNum(operandRegName(ops[2])) - if rs < 0 || rn < 0 || rt < 0 { + if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil @@ -1272,9 +1402,15 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs := arm64RegNum(operandRegName(ops[0])) - rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{}) + if rs < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + rn, err := arm64ExclMem(mnem, ops[1]) + if err != nil { + return nil, err + } rt := arm64RegNum(operandRegName(ops[2])) - if rs < 0 || rn < 0 || rt < 0 { + if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil @@ -1283,43 +1419,25 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, // ---- Bitfield/EXTR encoding ---- // encodeARM64Bitfield encodes a bitfield instruction. -// ASR/LSL/LSR/ROR $shamt, Rn, Rd → 3 operands: $imm, Rn, Rd // BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { - isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" || - mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW" - - if isShift { - // ASR $shamt, Rn, Rd → SBFM with immr=shamt, imms=31/63 - if len(ops) != 3 { - return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) - } - shamt := int(immFromOperand(ops[0])) - rn := arm64RegNum(operandRegName(ops[1])) - rd := arm64RegNum(operandRegName(ops[2])) - if rn < 0 || rd < 0 { - return nil, fmt.Errorf("invalid register operand in %s", mnem) - } - // ASR: SBFM with immr=shamt, imms=31(32-bit) or 63(64-bit) - is64 := mnem == "ASR" - imms := 31 - if is64 { - imms = 63 - } - return a64wordLE(baseOp | uint32(shamt)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil - } - // BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd) if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } - immr := int(immFromOperand(ops[0])) + immr := arm64Imm64(ops[0]) rn := arm64RegNum(operandRegName(ops[1])) - imms := int(immFromOperand(ops[2])) + imms := arm64Imm64(ops[2]) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } + // The toolchain rejects bit numbers at or above the operand width, which + // sf (bit 31 of the base) selects: 64 when set, 32 otherwise. + width := uint32(32) << (baseOp >> 31 & 1) + if immr < 0 || uint32(immr) >= width || imms < 0 || uint32(imms) >= width { + return nil, fmt.Errorf("%s: bit number out of range (immr=%d imms=%d, width=%d)", mnem, immr, imms, width) + } return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil } @@ -1329,13 +1447,19 @@ func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } - lsb := int(immFromOperand(ops[0])) + lsb := arm64Imm64(ops[0]) rm := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } + // The imms field is 6 bits and must stay below the operand width, which + // sf (bit 31 of the base) selects: 64 when set, 32 otherwise. + width := int64(32) << (baseOp >> 31 & 1) + if lsb < 0 || lsb >= width { + return nil, fmt.Errorf("%s: bit number %d out of range (width=%d)", mnem, lsb, width) + } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil } @@ -1367,7 +1491,7 @@ func AssembleFileARM64(f *ast.File) (*Image, error) { return nil, err } - img := &Image{Symbols: map[string]int{}} + img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index a09bf70..d6a68c2 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -27,6 +27,8 @@ package asm // Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET) // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd +import "maps" + // arm64RegNum returns the 5-bit register number for an AArch64 register name: // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the // runtime's assembly uses. Returns -1 for an unrecognised name. @@ -262,19 +264,15 @@ type a64Format uint8 const ( a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. - a64FDPIR // data-processing (immediate): ADD/SUB $imm - a64FLogImm // logical (immediate): AND/ORR/EOR $imm a64FMovWide // move wide: MOVZ, MOVN, MOVK - a64FLSU // load/store (unsigned immediate, scaled) - a64FLSUnscaled // load/store (unscaled immediate) - a64FLSPair // load/store pair a64FBranch // unconditional branch (B/BL) a64FBranchCond // conditional branch (B.cond) a64FUncondBranch // unconditional branch register (BR/BLR/RET) a64FADR // ADR/ADRP a64FEXTR // EXTR a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM - a64FSystem // system: NOP, BRK, etc. + a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source + a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. @@ -282,7 +280,6 @@ const ( a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL - a64FFMovGR // FMOV between GP and FP registers a64FCRC32 // CRC32 a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR @@ -332,30 +329,14 @@ func init() { "ANDSW": 0<<31 | 3<<29 | 0x0a<<24, "BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21, "BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21, - // Shift - "LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM - "LSLW": 0<<31 | 0<<29 | 0x0a<<24, - "LSR": 1<<31 | 0<<29 | 0x0a<<24, - "LSRW": 0<<31 | 0<<29 | 0x0a<<24, - "ASR": 1<<31 | 0<<29 | 0x0a<<24, - "ASRW": 0<<31 | 0<<29 | 0x0a<<24, - "ROR": 1<<31 | 0<<29 | 0x0a<<24, - "RORW": 0<<31 | 0<<29 | 0x0a<<24, - // Multiply - "MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21, - "MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21, - "MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21, - "MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21, - // Divide - "SDIV": 1<<31 | 0<<29 | 0x0d<<24, - "SDIVW": 0<<31 | 0<<29 | 0x0d<<24, - "UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10, - "UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10, - // CRC - "CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10, - "CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10, - "CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10, - "CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10, + // Divide (data-processing 2 source): the opcode occupies bits 15:10 + // of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM + // "Data-processing (2 source)"; the toolchain spells them OPDP2(2) + // and OPDP2(3)). sf=1 selects the X forms. + "SDIV": 1<<31 | 0xd6<<21 | 3<<10, + "SDIVW": 0<<31 | 0xd6<<21 | 3<<10, + "UDIV": 1<<31 | 0xd6<<21 | 2<<10, + "UDIVW": 0<<31 | 0xd6<<21 | 2<<10, // Conditional select "CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10, "CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10, @@ -385,14 +366,37 @@ func init() { a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]} a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]} - // ---- data-processing (immediate) ---- - // ADD/SUB $imm, Rn, Rd - a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24} - a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24} - a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24} - a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24} - a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24} - a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24} + // ---- shifts ---- + // The mnemonic serves both forms: with an immediate the aliases of the + // data-processing (immediate) group apply (ARM ARM "Shifts"), with a + // register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op + // field carries the immediate-alias base; encodeARM64Shift derives both + // it and the two-source opcode. Identities, W = 64 (X) or 32 (W): + // + // LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh + // LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1) + // ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1) + // ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh + shifts := map[string]a64Enc{ + "LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X + "LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W + "LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X + "LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W + "ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X + "ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W + "ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X + "RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W + } + maps.Copy(a64InstrTable, shifts) + + // ---- multiply accumulate ---- + // MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The + // toolchain's optab has no shorter row, so all four operands are + // mandatory, and Ra is the SECOND operand. + a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24} + a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24} + a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15} + a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15} // ---- move wide ---- // MOVZ/MOVN/MOVK @@ -407,22 +411,9 @@ func init() { a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0} a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1} - // ---- load/store (unsigned immediate) ---- - a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit - a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned - a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned - a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned - a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64) - a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half) - a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte) - a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP - a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP - - // Store opcodes (load ^ (1<<22)): - // STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22 - // STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22 - // STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22 - // STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22 + // Load/store mnemonics never enter this table: the MOV pseudo-instruction + // dispatch handles them through a64LoadTable, which also carries the store + // opcode (integer and FP stores both use opc=00, differing only in V). // ---- branches ---- a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26} @@ -445,10 +436,8 @@ func init() { a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21} // ---- system ---- - a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP} - a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP} - a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000} - a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)} + // NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch, + // so they carry no table entry; a64NOP and a64BRK are the encoders. // ---- EXTR ---- a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22} @@ -549,8 +538,8 @@ func init() { a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op} } - // ---- FMOV between GP and FP registers ---- - a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction + // FMOV between GP and FP registers needs no table entry: the MOV + // pseudo-instruction dispatches it by operand class (encodeARM64RegMove). // ---- conditional select: CSEL, CSINC, CSINV, CSNEG ---- csel := map[string]uint32{ @@ -642,14 +631,11 @@ var a64LoadTable = map[string]a64LSType{ "FMOVD": {3, 1, 1}, // LDR D (64-bit FP) } -// a64StoreOpc returns the store opc for a given load type. -// For integer: store opc = 00 (the load opc bits cleared). -// For FP: store opc = 00 (same pattern). +// a64StoreOpc returns the store opc for a given load type: integer and FP +// stores both encode opc=00 (the load's signedness bit sits in opc[1], which +// the store form clears; FP registers are selected by V, not opc). func a64StoreOpc(t a64LSType) int { - if t.V == 1 { - return 0 // FP store - } - return 0 // integer store + return 0 } // arm64RegClass discriminates integer (R), floating-point (F) registers for diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 5250fd6..9e72da1 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -611,3 +611,312 @@ TEXT ·f(SB), NOSPLIT, $0-0 } } } + +// arm64Words assembles a single NOSPLIT leaf body and returns its words. +func arm64Words(t *testing.T, body string) []uint32 { + t.Helper() + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + return leWords(img.Code) +} + +// TestArm64ShiftEncodings pins the shift words against `go tool asm -S` +// output (Go 1.27, arm64): immediate forms alias SBFM/UBFM with ROR as EXTR, +// register forms are the two-source LSLV/LSRV/ASRV/RORV. +func TestArm64ShiftEncodings(t *testing.T) { + got := arm64Words(t, "\tLSL $4, R0, R1\n\tLSR $8, R0, R2\n\tASR $4, R0, R3\n\tROR $12, R0, R4\n"+ + "\tLSLW $4, R0, R5\n\tLSRW $8, R0, R6\n\tASRW $4, R0, R7\n\tRORW $12, R0, R8\n") + want := []uint32{ + 0xd37cec01, // LSL $4 = UBFM X1, X0, #60, #59 + 0xd348fc02, // LSR $8 = UBFM X2, X0, #8, #63 + 0x9344fc03, // ASR $4 = SBFM X3, X0, #4, #63 + 0x93c03004, // ROR $12 = EXTR X4, X0, X0, #12 + 0x531c6c05, // LSLW $4 = UBFM W5, W0, #28, #27 + 0x53087c06, // LSRW $8 = UBFM W6, W0, #8, #31 + 0x13047c07, // ASRW $4 = SBFM W7, W0, #4, #31 + 0x13803008, // RORW $12 = EXTR W8, W0, W0, #12 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("imm shift word %d = %08x, want %08x", i, got[i], want[i]) + } + } + + got = arm64Words(t, "\tLSL R9, R0, R10\n\tLSR R9, R0, R11\n\tASR R9, R0, R12\n\tROR R9, R0, R13\n"+ + "\tLSLW R9, R0, R14\n\tLSRW R9, R0, R15\n\tASRW R9, R0, R16\n\tRORW R9, R0, R17\n") + want = []uint32{ + 0x9ac9200a, // LSLV X10, X0, X9 + 0x9ac9240b, // LSRV X11, X0, X9 + 0x9ac9280c, // ASRV X12, X0, X9 + 0x9ac92c0d, // RORV X13, X0, X9 + 0x1ac9200e, // LSLV W14, W0, W9 + 0x1ac9240f, // LSRV W15, W0, W9 + 0x1ac92810, // ASRV W16, W0, W9 + 0x1ac92c11, // RORV W17, W0, W9 + 0xd65f03c0, // RET + } + for i := range want { + if got[i] != want[i] { + t.Errorf("reg shift word %d = %08x, want %08x", i, got[i], want[i]) + } + } + + // Two-operand spellings fold to Rn = Rd. + got = arm64Words(t, "\tLSL $4, R1\n\tLSR R9, R1\n\tASR $4, R1\n\tROR R9, R1\n\tLSLW $4, R1\n\tRORW R9, R1\n") + want = []uint32{ + 0xd37cec21, // LSL $4, R1 = UBFM X1, X1, #60, #59 + 0x9ac92421, // LSRV X1, X1, X9 + 0x9344fc21, // ASR $4, R1 = SBFM X1, X1, #4, #63 + 0x9ac92c21, // RORV X1, X1, X9 + 0x531c6c21, // LSLW $4, R1 = UBFM W1, W1, #28, #27 + 0x1ac92c21, // RORV W1, W1, W9 + 0xd65f03c0, // RET + } + for i := range want { + if got[i] != want[i] { + t.Errorf("2op shift word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64ShiftRangeErrors: the toolchain reports "illegal bit number" for +// shift amounts at or above the operand width. +func TestArm64ShiftRangeErrors(t *testing.T) { + for _, src := range []string{ + "\tLSL $64, R0, R1\n", + "\tLSRW $32, R0, R1\n", + "\tRORW $32, R0, R1\n", + "\tASR $-1, R0, R1\n", + } { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%s: expected an error, got none", src) + } + } +} + +// TestArm64DivEncodings pins SDIV/UDIV in both widths: the 2-source opcode +// field (bits 15:10 of the 0xd6<<21 fixed field) is UDIV=0b0010, SDIV=0b0011. +func TestArm64DivEncodings(t *testing.T) { + got := arm64Words(t, "\tSDIV R1, R2, R3\n\tUDIV R1, R2, R3\n\tSDIVW R1, R2, R3\n\tUDIVW R1, R2, R3\n") + want := []uint32{ + 0x9ac10c43, // SDIV X3, X2, X1 + 0x9ac10843, // UDIV X3, X2, X1 + 0x1ac10c43, // SDIV W3, W2, W1 + 0x1ac10843, // UDIV W3, W2, W1 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("div word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64MAddSub pins the four-operand MADD/MSUB words (Rm, Ra, Rn, Rd, +// with Ra in bits 14:10) and rejects the shorter spellings the toolchain +// also rejects. +func TestArm64MAddSub(t *testing.T) { + got := arm64Words(t, "\tMADD R1, R2, R3, R4\n\tMSUB R1, R2, R3, R4\n\tMADDW R1, R2, R3, R5\n\tMSUBW R1, R2, R3, R5\n") + want := []uint32{ + 0x9b010864, // MADD X4, X3, X1, X2 (Rm=1, Ra=2, Rn=3) + 0x9b018864, // MSUB X4, X3, X1, X2 + 0x1b010865, // MADD W5, W3, W1, W2 + 0x1b018865, // MSUB W5, W3, W1, W2 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("madd word %d = %08x, want %08x", i, got[i], want[i]) + } + } + + // The accumulate operand is mandatory: 2- and 3-operand forms error + // rather than silently reading R0 or ZR as the accumulator. + for _, body := range []string{ + "\tMADD R1, R2\n", + "\tMADD R1, R2, R3\n", + "\tMSUBW R1, R2, R3\n", + } { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%s: expected an error, got none", body) + } + } +} + +// TestArm64MovImmWidth pins the immediate classifications whose size pass +// once disagreed with the encoder: negative and 0xFFFFFFFF W values go +// through MOVN after 32-bit truncation, and 3- to 4-chunk constants expand +// to one word per non-zero chunk. +func TestArm64MovImmWidth(t *testing.T) { + got := arm64Words(t, "\tMOVW $-1, R0\n\tMOVW $0xFFFFFFFF, R3\n") + want := []uint32{ + 0x12800000, // MOVN W0, #0 + 0x12800003, // MOVN W3, #0 + 0xd65f03c0, // RET + } + for i := range want { + if got[i] != want[i] { + t.Errorf("movw word %d = %08x, want %08x", i, got[i], want[i]) + } + } + + for _, tt := range []struct { + body string + words int + }{ + {"\tMOVD $0x0001000200030000, R2\n", 3}, // three chunks + {"\tMOVD $0x0001000200030004, R1\n", 4}, // four chunks + {"\tMOVW $-1, R0\n", 1}, // MOVN after truncation + } { + if got := arm64Words(t, tt.body); len(got) != tt.words+1 { + t.Errorf("%s: %d words, want %d (including RET)", tt.body, len(got), tt.words+1) + } + } +} + +// TestArm64ExclOffsetErrors: exclusive and atomic encodings carry no +// immediate field, so a non-zero offset is rejected the way the toolchain +// reports "illegal combination" for it, never silently dropped. +func TestArm64ExclOffsetErrors(t *testing.T) { + for _, body := range []string{ + "\tLDXR 8(R1), R2\n", + "\tLDAXR 8(R1), R2\n", + "\tSTXR R3, 8(R1), R4\n", + "\tSTLXR R3, 8(R1), R4\n", + "\tCASD R3, 8(R1), R4\n", + "\tLDADDD R3, 8(R1), R4\n", + } { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%s: expected an error, got none", body) + } + } +} + +// TestArm64ExclNoOffset pins the plain (Rn) forms. gasm parses the store +// with the status register first (ARM ARM order); go tool asm parses the +// same text with the data register first, so the two spellings differ and +// the store word below is gasm's own. +func TestArm64ExclNoOffset(t *testing.T) { + got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n") + want := []uint32{ + 0xc85f7c22, // LDXR X2, [X1] + 0xc8037c24, // STXR W3, X4, [X1] with Rs = R3, Rt = R4 + 0xd65f03c0, // RET + } + for i := range want { + if got[i] != want[i] { + t.Errorf("excl word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are +// rejected instead of wrapping through int32. +func TestArm64AddSubImmRange(t *testing.T) { + for _, body := range []string{ + "\tADD $0x100000000, R0, R1\n", + "\tSUB $-0x100000000, R0, R1\n", + "\tCMP $0x100000000, R0\n", + } { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%s: expected an error, got none", body) + } + } +} + +// TestArm64LargeRegisterOffset pins the large-offset path for a register +// base: the ADD offsets from the operand's own base, not from SP, matching +// the toolchain's `ADD $(256<<12), R2, R27; MOVD (R27), R3`. +func TestArm64LargeRegisterOffset(t *testing.T) { + got := arm64Words(t, "\tMOVD 0x100000(R2), R3\n\tMOVD R3, 0x100000(R2)\n") + want := []uint32{ + 0x9144005b, // ADD $(256<<12), R2, R27 + 0xf9400363, // MOVD (R27), R3 + 0x9144005b, // ADD $(256<<12), R2, R27 + 0xf9000363, // MOVD R3, (R27) + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("large offset word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64LargeFrameSpadj checks the stack-adjustment boundaries of a frame +// whose autosize must be materialised into REGTMP: $5000 rounds the autosize +// to 5024, so the prologue is [MOVD $5024, R27][SUB R27, RSP, R20][STP][ADD +// R20, SP][SUB $8] and SP moves only at its fourth word, while the RET's +// epilogue is [LDP][MOVD $5024, R27][ADD R27, RSP, RSP] before the final +// RET. These PCs feed the DWARF CFA rules and the goobj stack maps. +func TestArm64LargeFrameSpadj(t *testing.T) { + f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), $5000-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + fn := img.Funcs[0] + // autosize 5024: class-2 guard of 6 words (24 bytes), a 5-word prologue + // whose ADD R20, SP sits at byte 8 inside it, a one-instruction body, + // then a 3-word epilogue before the final RET. + wantSpadj := []SpadjStep{{PC: 24 + 12, Value: 5024}, {PC: 24 + 20 + 4 + 12, Value: 0}} + if len(fn.Spadj) != len(wantSpadj) { + t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj) + } + for i := range wantSpadj { + if fn.Spadj[i] != wantSpadj[i] { + t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i]) + } + } + // The words those PCs point between: the prologue's ADD R20, SP at byte + // 36, and the epilogue's materialised ADD R27, RSP, RSP right before the + // final RET at byte 60. + words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size]) + if got := words[(24+12)/4]; got != 0x9100029f { + t.Errorf("prologue word at byte 36 = %08x, want 9100029f (ADD R20, SP)", got) + } + if got := words[(24+20+4+8)/4]; got != 0x8b3b63ff { + t.Errorf("epilogue word at byte 56 = %08x, want 8b3b63ff (ADD R27, RSP, RSP)", got) + } + if got := words[(24+20+4+12)/4]; got != 0xd65f03c0 { + t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got) + } +} diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go index bc8741a..fc1c636 100644 --- a/asm/arm64_frame.go +++ b/asm/arm64_frame.go @@ -258,22 +258,30 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int { if fi.autosize <= 0xf0 { return 4 // MOVD.W instruction decrements SP } - return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD) + // Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose + // position depends on how many words the SUB itself took (immediate, + // shifted immediate, or a materialised REGTMP sequence). + return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1) } // arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to -// (but not including) the final RET instruction. +// (but not including) the final RET instruction. The ADD sequences share the +// prologue's immediate ladder, so their length is read from the same helper +// rather than assumed: a materialised autosize costs its MOV words plus the +// ADD itself. func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { if fi.autosize == 0 { return 0 } if fi.leaf { - return 8 // ADD + ADD + return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) + + len(arm64AddImmWords(uint32(fi.autosize), 31))) } if fi.autosize <= 0xf0 { return 8 // LDR + LDR.P } - return 8 // LDP + ADD + // LDP + the ADD ladder that deallocates the frame. + return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31)) } // arm64ResolvePseudo translates a pseudo-register memory reference into a @@ -382,9 +390,12 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte { ws = append(ws, wordsOf(mov)...) ml := len(mov) / 4 ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27 - ws = append(ws, br(8+ml, a64CondLO)) + // The branches sit at fixed byte offsets in the guard prefix: after + // the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO, + // then a further B.LO word and the CMP for B.LS. + ws = append(ws, br(8+4*ml, a64CondLO)) ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17 - ws = append(ws, br(8+ml+8, a64CondLS)) + ws = append(ws, br(16+4*ml, a64CondLS)) } return a64WordsLE(ws...) } diff --git a/asm/guard_test.go b/asm/guard_test.go index f84fb5d..c3dc134 100644 --- a/asm/guard_test.go +++ b/asm/guard_test.go @@ -152,6 +152,45 @@ func TestStackGuardBytesARM64(t *testing.T) { } } +// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch +// positions for a frame whose guard constant needs two MOV words: the +// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so +// both branches land on the morestack block rather than inside the body. +// The frame size makes the toolchain switch its own prologue decomposition, +// so the assertion is on the branch targets, not pinned bytes. +func TestStackGuardBranchTargetsARM64(t *testing.T) { + f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("assemble: %v", err) + } + fn := img.Funcs[0] + code := img.Code[fn.Offset : fn.Offset+fn.Size] + if len(code)%4 != 0 { + t.Fatalf("function size %d is not a word multiple", len(code)) + } + // autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2 + // and the branches sit at bytes 16 and 24 of the guard prefix. + const morestackBlock = 12 // MOVD R30, R3; BL; B back + blockStart := len(code) - morestackBlock + check := func(name string, off int) { + t.Helper() + w := leWord(code[off:]) + imm19 := int32(w>>5) & 0x7FFFF + if imm19&(1<<18) != 0 { + imm19 -= 1 << 19 + } + if target := off + int(imm19)*4; target != blockStart { + t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart) + } + } + check("B.LO", 16) + check("B.LS", 24) +} + // The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27, // riscv64): the morestack call sits between the guard and the body, and the // guard branches forward over it. Relocation fields are masked. diff --git a/testdata/verify/movimm2_arm64.s b/testdata/verify/movimm2_arm64.s new file mode 100644 index 0000000..ee5709f --- /dev/null +++ b/testdata/verify/movimm2_arm64.s @@ -0,0 +1,28 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Wide MOV immediates that expand past one instruction, each followed by a +// branch to a later label: the displacement is what the two passes must agree +// on, so a size pass that disagrees with the encoder corrupts the branch. +// Byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·movimm2(SB), NOSPLIT, $0-24 + MOVW $-1, R0 + B after1 + MOVD $0x0001000200030004, R1 + B after2 + MOVD $0x0001000200030000, R2 + B after3 + MOVW $0xFFFFFFFF, R3 + +after1: + MOVD R0, r0+0(FP) + +after2: + MOVD R1, r1+8(FP) + +after3: + MOVD R2, r2+16(FP) + RET diff --git a/testdata/verify/shifts_arm64.s b/testdata/verify/shifts_arm64.s new file mode 100644 index 0000000..47c9a8b --- /dev/null +++ b/testdata/verify/shifts_arm64.s @@ -0,0 +1,85 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Shifts, division and multiply-accumulate, byte-parity-checked against +// go tool asm: the immediate shift aliases (SBFM/UBFM, ROR via EXTR), the +// register two-source forms (LSLV/LSRV/ASRV/RORV), SDIV/UDIV and the +// four-operand MADD/MSUB. Every function body is padded to a whole number +// of 16 bytes so the toolchain's object adds no trailing alignment. + +#include "textflag.h" + +// shiftimm exercises the immediate shift aliases in both widths. +TEXT ·shiftimm(SB), NOSPLIT, $0-48 + MOVD x+0(FP), R0 + LSL $4, R0, R1 + LSR $8, R0, R2 + ASR $4, R0, R3 + ROR $12, R0, R4 + LSLW $4, R0, R5 + LSRW $8, R0, R6 + ASRW $4, R0, R7 + RORW $12, R0, R8 + MOVD R1, r1+0(FP) + MOVD R2, r2+8(FP) + MOVD R3, r3+16(FP) + MOVD R4, r4+24(FP) + MOVW R5, r5+32(FP) + MOVW R6, r6+40(FP) + RET + +// shiftreg exercises the register shift forms in both widths: the count +// comes from a register, encoding as LSLV/LSRV/ASRV/RORV. +TEXT ·shiftreg(SB), NOSPLIT, $0-40 + MOVD x+0(FP), R0 + MOVD c+8(FP), R20 + LSL R20, R0, R1 + LSR R20, R0, R2 + ASR R20, R0, R3 + ROR R20, R0, R4 + LSLW R20, R0, R5 + LSRW R20, R0, R6 + ASRW R20, R0, R7 + RORW R20, R0, R8 + MOVD R1, r1+0(FP) + MOVD R2, r2+8(FP) + MOVD R3, r3+16(FP) + MOVD R4, r4+24(FP) + MOVW R5, r5+32(FP) + RET + +// shift2op exercises the two-operand spellings, which fold to Rn = Rd. +TEXT ·shift2op(SB), NOSPLIT, $0-24 + MOVD x+0(FP), R1 + MOVD c+8(FP), R20 + LSL $4, R1 + LSR R20, R1 + LSL $12, R2 + ROR $8, R2 + LSLW $4, R3 + RORW $8, R3 + MOVD R1, r1+0(FP) + MOVD R2, r2+8(FP) + MOVD R3, r3+16(FP) + RET + +// divmul exercises SDIV/UDIV in both widths and the four-operand +// MADD/MSUB, whose accumulate register is the second operand (Rm, Ra, Rn, +// Rd) and rides bits 14:10. +TEXT ·divmul(SB), NOSPLIT, $0-40 + MOVD x+0(FP), R0 + MOVD y+8(FP), R1 + SDIV R1, R0, R2 + UDIV R1, R0, R3 + SDIVW R1, R0, R4 + UDIVW R1, R0, R5 + MADD R1, R0, R2, R6 + MSUB R1, R0, R2, R7 + MADDW R1, R0, R4, R8 + MSUBW R1, R0, R4, R9 + MOVD R2, r2+0(FP) + MOVD R3, r3+8(FP) + MOVD R6, r6+16(FP) + MOVD R7, r7+24(FP) + MOVD R8, r8+32(FP) + RET