From 40476546df39f022377751459138cec2c21a9155 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Mon, 14 Sep 2026 23:25:14 +0200 Subject: [PATCH] fix(asm): close the oracle parity gaps in frame addressing and calls --- asm/arm64_assemble.go | 92 +++++++++++++++++++----------- asm/arm64_frame.go | 27 +++++++-- asm/assemble.go | 16 ++++-- asm/loong64_assemble.go | 13 +++-- asm/riscv_assemble.go | 60 +++++++++++++++++-- asm/riscv_frame.go | 23 ++++---- testdata/verify/bigframe_amd64.s | 19 ++++++ testdata/verify/bigframe_arm64.s | 35 ++++++++++++ testdata/verify/bigframe_loong64.s | 18 ++++++ testdata/verify/bigframe_riscv64.s | 25 ++++++++ testdata/verify/guard_amd64.s | 25 ++++++++ testdata/verify/guard_arm64.s | 30 ++++++++++ testdata/verify/guard_loong64.s | 44 ++++++++++++++ testdata/verify/guard_riscv64.s | 25 ++++++++ verify/arm64_groundtruth_test.go | 2 + verify/groundtruth_test.go | 58 +++++++++++++++++++ verify/l64_groundtruth_test.go | 2 + verify/riscv_groundtruth_test.go | 2 + 18 files changed, 456 insertions(+), 60 deletions(-) create mode 100644 testdata/verify/bigframe_amd64.s create mode 100644 testdata/verify/bigframe_arm64.s create mode 100644 testdata/verify/bigframe_loong64.s create mode 100644 testdata/verify/bigframe_riscv64.s create mode 100644 testdata/verify/guard_amd64.s create mode 100644 testdata/verify/guard_arm64.s create mode 100644 testdata/verify/guard_loong64.s create mode 100644 testdata/verify/guard_riscv64.s diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index f2643e6..51d3ba7 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -222,7 +222,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } return a64wordLE(uint32(immFromOperand(ops[0]))), nil - case "B": + case "B", "JMP": return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve) case "BL", "CALL": return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve) @@ -337,8 +337,9 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri } op := ops[0] - // External symbol reference: BL sym(SB). - if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { + // Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a + // relocation (R_CALLARM64 either way). + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { if relocs != nil { *relocs = append(*relocs, Reloc{ Off: 0, @@ -348,8 +349,12 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri Kind: RelArm64Branch, }) } - // Emit BL with zero offset; the linker fills in the target. - return a64wordLE(a64Branch(1, 0)), nil + // Emit B/BL with zero offset; the linker fills in the target. + bop := uint32(0) // B + if link { + bop = 1 // BL + } + return a64wordLE(a64Branch(bop, 0)), nil } target := resolve(arm64Label(op)) @@ -593,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { } _, off := arm64MemWithFrame(mem, fi) // Scaled unsigned offset fits if aligned and in range. - lt := a64LoadTable[mnem] - if lt.size == 0 { - lt.size = 3 // default to64-bit for MOV + lt, ok := a64LoadTable[mnem] + if !ok { + lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access } scale := int32(1) << uint(lt.size) if off >= 0 && off%scale == 0 && off/scale < 4096 { @@ -604,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if off >= -256 && off <= 255 { return 4 // unscaled } - return 12 // materialise offset + LDR/STR + if _, _, _, ok := arm64SplitOffset(off, scale); ok { + return 8 // ADD base, REGTMP + access + } + return 12 // literal pool range: encoding reports it as unsupported default: return 4 // register move } @@ -824,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 } scale := int32(1) << uint(lt.size) - if load { - // Try scaled unsigned offset first. - if off >= 0 && off%scale == 0 { - imm12 := uint32(off / scale) - if imm12 < 4096 { - return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil - } - } - // Try unscaled (9-bit signed). - if off >= -256 && off <= 255 { - return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil - } - // Large offset: materialise in R20 (TMP) and use register-offset. - return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) - } - // Store: same encoding but opc bits indicate store. storeOpc := a64StoreOpc(lt) - if off >= 0 && off%scale == 0 { - imm12 := uint32(off / scale) - if imm12 < 4096 { - return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil - } + var opc int + if load { + opc = lt.opc + } else { + opc = storeOpc + } + // Scaled unsigned offset first, then the unscaled ±255 form. + if off >= 0 && off%scale == 0 && off/scale < 4096 { + return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil } if off >= -256 && off <= 255 { - return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil + return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil } - return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) + // Large offset: materialise the base in REGTMP (R27) the way the + // toolchain does and access what remains. + addImm, addShift, access, ok := arm64SplitOffset(off, scale) + if !ok { + return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) + } + return a64WordsLE( + a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<= 0 && rest>>12 <= 4095 { + return uint32(rest >> 12), 1, off - rest, true + } + return 0, 0, 0, false } // ---- static symbol references (ADRP + offset) ---- diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go index 1ca20ef..bc8741a 100644 --- a/asm/arm64_frame.go +++ b/asm/arm64_frame.go @@ -86,9 +86,16 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo { if fi.frame != 0 || !fi.leaf { fi.autosize = fi.frame + 8 // space for the saved LR - if fi.autosize%16 != 0 { - // The toolchain aligns to 16: if autosize%16 == 8, add 8; - // otherwise add whatever is needed. + // The toolchain always adds an extrasize: 8 when the total leaves a + // 16-byte alignment gap, another 16 when already aligned. + switch fi.autosize % 16 { + case 8: + fi.autosize += 8 + case 0: + fi.autosize += 16 + default: + // The toolchain rejects unaligned frames; round up so such + // sources still assemble. fi.autosize += 16 - (fi.autosize % 16) } } @@ -181,12 +188,16 @@ func arm64Prologue(fi arm64FrameInfo) []byte { } // arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value -// fits the imm12 field, otherwise the toolchain materialises it into REGTMP -// (R27) and subtracts the register in the extended-register form. +// fits the imm12 field (plain, or shifted left by 12 when it is a multiple +// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and +// subtracts the register in the extended-register form. func arm64SubImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)} } + if imm <= 4095<<12 && imm&0xFFF == 0 { + return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)} + } mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if err != nil { mov = nil @@ -194,11 +205,15 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 { return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd)) } -// arm64AddImmWords emits ADD $imm, SP, Rd with the same REGTMP fallback. +// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12 +// and REGTMP fallback ladder. func arm64AddImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)} } + if imm <= 4095<<12 && imm&0xFFF == 0 { + return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)} + } mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if err != nil { mov = nil diff --git a/asm/assemble.go b/asm/assemble.go index 6dd08fb..79be2dd 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -515,6 +515,9 @@ func addSP(size int) []byte { // ADDQ $size, SP func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) { mnem := strings.ToUpper(s.Mnemonic.Text) if isJumpMnemonic(mnem) { + if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { + return 5, nil // opcode + rel32, always the long form + } return jumpSize(mnem, long), nil } code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) @@ -564,9 +567,10 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon var ps []sbPatch var err error if isJumpMnemonic(mnem) { - if mnem == "CALL" && isSBCall(s) { - // CALL sym(SB): a rel32 call against a static or external - // symbol, resolved by the file-level layout or the linker. + if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { + // CALL/JMP sym(SB): a rel32 call (or tail call) against a + // static or external symbol, resolved by the file-level layout + // or the linker. code, ps, err = encodeSBCall(s, link) if err != nil { return nil, nil, err @@ -678,8 +682,12 @@ func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) { if !ok { return nil, nil, fmt.Errorf("CALL: unsupported operand") } + opcode := []byte{0xE8} + if strings.ToUpper(s.Mnemonic.Text) == "JMP" { + opcode = []byte{0xE9} // a tail call, no return address pushed + } e := &enc{} - if err := e.emit(&instr{opcode: []byte{0xE8}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil { + if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil { return nil, nil, err } ps := make([]sbPatch, len(e.patches)) diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index 4e57c88..18fcb08 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -527,13 +527,18 @@ func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[stri } return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil } - // Direct symbol: sym+off(SB) → bl with an R_CALLLOONG64 relocation (the - // linker fills the offset), as the toolchain does for CALL/BL/JAL. - if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" && link { + // Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation + // (the linker fills the offset), as the toolchain does for CALL/BL/JAL + // and for tail-calling JMP. + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { + opc := l64jumpTable["B"] + if link { + opc = l64jumpTable["BL"] + } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset}) } - return l64wordLE(l64bbl(l64jumpTable["BL"], 0)), nil + return l64wordLE(l64bbl(opc, 0)), nil } // Direct: label → b/bl. target := resolve(l64Label(op)) diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index 5ee6fc2..043c3c1 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -162,6 +162,14 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil { return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0])) } + // Frame-relative loads and stores: a frame offset beyond the signed + // 12-bit range materialises the address in X31 first. + if isMemOperand(ops[0]) && !isMemOperand(ops[1]) { + return riscvFrameMemSize(ops[0], fi) + } + if isMemOperand(ops[1]) && !isMemOperand(ops[0]) { + return riscvFrameMemSize(ops[1], fi) + } } // I-type arithmetic with a large immediate expands to several instructions. if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) { @@ -212,6 +220,14 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv // C.J, so always emit the 32-bit JAL. var target string if len(ops) >= 1 { + // JMP sym(SB): a tail call, JAL X0 against a symbol relocation. + if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" { + if relocs != nil { + *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset}) + } + word = riscvJType(0, 0) + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + } target = labelFromOperand(ops[0]) } targetOff, ok := offsets[target] @@ -586,8 +602,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV load: invalid operand") } - word := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rs1, off) - return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil } // Register → memory (store). @@ -604,8 +619,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt if rs2 < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV store: invalid operand") } - word := riscvSType(riscvEnc{0x23, 0x3, 0x00}, rs1, rs2, off) - return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil } // Register → register (ADDI $0, src, dst). @@ -620,6 +634,44 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt } } +// riscvFrameMemOp encodes a register-relative load (store=false, I-type +// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width +// at off(rs1). Offsets beyond the signed 12-bit range materialise the +// address in X31 first: LUI hi (the rounding split), then ADD X31, rs1, +// matching the toolchain's large-frame addressing; the access uses the +// sign-extended low part, which always fits. +func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte { + if fits12(off) { + var word uint32 + if store { + word = riscvSType(enc, rs1, reg, off) + } else { + word = riscvIType(enc, reg, rs1, off) + } + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} + } + lo := off - (splitHi(off) << 12) + out := riscvAddressInX31WithBase(off, rs1) + var word uint32 + if store { + word = riscvSType(enc, 31, reg, lo) + } else { + word = riscvIType(enc, reg, 31, lo) + } + return append(out, wordLE(word)...) +} + +// riscvFrameMemSize returns the encoded size of a frame-relative MOV for the +// layout pass: 4 bytes when the offset fits, otherwise the X31 +// materialisation plus the access. +func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int { + rs1, off := memFromOperandWithFrame(op, fi) + if fits12(off) { + return 4 + } + return len(riscvAddressInX31WithBase(off, rs1)) + 4 +} + // encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm, // rd), matching the toolchain's instructionsForMOVConst. For 12-bit // immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits diff --git a/asm/riscv_frame.go b/asm/riscv_frame.go index 1b94e65..66e5e66 100644 --- a/asm/riscv_frame.go +++ b/asm/riscv_frame.go @@ -146,10 +146,18 @@ func splitHi(v int32) int32 { return high } -// riscvAddressInX31 materialises hi(v) into X31 and leaves the caller to add -// the low part, matching the toolchain's large-frame addressing: C.LUI (or -// LUI) X31, hi; C.ADD (or ADD) X31, SP. +// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer, +// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi; +// C.ADD (or ADD) X31, SP. func riscvAddressInX31(v int32) []byte { + return riscvAddressInX31WithBase(v, 2) +} + +// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary +// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field +// carries the full 5-bit register, so the compressed form is always +// available. +func riscvAddressInX31WithBase(v int32, rs1 int) []byte { hi := splitHi(v) var out []byte if hi >= -32 && hi <= 31 { @@ -158,13 +166,8 @@ func riscvAddressInX31(v int32) []byte { } else { out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...) } - if hi >= -32 && hi <= 31 { - c := rvcCR(0x9, 31, 2) - out = append(out, byte(c), byte(c>>8)) - } else { - out = append(out, wordLE(riscvRType(riscvEnc{0x33, 0x0, 0x00}, 31, 2, 31))...) - } - return out + c := rvcCR(0x9, 31, uint32(rs1)) + return append(out, byte(c), byte(c>>8)) } // riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry: diff --git a/testdata/verify/bigframe_amd64.s b/testdata/verify/bigframe_amd64.s new file mode 100644 index 0000000..278093b --- /dev/null +++ b/testdata/verify/bigframe_amd64.s @@ -0,0 +1,19 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Large frame offsets and tail calls, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·bigframe(SB), $9000-32 + MOVQ x+0(FP), R8 + MOVQ y+8(FP), R9 + MOVQ R8, z+24(FP) + MOVQ R9, w+8992(FP) + RET + +TEXT ·tail(SB), $0-0 + JMP ·other(SB) + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/bigframe_arm64.s b/testdata/verify/bigframe_arm64.s new file mode 100644 index 0000000..21eaf27 --- /dev/null +++ b/testdata/verify/bigframe_arm64.s @@ -0,0 +1,35 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Large frame offsets and tail calls, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·bigframe(SB), $9000-32 + MOVD x+0(FP), R5 + MOVD y+8(FP), R6 + MOVD R5, z+24(FP) + MOVB R5, b+16(FP) + MOVD R5, w+30000(FP) + RET + +TEXT ·unaligned(SB), $9000-16 + MOVD x+4(FP), R5 + MOVD R5, ret+8(FP) + RET + +TEXT ·b32k(SB), $32744-8 + MOVD x+0(FP), R5 + MOVD R5, ret+0(FP) + RET + +TEXT ·oddframe(SB), $8-8 + MOVD x+0(FP), R5 + MOVD R5, ret+0(FP) + RET + +TEXT ·tail(SB), $0-0 + JMP ·other(SB) + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/bigframe_loong64.s b/testdata/verify/bigframe_loong64.s new file mode 100644 index 0000000..ae9b90d --- /dev/null +++ b/testdata/verify/bigframe_loong64.s @@ -0,0 +1,18 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Large frame offsets and tail calls, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·bigframe(SB), $9000-32 + MOVV x+0(FP), R5 + MOVB y+8990(FP), R6 + MOVV R5, z+16(FP) + RET + +TEXT ·tail(SB), $0-0 + JMP ·other(SB) + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/bigframe_riscv64.s b/testdata/verify/bigframe_riscv64.s new file mode 100644 index 0000000..61ca2f5 --- /dev/null +++ b/testdata/verify/bigframe_riscv64.s @@ -0,0 +1,25 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Large frame offsets and tail calls, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·bigframe(SB), $9000-32 + MOV x+0(FP), X5 + MOV y+8(FP), X6 + MOV X5, z+24(FP) + MOV $77, X7 + MOV X7, w+8992(FP) + RET + +TEXT ·oddsp(SB), $16-8 + MOV X5, x-9000(SP) + MOV X5, ret+0(FP) + RET + +TEXT ·tail(SB), $0-0 + JMP ·other(SB) + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/guard_amd64.s b/testdata/verify/guard_amd64.s new file mode 100644 index 0000000..6e17144 --- /dev/null +++ b/testdata/verify/guard_amd64.s @@ -0,0 +1,25 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Stack-split guard classes, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·leafsmall(SB), $16-0 + RET + +TEXT ·leafmed(SB), $256-0 + RET + +TEXT ·leafbig(SB), $8192-0 + RET + +TEXT ·callsmall(SB), $16-0 + CALL ·other(SB) + RET + +TEXT ·nosplit(SB), NOSPLIT, $16-0 + RET + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/guard_arm64.s b/testdata/verify/guard_arm64.s new file mode 100644 index 0000000..4676856 --- /dev/null +++ b/testdata/verify/guard_arm64.s @@ -0,0 +1,30 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Stack-split guard classes, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·leafsmall(SB), $16-0 + RET + +TEXT ·leafmed(SB), $256-0 + RET + +TEXT ·leafbig(SB), $8192-0 + RET + +TEXT ·oddframe(SB), $8-8 + MOVD x+0(FP), R5 + MOVD R5, ret+0(FP) + RET + +TEXT ·callsmall(SB), $16-0 + CALL ·other(SB) + RET + +TEXT ·nosplit(SB), NOSPLIT, $16-0 + RET + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/guard_loong64.s b/testdata/verify/guard_loong64.s new file mode 100644 index 0000000..5cda9ee --- /dev/null +++ b/testdata/verify/guard_loong64.s @@ -0,0 +1,44 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Stack-split guard classes, byte-parity-checked against go tool asm: every +// class plus the materialised-constant and zero-low-bit variants. + +#include "textflag.h" + +TEXT ·leafsmall(SB), $16-0 + RET + +TEXT ·leafmed(SB), $256-0 + RET + +TEXT ·fit2048(SB), $2040-0 + RET + +TEXT ·med2048off(SB), $2168-0 + RET + +TEXT ·medmat(SB), $2176-0 + RET + +TEXT ·leafbig(SB), $8192-0 + RET + +TEXT ·bigzero(SB), $4088-0 + RET + +TEXT ·big3976(SB), $4096-0 + RET + +TEXT ·giantlo0(SB), $4216-0 + RET + +TEXT ·callbig(SB), $8192-0 + CALL ·other(SB) + RET + +TEXT ·nosplit(SB), NOSPLIT, $16-0 + RET + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/testdata/verify/guard_riscv64.s b/testdata/verify/guard_riscv64.s new file mode 100644 index 0000000..ee02d80 --- /dev/null +++ b/testdata/verify/guard_riscv64.s @@ -0,0 +1,25 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Stack-split guard classes, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·leafsmall(SB), $16-0 + RET + +TEXT ·leafmed(SB), $256-0 + RET + +TEXT ·leafbig(SB), $8192-0 + RET + +TEXT ·frameless(SB), $0-0 + CALL ·other(SB) + RET + +TEXT ·nosplit(SB), NOSPLIT, $16-0 + RET + +TEXT ·other(SB), NOSPLIT, $0 + RET diff --git a/verify/arm64_groundtruth_test.go b/verify/arm64_groundtruth_test.go index b4b391f..3be96f9 100644 --- a/verify/arm64_groundtruth_test.go +++ b/verify/arm64_groundtruth_test.go @@ -23,6 +23,8 @@ func TestGroundTruthARM64(t *testing.T) { "../testdata/verify/movimm_arm64.s", "../testdata/verify/branch_arm64.s", "../testdata/verify/call_arm64.s", + "../testdata/verify/bigframe_arm64.s", + "../testdata/verify/guard_arm64.s", } { t.Run(path, func(t *testing.T) { src, err := os.ReadFile(path) diff --git a/verify/groundtruth_test.go b/verify/groundtruth_test.go index 7bfaf98..73d6a95 100644 --- a/verify/groundtruth_test.go +++ b/verify/groundtruth_test.go @@ -4,9 +4,23 @@ package verify import ( + "bytes" + "os" "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/asm" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" ) +func mustRead(t *testing.T, path string) string { + t.Helper() + b, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + return string(b) +} + func TestGroundTruthBasic(t *testing.T) { // Use the simple test kernel — it assembles with go tool asm. gt, err := GroundTruth("../testdata/verify/basic_amd64.s") @@ -75,3 +89,47 @@ func keys(m map[string][]byte) []string { } return out } + +// TestGroundTruthAMD64 runs the amd64 kernels through the same live +// comparison the other arches use: gasm output versus go tool asm output, +// with relocation fields masked on both sides. +func TestGroundTruthAMD64(t *testing.T) { + for _, path := range []string{ + "../testdata/verify/basic_amd64.s", + "../testdata/verify/bigframe_amd64.s", + "../testdata/verify/guard_amd64.s", + } { + t.Run(path, func(t *testing.T) { + f, errs := parser.Parse(path, mustRead(t, path)) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := asm.AssembleFile(f) + if err != nil { + t.Fatalf("AssembleFile: %v", err) + } + gt, err := GroundTruth(path) + if err != nil { + t.Fatalf("GroundTruth: %v", err) + } + matched := 0 + for _, fn := range img.Funcs { + gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs) + goCode, ok := gt[fn.Name] + if !ok { + t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt)) + continue + } + goCode = maskRelocs(goCode, fn.Relocs) + if !bytes.Equal(gasmCode, goCode) { + t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode)) + continue + } + matched++ + } + if matched == 0 { + t.Fatal("no functions matched") + } + }) + } +} diff --git a/verify/l64_groundtruth_test.go b/verify/l64_groundtruth_test.go index aeffe22..d59c4ee 100644 --- a/verify/l64_groundtruth_test.go +++ b/verify/l64_groundtruth_test.go @@ -22,6 +22,8 @@ func TestGroundTruthLOONG64(t *testing.T) { for _, path := range []string{ "../testdata/verify/basic_loong64.s", "../testdata/verify/fp_loong64.s", + "../testdata/verify/bigframe_loong64.s", + "../testdata/verify/guard_loong64.s", } { t.Run(path, func(t *testing.T) { src, err := os.ReadFile(path) diff --git a/verify/riscv_groundtruth_test.go b/verify/riscv_groundtruth_test.go index 300d110..5c4ec36 100644 --- a/verify/riscv_groundtruth_test.go +++ b/verify/riscv_groundtruth_test.go @@ -25,6 +25,8 @@ func TestGroundTruthRISCV(t *testing.T) { "../testdata/verify/movimm_riscv64.s", "../testdata/verify/branch_riscv64.s", "../testdata/verify/call_riscv64.s", + "../testdata/verify/bigframe_riscv64.s", + "../testdata/verify/guard_riscv64.s", } { t.Run(path, func(t *testing.T) { testGroundTruthRISCVFile(t, path)