From 1040739fbb24ed75dbc12d38e8a24eb0e1a66214 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Tue, 6 Oct 2026 20:31:18 +0200 Subject: [PATCH] fix(asm): validate the loong64 ll/sc offset span like the toolchain Assisted-by: GLM 5.3 Flash --- asm/loong64_assemble.go | 15 ++++++- asm/loong64_encode.go | 32 ++++++++++++++ asm/loong64_more_test.go | 92 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 138 insertions(+), 1 deletion(-) diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index 95c17aa..e16070e 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -387,6 +387,16 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int { return 8 // bne/beq over the BREAK, then BREAK case "PRELDX": return 20 // the four-instruction constant materialisation + preldx + case "LL", "LLW", "LLV", "SC", "SCW", "SCV", "MOVWP", "MOVVP": + // The 2RI14 families: one word inside the signed 16-bit offset + // span, otherwise the toolchain's materialisation sequences (see + // l64firr14Words, which decides the width). A memory operand that + // will not parse leaves the single-word size: the encode pass + // reports the error. + if _, _, off, _, err := l64MemOperands(ops, fi); err == nil { + return len(l64firr14Words(0, int(off), 0, 0)) + } + return 4 case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": return loong64MovSize(mnem, ops, fi) case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU": @@ -797,7 +807,10 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo // ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field. op -= 1 << 24 } - return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil + if off&3 != 0 { + return nil, fmt.Errorf("%s: offset must be a multiple of 4", mnem) + } + return l64firr14Words(op, int(off), rj, rd), nil case l64Fir20: // LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm. diff --git a/asm/loong64_encode.go b/asm/loong64_encode.go index 2cd1691..73cd035 100644 --- a/asm/loong64_encode.go +++ b/asm/loong64_encode.go @@ -257,6 +257,38 @@ func l64WordsLE(ws ...uint32) []byte { return out } +// l64firr14Words returns the words the 2RI14 families (LL/LLW/LLV, SC/SCW/ +// SCV, MOVWP/MOVVP) encode to for a 4-aligned byte offset. The toolchain +// classifies the memory operand by span, with the class constants' -2 slop +// included verbatim: a signed 16-bit offset (C_SOREG_16) rides in si14 alone; +// a 32-bit one (C_LOREG_32) splits between addu16i.d (bits 31:16, materialised +// in R30, the assembler temp) and si14 (bits 15:2); anything wider +// (C_LOREG_64) builds the whole constant in R30 through lu12i.w + ori + +// lu32i.d + lu52i.d and folds the base into it. The loong64 backend resolves +// memory offsets as int32, so the third span answers nothing the encoder can +// ask today: it stands so the mapping stays complete if that domain widens. +func l64firr14Words(op uint32, off, rj, rd int) []byte { + switch { + case off >= -32766 && off < 32766: + return l64wordLE(l64irr14(op, off>>2, rj, rd)) + case off >= -2147483650 && off < 2147483646: + return l64WordsLE( + l64irr16(l64InstrTable["ADDV16"].op, off>>16, 0, 30), + l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30), + l64irr14(op, off>>2, 30, rd), + ) + default: + return l64WordsLE( + l64ir(l64InstrTable["LU12IW"].op, off>>12, 30), + l64irr(l64DualTable["OR"].imm, off&0xFFF, 30, 30), + l64ir(l64InstrTable["LU32ID"].op, off>>32, 30), + l64irr(l64InstrTable["LU52ID"].op, off>>52, 30, 30), + l64rrr(l64DualTable["ADDV"].rrr, 30, rj, rj), + l64irr14(op, 0, rj, rd), + ) + } +} + // ---- instruction formats ---- type l64Format uint8 diff --git a/asm/loong64_more_test.go b/asm/loong64_more_test.go index d517a6b..b7f5363 100644 --- a/asm/loong64_more_test.go +++ b/asm/loong64_more_test.go @@ -131,6 +131,98 @@ TEXT ·ptr(SB), NOSPLIT, $0 ) } +// TestLOONG64_firr14Spans pins the 2RI14 offset spans of the LL/SC/MOVWP +// families against the toolchain's operand classes: an unaligned offset is +// rejected ("offset must be a multiple of 4"), a signed 16-bit offset rides +// in si14 alone, and anything wider materialises its high half in R30 with +// addu16i.d before the base folds in, with si14 keeping bits 15:2. The +// pinned words are GOARCH=loong64 go tool asm's own bytes for the same +// sources, boundaries included (-32766 and 32766 are the class edges). +func TestLOONG64_firr14Spans(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·spans(SB), NOSPLIT, $0 + SC R4, 32764(R5) + SC R4, -32764(R5) + SC R4, 32768(R5) + SC R4, -32768(R5) + SC R4, -32772(R5) + LL 65540(R5), R4 + LLV 65540(R5), R4 + SCV R4, 65540(R5) + MOVWP R4, 65540(R5) + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x217FFCA4, // sc.w r4, 8191(r5) — inside si14 + 0x218004A4, // sc.w r4, -8191(r5) + 0x1000001E, // addu16i.d r30, r0, 0 + 0x001097DE, // add.d r30, r30, r5 + 0x218003C4, // sc.w r4, 8192(r30) + 0x13FFFC1E, // addu16i.d r30, r0, -1 + 0x001097DE, // add.d r30, r30, r5 + 0x218003C4, // sc.w r4, 8192(r30) + 0x13FFFC1E, // addu16i.d r30, r0, -1 + 0x001097DE, // add.d r30, r30, r5 + 0x217FFFC4, // sc.w r4, 8191(r30) + 0x1000041E, // addu16i.d r30, r0, 1 + 0x001097DE, // add.d r30, r30, r5 + 0x200007C4, // ll.w r4, 1(r30) + 0x1000041E, // addu16i.d r30, r0, 1 + 0x001097DE, // add.d r30, r30, r5 + 0x220007C4, // ldptr.d r4, 1(r30) + 0x1000041E, // addu16i.d r30, r0, 1 + 0x001097DE, // add.d r30, r30, r5 + 0x230007C4, // sc.d r4, 1(r30) + 0x1000041E, // addu16i.d r30, r0, 1 + 0x001097DE, // add.d r30, r30, r5 + 0x250007C4, // stptr.w r4, 1(r30) + 0x4C000020, + ) + + // The expansion participates in layout: the function size carries the + // six three-word forms plus the single-word pair and the RET. + f, errs := parser.Parse("spans_loong64.s", `#include "textflag.h" +TEXT ·spans(SB), NOSPLIT, $0 + SC R4, 32768(R5) + LL 65540(R5), R4 + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + if img.Funcs[0].Size != 8*3+4 { + t.Errorf("function size = %d, want %d", img.Funcs[0].Size, 8*3+4) + } +} + +// TestLOONG64_firr14Alignment checks the LL/SC/MOVWP alignment rule the +// toolchain enforces as "offset must be a multiple of 4": the LoongArch +// ll/sc family requires a naturally scaled offset, and si14 cannot express +// the remainder. GOARCH=loong64 go tool asm rejects every case here. +func TestLOONG64_firr14Alignment(t *testing.T) { + cases := []string{ + "SC R4, 1(R5)", + "SCW R4, -1(R5)", + "SCV R4, 2(R5)", + "LL 1(R5), R4", + "LLW 3(R5), R4", + "LLV 6(R5), R4", + "MOVWP R4, 1(R5)", + "MOVVP R4, 2(R5)", + } + for _, src := range cases { + fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n") + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("%s: expected an error, got none", src) + } + } +} + // TestLOONG64_atomics exercises the AM* read-modify-write forms and // RDTIME, plus the MOVV FP→GP move. func TestLOONG64_atomics(t *testing.T) {