From ebdf14939fc8c7f8cb62633ae859f18ff81a0601 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sat, 19 Sep 2026 23:49:13 +0200 Subject: [PATCH] fix(loong64): FP immediates through R30 and unsigned branch forms Assisted-by: GLM 5.3 --- arch/arch.go | 3 ++ arch/loong64.go | 4 +- asm/loong64_assemble.go | 61 ++++++++++++++++------- asm/loong64_encode.go | 4 +- asm/loong64_encode_test.go | 8 +-- asm/loong64_more_test.go | 81 +++++++++++++++++++++++++++++-- testdata/verify/branchu_loong64.s | 18 +++++++ testdata/verify/movwfp_loong64.s | 17 +++++++ 8 files changed, 168 insertions(+), 28 deletions(-) create mode 100644 testdata/verify/branchu_loong64.s create mode 100644 testdata/verify/movwfp_loong64.s diff --git a/arch/arch.go b/arch/arch.go index b24cd6c..3af6a91 100644 --- a/arch/arch.go +++ b/arch/arch.go @@ -55,6 +55,7 @@ const ( Mask // AVX-512 mask register (K) Float // arm64 floating-point register (F) VecARM // arm64 SIMD/vector register (V) + VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX) Special // architecture-special register ) @@ -73,6 +74,8 @@ func (c RegClass) String() string { return "float" case VecARM: return "vector (arm64)" + case VecSIMD: + return "vector" case Special: return "special" default: diff --git a/arch/loong64.go b/arch/loong64.go index 97e3187..e1fd67b 100644 --- a/arch/loong64.go +++ b/arch/loong64.go @@ -31,10 +31,10 @@ func loong64Registers() []Register { add(fmt.Sprintf("F%d", i), Float, "floating-point register") } for i := 0; i <= 31; i++ { - add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register") + add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register") } for i := 0; i <= 31; i++ { - add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register") + add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register") } return regs } diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index aac8468..0617d19 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -442,6 +442,15 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } + // The toolchain validates the bit numbers ("illegal bit number"): + // 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb. + b := 64 + if strings.HasSuffix(mnem, "W") { + b = 32 + } + if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb { + return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb) + } return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil case l64Firrr: @@ -618,6 +627,15 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o if rj < 0 { return nil, fmt.Errorf("invalid register operand") } + if mnem == "BLTU" || mnem == "BGEU" { + // The unsigned compares have no single-register pseudo: the + // toolchain keeps the register-register form with rd = R0 + // (bltu rj, r0 is never taken), not a sometimes-taken beqz. + if (v<<16)>>16 != v { + return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) + } + return l64wordLE(l64irr16(op, v, rj, 0)), nil + } if (v<<11)>>11 != v { return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) } @@ -857,9 +875,16 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } - // MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}. - if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP { - return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil + // MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab + // accepts (AMOVW's C_12CON against C_FREG): it materialises the + // constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/ + // MOVD are illegal combinations there, and are diagnosed here rather + // than silently written into the GPR of the register's number. + if loong64RegClass(operandRegName(dst)) == l64ClsFP { + if mnem != "MOVW" { + return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem) + } + return encodeLOONG64ImmToFp(rd, l64Imm64(src)) } return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil } @@ -940,8 +965,8 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // pcalau12i + addi.d } - if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP { - return 8 // addi/ori r30 + movgr2fr + if loong64RegClass(operandRegName(dst)) == l64ClsFP { + return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid) } v := l64Imm64(src) if v == 0 { @@ -982,22 +1007,24 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int { } } -// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to -// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d). -func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte { - // ori for positive constants, addi.d for zero/negative. - op := uint32(0x00b << 22) - if v > 0 { - op = 0x00e << 22 +// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it +// to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which +// zero-extends) for the positive span, addi.w for zero and the negative +// span, then movgr2fr.w. The toolchain's optab accepts no wider constant on +// this path (it never materialises one fully first), so values outside +// [-2048, 4095] are diagnosed rather than masked into si12. +func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) { + if v < -2048 || v > 4095 { + return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v) } - mov := uint32(0x452a << 10) // movgr2fr.d - if mnem == "MOVF" { - mov = 0x4529 << 10 // movgr2fr.w + op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends) + if v > 0 { + op = 0x00e << 22 // ori r30, r0, v (zero-extends) } return l64WordsLE( l64irr(op, int(v), 0, 30), - l64rr(mov, 30, fd), - ) + l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30 + ), nil } // ---- 64-bit immediate classification ---- diff --git a/asm/loong64_encode.go b/asm/loong64_encode.go index e57dd15..28f10b8 100644 --- a/asm/loong64_encode.go +++ b/asm/loong64_encode.go @@ -199,7 +199,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 { } // l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. -// The msb/lsb fields are 6 bits wide (0-63) and are validated by the caller. +// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller +// must have validated them (0..31 for the .w forms, 0..63 for the .d forms, +// lsb <= msb), the same rule the toolchain enforces as "illegal bit number". func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) } diff --git a/asm/loong64_encode_test.go b/asm/loong64_encode_test.go index 8a20117..2cf8bfb 100644 --- a/asm/loong64_encode_test.go +++ b/asm/loong64_encode_test.go @@ -319,11 +319,11 @@ TEXT ·f(SB), NOSPLIT, $0-0 `) code = assembleLOONG64Helper(t, fn) wantWords(t, code, - 0x29FFE061, // addi.d r1, r2, -8 (prologue) - 0x02FFE063, // addi.d r3, r3, -8 - 0x29C00061, // st.d r1, r2, 0 (prologue saves RA) + 0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP) + 0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame) + 0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP) 0x4C0000A1, // jirl r1, r5, 0 - 0x28C00061, // ld.d r1, r2, 0 (epilogue restores RA) + 0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA) 0x02C02063, // addi.d r3, r3, 8 0x4C000020, // jirl r0, r1, 0 (RET) ) diff --git a/asm/loong64_more_test.go b/asm/loong64_more_test.go index 09913a7..88b9a83 100644 --- a/asm/loong64_more_test.go +++ b/asm/loong64_more_test.go @@ -347,21 +347,94 @@ TEXT ·sb(SB), NOSPLIT, $0 } } -// TestLOONG64_movImmToFp checks the immediate-to-FP move forms. +// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the +// only spelling the toolchain accepts, expanding to ori (or addi.w for the +// negative span) into R30 plus movgr2fr.w. The pinned words are the +// toolchain's own bytes; the other widths and out-of-range constants are +// illegal combinations there and are diagnosed here. func TestLOONG64_movImmToFp(t *testing.T) { fn := firstTextLOONG64(t, `#include "textflag.h" TEXT ·fpmov(SB), NOSPLIT, $0 - MOVV $0x1, F0 + MOVW $0x1, F0 MOVW $0x2, F4 + MOVW $-1, F4 RET `) code := assembleLOONG64Helper(t, fn) want := []byte{ - 0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1 - 0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2 + 0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1 + 0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30 + 0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2 + 0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30 + 0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1 + 0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30 0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0 } if !bytes.Equal(code, want) { t.Errorf("code = % x\nwant % x", code, want) } } + +// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the +// widths the toolchain rejects as illegal combinations, and constants beyond +// the 12-bit ori/addi.w span (the toolchain never materialises a wider +// constant on this path). +func TestLOONG64_movImmToFpErrors(t *testing.T) { + cases := []string{ + "MOVV $1, F0", + "MOVF $2, F4", + "MOVD $2, F4", + "MOVW $100000, F1", + "MOVW $-2049, F1", + "MOVW $4096, F1", + } + for _, src := range cases { + fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n") + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("%s: expected an error, got none", src) + } + } +} + +// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with +// one register BLTU/BGEU keep the register-register form against R0 (never +// taken), the toolchain's encoding, where a beqz would test the wrong +// condition; the three-operand forms are unchanged. +func TestLOONG64_branch16Unsigned(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·u(SB), NOSPLIT, $0 + BLTU R4, done + BGEU R5, done + BLTU R6, R7, done + BGEU R8, R9, done +done: + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x68001080, // bltu r4, r0, +4 + 0x6C000CA0, // bgeu r5, r0, +3 + 0x680008C7, // bltu r6, r7, +2 + 0x6C000509, // bgeu r8, r9, +1 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number +// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for +// the .w forms, 0..63 for the .d forms, and lsb <= msb. +func TestLOONG64_bitFieldRange(t *testing.T) { + cases := []string{ + "BSTRINSW $32, R4, $0, R5", + "BSTRPICKW $31, R4, $32, R5", + "BSTRINSV $64, R4, $0, R5", + "BSTRPICKV $3, R4, $4, R5", + "BSTRINSW $-1, R4, $0, R5", + } + for _, src := range cases { + fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n") + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("%s: expected an error, got none", src) + } + } +} diff --git a/testdata/verify/branchu_loong64.s b/testdata/verify/branchu_loong64.s new file mode 100644 index 0000000..4a6f995 --- /dev/null +++ b/testdata/verify/branchu_loong64.s @@ -0,0 +1,18 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// branchu exercises the unsigned 16-bit branches in their two- and +// three-operand spellings: with a single register the toolchain keeps the +// register-register form against R0 (never taken) rather than a zero-form +// pseudo. Byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·unsigned(SB), NOSPLIT, $0 + BLTU R4, done + BGEU R5, done + BLTU R6, R7, done + BGEU R8, R9, done + +done: + RET diff --git a/testdata/verify/movwfp_loong64.s b/testdata/verify/movwfp_loong64.s new file mode 100644 index 0000000..ab0433c --- /dev/null +++ b/testdata/verify/movwfp_loong64.s @@ -0,0 +1,17 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// movwfp exercises MOVW $c, Fd, the immediate-to-F-register form the +// toolchain's optab accepts: the constant is materialised in R30 (ori for +// the positive span, addi.w for the negative span) and moved across with +// movgr2fr.w. Byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·movwfp(SB), NOSPLIT, $0 + MOVW $1, F0 + MOVW $0x2, F4 + MOVW $0x7ff, F8 + MOVW $-1, F12 + MOVW $-2048, F16 + RET