feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent ca3fdce0e0
commit de5d9f358e
12 changed files with 2059 additions and 37 deletions
+223
View File
@@ -5,6 +5,8 @@ package asm
import (
"bytes"
"encoding/binary"
"encoding/hex"
"strings"
"testing"
@@ -971,3 +973,224 @@ TEXT ·edge(SB), NOSPLIT, $0
t.Errorf("int32-span immediates must assemble: %v", err)
}
}
// riscvWants decodes code as little-endian words and pins each one; the
// expected values below were read off GOARCH=riscv64 go tool objdump of
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWants(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) < len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// riscvWantsHex pins the exact hex encoding of a function's instruction
// bytes, including any 2-byte compressed instructions in the stream; the
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
// assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
t.Helper()
got := hex.EncodeToString(code)
if got != wantHex {
t.Errorf("code = %s, want %s", got, wantHex)
}
}
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
// the time CSR).
func TestRISCV_extendedPseudos(t *testing.T) {
t.Run("logic and minmax", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·l(SB), NOSPLIT, $0
ANDN X19, X20, X21
ANDN X19, X20
ORN X20, X19
MAX X26, X28, X29
MIN X29, X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NOT X5
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
// then SEQZ, NEG, NOT and RDTIME.
riscvWantsHex(t, code,
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
"9682"+"1683"+
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
})
t.Run("rotate", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·r(SB), NOSPLIT, $0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
`)
code := assembleRISCVHelper(t, fn)
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
riscvWantsHex(t, code,
"b30fa040b39ff50133d6a50033e6cf00"+
"b30fa040b39ff501b3d5a500b3e5bf00"+
"93dff5038605b3e5bf00"+
"9bdff6011b97160033e7ef00"+
"9b5f17009b17f701b3e7ff00"+
"9b5f37001b17d70133e7ef00"+
"b30ff040bb1ff801bb58f800b3e81f01"+
"9bdff6019b961600b3e6df00"+"67800000")
})
t.Run("fp and branches", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWantsHex(t, code,
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
"f32210c0"+"67800000")
})
}
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
// exactly as GOARCH=riscv64 go tool asm encodes them.
func TestRISCV_amoWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0E5323AF, // amoswap.w
0x0E5333AF, // amoswap.d
0x065323AF, // amoaddd.w
0x065333AF, // amoadd.d
0x665323AF, // amoand.w
0x665333AF, // amoand.d
0x465323AF, // amoor.w
0x465333AF, // amoor.d
0x265323AF, // amoxor.w
0x265333AF, // amoxor.d
0xA65323AF, // amomax.w
0xA65333AF, // amomax.d
0xE65323AF, // amomaxu.w
0xE65333AF, // amomaxu.d
0xC65323AF, // amominu.w
0xC65333AF, // amominu.d
0x1402A32F, // lr.w (aq)
0x1402B32F, // lr.d
0x1A5323AF, // sc.w (rl)
0x1A5333AF, // sc.d
)
}
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
// toolchain canonicalises an immediate avl to vsetivli even under the
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
// identical), which is what the 0xC00 bit of the first word carries.
func TestRISCV_vectorWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETVLI X5, E8, M8, TA, MA, X6
VSETIVLI $4, E32, M1, TA, MA, X0
VSETVLI $15, E32, M1, TA, MA, X12
VADDVV V1, V2, V3
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
VFIRSTM V0, X6
VIDV V12
VMV4RV V8, V24
VLE8V (X10), V8
VSE8V V24, (X10)
VSE32V V9, (X11)
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
0xCD027057, // vsetivli x0, 4
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
0x022081D7, // vadd.vv v3, v2, v1
0x02C64657, // vadd.vx v12, v12, x12
0x2F040C57, // vxor.vv v24, v16, v8
0x62864057, // vmseq.vx v0, v8, x12
0x67040057, // vmsne.vv v0, v16, v8
0x97C43F57, // vsll.vi v30, v28, 8
0xA3DCBED7, // vsrl.vi v29, v29, 25
0x4208A357, // vmfirst.m x6, v0
0x5208A657, // vid.v v12
0x9E81BC57, // vmv4r.v v24, v8
0x02050407, // vle8.v v8, (x10)
0x02050C27, // vse8.v v24, (x10)
0x0205E4A7, // vse32.v v9, (x11)
0x6A076007, // vlsseg4e32.v v0, (x14), x0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}