feat(asm): encode the amd64 and loong64 tails of the corpus testdata

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:43 +02:00
parent 2f679326c2
commit bafb2fd130
12 changed files with 1404 additions and 81 deletions
+32 -5
View File
@@ -279,6 +279,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b).
var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +353,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
}
var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
},
}
@@ -373,7 +386,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +480,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
@@ -868,7 +895,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +904,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
"XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},