feat(asm): encode the amd64 and loong64 tails of the corpus testdata

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:43 +02:00
parent 2f679326c2
commit bafb2fd130
12 changed files with 1404 additions and 81 deletions
+257 -6
View File
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
if mnem == "RET" {
return len(loong64Return(fi))
}
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem {
case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
return rd, rj, off, load, nil
}
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op)
}
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false
}
v.width, v.hasSuf = w, true
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil
}
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 {
return nil, fmt.Errorf("VMOVQ: invalid load operand")
}
op := enc.ld
if dst.hasSuf {
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2:
op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default:
op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
}
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}