feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
This commit is contained in:
+257
-6
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
|
||||
if mnem == "RET" {
|
||||
return len(loong64Return(fi))
|
||||
}
|
||||
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
|
||||
// toolchain spells only on x86 but accepts here the same way the arm64
|
||||
// and riscv64 encoders do (a superset spelling, shippable via the goobj
|
||||
// path).
|
||||
if mnem == "BYTE" {
|
||||
return len(ops)
|
||||
}
|
||||
switch mnem {
|
||||
case "END", "FUNCDATA", "PCDATA":
|
||||
return 0 // bookkeeping statements contribute no bytes
|
||||
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
||||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
|
||||
case "BYTE":
|
||||
// BYTE $b lays down one raw byte per operand, the same front-end
|
||||
// pseudo-op the arm64 and riscv64 encoders accept.
|
||||
var out []byte
|
||||
for _, op := range ops {
|
||||
b := l64Imm64(op)
|
||||
if b < 0 || b > 0xFF {
|
||||
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
|
||||
}
|
||||
out = append(out, byte(b))
|
||||
}
|
||||
return out, nil
|
||||
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
|
||||
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
|
||||
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
|
||||
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
|
||||
}
|
||||
return l64wordLE(l64rr(enc.op, rj, rd)), nil
|
||||
|
||||
case l64Fllsc:
|
||||
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
|
||||
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
|
||||
// (the toolchain's C_ZOREG, which rejects any displacement).
|
||||
rd, rj, off, _, err := l64MemOperands(ops, fi)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", mnem, err)
|
||||
}
|
||||
if off != 0 {
|
||||
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
|
||||
}
|
||||
return l64wordLE(l64rr(enc.op, rj, rd)), nil
|
||||
|
||||
case l64Fscq:
|
||||
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
|
||||
// against a zero-offset memory operand as with the LL/SC pair.
|
||||
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
|
||||
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
|
||||
}
|
||||
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
|
||||
rj, off := l64MemWithFrame(ops[2], fi)
|
||||
if first < 0 || middle < 0 || rj < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid register operand", mnem)
|
||||
}
|
||||
if off != 0 {
|
||||
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
|
||||
}
|
||||
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
|
||||
|
||||
case l64Firr:
|
||||
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
|
||||
if len(ops) < 2 || !isImmOperand(ops[0]) {
|
||||
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
|
||||
return rd, rj, off, load, nil
|
||||
}
|
||||
|
||||
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
|
||||
// the shared immediate parse reduces it to the bare number and keeps only
|
||||
// the text as a witness of the base register. ok reports the form was
|
||||
// found, with the base's register number (or -1 when the name is not a
|
||||
// general register).
|
||||
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
|
||||
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
|
||||
return 0, 0, false
|
||||
}
|
||||
raw := strings.ReplaceAll(op.Raw, " ", "")
|
||||
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
|
||||
return 0, 0, false
|
||||
}
|
||||
open := strings.LastIndexByte(raw, '(')
|
||||
if open < 2 {
|
||||
return 0, 0, false
|
||||
}
|
||||
base = loong64RegNum(raw[open+1 : len(raw)-1])
|
||||
v := op.Imm.Val
|
||||
if op.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
return int32(v), base, base >= 0
|
||||
}
|
||||
|
||||
// ---- the MOV pseudo-instruction ----
|
||||
|
||||
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
|
||||
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
|
||||
}
|
||||
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
|
||||
}
|
||||
// MOVx $off(rj), rd computes an address: the toolchain's `mov
|
||||
// $soreg, r` case, a plain addi.d whatever the move's width (both
|
||||
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
|
||||
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
|
||||
// its case 10). The immediate's Raw carries the base register,
|
||||
// which the shared immediate parse reduces to the bare number.
|
||||
if off, base, ok := l64ImmMem(src); ok {
|
||||
rd := l64Reg(dst)
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
|
||||
}
|
||||
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||||
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
|
||||
}
|
||||
if off >= -2048 && off <= 2047 {
|
||||
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
|
||||
}
|
||||
return l64WordsLE(
|
||||
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
|
||||
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
|
||||
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
|
||||
), nil
|
||||
}
|
||||
rd := l64Reg(dst)
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
||||
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
|
||||
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
||||
return 8 // pcalau12i + addi.d
|
||||
}
|
||||
// The $off(rj) address immediate: addi.d in the 12-bit window,
|
||||
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
|
||||
if off, _, ok := l64ImmMem(src); ok {
|
||||
if off >= -2048 && off <= 2047 {
|
||||
return 4
|
||||
}
|
||||
return 12
|
||||
}
|
||||
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||||
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
|
||||
}
|
||||
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
|
||||
return l64Mem(op)
|
||||
}
|
||||
|
||||
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
|
||||
// vector table falls back to the zero register as the FP-relative base
|
||||
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
|
||||
// through R3), so the vector moves keep the resolved offset but the zero
|
||||
// base, exactly as `go tool asm` emits them.
|
||||
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
|
||||
rj, off = l64MemWithFrame(op, fi)
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
|
||||
rj = 0
|
||||
}
|
||||
return rj, off
|
||||
}
|
||||
|
||||
// l64MemOffset returns the resolved byte offset of a memory operand.
|
||||
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
|
||||
_, off := l64MemWithFrame(op, fi)
|
||||
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
|
||||
type l64VecOperand struct {
|
||||
num int // 5-bit register number
|
||||
lasx bool // X bank (LASX) rather than V (LSX)
|
||||
width byte // suffix width letter (B/H/W/V), 0 on a bare register
|
||||
width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
|
||||
lanes int // lane count of a width suffix (B16 → 16)
|
||||
elem int // element index of a .T[i] suffix
|
||||
hasEl bool // the suffix names an element (.T[i])
|
||||
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
|
||||
}
|
||||
i++
|
||||
w := name[i]
|
||||
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
|
||||
if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
|
||||
return v, false
|
||||
}
|
||||
v.width, v.hasSuf = w, true
|
||||
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
|
||||
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
|
||||
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
|
||||
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
|
||||
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
|
||||
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
|
||||
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
|
||||
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
|
||||
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
|
||||
enc := l64VmovqTable[lasx]
|
||||
bank := "V"
|
||||
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
|
||||
return loong64RegNum(name), nil
|
||||
}
|
||||
|
||||
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
|
||||
// source element width matching the destination arrangement. An LSX-only
|
||||
// form: the toolchain's table gives vreplvei no LASX counterpart.
|
||||
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
|
||||
if lasx || src.lasx || dst.lasx {
|
||||
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
|
||||
}
|
||||
if src.unsig {
|
||||
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
|
||||
}
|
||||
if src.width != dst.width {
|
||||
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
|
||||
}
|
||||
if _, ok := l64VecSuffixWidth(false, dst); !ok {
|
||||
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
var op uint32
|
||||
limit := 0
|
||||
switch src.width {
|
||||
case 'B':
|
||||
op, limit = enc.rveiB, 15
|
||||
case 'H':
|
||||
op, limit = enc.rveiH, 7
|
||||
case 'W':
|
||||
op, limit = enc.rveiW, 3
|
||||
default:
|
||||
op, limit = enc.rveiD, 1
|
||||
}
|
||||
if src.elem > limit {
|
||||
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
|
||||
}
|
||||
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
|
||||
}
|
||||
|
||||
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
|
||||
// a bare X source into an arranged X destination. LASX only.
|
||||
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
|
||||
if !lasx || src.lasx != lasx || dst.lasx != lasx {
|
||||
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
|
||||
}
|
||||
var op uint32
|
||||
switch dst.width {
|
||||
case 'B':
|
||||
if dst.lanes != 32 {
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
op = enc.rve0B
|
||||
case 'H':
|
||||
if dst.lanes != 16 {
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
op = enc.rve0H
|
||||
case 'W':
|
||||
if dst.lanes != 8 {
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
op = enc.rve0W
|
||||
case 'V':
|
||||
if dst.lanes != 4 {
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
op = enc.rve0D
|
||||
case 'Q':
|
||||
if dst.lanes != 2 {
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
op = enc.rve0Q
|
||||
default:
|
||||
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
|
||||
}
|
||||
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
|
||||
}
|
||||
|
||||
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
|
||||
// source into one word or double-word lane. LASX only.
|
||||
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
|
||||
if !lasx || src.lasx != lasx || dst.lasx != lasx {
|
||||
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
|
||||
}
|
||||
op, limit := enc.xinsW, 7
|
||||
if dst.width != 'W' {
|
||||
op, limit = enc.xinsD, 3
|
||||
if dst.width != 'V' {
|
||||
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
|
||||
}
|
||||
}
|
||||
if dst.elem > limit {
|
||||
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
|
||||
}
|
||||
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
|
||||
}
|
||||
|
||||
// Element extract into a vector register: XVMOVQ xj.T[i], xd
|
||||
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
|
||||
// register. LASX only.
|
||||
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
|
||||
if !lasx || src.lasx != lasx || dst.lasx != lasx {
|
||||
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
|
||||
}
|
||||
op, limit := enc.xpickW, 7
|
||||
if src.width != 'W' {
|
||||
op, limit = enc.xpickD, 3
|
||||
if src.width != 'V' {
|
||||
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
|
||||
}
|
||||
}
|
||||
if src.elem > limit {
|
||||
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
|
||||
}
|
||||
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
|
||||
}
|
||||
|
||||
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
|
||||
// both operands bare registers of the same bank.
|
||||
if srcVec && dstVec {
|
||||
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
|
||||
}
|
||||
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
|
||||
}
|
||||
rj, off := l64MemWithFrame(ops[1], fi)
|
||||
rj, off := l64VmovqMem(ops[1], fi)
|
||||
if rj < 0 || off < -2048 || off > 2047 {
|
||||
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
|
||||
}
|
||||
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
|
||||
}
|
||||
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
|
||||
}
|
||||
rj, off := l64MemWithFrame(ops[0], fi)
|
||||
if rj < 0 || off < -2048 || off > 2047 {
|
||||
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
|
||||
rj, off := l64VmovqMem(ops[0], fi)
|
||||
if rj < 0 {
|
||||
return nil, fmt.Errorf("VMOVQ: invalid load operand")
|
||||
}
|
||||
op := enc.ld
|
||||
if dst.hasSuf {
|
||||
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
|
||||
}
|
||||
// vldrepl keeps the byte offset raw for bytes and scales it by
|
||||
// the element width for the wider forms, the immediate field
|
||||
// shrinking a bit per scale exactly as the toolchain encodes it
|
||||
// (the field mask keeps the two's complement inside its width).
|
||||
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
|
||||
switch w {
|
||||
case 0:
|
||||
op = enc.replB
|
||||
case 1:
|
||||
op = enc.replH
|
||||
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
|
||||
case 2:
|
||||
op = enc.replW
|
||||
scale, mask, lo, hi = 4, 0x3FF, -512, 511
|
||||
default:
|
||||
op = enc.replD
|
||||
scale, mask, lo, hi = 8, 0x1FF, -256, 255
|
||||
}
|
||||
if off%int32(scale) != 0 {
|
||||
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
|
||||
}
|
||||
off /= int32(scale)
|
||||
if off < int32(lo) || off > int32(hi) {
|
||||
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
|
||||
}
|
||||
off &= mask
|
||||
} else if off < -2048 || off > 2047 {
|
||||
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
|
||||
}
|
||||
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user