feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+130
-30
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
return 4
|
||||
}
|
||||
}
|
||||
// The funcdata pseudo-statements contribute no bytes.
|
||||
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||
// FUNCDATA/PCDATA forms included.
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
|
||||
return 0
|
||||
}
|
||||
switch mnem {
|
||||
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
||||
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
||||
// code: the toolchain emits zero instruction bytes for them, and so does
|
||||
// the encoder here.
|
||||
// the encoder here. Files that include funcdata.h spell them after
|
||||
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
|
||||
// bookkeeping too (the same treatment the loong64 encoder applies).
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
return nil, nil
|
||||
case "END":
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
|
||||
}
|
||||
return nil, nil
|
||||
case "FUNCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
|
||||
}
|
||||
return nil, nil
|
||||
case "PCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
||||
return nil, fmt.Errorf("PCDATA expects $n, $n")
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
||||
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
||||
// over the plain arrangement paths, which carry no index.
|
||||
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
if mnem == "VDUP" || mnem == "VMOV" {
|
||||
if arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
}
|
||||
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
|
||||
// vector (asm7.go case 82, shared by both mnemonics). The element
|
||||
// paths above only run when a lane index is spelled, so this is the
|
||||
// whole-vector shape's only route.
|
||||
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
|
||||
return b, err
|
||||
}
|
||||
}
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// Integer → integer: ORR Rd, ZR, Rs.
|
||||
// Integer → integer. Every truncating register move lowers to an
|
||||
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
|
||||
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
|
||||
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
|
||||
// ORR Xd, XZR, Xm.
|
||||
if rs != 31 {
|
||||
switch mnem {
|
||||
case "MOVB":
|
||||
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVH":
|
||||
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVW":
|
||||
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVBU":
|
||||
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVHU":
|
||||
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
}
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
|
||||
mnem == "MOVH" || mnem == "MOVHU" {
|
||||
if mnem == "MOVWU" {
|
||||
sf = 0
|
||||
}
|
||||
// A narrow move out of the zero register loses its width: the
|
||||
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
|
||||
// WZR. MOVD and MOV keep the 64-bit form.
|
||||
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
|
||||
sf = 0
|
||||
}
|
||||
op := uint32(1<<29 | 0x0a<<24) // ORR
|
||||
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
|
||||
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
|
||||
// for bare V registers alone: the arranged forms and every other
|
||||
// three-register mnemonic are rejected outright.
|
||||
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
|
||||
vm, ok1 := arm64VecOf(ops[0])
|
||||
vn, ok2 := arm64VecOf(ops[1])
|
||||
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
|
||||
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
|
||||
}
|
||||
base := uint32(0x5ee08400) // VADD
|
||||
if mnem == "VSUB" {
|
||||
base = 0x7ee08400
|
||||
}
|
||||
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
|
||||
// general register into an arranged vector, the spelling asm7.go's case 82
|
||||
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
|
||||
// shape, so the caller falls through to the arrangement and element paths;
|
||||
// the toolchain rejects the bare spellings outright, and the reverse
|
||||
// Vd.<T>, Rs with them.
|
||||
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, false, nil
|
||||
}
|
||||
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
|
||||
return nil, false, nil
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, false, nil
|
||||
}
|
||||
dst, ok := arm64VecOf(ops[1])
|
||||
if !ok || dst.hasIdx || dst.arr == "" {
|
||||
return nil, false, nil
|
||||
}
|
||||
b, err := a64GPVecWhole(mnem, rs, dst)
|
||||
return b, true, err
|
||||
}
|
||||
|
||||
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
|
||||
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
|
||||
// lane width and Q the vector length. Both VMOV and VDUP take this form
|
||||
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
|
||||
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
|
||||
// spelling for this form, so the encoder refuses it too.
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
// lane indices:
|
||||
//
|
||||
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
||||
}
|
||||
if !dst.hasIdx {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D1":
|
||||
imm5, q = 8, 0
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
// Duplicates the register across every lane (DUP Vd.T, Rn).
|
||||
return a64GPVecWhole(mnem, rs, dst)
|
||||
}
|
||||
f, ok := a64ElemField(dst.arr, dst.idx)
|
||||
if !ok {
|
||||
|
||||
Reference in New Issue
Block a user