feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 81e2673923
commit b0f9071bf5
6 changed files with 318 additions and 30 deletions
+130 -30
View File
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
return 4
}
}
// The funcdata pseudo-statements contribute no bytes.
// The funcdata pseudo-statements contribute no bytes, the expanded
// FUNCDATA/PCDATA forms included.
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
return 0
}
switch mnem {
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
// code: the toolchain emits zero instruction bytes for them, and so does
// the encoder here.
// the encoder here. Files that include funcdata.h spell them after
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
// bookkeeping too (the same treatment the loong64 encoder applies).
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
return nil, nil
case "END":
if len(ops) != 0 {
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
}
return nil, nil
case "FUNCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $n")
}
return nil, nil
}
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
// over the plain arrangement paths, which carry no index.
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
if mnem == "VDUP" || mnem == "VMOV" {
if arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
}
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
// vector (asm7.go case 82, shared by both mnemonics). The element
// paths above only run when a lane index is spelled, so this is the
// whole-vector shape's only route.
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
return b, err
}
}
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
}
// Integer → integer: ORR Rd, ZR, Rs.
// Integer → integer. Every truncating register move lowers to an
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
// ORR Xd, XZR, Xm.
if rs != 31 {
switch mnem {
case "MOVB":
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVH":
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVW":
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVBU":
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVHU":
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
}
}
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
mnem == "MOVH" || mnem == "MOVHU" {
if mnem == "MOVWU" {
sf = 0
}
// A narrow move out of the zero register loses its width: the
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
// WZR. MOVD and MOV keep the 64-bit form.
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
sf = 0
}
op := uint32(1<<29 | 0x0a<<24) // ORR
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
// for bare V registers alone: the arranged forms and every other
// three-register mnemonic are rejected outright.
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
vm, ok1 := arm64VecOf(ops[0])
vn, ok2 := arm64VecOf(ops[1])
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
}
base := uint32(0x5ee08400) // VADD
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
// general register into an arranged vector, the spelling asm7.go's case 82
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
// shape, so the caller falls through to the arrangement and element paths;
// the toolchain rejects the bare spellings outright, and the reverse
// Vd.<T>, Rs with them.
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
if len(ops) != 2 {
return nil, false, nil
}
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
return nil, false, nil
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, false, nil
}
dst, ok := arm64VecOf(ops[1])
if !ok || dst.hasIdx || dst.arr == "" {
return nil, false, nil
}
b, err := a64GPVecWhole(mnem, rs, dst)
return b, true, err
}
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
// lane width and Q the vector length. Both VMOV and VDUP take this form
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D2":
imm5, q = 8, 1<<30
default:
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
// spelling for this form, so the encoder refuses it too.
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
// lane indices:
//
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: source must be a general register", mnem)
}
if !dst.hasIdx {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D1":
imm5, q = 8, 0
case "D2":
imm5, q = 8, 1<<30
default:
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
// Duplicates the register across every lane (DUP Vd.T, Rn).
return a64GPVecWhole(mnem, rs, dst)
}
f, ok := a64ElemField(dst.arr, dst.idx)
if !ok {