feat(asm): encode the arm64 Q-width FP load and store

FMOVQ routes through the MOV load/store machinery in the plain, post-index,
pre-index and static-symbol forms.  The Q width carries its size in the opc
field, so the store spelling is opc=10 and the access scales by sixteen;
both come from helpers now instead of the size exponent.  The static-symbol
form takes the toolchain's twelve-byte ADRP + ADD + access fallback with the
R_ADDRARM64 pair.  The register-to-register and immediate forms stay
rejected, matching the toolchain's own table.

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent a6bd9c1ebe
commit daf7fad5b9
5 files changed
+161 -9

No files matched your search

+59 -8
View File
@@ -228,11 +228,12 @@ func arm64PCAlignPad(pos int, instr *ast.Instr) int {
}
// isARM64MovMnemonic reports whether m is one of the MOV-family spellings the
// arm64 encoder treats as the MOV pseudo-instruction.
// arm64 encoder treats as the MOV pseudo-instruction. FMOVQ rides the same
// load/store machinery but carries no register-move or immediate form.
func isARM64MovMnemonic(m string) bool {
switch m {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
"FMOVS", "FMOVD", "FMOVQ":
return true
}
return false
@@ -279,7 +280,7 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
}
switch mnem {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
"FMOVS", "FMOVD", "FMOVQ":
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
"ADDS", "ADDSW", "SUBS", "SUBSW":
@@ -354,7 +355,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
case "BL", "CALL":
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
"FMOVS", "FMOVD", "FMOVQ":
return encodeARM64Mov(instr, mnem, "", fi, relocs)
}
@@ -1479,6 +1480,20 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
}
src, dst := ops[0], ops[1]
if mnem == "FMOVQ" {
// The 128-bit FP move is a pure memory access in the toolchain: its
// table carries no register-to-register or immediate row, and both
// spellings are rejected ("illegal combination"), so only the
// load/store and static-symbol forms exist here either.
loadOK := arm64IsMemOperand(src) && !arm64IsMemOperand(dst) && !isImmOperand(src)
storeOK := arm64IsMemOperand(dst) && !arm64IsMemOperand(src) && !isImmOperand(src)
sbLoad := src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB"
sbStore := dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB"
if !loadOK && !storeOK && !sbLoad && !sbStore {
return nil, fmt.Errorf("%s: only the memory load and store forms exist", mnem)
}
}
if wb != "" {
switch {
case isMemOperand(src) && !isMemOperand(dst):
@@ -1609,8 +1624,14 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
}
return len(b)
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
if mnem == "FMOVQ" {
return 12 // ADRP + ADD + LDR: the unaligned-access fallback
}
return 8 // ADRP + LDR
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
if mnem == "FMOVQ" {
return 12 // ADRP + ADD + STR: the unaligned-access fallback
}
return 8 // ADRP + STR
case isMemOperand(src) || isMemOperand(dst):
mem := src
@@ -1623,7 +1644,7 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if !ok {
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
}
scale := int64(1) << uint(lt.size)
scale := a64LSScale(lt)
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return 4
}
@@ -1876,7 +1897,7 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
lt = a64LoadTable["MOVD"]
}
scale := int64(1) << uint(lt.size)
scale := a64LSScale(lt)
storeOpc := a64StoreOpc(lt)
var opc int
if load {
@@ -1963,12 +1984,27 @@ func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
// encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations,
// matching the toolchain: the scratch register is REGTMP (R27) and the pair
// carries R_ARM64_PCREL_LDST64.
// carries R_ARM64_PCREL_LDST64. FMOVQ has no LDST relocation width, so it
// takes the toolchain's unaligned-access fallback instead (asm7.go case 65):
// ADRP R27, 0; ADD R27, R27, 0; LDR with the R_ADDRARM64 pair.
func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"]
}
if mnem == "FMOVQ" {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, (R27)
), nil
}
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
@@ -1981,13 +2017,28 @@ func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([
}
// encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations,
// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair.
// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair. FMOVQ
// takes the twelve-byte ADRP + ADD + STR fallback with the R_ADDRARM64 pair,
// the store-side twin of the load above (asm7.go case 64).
func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"]
}
storeOpc := a64StoreOpc(lt)
if mnem == "FMOVQ" {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 27), // ADRP R27, 0
a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, (R27)
), nil
}
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
+16 -1
View File
@@ -1518,11 +1518,26 @@ var a64LoadTable = map[string]a64LSType{
// a64StoreOpc returns the store opc for a given load type: integer and FP
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
// the store form clears; FP registers are selected by V, not opc).
// the store form clears; FP registers are selected by V, not opc). The one
// exception is the 128-bit Q width, whose store is the opc=10 spelling: with
// opc=00 the size field selects STR B instead (STR Qt is size=00, opc=10).
func a64StoreOpc(t a64LSType) int {
if t.size == 0 && t.V == 1 {
return 2
}
return 0
}
// a64LSScale returns the byte width an access's unsigned offset divides by.
// The Q width carries its size in opc (the size field stays 0), yet scales
// by 16 like any other 128-bit access, so the exponent alone does not answer.
func a64LSScale(t a64LSType) int64 {
if t.size == 0 && t.V == 1 {
return 16
}
return int64(1) << uint(t.size)
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
+36
View File
@@ -679,6 +679,42 @@ func TestArm64AcquireRelease(t *testing.T) {
}
}
// TestArm64QMove pins the Q-width FP move against the toolchain words the
// corpus records: the plain, writeback and static-symbol load/store forms.
// FMOVQ carries no register-to-register or immediate form, and both are
// rejected the way the toolchain rejects them.
func TestArm64QMove(t *testing.T) {
got := arm64Words(t, "\tFMOVQ.P F13, 11(R10)\n\tFMOVQ.W F15, 11(R20)\n\tFMOVQ.P 11(R10), F13\n\tFMOVQ.W 11(R20), F15\n"+
"\tFMOVQ F0, 32(R5)\n\tFMOVQ F10, 65520(R10)\n\tFMOVQ 32(R5), F2\n")
want := []uint32{
0x3c80b54d, // FMOVQ.P F13, 11(R10)
0x3c80be8f, // FMOVQ.W F15, 11(R20)
0x3cc0b54d, // FMOVQ.P 11(R10), F13
0x3cc0be8f, // FMOVQ.W 11(R20), F15
0x3d8008a0, // FMOVQ F0, 32(R5)
0x3dbffd4a, // FMOVQ F10, 65520(R10)
0x3dc008a2, // FMOVQ 32(R5), F2
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, src := range []string{"\tFMOVQ F1, F2\n", "\tFMOVQ $1, R2\n", "\tFMOVQ $0, 8(R3)\n"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %q: %v", src, errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%q: assembled, want rejection", strings.TrimSpace(src))
}
}
}
// TestArm64BTI pins the landing-pad family against the toolchain words:
// only the uppercase C/J/JC spellings assemble, and bare BTI is a
// diagnostic, never a panic.
+1
View File
@@ -142,6 +142,7 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "qmov_arm64.s"), "arm64", true},
} {
t.Run(filepath.Base(k.path), func(t *testing.T) {
src, err := os.ReadFile(k.path)
+49
View File
@@ -0,0 +1,49 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 Q-width FP move: the FMOVQ load and
// store in the plain, post-index and pre-index forms, against scaled and
// unscaled offsets, and against static symbols. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func writeback()
TEXT ·writeback(SB), NOSPLIT, $0-0
FMOVQ.P F13, 11(R10)
FMOVQ.W F15, 11(R20)
FMOVS.P F20, 4(R0)
FMOVS.W F20, 4(R0)
FMOVD.P F20, 8(R1)
FMOVD.W F20, 8(R1)
FMOVQ.P 11(R10), F13
FMOVQ.W 11(R20), F15
FMOVS.P 8(R0), F20
FMOVS.W 8(R0), F20
FMOVD.W 8(R1), F20
RET
// func plain()
TEXT ·plain(SB), NOSPLIT, $0-0
FMOVQ F0, 32(R5)
FMOVQ F10, 65520(R10)
FMOVQ F11, 64(RSP)
FMOVQ F11, 8(R20)
FMOVQ F11, 4(R20)
FMOVQ 32(R5), F2
FMOVQ 65520(R10), F10
FMOVQ 64(RSP), F11
FMOVD F1, 8(R2)
FMOVD 8(R2), F1
RET
// func symbols()
TEXT ·symbols(SB), NOSPLIT, $0-0
FMOVQ F5, x(SB)
FMOVQ F5, x+8(SB)
FMOVQ x+8(SB), F5
FMOVQ x(SB), F5
RET
// data the symbol forms refer to
GLOBL x(SB), NOPTR, $16