diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 34cade3..66c0a6b 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -228,11 +228,12 @@ func arm64PCAlignPad(pos int, instr *ast.Instr) int { } // isARM64MovMnemonic reports whether m is one of the MOV-family spellings the -// arm64 encoder treats as the MOV pseudo-instruction. +// arm64 encoder treats as the MOV pseudo-instruction. FMOVQ rides the same +// load/store machinery but carries no register-move or immediate form. func isARM64MovMnemonic(m string) bool { switch m { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", - "FMOVS", "FMOVD": + "FMOVS", "FMOVD", "FMOVQ": return true } return false @@ -279,7 +280,7 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { } switch mnem { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", - "FMOVS", "FMOVD": + "FMOVS", "FMOVD", "FMOVQ": return arm64MovSize(mnem, ops, fi) case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW", "ADDS", "ADDSW", "SUBS", "SUBSW": @@ -354,7 +355,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 case "BL", "CALL": return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve) case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", - "FMOVS", "FMOVD": + "FMOVS", "FMOVD", "FMOVQ": return encodeARM64Mov(instr, mnem, "", fi, relocs) } @@ -1479,6 +1480,20 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, } src, dst := ops[0], ops[1] + if mnem == "FMOVQ" { + // The 128-bit FP move is a pure memory access in the toolchain: its + // table carries no register-to-register or immediate row, and both + // spellings are rejected ("illegal combination"), so only the + // load/store and static-symbol forms exist here either. + loadOK := arm64IsMemOperand(src) && !arm64IsMemOperand(dst) && !isImmOperand(src) + storeOK := arm64IsMemOperand(dst) && !arm64IsMemOperand(src) && !isImmOperand(src) + sbLoad := src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" + sbStore := dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" + if !loadOK && !storeOK && !sbLoad && !sbStore { + return nil, fmt.Errorf("%s: only the memory load and store forms exist", mnem) + } + } + if wb != "" { switch { case isMemOperand(src) && !isMemOperand(dst): @@ -1609,8 +1624,14 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { } return len(b) case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": + if mnem == "FMOVQ" { + return 12 // ADRP + ADD + LDR: the unaligned-access fallback + } return 8 // ADRP + LDR case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": + if mnem == "FMOVQ" { + return 12 // ADRP + ADD + STR: the unaligned-access fallback + } return 8 // ADRP + STR case isMemOperand(src) || isMemOperand(dst): mem := src @@ -1623,7 +1644,7 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if !ok { lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access } - scale := int64(1) << uint(lt.size) + scale := a64LSScale(lt) if off >= 0 && off%scale == 0 && off/scale < 4096 { return 4 } @@ -1876,7 +1897,7 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 lt = a64LoadTable["MOVD"] } - scale := int64(1) << uint(lt.size) + scale := a64LSScale(lt) storeOpc := a64StoreOpc(lt) var opc int if load { @@ -1963,12 +1984,27 @@ func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { // encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations, // matching the toolchain: the scratch register is REGTMP (R27) and the pair -// carries R_ARM64_PCREL_LDST64. +// carries R_ARM64_PCREL_LDST64. FMOVQ has no LDST relocation width, so it +// takes the toolchain's unaligned-access fallback instead (asm7.go case 65): +// ADRP R27, 0; ADD R27, R27, 0; LDR with the R_ADDRARM64 pair. func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } + if mnem == "FMOVQ" { + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 27), // ADRP R27, 0 + a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27 + a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, (R27) + ), nil + } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset}, @@ -1981,13 +2017,28 @@ func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([ } // encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations, -// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair. +// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair. FMOVQ +// takes the twelve-byte ADRP + ADD + STR fallback with the R_ADDRARM64 pair, +// the store-side twin of the load above (asm7.go case 64). func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } storeOpc := a64StoreOpc(lt) + if mnem == "FMOVQ" { + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 27), // ADRP R27, 0 + a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27 + a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, (R27) + ), nil + } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset}, diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 70f24de..f664fcf 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -1518,11 +1518,26 @@ var a64LoadTable = map[string]a64LSType{ // a64StoreOpc returns the store opc for a given load type: integer and FP // stores both encode opc=00 (the load's signedness bit sits in opc[1], which -// the store form clears; FP registers are selected by V, not opc). +// the store form clears; FP registers are selected by V, not opc). The one +// exception is the 128-bit Q width, whose store is the opc=10 spelling: with +// opc=00 the size field selects STR B instead (STR Qt is size=00, opc=10). func a64StoreOpc(t a64LSType) int { + if t.size == 0 && t.V == 1 { + return 2 + } return 0 } +// a64LSScale returns the byte width an access's unsigned offset divides by. +// The Q width carries its size in opc (the size field stays 0), yet scales +// by 16 like any other 128-bit access, so the exponent alone does not answer. +func a64LSScale(t a64LSType) int64 { + if t.size == 0 && t.V == 1 { + return 16 + } + return int64(1) << uint(t.size) +} + // arm64RegClass discriminates integer (R), floating-point (F) registers for // the MOV pseudo-instruction. type arm64RegClass int diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 879bcf3..611b046 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -679,6 +679,42 @@ func TestArm64AcquireRelease(t *testing.T) { } } +// TestArm64QMove pins the Q-width FP move against the toolchain words the +// corpus records: the plain, writeback and static-symbol load/store forms. +// FMOVQ carries no register-to-register or immediate form, and both are +// rejected the way the toolchain rejects them. +func TestArm64QMove(t *testing.T) { + got := arm64Words(t, "\tFMOVQ.P F13, 11(R10)\n\tFMOVQ.W F15, 11(R20)\n\tFMOVQ.P 11(R10), F13\n\tFMOVQ.W 11(R20), F15\n"+ + "\tFMOVQ F0, 32(R5)\n\tFMOVQ F10, 65520(R10)\n\tFMOVQ 32(R5), F2\n") + want := []uint32{ + 0x3c80b54d, // FMOVQ.P F13, 11(R10) + 0x3c80be8f, // FMOVQ.W F15, 11(R20) + 0x3cc0b54d, // FMOVQ.P 11(R10), F13 + 0x3cc0be8f, // FMOVQ.W 11(R20), F15 + 0x3d8008a0, // FMOVQ F0, 32(R5) + 0x3dbffd4a, // FMOVQ F10, 65520(R10) + 0x3dc008a2, // FMOVQ 32(R5), F2 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } + for _, src := range []string{"\tFMOVQ F1, F2\n", "\tFMOVQ $1, R2\n", "\tFMOVQ $0, 8(R3)\n"} { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse %q: %v", src, errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%q: assembled, want rejection", strings.TrimSpace(src)) + } + } +} + // TestArm64BTI pins the landing-pad family against the toolchain words: // only the uppercase C/J/JC spellings assemble, and bare BTI is a // diagnostic, never a panic. diff --git a/asm/kernels_differential_test.go b/asm/kernels_differential_test.go index e7809c7..e07b836 100644 --- a/asm/kernels_differential_test.go +++ b/asm/kernels_differential_test.go @@ -142,6 +142,7 @@ func TestDifferentialKernels(t *testing.T) { {filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true}, + {filepath.Join("..", "testdata", "verify", "qmov_arm64.s"), "arm64", true}, } { t.Run(filepath.Base(k.path), func(t *testing.T) { src, err := os.ReadFile(k.path) diff --git a/testdata/verify/qmov_arm64.s b/testdata/verify/qmov_arm64.s new file mode 100644 index 0000000..8102fe9 --- /dev/null +++ b/testdata/verify/qmov_arm64.s @@ -0,0 +1,49 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 Q-width FP move: the FMOVQ load and +// store in the plain, post-index and pre-index forms, against scaled and +// unscaled offsets, and against static symbols. Every function is +// byte-compared against go tool asm. + +#include "textflag.h" + +// func writeback() +TEXT ·writeback(SB), NOSPLIT, $0-0 + FMOVQ.P F13, 11(R10) + FMOVQ.W F15, 11(R20) + FMOVS.P F20, 4(R0) + FMOVS.W F20, 4(R0) + FMOVD.P F20, 8(R1) + FMOVD.W F20, 8(R1) + FMOVQ.P 11(R10), F13 + FMOVQ.W 11(R20), F15 + FMOVS.P 8(R0), F20 + FMOVS.W 8(R0), F20 + FMOVD.W 8(R1), F20 + RET + +// func plain() +TEXT ·plain(SB), NOSPLIT, $0-0 + FMOVQ F0, 32(R5) + FMOVQ F10, 65520(R10) + FMOVQ F11, 64(RSP) + FMOVQ F11, 8(R20) + FMOVQ F11, 4(R20) + FMOVQ 32(R5), F2 + FMOVQ 65520(R10), F10 + FMOVQ 64(RSP), F11 + FMOVD F1, 8(R2) + FMOVD 8(R2), F1 + RET + +// func symbols() +TEXT ·symbols(SB), NOSPLIT, $0-0 + FMOVQ F5, x(SB) + FMOVQ F5, x+8(SB) + FMOVQ x+8(SB), F5 + FMOVQ x(SB), F5 + RET + +// data the symbol forms refer to +GLOBL x(SB), NOPTR, $16