feat(asm): PCALIGN alignment on amd64

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 22:00:30 +02:00
parent 82ef289d3a
commit 2c9042d62c
3 changed files with 88 additions and 7 deletions
+8 -7
View File
@@ -22,13 +22,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
relocation, per-GOOS); arm64 accepts the bare-register indirect branch relocation, per-GOOS); arm64 accepts the bare-register indirect branch
(`BL R9` beside `BL (R9)`, both BLR) and the zero-immediate store (`BL R9` beside `BL (R9)`, both BLR) and the zero-immediate store
(`MOVD $0, mem` through the zero register, rejecting non-zero immediates (`MOVD $0, mem` through the zero register, rejecting non-zero immediates
as the toolchain does); and `gasm asm` predefines the `GOARCH_<arch>` as the toolchain does); `PCALIGN` now aligns on amd64, padding with the
and `GOOS_<goos>` macros the go command passes to `go tool asm`, so toolchain's greedy single-instruction NOPs; and `gasm asm` predefines
GOROOT headers' `#ifdef GOARCH_amd64` platform blocks (`go_tls.h`'s the `GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
`get_tls` and friends) select as intended. The GOROOT corpus measure `go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform blocks
moves to 276 of 353 files assembling for every target architecture (`go_tls.h`'s `get_tls` and friends) select as intended. The GOROOT
(78.2 %), 91.4 % of the real-code corpus (280 of 303), from 70.8 % corpus measure moves to 285 of 353 files assembling for every target
and 82.2 %. architecture (80.7 %), 91.4 % of the real-code corpus (280 of 303),
from 70.8 % and 82.2 %.
### Added ### Added
+71
View File
@@ -122,6 +122,19 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
case *ast.Label: case *ast.Label:
offsets[s.Name.Text] = pos offsets[s.Name.Text] = pos
case *ast.Instr: case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
// The alignment pseudo-statement: its size is the
// padding to the next boundary at this very position,
// filled with NOPs at emission.
pad, err := pcAlignPad(pcAlignValue(s), pos)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("PCALIGN: %w", err)
}
sizes[i] = pad
pcs[i] = pos
pos += pad
continue
}
sz, err := instrSize(s, fi, long[i], link) sz, err := instrSize(s, fi, long[i], link)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
@@ -548,6 +561,52 @@ func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
return 0, false return 0, false
} }
// x86 NOP encodings, single-instruction no-ops of lengths 1 to 9 (the
// toolchain's asm6.go nop table); longer padding repeats the largest that
// fits, greedy from the end.
var x86Nops = [][]byte{
{0x90},
{0x66, 0x90},
{0x0F, 0x1F, 0x00},
{0x0F, 0x1F, 0x40, 0x00},
{0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
}
// fillNOPs fills p with the greedy largest single-instruction NOPs, exactly
// the toolchain's fillnop.
func fillNOPs(p []byte) {
for len(p) > 0 {
m := min(len(p), len(x86Nops))
copy(p[:m], x86Nops[m-1])
p = p[m:]
}
}
// pcAlignPad computes the padding PCALIGN $align inserts at pos: the
// alignment must be a power of two in [8, 2048] and the padding runs to the
// next boundary (zero when the position is already aligned).
func pcAlignPad(align, pos int) (int, error) {
if align <= 0 || align&(align-1) != 0 || align < 8 || align > 2048 {
return 0, fmt.Errorf("alignment value of an instruction must be a power of two and in the range [8, 2048], got %d", align)
}
if lob := pos & (align - 1); lob != 0 {
return align - lob, nil
}
return 0, nil
}
// pcAlignValue reads a PCALIGN statement's alignment operand.
func pcAlignValue(s *ast.Instr) int {
if len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.HasVal {
return int(s.Operands[0].Imm.Val)
}
return 0 // rejected by pcAlignPad's range check
}
// hasCall reports whether the function body contains a CALL instruction. // hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool { func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body { for _, stmt := range t.Body {
@@ -760,6 +819,18 @@ func jumpSize(mnem string, long bool) int {
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) { func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text) mnem := strings.ToUpper(s.Mnemonic.Text)
if mnem == "PCALIGN" {
// The layout pass already accounted the padding; emit the same
// amount of NOP bytes for the statement's own position.
pad, err := pcAlignPad(pcAlignValue(s), pc)
if err != nil {
return nil, nil, nil, err
}
out := make([]byte, pad)
fillNOPs(out)
return out, nil, nil, nil
}
var prefix []byte var prefix []byte
if mnem == "RET" && fi.useFP { if mnem == "RET" && fi.useFP {
prefix = fi.epilogue prefix = fi.epilogue
+9
View File
@@ -41,3 +41,12 @@ TEXT ·Tls(SB), NOSPLIT, $0-8
MOVQ 0(BX)(TLS*1), AX MOVQ 0(BX)(TLS*1), AX
MOVQ AX, ret+0(FP) MOVQ AX, ret+0(FP)
RET RET
// func Aligned() int64
TEXT ·Aligned(SB), NOSPLIT, $0-8
MOVQ $1, AX
PCALIGN $16
MOVQ $2, AX
PCALIGN $32
MOVQ AX, ret+0(FP)
RET