feat(amd64): floating-point immediates through a synthesised pool

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 02:04:44 +02:00
parent bfb7701db1
commit 29ac03468e
10 changed files with 761 additions and 41 deletions
+138 -31
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -28,7 +29,7 @@ import (
// emitted: the bytes match go tool asm only for NOSPLIT functions or
// zero-frame leaves, where the toolchain emits no guard either.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, _, _, err := assemble(t, nil)
code, _, labels, _, _, _, err := assemble(t, nil)
return code, labels, err
}
@@ -66,9 +67,9 @@ type spadjStep struct {
// assemble encodes a TEXT body, returning the machine code, the static-symbol
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) {
if err := checkAdjspBalance(t); err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
fi := computeFrame(t)
chain := jumpChain(t)
@@ -88,6 +89,8 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
offsets := map[string]int{}
pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool
poolSeen := map[string]bool{}
var poolList []floatPoolEntry
for {
guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue)
@@ -98,7 +101,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
case *ast.Instr:
sz, err := instrSize(s, fi, long[i], link)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pcs[i] = pos
@@ -229,12 +232,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{pos + epi, 0},
)
}
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
for _, entry := range pool {
if !poolSeen[entry.name] {
poolSeen[entry.name] = true
poolList = append(poolList, entry)
}
}
if len(code) != sizes[i] {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
for k := range ps {
@@ -272,7 +281,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
pos += len(suffix)
}
_ = pos
return out, patches, offsets, steps, lines, nil
return out, patches, offsets, steps, lines, poolList, nil
}
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
@@ -593,7 +602,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
}
return jumpSize(mnem, long), nil
}
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
if err != nil {
return 0, err
}
@@ -628,7 +637,7 @@ func jumpSize(mnem string, long bool) int {
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
var prefix []byte
@@ -638,6 +647,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
var code []byte
var ps []sbPatch
var pool []floatPoolEntry
var err error
if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
@@ -646,7 +656,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
// or the linker.
code, ps, err = encodeSBCall(s, link)
if err != nil {
return nil, nil, err
return nil, nil, nil, err
}
for i := range ps {
ps[i].kind = RelCall
@@ -656,23 +666,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
return append(prefix, code...), ps, nil, nil
}
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
// JMP/CALL through a register or memory: no relocation and no
// label to resolve, the operand fully determines the bytes.
code, err = encodeIndirectJump(s, mnem)
if err != nil {
return nil, nil, err
return nil, nil, nil, err
}
return append(prefix, code...), nil, nil
return append(prefix, code...), nil, nil, nil
}
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else {
code, ps, err = encodeNormal(s, fi, link)
code, ps, pool, err = encodeNormal(s, fi, link)
}
if err != nil {
return nil, nil, err
return nil, nil, nil, err
}
// Anchor the patch fields at function-relative positions: off indexes the
// disp32 field, after is the address just past the instruction.
@@ -681,31 +691,65 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
return append(prefix, code...), ps, pool, nil
}
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnemUpper := strings.ToUpper(s.Mnemonic.Text)
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
code, err := encodeBookkeeping(mnemUpper, s)
if err != nil {
return nil, nil, nil, err
}
return code, nil, nil, nil
}
_, size := splitSize(mnemUpper)
if size == 0 {
size = 8
}
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
o, err := operandFromAST(op, size, fi, link)
o, err := operandFromAST(mnemUpper, op, size, fi, link)
if err != nil {
return nil, nil, err
return nil, nil, nil, err
}
ops[i] = o
}
e := &enc{}
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
return nil, nil, err
return nil, nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
return e.out, ps, e.floatPoolList(), nil
}
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
// level, before operand conversion: the toolchain's shapes are FUNCDATA
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
// function body. The symbol reference must not run through the SB-operand
// path, which demands file-level resolution the statement never needs.
func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
if len(s.Operands) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands))
}
a, b := s.Operands[0], s.Operands[1]
if a.Kind != ast.OpImmediate || !a.Imm.HasVal {
return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" {
return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if b.Kind != ast.OpImmediate || !b.Imm.HasVal {
return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil, nil
}
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
@@ -756,7 +800,7 @@ func isSBCall(s *ast.Instr) bool {
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link)
if err != nil {
return nil, nil, err
}
@@ -813,7 +857,7 @@ func indirectJumpTarget(s *ast.Instr) bool {
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
o, err := operandFromAST(op, 8, frameInfo{}, nil)
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
if err != nil {
return nil, err
}
@@ -830,8 +874,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
var spReg = Reg{idx: 4, size: 8}
// operandFromAST converts a parsed operand into an encoder Operand, applying
// the frame translation to FP/SP pseudo-register operands.
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
// the frame translation to FP/SP pseudo-register operands. mnemUpper is the
// instruction's upper-case mnemonic, which the floating-point immediate gate
// needs: only the SSE mnemonics whose encoding takes an XMM/memory source
// accept one.
func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
switch op.Kind {
case ast.OpImmediate:
if op.Imm.HasVal {
@@ -841,17 +888,43 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
}
return Imm(v), nil
}
// A floating-point immediate: $1.5, $-1.0 or the parenthesised
// $(-1.0) spelling (the constant-expression folder only folds
// integers, so that shape arrives with an empty Immediate and only
// the raw spelling carries the value). The toolchain rewrites it
// into a pooled-constant read on the SSE scalar paths and rejects
// it everywhere else.
if text, neg, ok := floatImmText(op); ok {
if !sseFloatImm[mnemUpper] {
return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper)
}
return FloatImm{Text: text, Neg: neg}, nil
}
return nil, fmt.Errorf("non-integer immediate not supported")
case ast.OpAddr:
a := op.Addr
// A bracketed register range, [Z0-Z3]: the four-register source of
// the 4FMAPS/4VNNIW families. The EVEX quad-register emit path
// needs an encoder operand of its own, so the shape stays a named
// gap rather than an encoding.
// the 4FMAPS/4VNNIW families. The range must span four consecutive
// same-width vector registers, exactly what the toolchain's parser
// takes; the EVEX quad-register emit path reads the low end.
if a.Range != nil {
return nil, fmt.Errorf("register range %q needs quad-register encoder support", op.Raw)
lo, ok := ParseReg(a.Range.Lo)
if !ok {
return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo)
}
hi, ok := ParseReg(a.Range.Hi)
if !ok {
return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi)
}
if !lo.isVec() || lo.size != hi.size {
return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw)
}
if hi.idx != lo.idx+3 {
return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw)
}
return RegList{Lo: lo, Hi: hi}, nil
}
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
@@ -921,3 +994,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
}
return nil, fmt.Errorf("unsupported operand")
}
// floatImmText recovers a floating-point immediate's magnitude and sign from
// the parsed operand. The ordinary spellings arrive in Imm.Float; the
// parenthesised $(-1.0) leaves the Immediate empty, because the integer
// folder cannot read it, and only the verbatim operand text still carries
// the value. Anything that is not a number a float parser accepts reports
// not-ok, so every other shape keeps its existing diagnostic.
func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) {
if op.Imm.Float != "" {
return op.Imm.Float, op.Imm.Neg, true
}
if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil {
return "", false, false
}
// joinRaw spaced the token texts; the compact spelling is what matters.
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "$(")
if !ok || !strings.HasSuffix(inner, ")") {
return "", false, false
}
inner = strings.TrimSuffix(inner, ")")
inner = strings.TrimPrefix(inner, "+")
if s, ok := strings.CutPrefix(inner, "-"); ok {
neg = true
inner = s
}
if inner == "" || !strings.ContainsAny(inner, "0123456789") {
return "", false, false
}
if _, err := strconv.ParseFloat(inner, 64); err != nil {
return "", false, false
}
return inner, neg, true
}
+25
View File
@@ -568,3 +568,28 @@ TEXT ·framed(SB), $16-8
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleRegRange pins the bracketed register range at the statement
// level: exactly four consecutive same-width vector registers assemble, the
// toolchain's rejected shapes all report an error.
func TestAssembleRegRange(t *testing.T) {
asm := func(t *testing.T, op string) ([]byte, error) {
t.Helper()
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %s: %v", op, errs)
}
code, _, err := Assemble(f.Decls[0].(*ast.Text))
return code, err
}
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
if _, err := asm(t, op); err != nil {
t.Errorf("%s: %v", op, err)
}
}
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
if _, err := asm(t, op); err == nil {
t.Errorf("%s: assembled, want an error", op)
}
}
}
+195 -2
View File
@@ -5,6 +5,8 @@ package asm
import (
"fmt"
"math"
"strconv"
"strings"
)
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
type enc struct {
out []byte
patches []encPatch // disp32 fields awaiting static-symbol resolution
// FloatPool collects the pooled constants the floating-point
// immediates reference, in first-use order.
floatPool []floatPoolEntry
floatPoolSeen map[string]bool
}
// floatPoolEntry is one pooled floating-point constant: the symbol name
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
type floatPoolEntry struct {
name string
data []byte
}
// addFloatPool records a pooled constant, deduplicated by symbol name.
func (e *enc) addFloatPool(name string, bits uint64, width int) {
if e.floatPoolSeen == nil {
e.floatPoolSeen = map[string]bool{}
}
if e.floatPoolSeen[name] {
return
}
e.floatPoolSeen[name] = true
data := make([]byte, width)
for i := range width {
data[i] = byte(bits >> (8 * i))
}
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
}
// floatPoolList returns the pooled constants in first-use order.
func (e *enc) floatPoolList() []floatPoolEntry {
return e.floatPool
}
// encPatch marks a 4-byte displacement field in enc.out that must receive the
@@ -102,6 +137,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
// The runtime's bookkeeping statements carry no text bytes: go tool asm
// records FUNCDATA and PCDATA in the program list only, so the encoded
// body shows nothing, on every architecture.
case "FUNCDATA", "PCDATA":
return e.encodeFuncdata(upper, ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
@@ -141,11 +181,18 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
// Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
// which the size split must not eat. A floating-point immediate
// rewrites into a pooled-constant read on the scalar members.
if m, ok := sseBinTable[upper]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
@@ -222,7 +269,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "MOVSD", "MOVSS":
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatMove(upper, f, ops)
}
return e.encodeSSEMove(sseMoveTable[base], ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -285,6 +337,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error {
return nil
}
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
// for either (the entries live in the object's ancillary tables, not the
// function body), and the operand shapes it takes are exactly these: an
// integer count first, then a symbol reference for FUNCDATA and an integer
// value for PCDATA. The other architectures accept-and-ignore the same
// statements; amd64 now matches.
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if _, ok := ops[0].(Imm); !ok {
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if _, ok := ops[1].(sbMem); !ok {
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if _, ok := ops[1].(Imm); !ok {
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
@@ -319,6 +398,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error {
return nil
}
// --- floating-point immediates ----------------------------------------------
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
// immediate, the set go tool asm rewrites into a pooled-constant read: the
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
// reject the immediate in the toolchain and are absent here on purpose.
var sseFloatImm = map[string]bool{
"MOVSD": true, "MOVSS": true,
"ADDSD": true, "ADDSS": true,
"SUBSD": true, "SUBSS": true,
"MULSD": true, "MULSS": true,
"DIVSD": true, "DIVSS": true,
"COMISD": true, "COMISS": true,
"UCOMISD": true, "UCOMISS": true,
}
// floatImmOperand reports whether the operand list opens with a
// floating-point immediate in the two-operand spelling (imm, dst).
func floatImmOperand(ops []Operand) (FloatImm, bool) {
if len(ops) != 2 {
return FloatImm{}, false
}
f, ok := ops[0].(FloatImm)
return f, ok
}
// floatPoolValue evaluates a floating-point immediate at the width its
// mnemonic encodes and names the pool constant the toolchain synthesises:
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
// rounding of the parsed value). The name carries the IEEE-754 bits; the
// section holds them little-endian.
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
v, err := strconv.ParseFloat(f.Text, 64)
if err != nil {
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
}
if f.Neg {
v = -v
}
if strings.HasSuffix(mnem, "D") {
bits = math.Float64bits(v)
return bits, fmt.Sprintf("$f64.%016x", bits), nil
}
bits = uint64(math.Float32bits(float32(v)))
return bits, fmt.Sprintf("$f32.%08x", bits), nil
}
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
// source. A positive zero needs no memory read: the toolchain emits
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
// file-level layout or the linker resolves.
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
if bits == 0 {
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
if err := setRM(i, dst, dst, 8); err != nil {
return err
}
return e.emit(i)
}
m := sseMoveTable[mnem]
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
// a floating-point immediate source: the constant is read from the pool into
// the instruction's r/m side (reg = destination), the rewrite go tool asm
// performs at the source level.
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
// spellings end in D, the single spellings in S.
func mwidth(mnem string) int {
if strings.HasSuffix(mnem, "D") {
return 8
}
return 4
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
+157
View File
@@ -9,6 +9,9 @@ import (
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// decode encodes an instruction and decodes it back, returning the decoded
@@ -1064,3 +1067,157 @@ func TestAdjsp(t *testing.T) {
t.Error("ADJSP AX assembled, want an error")
}
}
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
// byte for byte against go tool asm: the scalar moves and the scalar
// arithmetic read the constant from a synthesised read-only pool symbol
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
func TestFloatImmediateGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
// single spellings; negative zero keeps its sign bit and never takes the
// XORPS shortcut.
for _, c := range []struct {
mnem string
imm FloatImm
want string
}{
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
} {
_, name, err := floatPoolValue(c.mnem, c.imm)
if err != nil {
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
continue
}
if name != c.want {
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
}
}
// The shapes the toolchain's parser rejects: the packed and uniform
// forms, a non-vector destination, and the integer spellings.
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
// go tool asm emits no text bytes for either, on every architecture.
func TestBookkeepingGroundTruth(t *testing.T) {
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
} {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if len(code) != 0 {
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
// At the statement level the bookkeeping lines sit between real
// instructions and contribute nothing to the body, symbol reference
// included: the FUNCDATA operand never needs file-level resolution.
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "90c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("FUNCDATA $1, X0 assembled, want an error")
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("PCDATA $1, X0 assembled, want an error")
}
// Encodable mirrors Encode for the names this work touched.
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
if !Encodable(mnem) {
t.Errorf("Encodable(%s) = false, want true", mnem)
}
}
}
// mustParse parses src or fails the test.
func mustParse(t *testing.T, src string) *ast.File {
t.Helper()
f, errs := parser.Parse("t_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return f
}
+124
View File
@@ -810,3 +810,127 @@ func TestAvx512CorpusFamilies(t *testing.T) {
}
}
}
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
// source keeps r/m, the bracketed list's LOW register travels the inverted
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
// the vector length follows the destination (L'L=512 for the ZMM forms,
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
// member. The x86 decoder has no view of these forms, so no decode check
// runs.
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
sp := vreg(t, "RSP")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a842411000000"},
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f22f4a9a842411000000"},
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f25f429a842411000000"},
{"V4FMADDPS Z8 dst", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
"62727f4a9a842411000000"},
{"V4FMADDPS disp8x16", "V4FMADDPS",
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a442404"},
{"V4FMADDPS unmasked", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f489a842411000000"},
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9bb007000000"},
{"V4FMADDSS (DI)", "V4FMADDSS",
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9b37"},
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
"62e22f0d9bb007000000"},
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
"62e25f059bb007000000"},
{"V4FMADDSS X30 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
"62627f0d9bb007000000"},
{"V4FMADDSS X3 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
"62f27f0d9b9807000000"},
{"V4FMADDSS disp8x16", "V4FMADDSS",
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
"62625f059b7001"},
{"V4FNMADDPS", "V4FNMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4aaa842411000000"},
{"V4FNMADDSS", "V4FNMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0dabb007000000"},
{"VP4DPWSSD", "VP4DPWSSD",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a52842411000000"},
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f4853842411000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
// the register class the list and the destination take is fixed per
// instruction, the source is memory only, the opmask slot is positional and
// the list's low register owns V'VVVV.
func TestEvexQuadRegisterErrors(t *testing.T) {
sp := vreg(t, "RSP")
list := func(lo, hi string) RegList {
return RegList{vreg(t, lo), vreg(t, hi)}
}
cases := []struct {
name string
mnem string
ops []Operand
}{
{"X list on the PS form", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"Z list on the SS form", "V4FMADDSS",
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
{"Y destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
{"register source", "V4FMADDPS",
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"non-mask third operand", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
{"k0 mask", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
{"K after the destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
{"zeroing without a mask", "V4FMADDPS.Z",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
{"SAE suffix", "V4FMADDPS.SAE",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"high index source", "VP4DPWSSD",
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"short operand list", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
+8 -7
View File
@@ -63,7 +63,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte {
}
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
// toolchain object, keyed by the name the object records (pkg.name).
// toolchain object, keyed by the name the object records (pkg.name). Each
// function's span is its own symbol size: a toolchain object that follows
// the text with data symbols (the synthesised float-constant pool) would
// otherwise fold them into the last function's bytes.
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
t.Helper()
v := openGoobj(t, obj)
@@ -76,18 +79,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
preceding += len(v.blk(bi)) / symSize
}
total := preceding + len(nps)
out := make(map[string][]byte, len(nps))
for i, s := range nps {
if s.typ != kindSTEXT {
continue
}
start := le.Uint32(didx[4*(preceding+i):])
end := uint32(len(data))
if preceding+i+1 < total {
end = le.Uint32(didx[4*(preceding+i+1):])
}
out[s.name] = data[start:end]
out[s.name] = data[start : start+s.size]
}
return out
}
@@ -129,6 +127,9 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} {
+21 -1
View File
@@ -167,6 +167,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
known[d.name] = true
}
link := &linkInfo{symbols: known, allowExternal: true}
poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{}
@@ -180,7 +181,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, patches, labels, steps, lines, err := assemble(t, link)
code, patches, labels, steps, lines, pool, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
// The pooled floating-point constants join the declared data as
// read-only symbols, deduplicated across the file (the toolchain
// synthesises the same symbols into its rodata).
for _, entry := range pool {
if poolSeen[entry.name] {
continue
}
poolSeen[entry.name] = true
dataSyms = append(dataSyms, dataSym{
name: entry.name,
buf: entry.data,
size: len(entry.data),
rodata: true,
dupok: true,
})
}
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}