Compare commits

...
2 Commits
Author SHA1 Message Date
petrbalvin 9a5e90ba98 feat(amd64): floating-point immediates through a synthesised pool
Test / test (push) Canceled after 42s
Assisted-by: GLM 5.3 Flash
2026-09-21 02:02:19 +02:00
petrbalvin bfb7701db1 feat(amd64): emit the quad-register EVEX families
Assisted-by: GLM 5.3 Flash
2026-09-21 02:02:19 +02:00
13 changed files with 803 additions and 46 deletions
+138 -31
View File
@@ -5,6 +5,7 @@ package asm
import ( import (
"fmt" "fmt"
"strconv"
"strings" "strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -28,7 +29,7 @@ import (
// emitted: the bytes match go tool asm only for NOSPLIT functions or // emitted: the bytes match go tool asm only for NOSPLIT functions or
// zero-frame leaves, where the toolchain emits no guard either. // zero-frame leaves, where the toolchain emits no guard either.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) { func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, _, _, err := assemble(t, nil) code, _, labels, _, _, _, err := assemble(t, nil)
return code, labels, err return code, labels, err
} }
@@ -66,9 +67,9 @@ type spadjStep struct {
// assemble encodes a TEXT body, returning the machine code, the static-symbol // assemble encodes a TEXT body, returning the machine code, the static-symbol
// patch sites (for the file-level layout to resolve), the label table and the // patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries. // stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) { func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) {
if err := checkAdjspBalance(t); err != nil { if err := checkAdjspBalance(t); err != nil {
return nil, nil, nil, nil, nil, err return nil, nil, nil, nil, nil, nil, err
} }
fi := computeFrame(t) fi := computeFrame(t)
chain := jumpChain(t) chain := jumpChain(t)
@@ -88,6 +89,8 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
offsets := map[string]int{} offsets := map[string]int{}
pcs := make([]int, len(t.Body)) pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool var guardJBlong, guardJBElong, moreJMPlong bool
poolSeen := map[string]bool{}
var poolList []floatPoolEntry
for { for {
guard := fi.guardLen(guardJBlong, guardJBElong) guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue) pos := guard + len(fi.prologue)
@@ -98,7 +101,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
case *ast.Instr: case *ast.Instr:
sz, err := instrSize(s, fi, long[i], link) sz, err := instrSize(s, fi, long[i], link)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
} }
sizes[i] = sz sizes[i] = sz
pcs[i] = pos pcs[i] = pos
@@ -229,12 +232,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{pos + epi, 0}, spadjStep{pos + epi, 0},
) )
} }
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
if err != nil { if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
for _, entry := range pool {
if !poolSeen[entry.name] {
poolSeen[entry.name] = true
poolList = append(poolList, entry)
}
} }
if len(code) != sizes[i] { if len(code) != sizes[i] {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i]) return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
} }
if strings.ToUpper(s.Mnemonic.Text) == "CALL" { if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
for k := range ps { for k := range ps {
@@ -272,7 +281,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
pos += len(suffix) pos += len(suffix)
} }
_ = pos _ = pos
return out, patches, offsets, steps, lines, nil return out, patches, offsets, steps, lines, poolList, nil
} }
// jumpChain precomputes jump-to-jump folding: a label whose first instruction // jumpChain precomputes jump-to-jump folding: a label whose first instruction
@@ -593,7 +602,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
} }
return jumpSize(mnem, long), nil return jumpSize(mnem, long), nil
} }
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
if err != nil { if err != nil {
return 0, err return 0, err
} }
@@ -628,7 +637,7 @@ func jumpSize(mnem string, long bool) int {
// (relative to pc, the instruction's own offset). A RET in a frame-pointer // (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue. resolve, when non-nil, redirects a // function is prefixed with the epilogue. resolve, when non-nil, redirects a
// jump label through the jump-to-jump chain before the offset lookup. // jump label through the jump-to-jump chain before the offset lookup.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) { func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnem := strings.ToUpper(s.Mnemonic.Text) mnem := strings.ToUpper(s.Mnemonic.Text)
var prefix []byte var prefix []byte
@@ -638,6 +647,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
var code []byte var code []byte
var ps []sbPatch var ps []sbPatch
var pool []floatPoolEntry
var err error var err error
if isJumpMnemonic(mnem) { if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
@@ -646,7 +656,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
// or the linker. // or the linker.
code, ps, err = encodeSBCall(s, link) code, ps, err = encodeSBCall(s, link)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, nil, err
} }
for i := range ps { for i := range ps {
ps[i].kind = RelCall ps[i].kind = RelCall
@@ -656,23 +666,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
ps[i].off += body ps[i].off += body
ps[i].after = body + len(code) ps[i].after = body + len(code)
} }
return append(prefix, code...), ps, nil return append(prefix, code...), ps, nil, nil
} }
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) { if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
// JMP/CALL through a register or memory: no relocation and no // JMP/CALL through a register or memory: no relocation and no
// label to resolve, the operand fully determines the bytes. // label to resolve, the operand fully determines the bytes.
code, err = encodeIndirectJump(s, mnem) code, err = encodeIndirectJump(s, mnem)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, nil, err
} }
return append(prefix, code...), nil, nil return append(prefix, code...), nil, nil, nil
} }
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else { } else {
code, ps, err = encodeNormal(s, fi, link) code, ps, pool, err = encodeNormal(s, fi, link)
} }
if err != nil { if err != nil {
return nil, nil, err return nil, nil, nil, err
} }
// Anchor the patch fields at function-relative positions: off indexes the // Anchor the patch fields at function-relative positions: off indexes the
// disp32 field, after is the address just past the instruction. // disp32 field, after is the address just past the instruction.
@@ -681,31 +691,65 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
ps[i].off += body ps[i].off += body
ps[i].after = body + len(code) ps[i].after = body + len(code)
} }
return append(prefix, code...), ps, nil return append(prefix, code...), ps, pool, nil
} }
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) { func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text)) mnemUpper := strings.ToUpper(s.Mnemonic.Text)
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
code, err := encodeBookkeeping(mnemUpper, s)
if err != nil {
return nil, nil, nil, err
}
return code, nil, nil, nil
}
_, size := splitSize(mnemUpper)
if size == 0 { if size == 0 {
size = 8 size = 8
} }
ops := make([]Operand, len(s.Operands)) ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands { for i, op := range s.Operands {
o, err := operandFromAST(op, size, fi, link) o, err := operandFromAST(mnemUpper, op, size, fi, link)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, nil, err
} }
ops[i] = o ops[i] = o
} }
e := &enc{} e := &enc{}
if err := e.encode(s.Mnemonic.Text, ops); err != nil { if err := e.encode(s.Mnemonic.Text, ops); err != nil {
return nil, nil, err return nil, nil, nil, err
} }
ps := make([]sbPatch, len(e.patches)) ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches { for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend} ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
} }
return e.out, ps, nil return e.out, ps, e.floatPoolList(), nil
}
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
// level, before operand conversion: the toolchain's shapes are FUNCDATA
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
// function body. The symbol reference must not run through the SB-operand
// path, which demands file-level resolution the statement never needs.
func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
if len(s.Operands) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands))
}
a, b := s.Operands[0], s.Operands[1]
if a.Kind != ast.OpImmediate || !a.Imm.HasVal {
return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" {
return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if b.Kind != ast.OpImmediate || !b.Imm.HasVal {
return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil, nil
} }
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
@@ -756,7 +800,7 @@ func isSBCall(s *ast.Instr) bool {
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site. // encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) { func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link) o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, err
} }
@@ -813,7 +857,7 @@ func indirectJumpTarget(s *ast.Instr) bool {
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands)) ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands { for i, op := range s.Operands {
o, err := operandFromAST(op, 8, frameInfo{}, nil) o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -830,8 +874,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
var spReg = Reg{idx: 4, size: 8} var spReg = Reg{idx: 4, size: 8}
// operandFromAST converts a parsed operand into an encoder Operand, applying // operandFromAST converts a parsed operand into an encoder Operand, applying
// the frame translation to FP/SP pseudo-register operands. // the frame translation to FP/SP pseudo-register operands. mnemUpper is the
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) { // instruction's upper-case mnemonic, which the floating-point immediate gate
// needs: only the SSE mnemonics whose encoding takes an XMM/memory source
// accept one.
func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
switch op.Kind { switch op.Kind {
case ast.OpImmediate: case ast.OpImmediate:
if op.Imm.HasVal { if op.Imm.HasVal {
@@ -841,17 +888,43 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
} }
return Imm(v), nil return Imm(v), nil
} }
// A floating-point immediate: $1.5, $-1.0 or the parenthesised
// $(-1.0) spelling (the constant-expression folder only folds
// integers, so that shape arrives with an empty Immediate and only
// the raw spelling carries the value). The toolchain rewrites it
// into a pooled-constant read on the SSE scalar paths and rejects
// it everywhere else.
if text, neg, ok := floatImmText(op); ok {
if !sseFloatImm[mnemUpper] {
return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper)
}
return FloatImm{Text: text, Neg: neg}, nil
}
return nil, fmt.Errorf("non-integer immediate not supported") return nil, fmt.Errorf("non-integer immediate not supported")
case ast.OpAddr: case ast.OpAddr:
a := op.Addr a := op.Addr
// A bracketed register range, [Z0-Z3]: the four-register source of // A bracketed register range, [Z0-Z3]: the four-register source of
// the 4FMAPS/4VNNIW families. The EVEX quad-register emit path // the 4FMAPS/4VNNIW families. The range must span four consecutive
// needs an encoder operand of its own, so the shape stays a named // same-width vector registers, exactly what the toolchain's parser
// gap rather than an encoding. // takes; the EVEX quad-register emit path reads the low end.
if a.Range != nil { if a.Range != nil {
return nil, fmt.Errorf("register range %q needs quad-register encoder support", op.Raw) lo, ok := ParseReg(a.Range.Lo)
if !ok {
return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo)
}
hi, ok := ParseReg(a.Range.Hi)
if !ok {
return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi)
}
if !lo.isVec() || lo.size != hi.size {
return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw)
}
if hi.idx != lo.idx+3 {
return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw)
}
return RegList{Lo: lo, Hi: hi}, nil
} }
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the // FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
@@ -921,3 +994,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
} }
return nil, fmt.Errorf("unsupported operand") return nil, fmt.Errorf("unsupported operand")
} }
// floatImmText recovers a floating-point immediate's magnitude and sign from
// the parsed operand. The ordinary spellings arrive in Imm.Float; the
// parenthesised $(-1.0) leaves the Immediate empty, because the integer
// folder cannot read it, and only the verbatim operand text still carries
// the value. Anything that is not a number a float parser accepts reports
// not-ok, so every other shape keeps its existing diagnostic.
func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) {
if op.Imm.Float != "" {
return op.Imm.Float, op.Imm.Neg, true
}
if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil {
return "", false, false
}
// joinRaw spaced the token texts; the compact spelling is what matters.
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "$(")
if !ok || !strings.HasSuffix(inner, ")") {
return "", false, false
}
inner = strings.TrimSuffix(inner, ")")
inner = strings.TrimPrefix(inner, "+")
if s, ok := strings.CutPrefix(inner, "-"); ok {
neg = true
inner = s
}
if inner == "" || !strings.ContainsAny(inner, "0123456789") {
return "", false, false
}
if _, err := strconv.ParseFloat(inner, 64); err != nil {
return "", false, false
}
return inner, neg, true
}
+25
View File
@@ -568,3 +568,28 @@ TEXT ·framed(SB), $16-8
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
} }
} }
// TestAssembleRegRange pins the bracketed register range at the statement
// level: exactly four consecutive same-width vector registers assemble, the
// toolchain's rejected shapes all report an error.
func TestAssembleRegRange(t *testing.T) {
asm := func(t *testing.T, op string) ([]byte, error) {
t.Helper()
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %s: %v", op, errs)
}
code, _, err := Assemble(f.Decls[0].(*ast.Text))
return code, err
}
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
if _, err := asm(t, op); err != nil {
t.Errorf("%s: %v", op, err)
}
}
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
if _, err := asm(t, op); err == nil {
t.Errorf("%s: assembled, want an error", op)
}
}
}
+3 -3
View File
@@ -20,9 +20,9 @@ func Encodable(mnemonic string) bool {
switch upper { switch upper {
case "RET", "NOP", "CALL", "JMP", case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the // The literal-data pseudo-ops, the accepted-and-ignored END and
// SP adjust. // bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP": "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true return true
} }
if _, ok := noOperandTable[upper]; ok { if _, ok := noOperandTable[upper]; ok {
+195 -2
View File
@@ -5,6 +5,8 @@ package asm
import ( import (
"fmt" "fmt"
"math"
"strconv"
"strings" "strings"
) )
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
type enc struct { type enc struct {
out []byte out []byte
patches []encPatch // disp32 fields awaiting static-symbol resolution patches []encPatch // disp32 fields awaiting static-symbol resolution
// FloatPool collects the pooled constants the floating-point
// immediates reference, in first-use order.
floatPool []floatPoolEntry
floatPoolSeen map[string]bool
}
// floatPoolEntry is one pooled floating-point constant: the symbol name
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
type floatPoolEntry struct {
name string
data []byte
}
// addFloatPool records a pooled constant, deduplicated by symbol name.
func (e *enc) addFloatPool(name string, bits uint64, width int) {
if e.floatPoolSeen == nil {
e.floatPoolSeen = map[string]bool{}
}
if e.floatPoolSeen[name] {
return
}
e.floatPoolSeen[name] = true
data := make([]byte, width)
for i := range width {
data[i] = byte(bits >> (8 * i))
}
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
}
// floatPoolList returns the pooled constants in first-use order.
func (e *enc) floatPoolList() []floatPoolEntry {
return e.floatPool
} }
// encPatch marks a 4-byte displacement field in enc.out that must receive the // encPatch marks a 4-byte displacement field in enc.out that must receive the
@@ -102,6 +137,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeEnd(ops) return e.encodeEnd(ops)
case "ADJSP": case "ADJSP":
return e.encodeAdjsp(ops) return e.encodeAdjsp(ops)
// The runtime's bookkeeping statements carry no text bytes: go tool asm
// records FUNCDATA and PCDATA in the program list only, so the encoded
// body shows nothing, on every architecture.
case "FUNCDATA", "PCDATA":
return e.encodeFuncdata(upper, ops)
} }
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing // VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
@@ -141,11 +181,18 @@ func (e *enc) encode(mnem string, ops []Operand) error {
} }
// Legacy SSE packed binaries dispatch on the full name: the packed // Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...), // integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat. // which the size split must not eat. A floating-point immediate
// rewrites into a pooled-constant read on the scalar members.
if m, ok := sseBinTable[upper]; ok { if m, ok := sseBinTable[upper]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops) return e.encodeSSEBin(m, ops)
} }
if m, ok := sseBinTable[base]; ok { if m, ok := sseBinTable[base]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops) return e.encodeSSEBin(m, ops)
} }
// The imm8-controlled legacy instructions, the lane extracts and inserts // The imm8-controlled legacy instructions, the lane extracts and inserts
@@ -222,7 +269,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeCvtInt(base, ops, size) return e.encodeCvtInt(base, ops, size)
case "FMOVD": case "FMOVD":
return e.encodeFmov(ops) return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS": case "MOVSD", "MOVSS":
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatMove(upper, f, ops)
}
return e.encodeSSEMove(sseMoveTable[base], ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
return e.encodeSSEMove(sseMoveTable[base], ops) return e.encodeSSEMove(sseMoveTable[base], ops)
} }
return fmt.Errorf("unsupported instruction %q", mnem) return fmt.Errorf("unsupported instruction %q", mnem)
@@ -285,6 +337,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error {
return nil return nil
} }
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
// for either (the entries live in the object's ancillary tables, not the
// function body), and the operand shapes it takes are exactly these: an
// integer count first, then a symbol reference for FUNCDATA and an integer
// value for PCDATA. The other architectures accept-and-ignore the same
// statements; amd64 now matches.
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if _, ok := ops[0].(Imm); !ok {
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if _, ok := ops[1].(sbMem); !ok {
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if _, ok := ops[1].(Imm); !ok {
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement // encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so // entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the // the statements after an END still belong to the same function and the
@@ -319,6 +398,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error {
return nil return nil
} }
// --- floating-point immediates ----------------------------------------------
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
// immediate, the set go tool asm rewrites into a pooled-constant read: the
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
// reject the immediate in the toolchain and are absent here on purpose.
var sseFloatImm = map[string]bool{
"MOVSD": true, "MOVSS": true,
"ADDSD": true, "ADDSS": true,
"SUBSD": true, "SUBSS": true,
"MULSD": true, "MULSS": true,
"DIVSD": true, "DIVSS": true,
"COMISD": true, "COMISS": true,
"UCOMISD": true, "UCOMISS": true,
}
// floatImmOperand reports whether the operand list opens with a
// floating-point immediate in the two-operand spelling (imm, dst).
func floatImmOperand(ops []Operand) (FloatImm, bool) {
if len(ops) != 2 {
return FloatImm{}, false
}
f, ok := ops[0].(FloatImm)
return f, ok
}
// floatPoolValue evaluates a floating-point immediate at the width its
// mnemonic encodes and names the pool constant the toolchain synthesises:
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
// rounding of the parsed value). The name carries the IEEE-754 bits; the
// section holds them little-endian.
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
v, err := strconv.ParseFloat(f.Text, 64)
if err != nil {
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
}
if f.Neg {
v = -v
}
if strings.HasSuffix(mnem, "D") {
bits = math.Float64bits(v)
return bits, fmt.Sprintf("$f64.%016x", bits), nil
}
bits = uint64(math.Float32bits(float32(v)))
return bits, fmt.Sprintf("$f32.%08x", bits), nil
}
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
// source. A positive zero needs no memory read: the toolchain emits
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
// file-level layout or the linker resolves.
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
if bits == 0 {
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
if err := setRM(i, dst, dst, 8); err != nil {
return err
}
return e.emit(i)
}
m := sseMoveTable[mnem]
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
// a floating-point immediate source: the constant is read from the pool into
// the instruction's r/m side (reg = destination), the rewrite go tool asm
// performs at the source level.
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
// spellings end in D, the single spellings in S.
func mwidth(mnem string) int {
if strings.HasSuffix(mnem, "D") {
return 8
}
return 4
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic. // splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) { func splitSize(upper string) (base string, size int) {
if upper == "" { if upper == "" {
+157
View File
@@ -9,6 +9,9 @@ import (
"testing" "testing"
"golang.org/x/arch/x86/x86asm" "golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
) )
// decode encodes an instruction and decodes it back, returning the decoded // decode encodes an instruction and decodes it back, returning the decoded
@@ -1064,3 +1067,157 @@ func TestAdjsp(t *testing.T) {
t.Error("ADJSP AX assembled, want an error") t.Error("ADJSP AX assembled, want an error")
} }
} }
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
// byte for byte against go tool asm: the scalar moves and the scalar
// arithmetic read the constant from a synthesised read-only pool symbol
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
func TestFloatImmediateGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
// single spellings; negative zero keeps its sign bit and never takes the
// XORPS shortcut.
for _, c := range []struct {
mnem string
imm FloatImm
want string
}{
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
} {
_, name, err := floatPoolValue(c.mnem, c.imm)
if err != nil {
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
continue
}
if name != c.want {
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
}
}
// The shapes the toolchain's parser rejects: the packed and uniform
// forms, a non-vector destination, and the integer spellings.
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
// go tool asm emits no text bytes for either, on every architecture.
func TestBookkeepingGroundTruth(t *testing.T) {
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
} {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if len(code) != 0 {
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
// At the statement level the bookkeeping lines sit between real
// instructions and contribute nothing to the body, symbol reference
// included: the FUNCDATA operand never needs file-level resolution.
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "90c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("FUNCDATA $1, X0 assembled, want an error")
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("PCDATA $1, X0 assembled, want an error")
}
// Encodable mirrors Encode for the names this work touched.
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
if !Encodable(mnem) {
t.Errorf("Encodable(%s) = false, want true", mnem)
}
}
}
// mustParse parses src or fails the test.
func mustParse(t *testing.T, src string) *ast.File {
t.Helper()
f, errs := parser.Parse("t_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return f
}
+97 -2
View File
@@ -737,6 +737,91 @@ var evexTable = map[string]evexSpec{
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}}, "VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
} }
// evexQuad describes one quad-register instruction: the opcode under
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
// registers the bracketed list and the destination take (512-bit ZMM for
// the packed forms, 128-bit XMM for the scalar ones).
type evexQuad struct {
opcode byte
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
}
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
// families) to their encoding. The operand shape is fixed: a single memory
// source in r/m, the bracketed register list whose LOW register travels the
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
// destination in reg. The vector length follows the destination (512-bit
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
// multiplier stays 16 for every member, the toolchain's own tuple choice.
var evexQuadTable = map[string]evexQuad{
"V4FMADDPS": {0x9A, 64},
"V4FMADDSS": {0x9B, 16},
"V4FNMADDPS": {0xAA, 64},
"V4FNMADDSS": {0xAB, 16},
"VP4DPWSSD": {0x52, 64},
"VP4DPWSSDS": {0x53, 64},
}
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
func isEvexQuad(upper string) bool {
_, ok := evexQuadTable[upper]
return ok
}
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
// The register list is the VVVV-side source: its low register fills the
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
// suffix applies.
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
if len(ops) != 3 && len(ops) != 4 {
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
}
mem, lst := ops[0], ops[1]
dst := ops[len(ops)-1]
mask := 0
if len(ops) == 4 {
k, ok := ops[2].(Reg)
if !ok || !k.mask {
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
}
if k.idx == 0 {
return fmt.Errorf("k0 is not a usable mask register")
}
mask = k.idx
}
list, ok := lst.(RegList)
if !ok {
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
}
if list.Lo.size != q.width {
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if dstReg.size != q.width {
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
}
if !memOperand(mem) {
return fmt.Errorf("%s: the source must be a memory operand", mnem)
}
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
// its fifth bit into the same field the list's low register occupies.
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
}
if sfx.zeroing && mask == 0 {
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
}
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
// The vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones), exactly as the oracle encodes it.
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind, a GPR source uses opReg, a memory source uses // depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n. // opMem with a disp8×N of n.
@@ -812,8 +897,10 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexBcastTable[mnemUpper]; ok { if _, ok := evexBcastTable[mnemUpper]; ok {
return true return true
} }
_, ok := evexMoveTable[mnemUpper] if _, ok := evexMoveTable[mnemUpper]; ok {
return ok return true
}
return isEvexQuad(mnemUpper)
} }
// evexRequired reports whether the operands force the EVEX encoding of a // evexRequired reports whether the operands force the EVEX encoding of a
@@ -995,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx) return e.encodeEvexRM(spec, ops, 0, sfx)
} }
spec, inTable := evexTable[mnemUpper] spec, inTable := evexTable[mnemUpper]
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
}
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
}
if inTable { if inTable {
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] { if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper) return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
+8 -7
View File
@@ -63,7 +63,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte {
} }
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a // oracleFuncCode extracts the non-package TEXT functions' code bytes from a
// toolchain object, keyed by the name the object records (pkg.name). // toolchain object, keyed by the name the object records (pkg.name). Each
// function's span is its own symbol size: a toolchain object that follows
// the text with data symbols (the synthesised float-constant pool) would
// otherwise fold them into the last function's bytes.
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte { func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
t.Helper() t.Helper()
v := openGoobj(t, obj) v := openGoobj(t, obj)
@@ -76,18 +79,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} { for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
preceding += len(v.blk(bi)) / symSize preceding += len(v.blk(bi)) / symSize
} }
total := preceding + len(nps)
out := make(map[string][]byte, len(nps)) out := make(map[string][]byte, len(nps))
for i, s := range nps { for i, s := range nps {
if s.typ != kindSTEXT { if s.typ != kindSTEXT {
continue continue
} }
start := le.Uint32(didx[4*(preceding+i):]) start := le.Uint32(didx[4*(preceding+i):])
end := uint32(len(data)) out[s.name] = data[start : start+s.size]
if preceding+i+1 < total {
end = le.Uint32(didx[4*(preceding+i+1):])
}
out[s.name] = data[start:end]
} }
return out return out
} }
@@ -129,6 +127,9 @@ func TestDifferentialKernels(t *testing.T) {
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} { } {
+21 -1
View File
@@ -167,6 +167,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
known[d.name] = true known[d.name] = true
} }
link := &linkInfo{symbols: known, allowExternal: true} link := &linkInfo{symbols: known, allowExternal: true}
poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{} textOff := map[string]int{}
@@ -180,7 +181,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok { if !ok {
continue continue
} }
code, patches, labels, steps, lines, err := assemble(t, link) code, patches, labels, steps, lines, pool, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
// The pooled floating-point constants join the declared data as
// read-only symbols, deduplicated across the file (the toolchain
// synthesises the same symbols into its rodata).
for _, entry := range pool {
if poolSeen[entry.name] {
continue
}
poolSeen[entry.name] = true
dataSyms = append(dataSyms, dataSym{
name: entry.name,
buf: entry.data,
size: len(entry.data),
rodata: true,
dupok: true,
})
}
if err != nil { if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err) return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
} }
+22
View File
@@ -14,6 +14,28 @@ type Imm int64
func (Imm) isOperand() {} func (Imm) isOperand() {}
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
// low register through the inverted 5-bit V'VVVV field; the three higher
// registers are implied by the instruction, so only the pair travels here.
type RegList struct {
Lo Reg
Hi Reg // implied by the encoding; Lo.idx+3 by construction
}
func (RegList) isOperand() {}
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
// encoding takes an XMM/memory source at that position rewrite it as a read
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
// own behaviour; every other instruction rejects it.
type FloatImm struct {
Text string // the numeric text as written, sign excluded
Neg bool // a leading minus
}
func (FloatImm) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale). // Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct { type Mem struct {
Base Reg Base Reg
+32
View File
@@ -0,0 +1,32 @@
// The runtime bookkeeping statements: FUNCDATA and PCDATA contribute no
// text bytes on any architecture, and amd64 now matches. They sit between
// real instructions here, with plain, static and offset symbol references
// on the FUNCDATA lines, so the byte counts prove the zero contribution.
#include "textflag.h"
// func bookkeep(x int64) int64
TEXT ·bookkeep(SB), NOSPLIT, $0-16
PCDATA $0, $-1
MOVQ x+0(FP), AX
PCDATA $1, $-2
FUNCDATA $0, args_stackmap(SB)
ADDQ $1, AX
FUNCDATA $5, arginfo0(SB)
PCDATA $1, $3
MOVQ AX, ret+8(FP)
FUNCDATA $1, externalfuncdata(SB)
PCDATA $0, $0
RET
// func bookkeepstatic() int64
TEXT ·bookkeepstatic(SB), NOSPLIT, $0-8
// A static symbol and a defined data symbol as the funcdata target.
// (A symbol+offset target the toolchain itself refuses.)
FUNCDATA $2, fdtable<>(SB)
FUNCDATA $3, undefsym(SB)
MOVQ $7, AX
MOVQ AX, ret+0(FP)
RET
GLOBL fdtable<>(SB), NOPTR, $16
+55
View File
@@ -0,0 +1,55 @@
// Floating-point immediates on the SSE scalar paths: the constant is
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
// displacement left to the relocation. A positive zero on the moves
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
// takes the pool. The parenthesised $(-1.0) spelling is the one
// math/floor_amd64.s uses. Every result is folded back so no
// instruction is dead.
#include "textflag.h"
// func floatimm(x float64) float64
TEXT ·floatimm(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
MOVQ AX, X0
// The floor kernel's sign fold: the parenthesised negative spelling.
MOVSD $ (-1.0), X2
ANDPD X2, X0
// Positive and fractional constants on the scalar moves.
MOVSD $1.5, X3
MOVSD $0.5, X4
MOVSS $2.5, X5
MOVSS $-0.5, X6
// A positive zero collapses to XORPS; a negative zero does not.
MOVSD $0.0, X7
MOVSS $0.0, X8
MOVSD $-0.0, X9
// The scalar arithmetic reads the pool through r/m (hypot's shape).
ADDSD $1.0, X3
SUBSD $0.5, X4
MULSD $-2.5, X4
DIVSD $2.0, X3
ADDSS $0.25, X5
// Fold everything into one double.
ADDSD X5, X3
ADDSD X6, X3
ADDSD X7, X3
ADDSD X8, X3
ADDSD X9, X3
ADDSD X4, X3
ADDSD X0, X3
MOVSD X3, ret+8(FP)
RET
// func floatimmfloat32() float32
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
// The single-width pool constants ride the F3 prefix.
MOVSS $1.0, X0
MOVSS $-1.0, X1
MOVSS $0.0, X2
ADDSS $0.5, X0
ADDSS X1, X0
ADDSS X2, X0
MOVSS X0, ret+0(FP)
RET
+44
View File
@@ -0,0 +1,44 @@
// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
// VP4DPWSSDS). The bracketed list's low register travels the inverted
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
// vector length follows the destination (512-bit for the ZMM forms,
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
// every member. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func quadf4(src *[16]uint32, n int) float32
TEXT ·quadf4(SB), NOSPLIT, $0-20
MOVQ src+0(FP), SI
MOVQ n+8(FP), CX
// The packed 4-FMA form over four consecutive ZMM accumulators,
// masked with K2, K3 and unmasked alike; the displacements exercise
// the disp32 form and the disp8x16 compressed form.
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
V4FMADDPS (SI), [Z20-Z23], Z2
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
// The scalar form reads XMM lists and takes the 128-bit length; the
// displacement compresses by 16.
V4FMADDSS 7(AX), [X0-X3], K5, X22
V4FMADDSS (DI), [X10-X13], K5, X23
V4FNMADDSS 16(SI), [X20-X23], K1, X24
// The 4-VNNI dot products, indexed source included.
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
VP4DPWSSD (SI), [Z12-Z15], Z6
// Zeroing keeps the usual rule: a mask register must ride along.
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
// Fold every accumulator into one scalar.
VPADDD Z0, Z1, Z9
VPADDD Z2, Z5, Z10
VPADDD Z9, Z17, Z11
VPADDD Z10, Z31, Z12
VPADDD Z11, Z12, Z13
VPADDD Z13, Z14, Z15
VADDSS X22, X23, X0
VADDSS X24, X0, X1
VADDSS X1, X2, X3
VMOVSS X3, ret+16(FP)
RET
+6
View File
@@ -124,10 +124,16 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/avx_amd64.s", "../testdata/verify/avx_amd64.s",
"../testdata/verify/pfx_amd64.s", "../testdata/verify/pfx_amd64.s",
"../testdata/verify/vsib_amd64.s", "../testdata/verify/vsib_amd64.s",
"../testdata/verify/floatimm_amd64.s",
"../testdata/verify/bookkeep_amd64.s",
"../testdata/verify/quadreg_amd64.s",
"../testdata/verify/rawdata_amd64.s", "../testdata/verify/rawdata_amd64.s",
"../testdata/verify/avx512_amd64.s", "../testdata/verify/avx512_amd64.s",
"../testdata/verify/pfx_amd64.s", "../testdata/verify/pfx_amd64.s",
"../testdata/verify/vsib_amd64.s", "../testdata/verify/vsib_amd64.s",
"../testdata/verify/floatimm_amd64.s",
"../testdata/verify/bookkeep_amd64.s",
"../testdata/verify/quadreg_amd64.s",
"../testdata/verify/rawdata_amd64.s", "../testdata/verify/rawdata_amd64.s",
"../testdata/verify/avx512_amd64.s", "../testdata/verify/avx512_amd64.s",
"../testdata/verify/doubleshift_amd64.s", "../testdata/verify/doubleshift_amd64.s",