feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+138
-31
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
@@ -28,7 +29,7 @@ import (
|
||||
// emitted: the bytes match go tool asm only for NOSPLIT functions or
|
||||
// zero-frame leaves, where the toolchain emits no guard either.
|
||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
code, _, labels, _, _, err := assemble(t, nil)
|
||||
code, _, labels, _, _, _, err := assemble(t, nil)
|
||||
return code, labels, err
|
||||
}
|
||||
|
||||
@@ -66,9 +67,9 @@ type spadjStep struct {
|
||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||
// patch sites (for the file-level layout to resolve), the label table and the
|
||||
// stack-adjustment boundaries.
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) {
|
||||
if err := checkAdjspBalance(t); err != nil {
|
||||
return nil, nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, nil, err
|
||||
}
|
||||
fi := computeFrame(t)
|
||||
chain := jumpChain(t)
|
||||
@@ -88,6 +89,8 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
offsets := map[string]int{}
|
||||
pcs := make([]int, len(t.Body))
|
||||
var guardJBlong, guardJBElong, moreJMPlong bool
|
||||
poolSeen := map[string]bool{}
|
||||
var poolList []floatPoolEntry
|
||||
for {
|
||||
guard := fi.guardLen(guardJBlong, guardJBElong)
|
||||
pos := guard + len(fi.prologue)
|
||||
@@ -98,7 +101,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi, long[i], link)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
sizes[i] = sz
|
||||
pcs[i] = pos
|
||||
@@ -229,12 +232,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
spadjStep{pos + epi, 0},
|
||||
)
|
||||
}
|
||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
for _, entry := range pool {
|
||||
if !poolSeen[entry.name] {
|
||||
poolSeen[entry.name] = true
|
||||
poolList = append(poolList, entry)
|
||||
}
|
||||
}
|
||||
if len(code) != sizes[i] {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
}
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
|
||||
for k := range ps {
|
||||
@@ -272,7 +281,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
pos += len(suffix)
|
||||
}
|
||||
_ = pos
|
||||
return out, patches, offsets, steps, lines, nil
|
||||
return out, patches, offsets, steps, lines, poolList, nil
|
||||
}
|
||||
|
||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||
@@ -593,7 +602,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
|
||||
}
|
||||
return jumpSize(mnem, long), nil
|
||||
}
|
||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
@@ -628,7 +637,7 @@ func jumpSize(mnem string, long bool) int {
|
||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
||||
// jump label through the jump-to-jump chain before the offset lookup.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
|
||||
var prefix []byte
|
||||
@@ -638,6 +647,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
|
||||
var code []byte
|
||||
var ps []sbPatch
|
||||
var pool []floatPoolEntry
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||
@@ -646,7 +656,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
// or the linker.
|
||||
code, ps, err = encodeSBCall(s, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
for i := range ps {
|
||||
ps[i].kind = RelCall
|
||||
@@ -656,23 +666,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
return append(prefix, code...), ps, nil, nil
|
||||
}
|
||||
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
||||
// JMP/CALL through a register or memory: no relocation and no
|
||||
// label to resolve, the operand fully determines the bytes.
|
||||
code, err = encodeIndirectJump(s, mnem)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
return append(prefix, code...), nil, nil
|
||||
return append(prefix, code...), nil, nil, nil
|
||||
}
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
||||
} else {
|
||||
code, ps, err = encodeNormal(s, fi, link)
|
||||
code, ps, pool, err = encodeNormal(s, fi, link)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
// Anchor the patch fields at function-relative positions: off indexes the
|
||||
// disp32 field, after is the address just past the instruction.
|
||||
@@ -681,31 +691,65 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
return append(prefix, code...), ps, pool, nil
|
||||
}
|
||||
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||
mnemUpper := strings.ToUpper(s.Mnemonic.Text)
|
||||
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
|
||||
code, err := encodeBookkeeping(mnemUpper, s)
|
||||
if err != nil {
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
return code, nil, nil, nil
|
||||
}
|
||||
_, size := splitSize(mnemUpper)
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, size, fi, link)
|
||||
o, err := operandFromAST(mnemUpper, op, size, fi, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
ops[i] = o
|
||||
}
|
||||
e := &enc{}
|
||||
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
ps := make([]sbPatch, len(e.patches))
|
||||
for i, p := range e.patches {
|
||||
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||
}
|
||||
return e.out, ps, nil
|
||||
return e.out, ps, e.floatPoolList(), nil
|
||||
}
|
||||
|
||||
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
|
||||
// level, before operand conversion: the toolchain's shapes are FUNCDATA
|
||||
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
|
||||
// function body. The symbol reference must not run through the SB-operand
|
||||
// path, which demands file-level resolution the statement never needs.
|
||||
func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
|
||||
if len(s.Operands) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands))
|
||||
}
|
||||
a, b := s.Operands[0], s.Operands[1]
|
||||
if a.Kind != ast.OpImmediate || !a.Imm.HasVal {
|
||||
return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||
}
|
||||
switch upper {
|
||||
case "FUNCDATA":
|
||||
if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" {
|
||||
return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||
}
|
||||
case "PCDATA":
|
||||
if b.Kind != ast.OpImmediate || !b.Imm.HasVal {
|
||||
return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||
}
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
||||
@@ -756,7 +800,7 @@ func isSBCall(s *ast.Instr) bool {
|
||||
|
||||
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
|
||||
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
|
||||
o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
@@ -813,7 +857,7 @@ func indirectJumpTarget(s *ast.Instr) bool {
|
||||
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, 8, frameInfo{}, nil)
|
||||
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -830,8 +874,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||
var spReg = Reg{idx: 4, size: 8}
|
||||
|
||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||
// the frame translation to FP/SP pseudo-register operands.
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
// the frame translation to FP/SP pseudo-register operands. mnemUpper is the
|
||||
// instruction's upper-case mnemonic, which the floating-point immediate gate
|
||||
// needs: only the SSE mnemonics whose encoding takes an XMM/memory source
|
||||
// accept one.
|
||||
func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
switch op.Kind {
|
||||
case ast.OpImmediate:
|
||||
if op.Imm.HasVal {
|
||||
@@ -841,17 +888,43 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
||||
}
|
||||
return Imm(v), nil
|
||||
}
|
||||
// A floating-point immediate: $1.5, $-1.0 or the parenthesised
|
||||
// $(-1.0) spelling (the constant-expression folder only folds
|
||||
// integers, so that shape arrives with an empty Immediate and only
|
||||
// the raw spelling carries the value). The toolchain rewrites it
|
||||
// into a pooled-constant read on the SSE scalar paths and rejects
|
||||
// it everywhere else.
|
||||
if text, neg, ok := floatImmText(op); ok {
|
||||
if !sseFloatImm[mnemUpper] {
|
||||
return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper)
|
||||
}
|
||||
return FloatImm{Text: text, Neg: neg}, nil
|
||||
}
|
||||
return nil, fmt.Errorf("non-integer immediate not supported")
|
||||
|
||||
case ast.OpAddr:
|
||||
a := op.Addr
|
||||
|
||||
// A bracketed register range, [Z0-Z3]: the four-register source of
|
||||
// the 4FMAPS/4VNNIW families. The EVEX quad-register emit path
|
||||
// needs an encoder operand of its own, so the shape stays a named
|
||||
// gap rather than an encoding.
|
||||
// the 4FMAPS/4VNNIW families. The range must span four consecutive
|
||||
// same-width vector registers, exactly what the toolchain's parser
|
||||
// takes; the EVEX quad-register emit path reads the low end.
|
||||
if a.Range != nil {
|
||||
return nil, fmt.Errorf("register range %q needs quad-register encoder support", op.Raw)
|
||||
lo, ok := ParseReg(a.Range.Lo)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo)
|
||||
}
|
||||
hi, ok := ParseReg(a.Range.Hi)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi)
|
||||
}
|
||||
if !lo.isVec() || lo.size != hi.size {
|
||||
return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw)
|
||||
}
|
||||
if hi.idx != lo.idx+3 {
|
||||
return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw)
|
||||
}
|
||||
return RegList{Lo: lo, Hi: hi}, nil
|
||||
}
|
||||
|
||||
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
||||
@@ -921,3 +994,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
||||
}
|
||||
return nil, fmt.Errorf("unsupported operand")
|
||||
}
|
||||
|
||||
// floatImmText recovers a floating-point immediate's magnitude and sign from
|
||||
// the parsed operand. The ordinary spellings arrive in Imm.Float; the
|
||||
// parenthesised $(-1.0) leaves the Immediate empty, because the integer
|
||||
// folder cannot read it, and only the verbatim operand text still carries
|
||||
// the value. Anything that is not a number a float parser accepts reports
|
||||
// not-ok, so every other shape keeps its existing diagnostic.
|
||||
func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) {
|
||||
if op.Imm.Float != "" {
|
||||
return op.Imm.Float, op.Imm.Neg, true
|
||||
}
|
||||
if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil {
|
||||
return "", false, false
|
||||
}
|
||||
// joinRaw spaced the token texts; the compact spelling is what matters.
|
||||
compact := strings.ReplaceAll(op.Raw, " ", "")
|
||||
inner, ok := strings.CutPrefix(compact, "$(")
|
||||
if !ok || !strings.HasSuffix(inner, ")") {
|
||||
return "", false, false
|
||||
}
|
||||
inner = strings.TrimSuffix(inner, ")")
|
||||
inner = strings.TrimPrefix(inner, "+")
|
||||
if s, ok := strings.CutPrefix(inner, "-"); ok {
|
||||
neg = true
|
||||
inner = s
|
||||
}
|
||||
if inner == "" || !strings.ContainsAny(inner, "0123456789") {
|
||||
return "", false, false
|
||||
}
|
||||
if _, err := strconv.ParseFloat(inner, 64); err != nil {
|
||||
return "", false, false
|
||||
}
|
||||
return inner, neg, true
|
||||
}
|
||||
|
||||
@@ -568,3 +568,28 @@ TEXT ·framed(SB), $16-8
|
||||
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleRegRange pins the bracketed register range at the statement
|
||||
// level: exactly four consecutive same-width vector registers assemble, the
|
||||
// toolchain's rejected shapes all report an error.
|
||||
func TestAssembleRegRange(t *testing.T) {
|
||||
asm := func(t *testing.T, op string) ([]byte, error) {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", op, errs)
|
||||
}
|
||||
code, _, err := Assemble(f.Decls[0].(*ast.Text))
|
||||
return code, err
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
|
||||
if _, err := asm(t, op); err != nil {
|
||||
t.Errorf("%s: %v", op, err)
|
||||
}
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
|
||||
if _, err := asm(t, op); err == nil {
|
||||
t.Errorf("%s: assembled, want an error", op)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+195
-2
@@ -5,6 +5,8 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
type enc struct {
|
||||
out []byte
|
||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||
|
||||
// FloatPool collects the pooled constants the floating-point
|
||||
// immediates reference, in first-use order.
|
||||
floatPool []floatPoolEntry
|
||||
floatPoolSeen map[string]bool
|
||||
}
|
||||
|
||||
// floatPoolEntry is one pooled floating-point constant: the symbol name
|
||||
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
|
||||
type floatPoolEntry struct {
|
||||
name string
|
||||
data []byte
|
||||
}
|
||||
|
||||
// addFloatPool records a pooled constant, deduplicated by symbol name.
|
||||
func (e *enc) addFloatPool(name string, bits uint64, width int) {
|
||||
if e.floatPoolSeen == nil {
|
||||
e.floatPoolSeen = map[string]bool{}
|
||||
}
|
||||
if e.floatPoolSeen[name] {
|
||||
return
|
||||
}
|
||||
e.floatPoolSeen[name] = true
|
||||
data := make([]byte, width)
|
||||
for i := range width {
|
||||
data[i] = byte(bits >> (8 * i))
|
||||
}
|
||||
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
|
||||
}
|
||||
|
||||
// floatPoolList returns the pooled constants in first-use order.
|
||||
func (e *enc) floatPoolList() []floatPoolEntry {
|
||||
return e.floatPool
|
||||
}
|
||||
|
||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||
@@ -102,6 +137,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeEnd(ops)
|
||||
case "ADJSP":
|
||||
return e.encodeAdjsp(ops)
|
||||
// The runtime's bookkeeping statements carry no text bytes: go tool asm
|
||||
// records FUNCDATA and PCDATA in the program list only, so the encoded
|
||||
// body shows nothing, on every architecture.
|
||||
case "FUNCDATA", "PCDATA":
|
||||
return e.encodeFuncdata(upper, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
@@ -141,11 +181,18 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
}
|
||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||
// which the size split must not eat.
|
||||
// which the size split must not eat. A floating-point immediate
|
||||
// rewrites into a pooled-constant read on the scalar members.
|
||||
if m, ok := sseBinTable[upper]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
if m, ok := sseBinTable[base]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||
@@ -222,7 +269,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeCvtInt(base, ops, size)
|
||||
case "FMOVD":
|
||||
return e.encodeFmov(ops)
|
||||
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
case "MOVSD", "MOVSS":
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatMove(upper, f, ops)
|
||||
}
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
@@ -285,6 +337,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
|
||||
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
|
||||
// for either (the entries live in the object's ancillary tables, not the
|
||||
// function body), and the operand shapes it takes are exactly these: an
|
||||
// integer count first, then a symbol reference for FUNCDATA and an integer
|
||||
// value for PCDATA. The other architectures accept-and-ignore the same
|
||||
// statements; amd64 now matches.
|
||||
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
if _, ok := ops[0].(Imm); !ok {
|
||||
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||
}
|
||||
switch upper {
|
||||
case "FUNCDATA":
|
||||
if _, ok := ops[1].(sbMem); !ok {
|
||||
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||
}
|
||||
case "PCDATA":
|
||||
if _, ok := ops[1].(Imm); !ok {
|
||||
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||
// the statements after an END still belong to the same function and the
|
||||
@@ -319,6 +398,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// --- floating-point immediates ----------------------------------------------
|
||||
|
||||
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
|
||||
// immediate, the set go tool asm rewrites into a pooled-constant read: the
|
||||
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
|
||||
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
|
||||
// reject the immediate in the toolchain and are absent here on purpose.
|
||||
var sseFloatImm = map[string]bool{
|
||||
"MOVSD": true, "MOVSS": true,
|
||||
"ADDSD": true, "ADDSS": true,
|
||||
"SUBSD": true, "SUBSS": true,
|
||||
"MULSD": true, "MULSS": true,
|
||||
"DIVSD": true, "DIVSS": true,
|
||||
"COMISD": true, "COMISS": true,
|
||||
"UCOMISD": true, "UCOMISS": true,
|
||||
}
|
||||
|
||||
// floatImmOperand reports whether the operand list opens with a
|
||||
// floating-point immediate in the two-operand spelling (imm, dst).
|
||||
func floatImmOperand(ops []Operand) (FloatImm, bool) {
|
||||
if len(ops) != 2 {
|
||||
return FloatImm{}, false
|
||||
}
|
||||
f, ok := ops[0].(FloatImm)
|
||||
return f, ok
|
||||
}
|
||||
|
||||
// floatPoolValue evaluates a floating-point immediate at the width its
|
||||
// mnemonic encodes and names the pool constant the toolchain synthesises:
|
||||
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
|
||||
// rounding of the parsed value). The name carries the IEEE-754 bits; the
|
||||
// section holds them little-endian.
|
||||
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
|
||||
v, err := strconv.ParseFloat(f.Text, 64)
|
||||
if err != nil {
|
||||
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
|
||||
}
|
||||
if f.Neg {
|
||||
v = -v
|
||||
}
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
bits = math.Float64bits(v)
|
||||
return bits, fmt.Sprintf("$f64.%016x", bits), nil
|
||||
}
|
||||
bits = uint64(math.Float32bits(float32(v)))
|
||||
return bits, fmt.Sprintf("$f32.%08x", bits), nil
|
||||
}
|
||||
|
||||
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
|
||||
// source. A positive zero needs no memory read: the toolchain emits
|
||||
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
|
||||
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
|
||||
// file-level layout or the linker resolves.
|
||||
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
if bits == 0 {
|
||||
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
|
||||
if err := setRM(i, dst, dst, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
m := sseMoveTable[mnem]
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
|
||||
// a floating-point immediate source: the constant is read from the pool into
|
||||
// the instruction's r/m side (reg = destination), the rewrite go tool asm
|
||||
// performs at the source level.
|
||||
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
|
||||
// spellings end in D, the single spellings in S.
|
||||
func mwidth(mnem string) int {
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
return 8
|
||||
}
|
||||
return 4
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
|
||||
@@ -9,6 +9,9 @@ import (
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// decode encodes an instruction and decodes it back, returning the decoded
|
||||
@@ -1064,3 +1067,157 @@ func TestAdjsp(t *testing.T) {
|
||||
t.Error("ADJSP AX assembled, want an error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
|
||||
// byte for byte against go tool asm: the scalar moves and the scalar
|
||||
// arithmetic read the constant from a synthesised read-only pool symbol
|
||||
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
|
||||
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
|
||||
func TestFloatImmediateGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
|
||||
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
|
||||
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
|
||||
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
|
||||
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
|
||||
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
|
||||
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
|
||||
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
|
||||
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
|
||||
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
|
||||
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
|
||||
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
|
||||
// single spellings; negative zero keeps its sign bit and never takes the
|
||||
// XORPS shortcut.
|
||||
for _, c := range []struct {
|
||||
mnem string
|
||||
imm FloatImm
|
||||
want string
|
||||
}{
|
||||
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
|
||||
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
|
||||
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
|
||||
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
|
||||
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
|
||||
} {
|
||||
_, name, err := floatPoolValue(c.mnem, c.imm)
|
||||
if err != nil {
|
||||
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
|
||||
continue
|
||||
}
|
||||
if name != c.want {
|
||||
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// The shapes the toolchain's parser rejects: the packed and uniform
|
||||
// forms, a non-vector destination, and the integer spellings.
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
|
||||
} {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
|
||||
// go tool asm emits no text bytes for either, on every architecture.
|
||||
func TestBookkeepingGroundTruth(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
|
||||
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
|
||||
} {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if len(code) != 0 {
|
||||
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
|
||||
}
|
||||
}
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
|
||||
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
|
||||
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
|
||||
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
|
||||
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
|
||||
} {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
|
||||
// At the statement level the bookkeeping lines sit between real
|
||||
// instructions and contribute nothing to the body, symbol reference
|
||||
// included: the FUNCDATA operand never needs file-level resolution.
|
||||
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
want := "90c3"
|
||||
if got := hexCompact(img.Code); got != want {
|
||||
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
|
||||
}
|
||||
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
|
||||
t.Error("FUNCDATA $1, X0 assembled, want an error")
|
||||
}
|
||||
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
|
||||
t.Error("PCDATA $1, X0 assembled, want an error")
|
||||
}
|
||||
|
||||
// Encodable mirrors Encode for the names this work touched.
|
||||
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
|
||||
if !Encodable(mnem) {
|
||||
t.Errorf("Encodable(%s) = false, want true", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// mustParse parses src or fails the test.
|
||||
func mustParse(t *testing.T, src string) *ast.File {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("t_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
@@ -810,3 +810,127 @@ func TestAvx512CorpusFamilies(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
|
||||
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
|
||||
// source keeps r/m, the bracketed list's LOW register travels the inverted
|
||||
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
|
||||
// the vector length follows the destination (L'L=512 for the ZMM forms,
|
||||
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
|
||||
// member. The x86 decoder has no view of these forms, so no decode check
|
||||
// runs.
|
||||
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
|
||||
sp := vreg(t, "RSP")
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a9a842411000000"},
|
||||
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f22f4a9a842411000000"},
|
||||
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f25f429a842411000000"},
|
||||
{"V4FMADDPS Z8 dst", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
|
||||
"62727f4a9a842411000000"},
|
||||
{"V4FMADDPS disp8x16", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a9a442404"},
|
||||
{"V4FMADDPS unmasked", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||
"62f27f489a842411000000"},
|
||||
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0d9bb007000000"},
|
||||
{"V4FMADDSS (DI)", "V4FMADDSS",
|
||||
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0d9b37"},
|
||||
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e22f0d9bb007000000"},
|
||||
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e25f059bb007000000"},
|
||||
{"V4FMADDSS X30 dst", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||
"62627f0d9bb007000000"},
|
||||
{"V4FMADDSS X3 dst", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
|
||||
"62f27f0d9b9807000000"},
|
||||
{"V4FMADDSS disp8x16", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||
"62625f059b7001"},
|
||||
{"V4FNMADDPS", "V4FNMADDPS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4aaa842411000000"},
|
||||
{"V4FNMADDSS", "V4FNMADDSS",
|
||||
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||
"62e27f0dabb007000000"},
|
||||
{"VP4DPWSSD", "VP4DPWSSD",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||
"62f27f4a52842411000000"},
|
||||
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
|
||||
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||
"62f27f4853842411000000"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
|
||||
// the register class the list and the destination take is fixed per
|
||||
// instruction, the source is memory only, the opmask slot is positional and
|
||||
// the list's low register owns V'VVVV.
|
||||
func TestEvexQuadRegisterErrors(t *testing.T) {
|
||||
sp := vreg(t, "RSP")
|
||||
list := func(lo, hi string) RegList {
|
||||
return RegList{vreg(t, lo), vreg(t, hi)}
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"X list on the PS form", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"Z list on the SS form", "V4FMADDSS",
|
||||
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
|
||||
{"Y destination", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
|
||||
{"register source", "V4FMADDPS",
|
||||
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"non-mask third operand", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
|
||||
{"k0 mask", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
|
||||
{"K after the destination", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
|
||||
{"zeroing without a mask", "V4FMADDPS.Z",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
|
||||
{"SAE suffix", "V4FMADDPS.SAE",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"high index source", "VP4DPWSSD",
|
||||
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||
{"short operand list", "V4FMADDPS",
|
||||
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,7 +63,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte {
|
||||
}
|
||||
|
||||
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
|
||||
// toolchain object, keyed by the name the object records (pkg.name).
|
||||
// toolchain object, keyed by the name the object records (pkg.name). Each
|
||||
// function's span is its own symbol size: a toolchain object that follows
|
||||
// the text with data symbols (the synthesised float-constant pool) would
|
||||
// otherwise fold them into the last function's bytes.
|
||||
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||
t.Helper()
|
||||
v := openGoobj(t, obj)
|
||||
@@ -76,18 +79,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||
preceding += len(v.blk(bi)) / symSize
|
||||
}
|
||||
total := preceding + len(nps)
|
||||
out := make(map[string][]byte, len(nps))
|
||||
for i, s := range nps {
|
||||
if s.typ != kindSTEXT {
|
||||
continue
|
||||
}
|
||||
start := le.Uint32(didx[4*(preceding+i):])
|
||||
end := uint32(len(data))
|
||||
if preceding+i+1 < total {
|
||||
end = le.Uint32(didx[4*(preceding+i+1):])
|
||||
}
|
||||
out[s.name] = data[start:end]
|
||||
out[s.name] = data[start : start+s.size]
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -129,6 +127,9 @@ func TestDifferentialKernels(t *testing.T) {
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
|
||||
} {
|
||||
|
||||
+21
-1
@@ -167,6 +167,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
known[d.name] = true
|
||||
}
|
||||
link := &linkInfo{symbols: known, allowExternal: true}
|
||||
poolSeen := map[string]bool{}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||
textOff := map[string]int{}
|
||||
@@ -180,7 +181,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||
code, patches, labels, steps, lines, pool, err := assemble(t, link)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
// The pooled floating-point constants join the declared data as
|
||||
// read-only symbols, deduplicated across the file (the toolchain
|
||||
// synthesises the same symbols into its rodata).
|
||||
for _, entry := range pool {
|
||||
if poolSeen[entry.name] {
|
||||
continue
|
||||
}
|
||||
poolSeen[entry.name] = true
|
||||
dataSyms = append(dataSyms, dataSym{
|
||||
name: entry.name,
|
||||
buf: entry.data,
|
||||
size: len(entry.data),
|
||||
rodata: true,
|
||||
dupok: true,
|
||||
})
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
|
||||
Vendored
+32
@@ -0,0 +1,32 @@
|
||||
// The runtime bookkeeping statements: FUNCDATA and PCDATA contribute no
|
||||
// text bytes on any architecture, and amd64 now matches. They sit between
|
||||
// real instructions here, with plain, static and offset symbol references
|
||||
// on the FUNCDATA lines, so the byte counts prove the zero contribution.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func bookkeep(x int64) int64
|
||||
TEXT ·bookkeep(SB), NOSPLIT, $0-16
|
||||
PCDATA $0, $-1
|
||||
MOVQ x+0(FP), AX
|
||||
PCDATA $1, $-2
|
||||
FUNCDATA $0, args_stackmap(SB)
|
||||
ADDQ $1, AX
|
||||
FUNCDATA $5, arginfo0(SB)
|
||||
PCDATA $1, $3
|
||||
MOVQ AX, ret+8(FP)
|
||||
FUNCDATA $1, externalfuncdata(SB)
|
||||
PCDATA $0, $0
|
||||
RET
|
||||
|
||||
// func bookkeepstatic() int64
|
||||
TEXT ·bookkeepstatic(SB), NOSPLIT, $0-8
|
||||
// A static symbol and a defined data symbol as the funcdata target.
|
||||
// (A symbol+offset target the toolchain itself refuses.)
|
||||
FUNCDATA $2, fdtable<>(SB)
|
||||
FUNCDATA $3, undefsym(SB)
|
||||
MOVQ $7, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL fdtable<>(SB), NOPTR, $16
|
||||
Vendored
+55
@@ -0,0 +1,55 @@
|
||||
// Floating-point immediates on the SSE scalar paths: the constant is
|
||||
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
|
||||
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
|
||||
// displacement left to the relocation. A positive zero on the moves
|
||||
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
|
||||
// takes the pool. The parenthesised $(-1.0) spelling is the one
|
||||
// math/floor_amd64.s uses. Every result is folded back so no
|
||||
// instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func floatimm(x float64) float64
|
||||
TEXT ·floatimm(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
MOVQ AX, X0
|
||||
// The floor kernel's sign fold: the parenthesised negative spelling.
|
||||
MOVSD $ (-1.0), X2
|
||||
ANDPD X2, X0
|
||||
// Positive and fractional constants on the scalar moves.
|
||||
MOVSD $1.5, X3
|
||||
MOVSD $0.5, X4
|
||||
MOVSS $2.5, X5
|
||||
MOVSS $-0.5, X6
|
||||
// A positive zero collapses to XORPS; a negative zero does not.
|
||||
MOVSD $0.0, X7
|
||||
MOVSS $0.0, X8
|
||||
MOVSD $-0.0, X9
|
||||
// The scalar arithmetic reads the pool through r/m (hypot's shape).
|
||||
ADDSD $1.0, X3
|
||||
SUBSD $0.5, X4
|
||||
MULSD $-2.5, X4
|
||||
DIVSD $2.0, X3
|
||||
ADDSS $0.25, X5
|
||||
// Fold everything into one double.
|
||||
ADDSD X5, X3
|
||||
ADDSD X6, X3
|
||||
ADDSD X7, X3
|
||||
ADDSD X8, X3
|
||||
ADDSD X9, X3
|
||||
ADDSD X4, X3
|
||||
ADDSD X0, X3
|
||||
MOVSD X3, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func floatimmfloat32() float32
|
||||
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
|
||||
// The single-width pool constants ride the F3 prefix.
|
||||
MOVSS $1.0, X0
|
||||
MOVSS $-1.0, X1
|
||||
MOVSS $0.0, X2
|
||||
ADDSS $0.5, X0
|
||||
ADDSS X1, X0
|
||||
ADDSS X2, X0
|
||||
MOVSS X0, ret+0(FP)
|
||||
RET
|
||||
@@ -124,10 +124,16 @@ func TestGroundTruthAMD64(t *testing.T) {
|
||||
"../testdata/verify/avx_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/vsib_amd64.s",
|
||||
"../testdata/verify/floatimm_amd64.s",
|
||||
"../testdata/verify/bookkeep_amd64.s",
|
||||
"../testdata/verify/quadreg_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/vsib_amd64.s",
|
||||
"../testdata/verify/floatimm_amd64.s",
|
||||
"../testdata/verify/bookkeep_amd64.s",
|
||||
"../testdata/verify/quadreg_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/doubleshift_amd64.s",
|
||||
|
||||
Reference in New Issue
Block a user