Compare commits
2
Commits
e8b6ff5d7c
...
9a5e90ba98
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9a5e90ba98 | ||
|
|
bfb7701db1 |
+138
-31
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
@@ -28,7 +29,7 @@ import (
|
||||
// emitted: the bytes match go tool asm only for NOSPLIT functions or
|
||||
// zero-frame leaves, where the toolchain emits no guard either.
|
||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
code, _, labels, _, _, err := assemble(t, nil)
|
||||
code, _, labels, _, _, _, err := assemble(t, nil)
|
||||
return code, labels, err
|
||||
}
|
||||
|
||||
@@ -66,9 +67,9 @@ type spadjStep struct {
|
||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||
// patch sites (for the file-level layout to resolve), the label table and the
|
||||
// stack-adjustment boundaries.
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) {
|
||||
if err := checkAdjspBalance(t); err != nil {
|
||||
return nil, nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, nil, err
|
||||
}
|
||||
fi := computeFrame(t)
|
||||
chain := jumpChain(t)
|
||||
@@ -88,6 +89,8 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
offsets := map[string]int{}
|
||||
pcs := make([]int, len(t.Body))
|
||||
var guardJBlong, guardJBElong, moreJMPlong bool
|
||||
poolSeen := map[string]bool{}
|
||||
var poolList []floatPoolEntry
|
||||
for {
|
||||
guard := fi.guardLen(guardJBlong, guardJBElong)
|
||||
pos := guard + len(fi.prologue)
|
||||
@@ -98,7 +101,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi, long[i], link)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
sizes[i] = sz
|
||||
pcs[i] = pos
|
||||
@@ -229,12 +232,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
spadjStep{pos + epi, 0},
|
||||
)
|
||||
}
|
||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
for _, entry := range pool {
|
||||
if !poolSeen[entry.name] {
|
||||
poolSeen[entry.name] = true
|
||||
poolList = append(poolList, entry)
|
||||
}
|
||||
}
|
||||
if len(code) != sizes[i] {
|
||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
}
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
|
||||
for k := range ps {
|
||||
@@ -272,7 +281,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
pos += len(suffix)
|
||||
}
|
||||
_ = pos
|
||||
return out, patches, offsets, steps, lines, nil
|
||||
return out, patches, offsets, steps, lines, poolList, nil
|
||||
}
|
||||
|
||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||
@@ -593,7 +602,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
|
||||
}
|
||||
return jumpSize(mnem, long), nil
|
||||
}
|
||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
@@ -628,7 +637,7 @@ func jumpSize(mnem string, long bool) int {
|
||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
||||
// jump label through the jump-to-jump chain before the offset lookup.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
|
||||
var prefix []byte
|
||||
@@ -638,6 +647,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
|
||||
var code []byte
|
||||
var ps []sbPatch
|
||||
var pool []floatPoolEntry
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||
@@ -646,7 +656,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
// or the linker.
|
||||
code, ps, err = encodeSBCall(s, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
for i := range ps {
|
||||
ps[i].kind = RelCall
|
||||
@@ -656,23 +666,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
return append(prefix, code...), ps, nil, nil
|
||||
}
|
||||
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
||||
// JMP/CALL through a register or memory: no relocation and no
|
||||
// label to resolve, the operand fully determines the bytes.
|
||||
code, err = encodeIndirectJump(s, mnem)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
return append(prefix, code...), nil, nil
|
||||
return append(prefix, code...), nil, nil, nil
|
||||
}
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
||||
} else {
|
||||
code, ps, err = encodeNormal(s, fi, link)
|
||||
code, ps, pool, err = encodeNormal(s, fi, link)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
// Anchor the patch fields at function-relative positions: off indexes the
|
||||
// disp32 field, after is the address just past the instruction.
|
||||
@@ -681,31 +691,65 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
ps[i].off += body
|
||||
ps[i].after = body + len(code)
|
||||
}
|
||||
return append(prefix, code...), ps, nil
|
||||
return append(prefix, code...), ps, pool, nil
|
||||
}
|
||||
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||
mnemUpper := strings.ToUpper(s.Mnemonic.Text)
|
||||
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
|
||||
code, err := encodeBookkeeping(mnemUpper, s)
|
||||
if err != nil {
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
return code, nil, nil, nil
|
||||
}
|
||||
_, size := splitSize(mnemUpper)
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, size, fi, link)
|
||||
o, err := operandFromAST(mnemUpper, op, size, fi, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
ops[i] = o
|
||||
}
|
||||
e := &enc{}
|
||||
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
||||
return nil, nil, err
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
ps := make([]sbPatch, len(e.patches))
|
||||
for i, p := range e.patches {
|
||||
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||
}
|
||||
return e.out, ps, nil
|
||||
return e.out, ps, e.floatPoolList(), nil
|
||||
}
|
||||
|
||||
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
|
||||
// level, before operand conversion: the toolchain's shapes are FUNCDATA
|
||||
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
|
||||
// function body. The symbol reference must not run through the SB-operand
|
||||
// path, which demands file-level resolution the statement never needs.
|
||||
func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
|
||||
if len(s.Operands) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands))
|
||||
}
|
||||
a, b := s.Operands[0], s.Operands[1]
|
||||
if a.Kind != ast.OpImmediate || !a.Imm.HasVal {
|
||||
return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||
}
|
||||
switch upper {
|
||||
case "FUNCDATA":
|
||||
if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" {
|
||||
return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||
}
|
||||
case "PCDATA":
|
||||
if b.Kind != ast.OpImmediate || !b.Imm.HasVal {
|
||||
return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||
}
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
||||
@@ -756,7 +800,7 @@ func isSBCall(s *ast.Instr) bool {
|
||||
|
||||
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
|
||||
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
|
||||
o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
@@ -813,7 +857,7 @@ func indirectJumpTarget(s *ast.Instr) bool {
|
||||
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, 8, frameInfo{}, nil)
|
||||
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -830,8 +874,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||
var spReg = Reg{idx: 4, size: 8}
|
||||
|
||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||
// the frame translation to FP/SP pseudo-register operands.
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
// the frame translation to FP/SP pseudo-register operands. mnemUpper is the
|
||||
// instruction's upper-case mnemonic, which the floating-point immediate gate
|
||||
// needs: only the SSE mnemonics whose encoding takes an XMM/memory source
|
||||
// accept one.
|
||||
func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||
switch op.Kind {
|
||||
case ast.OpImmediate:
|
||||
if op.Imm.HasVal {
|
||||
@@ -841,17 +888,43 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
||||
}
|
||||
return Imm(v), nil
|
||||
}
|
||||
// A floating-point immediate: $1.5, $-1.0 or the parenthesised
|
||||
// $(-1.0) spelling (the constant-expression folder only folds
|
||||
// integers, so that shape arrives with an empty Immediate and only
|
||||
// the raw spelling carries the value). The toolchain rewrites it
|
||||
// into a pooled-constant read on the SSE scalar paths and rejects
|
||||
// it everywhere else.
|
||||
if text, neg, ok := floatImmText(op); ok {
|
||||
if !sseFloatImm[mnemUpper] {
|
||||
return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper)
|
||||
}
|
||||
return FloatImm{Text: text, Neg: neg}, nil
|
||||
}
|
||||
return nil, fmt.Errorf("non-integer immediate not supported")
|
||||
|
||||
case ast.OpAddr:
|
||||
a := op.Addr
|
||||
|
||||
// A bracketed register range, [Z0-Z3]: the four-register source of
|
||||
// the 4FMAPS/4VNNIW families. The EVEX quad-register emit path
|
||||
// needs an encoder operand of its own, so the shape stays a named
|
||||
// gap rather than an encoding.
|
||||
// the 4FMAPS/4VNNIW families. The range must span four consecutive
|
||||
// same-width vector registers, exactly what the toolchain's parser
|
||||
// takes; the EVEX quad-register emit path reads the low end.
|
||||
if a.Range != nil {
|
||||
return nil, fmt.Errorf("register range %q needs quad-register encoder support", op.Raw)
|
||||
lo, ok := ParseReg(a.Range.Lo)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo)
|
||||
}
|
||||
hi, ok := ParseReg(a.Range.Hi)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi)
|
||||
}
|
||||
if !lo.isVec() || lo.size != hi.size {
|
||||
return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw)
|
||||
}
|
||||
if hi.idx != lo.idx+3 {
|
||||
return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw)
|
||||
}
|
||||
return RegList{Lo: lo, Hi: hi}, nil
|
||||
}
|
||||
|
||||
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
||||
@@ -921,3 +994,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
||||
}
|
||||
return nil, fmt.Errorf("unsupported operand")
|
||||
}
|
||||
|
||||
// floatImmText recovers a floating-point immediate's magnitude and sign from
|
||||
// the parsed operand. The ordinary spellings arrive in Imm.Float; the
|
||||
// parenthesised $(-1.0) leaves the Immediate empty, because the integer
|
||||
// folder cannot read it, and only the verbatim operand text still carries
|
||||
// the value. Anything that is not a number a float parser accepts reports
|
||||
// not-ok, so every other shape keeps its existing diagnostic.
|
||||
func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) {
|
||||
if op.Imm.Float != "" {
|
||||
return op.Imm.Float, op.Imm.Neg, true
|
||||
}
|
||||
if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil {
|
||||
return "", false, false
|
||||
}
|
||||
// joinRaw spaced the token texts; the compact spelling is what matters.
|
||||
compact := strings.ReplaceAll(op.Raw, " ", "")
|
||||
inner, ok := strings.CutPrefix(compact, "$(")
|
||||
if !ok || !strings.HasSuffix(inner, ")") {
|
||||
return "", false, false
|
||||
}
|
||||
inner = strings.TrimSuffix(inner, ")")
|
||||
inner = strings.TrimPrefix(inner, "+")
|
||||
if s, ok := strings.CutPrefix(inner, "-"); ok {
|
||||
neg = true
|
||||
inner = s
|
||||
}
|
||||
if inner == "" || !strings.ContainsAny(inner, "0123456789") {
|
||||
return "", false, false
|
||||
}
|
||||
if _, err := strconv.ParseFloat(inner, 64); err != nil {
|
||||
return "", false, false
|
||||
}
|
||||
return inner, neg, true
|
||||
}
|
||||
|
||||
@@ -568,3 +568,28 @@ TEXT ·framed(SB), $16-8
|
||||
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleRegRange pins the bracketed register range at the statement
|
||||
// level: exactly four consecutive same-width vector registers assemble, the
|
||||
// toolchain's rejected shapes all report an error.
|
||||
func TestAssembleRegRange(t *testing.T) {
|
||||
asm := func(t *testing.T, op string) ([]byte, error) {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", op, errs)
|
||||
}
|
||||
code, _, err := Assemble(f.Decls[0].(*ast.Text))
|
||||
return code, err
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
|
||||
if _, err := asm(t, op); err != nil {
|
||||
t.Errorf("%s: %v", op, err)
|
||||
}
|
||||
}
|
||||
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
|
||||
if _, err := asm(t, op); err == nil {
|
||||
t.Errorf("%s: assembled, want an error", op)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+3
-3
@@ -20,9 +20,9 @@ func Encodable(mnemonic string) bool {
|
||||
switch upper {
|
||||
case "RET", "NOP", "CALL", "JMP",
|
||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and the
|
||||
// SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
|
||||
// The literal-data pseudo-ops, the accepted-and-ignored END and
|
||||
// bookkeeping statements, and the SP adjust.
|
||||
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
|
||||
return true
|
||||
}
|
||||
if _, ok := noOperandTable[upper]; ok {
|
||||
|
||||
+195
-2
@@ -5,6 +5,8 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
type enc struct {
|
||||
out []byte
|
||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||
|
||||
// FloatPool collects the pooled constants the floating-point
|
||||
// immediates reference, in first-use order.
|
||||
floatPool []floatPoolEntry
|
||||
floatPoolSeen map[string]bool
|
||||
}
|
||||
|
||||
// floatPoolEntry is one pooled floating-point constant: the symbol name
|
||||
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
|
||||
type floatPoolEntry struct {
|
||||
name string
|
||||
data []byte
|
||||
}
|
||||
|
||||
// addFloatPool records a pooled constant, deduplicated by symbol name.
|
||||
func (e *enc) addFloatPool(name string, bits uint64, width int) {
|
||||
if e.floatPoolSeen == nil {
|
||||
e.floatPoolSeen = map[string]bool{}
|
||||
}
|
||||
if e.floatPoolSeen[name] {
|
||||
return
|
||||
}
|
||||
e.floatPoolSeen[name] = true
|
||||
data := make([]byte, width)
|
||||
for i := range width {
|
||||
data[i] = byte(bits >> (8 * i))
|
||||
}
|
||||
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
|
||||
}
|
||||
|
||||
// floatPoolList returns the pooled constants in first-use order.
|
||||
func (e *enc) floatPoolList() []floatPoolEntry {
|
||||
return e.floatPool
|
||||
}
|
||||
|
||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||
@@ -102,6 +137,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeEnd(ops)
|
||||
case "ADJSP":
|
||||
return e.encodeAdjsp(ops)
|
||||
// The runtime's bookkeeping statements carry no text bytes: go tool asm
|
||||
// records FUNCDATA and PCDATA in the program list only, so the encoded
|
||||
// body shows nothing, on every architecture.
|
||||
case "FUNCDATA", "PCDATA":
|
||||
return e.encodeFuncdata(upper, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||
@@ -141,11 +181,18 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
}
|
||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||
// which the size split must not eat.
|
||||
// which the size split must not eat. A floating-point immediate
|
||||
// rewrites into a pooled-constant read on the scalar members.
|
||||
if m, ok := sseBinTable[upper]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
if m, ok := sseBinTable[base]; ok {
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||
}
|
||||
return e.encodeSSEBin(m, ops)
|
||||
}
|
||||
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||
@@ -222,7 +269,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeCvtInt(base, ops, size)
|
||||
case "FMOVD":
|
||||
return e.encodeFmov(ops)
|
||||
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
case "MOVSD", "MOVSS":
|
||||
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||
return e.encodeSSEFloatMove(upper, f, ops)
|
||||
}
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
@@ -285,6 +337,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
|
||||
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
|
||||
// for either (the entries live in the object's ancillary tables, not the
|
||||
// function body), and the operand shapes it takes are exactly these: an
|
||||
// integer count first, then a symbol reference for FUNCDATA and an integer
|
||||
// value for PCDATA. The other architectures accept-and-ignore the same
|
||||
// statements; amd64 now matches.
|
||||
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
if _, ok := ops[0].(Imm); !ok {
|
||||
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||
}
|
||||
switch upper {
|
||||
case "FUNCDATA":
|
||||
if _, ok := ops[1].(sbMem); !ok {
|
||||
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||
}
|
||||
case "PCDATA":
|
||||
if _, ok := ops[1].(Imm); !ok {
|
||||
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||
// the statements after an END still belong to the same function and the
|
||||
@@ -319,6 +398,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// --- floating-point immediates ----------------------------------------------
|
||||
|
||||
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
|
||||
// immediate, the set go tool asm rewrites into a pooled-constant read: the
|
||||
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
|
||||
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
|
||||
// reject the immediate in the toolchain and are absent here on purpose.
|
||||
var sseFloatImm = map[string]bool{
|
||||
"MOVSD": true, "MOVSS": true,
|
||||
"ADDSD": true, "ADDSS": true,
|
||||
"SUBSD": true, "SUBSS": true,
|
||||
"MULSD": true, "MULSS": true,
|
||||
"DIVSD": true, "DIVSS": true,
|
||||
"COMISD": true, "COMISS": true,
|
||||
"UCOMISD": true, "UCOMISS": true,
|
||||
}
|
||||
|
||||
// floatImmOperand reports whether the operand list opens with a
|
||||
// floating-point immediate in the two-operand spelling (imm, dst).
|
||||
func floatImmOperand(ops []Operand) (FloatImm, bool) {
|
||||
if len(ops) != 2 {
|
||||
return FloatImm{}, false
|
||||
}
|
||||
f, ok := ops[0].(FloatImm)
|
||||
return f, ok
|
||||
}
|
||||
|
||||
// floatPoolValue evaluates a floating-point immediate at the width its
|
||||
// mnemonic encodes and names the pool constant the toolchain synthesises:
|
||||
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
|
||||
// rounding of the parsed value). The name carries the IEEE-754 bits; the
|
||||
// section holds them little-endian.
|
||||
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
|
||||
v, err := strconv.ParseFloat(f.Text, 64)
|
||||
if err != nil {
|
||||
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
|
||||
}
|
||||
if f.Neg {
|
||||
v = -v
|
||||
}
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
bits = math.Float64bits(v)
|
||||
return bits, fmt.Sprintf("$f64.%016x", bits), nil
|
||||
}
|
||||
bits = uint64(math.Float32bits(float32(v)))
|
||||
return bits, fmt.Sprintf("$f32.%08x", bits), nil
|
||||
}
|
||||
|
||||
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
|
||||
// source. A positive zero needs no memory read: the toolchain emits
|
||||
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
|
||||
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
|
||||
// file-level layout or the linker resolves.
|
||||
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
if bits == 0 {
|
||||
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
|
||||
if err := setRM(i, dst, dst, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
m := sseMoveTable[mnem]
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
|
||||
// a floating-point immediate source: the constant is read from the pool into
|
||||
// the instruction's r/m side (reg = destination), the rewrite go tool asm
|
||||
// performs at the source level.
|
||||
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
|
||||
if !sseFloatImm[mnem] {
|
||||
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||
}
|
||||
dst, ok := ops[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
bits, name, err := floatPoolValue(mnem, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.addFloatPool(name, bits, mwidth(mnem))
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
|
||||
// spellings end in D, the single spellings in S.
|
||||
func mwidth(mnem string) int {
|
||||
if strings.HasSuffix(mnem, "D") {
|
||||
return 8
|
||||
}
|
||||
return 4
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
|
||||
@@ -9,6 +9,9 @@ import (
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// decode encodes an instruction and decodes it back, returning the decoded
|
||||
@@ -1064,3 +1067,157 @@ func TestAdjsp(t *testing.T) {
|
||||
t.Error("ADJSP AX assembled, want an error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
|
||||
// byte for byte against go tool asm: the scalar moves and the scalar
|
||||
// arithmetic read the constant from a synthesised read-only pool symbol
|
||||
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
|
||||
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
|
||||
func TestFloatImmediateGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
|
||||
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
|
||||
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
|
||||
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
|
||||
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
|
||||
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
|
||||
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
|
||||
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
|
||||
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
|
||||
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
|
||||
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
|
||||
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
|
||||
// single spellings; negative zero keeps its sign bit and never takes the
|
||||
// XORPS shortcut.
|
||||
for _, c := range []struct {
|
||||
mnem string
|
||||
imm FloatImm
|
||||
want string
|
||||
}{
|
||||
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
|
||||
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
|
||||
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
|
||||
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
|
||||
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
|
||||
} {
|
||||
_, name, err := floatPoolValue(c.mnem, c.imm)
|
||||
if err != nil {
|
||||
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
|
||||
continue
|
||||
}
|
||||
if name != c.want {
|
||||
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
|
||||
}
|
||||
}
|
||||
|
||||
// The shapes the toolchain's parser rejects: the packed and uniform
|
||||
// forms, a non-vector destination, and the integer spellings.
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
|
||||
} {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
|
||||
// go tool asm emits no text bytes for either, on every architecture.
|
||||
func TestBookkeepingGroundTruth(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
|
||||
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
|
||||
} {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if len(code) != 0 {
|
||||
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
|
||||
}
|
||||
}
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
|
||||
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
|
||||
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
|
||||
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
|
||||
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
|
||||
} {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
|
||||
// At the statement level the bookkeeping lines sit between real
|
||||
// instructions and contribute nothing to the body, symbol reference
|
||||
// included: the FUNCDATA operand never needs file-level resolution.
|
||||
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("assemble: %v", err)
|
||||
}
|
||||
want := "90c3"
|
||||
if got := hexCompact(img.Code); got != want {
|
||||
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
|
||||
}
|
||||
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
|
||||
t.Error("FUNCDATA $1, X0 assembled, want an error")
|
||||
}
|
||||
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
|
||||
t.Error("PCDATA $1, X0 assembled, want an error")
|
||||
}
|
||||
|
||||
// Encodable mirrors Encode for the names this work touched.
|
||||
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
|
||||
if !Encodable(mnem) {
|
||||
t.Errorf("Encodable(%s) = false, want true", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// mustParse parses src or fails the test.
|
||||
func mustParse(t *testing.T, src string) *ast.File {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("t_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
+97
-2
@@ -737,6 +737,91 @@ var evexTable = map[string]evexSpec{
|
||||
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
}
|
||||
|
||||
// evexQuad describes one quad-register instruction: the opcode under
|
||||
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
|
||||
// registers the bracketed list and the destination take (512-bit ZMM for
|
||||
// the packed forms, 128-bit XMM for the scalar ones).
|
||||
type evexQuad struct {
|
||||
opcode byte
|
||||
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
|
||||
}
|
||||
|
||||
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
|
||||
// families) to their encoding. The operand shape is fixed: a single memory
|
||||
// source in r/m, the bracketed register list whose LOW register travels the
|
||||
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
|
||||
// destination in reg. The vector length follows the destination (512-bit
|
||||
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
|
||||
// multiplier stays 16 for every member, the toolchain's own tuple choice.
|
||||
var evexQuadTable = map[string]evexQuad{
|
||||
"V4FMADDPS": {0x9A, 64},
|
||||
"V4FMADDSS": {0x9B, 16},
|
||||
"V4FNMADDPS": {0xAA, 64},
|
||||
"V4FNMADDSS": {0xAB, 16},
|
||||
"VP4DPWSSD": {0x52, 64},
|
||||
"VP4DPWSSDS": {0x53, 64},
|
||||
}
|
||||
|
||||
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
|
||||
func isEvexQuad(upper string) bool {
|
||||
_, ok := evexQuadTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
|
||||
// The register list is the VVVV-side source: its low register fills the
|
||||
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
|
||||
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
|
||||
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
|
||||
// suffix applies.
|
||||
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
|
||||
if len(ops) != 3 && len(ops) != 4 {
|
||||
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
|
||||
}
|
||||
mem, lst := ops[0], ops[1]
|
||||
dst := ops[len(ops)-1]
|
||||
mask := 0
|
||||
if len(ops) == 4 {
|
||||
k, ok := ops[2].(Reg)
|
||||
if !ok || !k.mask {
|
||||
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
|
||||
}
|
||||
if k.idx == 0 {
|
||||
return fmt.Errorf("k0 is not a usable mask register")
|
||||
}
|
||||
mask = k.idx
|
||||
}
|
||||
list, ok := lst.(RegList)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
|
||||
}
|
||||
if list.Lo.size != q.width {
|
||||
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
if dstReg.size != q.width {
|
||||
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
|
||||
}
|
||||
if !memOperand(mem) {
|
||||
return fmt.Errorf("%s: the source must be a memory operand", mnem)
|
||||
}
|
||||
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
|
||||
// its fifth bit into the same field the list's low register occupies.
|
||||
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
|
||||
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
|
||||
}
|
||||
if sfx.zeroing && mask == 0 {
|
||||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
|
||||
}
|
||||
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
|
||||
// The vector length follows the destination (512-bit for the ZMM forms,
|
||||
// 128-bit for the scalar ones), exactly as the oracle encodes it.
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
|
||||
}
|
||||
|
||||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||||
// depends on the source kind, a GPR source uses opReg, a memory source uses
|
||||
// opMem with a disp8×N of n.
|
||||
@@ -812,8 +897,10 @@ func isEvex(mnemUpper string) bool {
|
||||
if _, ok := evexBcastTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
_, ok := evexMoveTable[mnemUpper]
|
||||
return ok
|
||||
if _, ok := evexMoveTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
return isEvexQuad(mnemUpper)
|
||||
}
|
||||
|
||||
// evexRequired reports whether the operands force the EVEX encoding of a
|
||||
@@ -995,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||
}
|
||||
spec, inTable := evexTable[mnemUpper]
|
||||
if q, ok := evexQuadTable[mnemUpper]; ok {
|
||||
// The quad-register family carries no rounding, SAE or broadcast;
|
||||
// only masking and zeroing apply.
|
||||
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
|
||||
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
|
||||
}
|
||||
if inTable {
|
||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
||||
|
||||
@@ -63,7 +63,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte {
|
||||
}
|
||||
|
||||
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
|
||||
// toolchain object, keyed by the name the object records (pkg.name).
|
||||
// toolchain object, keyed by the name the object records (pkg.name). Each
|
||||
// function's span is its own symbol size: a toolchain object that follows
|
||||
// the text with data symbols (the synthesised float-constant pool) would
|
||||
// otherwise fold them into the last function's bytes.
|
||||
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||
t.Helper()
|
||||
v := openGoobj(t, obj)
|
||||
@@ -76,18 +79,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||
preceding += len(v.blk(bi)) / symSize
|
||||
}
|
||||
total := preceding + len(nps)
|
||||
out := make(map[string][]byte, len(nps))
|
||||
for i, s := range nps {
|
||||
if s.typ != kindSTEXT {
|
||||
continue
|
||||
}
|
||||
start := le.Uint32(didx[4*(preceding+i):])
|
||||
end := uint32(len(data))
|
||||
if preceding+i+1 < total {
|
||||
end = le.Uint32(didx[4*(preceding+i+1):])
|
||||
}
|
||||
out[s.name] = data[start:end]
|
||||
out[s.name] = data[start : start+s.size]
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -129,6 +127,9 @@ func TestDifferentialKernels(t *testing.T) {
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
|
||||
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
|
||||
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
|
||||
} {
|
||||
|
||||
+21
-1
@@ -167,6 +167,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
known[d.name] = true
|
||||
}
|
||||
link := &linkInfo{symbols: known, allowExternal: true}
|
||||
poolSeen := map[string]bool{}
|
||||
|
||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||
textOff := map[string]int{}
|
||||
@@ -180,7 +181,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
||||
code, patches, labels, steps, lines, pool, err := assemble(t, link)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
// The pooled floating-point constants join the declared data as
|
||||
// read-only symbols, deduplicated across the file (the toolchain
|
||||
// synthesises the same symbols into its rodata).
|
||||
for _, entry := range pool {
|
||||
if poolSeen[entry.name] {
|
||||
continue
|
||||
}
|
||||
poolSeen[entry.name] = true
|
||||
dataSyms = append(dataSyms, dataSym{
|
||||
name: entry.name,
|
||||
buf: entry.data,
|
||||
size: len(entry.data),
|
||||
rodata: true,
|
||||
dupok: true,
|
||||
})
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||
}
|
||||
|
||||
@@ -14,6 +14,28 @@ type Imm int64
|
||||
|
||||
func (Imm) isOperand() {}
|
||||
|
||||
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
|
||||
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
|
||||
// low register through the inverted 5-bit V'VVVV field; the three higher
|
||||
// registers are implied by the instruction, so only the pair travels here.
|
||||
type RegList struct {
|
||||
Lo Reg
|
||||
Hi Reg // implied by the encoding; Lo.idx+3 by construction
|
||||
}
|
||||
|
||||
func (RegList) isOperand() {}
|
||||
|
||||
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
|
||||
// encoding takes an XMM/memory source at that position rewrite it as a read
|
||||
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
|
||||
// own behaviour; every other instruction rejects it.
|
||||
type FloatImm struct {
|
||||
Text string // the numeric text as written, sign excluded
|
||||
Neg bool // a leading minus
|
||||
}
|
||||
|
||||
func (FloatImm) isOperand() {}
|
||||
|
||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||
type Mem struct {
|
||||
Base Reg
|
||||
|
||||
Vendored
+32
@@ -0,0 +1,32 @@
|
||||
// The runtime bookkeeping statements: FUNCDATA and PCDATA contribute no
|
||||
// text bytes on any architecture, and amd64 now matches. They sit between
|
||||
// real instructions here, with plain, static and offset symbol references
|
||||
// on the FUNCDATA lines, so the byte counts prove the zero contribution.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func bookkeep(x int64) int64
|
||||
TEXT ·bookkeep(SB), NOSPLIT, $0-16
|
||||
PCDATA $0, $-1
|
||||
MOVQ x+0(FP), AX
|
||||
PCDATA $1, $-2
|
||||
FUNCDATA $0, args_stackmap(SB)
|
||||
ADDQ $1, AX
|
||||
FUNCDATA $5, arginfo0(SB)
|
||||
PCDATA $1, $3
|
||||
MOVQ AX, ret+8(FP)
|
||||
FUNCDATA $1, externalfuncdata(SB)
|
||||
PCDATA $0, $0
|
||||
RET
|
||||
|
||||
// func bookkeepstatic() int64
|
||||
TEXT ·bookkeepstatic(SB), NOSPLIT, $0-8
|
||||
// A static symbol and a defined data symbol as the funcdata target.
|
||||
// (A symbol+offset target the toolchain itself refuses.)
|
||||
FUNCDATA $2, fdtable<>(SB)
|
||||
FUNCDATA $3, undefsym(SB)
|
||||
MOVQ $7, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL fdtable<>(SB), NOPTR, $16
|
||||
Vendored
+55
@@ -0,0 +1,55 @@
|
||||
// Floating-point immediates on the SSE scalar paths: the constant is
|
||||
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
|
||||
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
|
||||
// displacement left to the relocation. A positive zero on the moves
|
||||
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
|
||||
// takes the pool. The parenthesised $(-1.0) spelling is the one
|
||||
// math/floor_amd64.s uses. Every result is folded back so no
|
||||
// instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func floatimm(x float64) float64
|
||||
TEXT ·floatimm(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
MOVQ AX, X0
|
||||
// The floor kernel's sign fold: the parenthesised negative spelling.
|
||||
MOVSD $ (-1.0), X2
|
||||
ANDPD X2, X0
|
||||
// Positive and fractional constants on the scalar moves.
|
||||
MOVSD $1.5, X3
|
||||
MOVSD $0.5, X4
|
||||
MOVSS $2.5, X5
|
||||
MOVSS $-0.5, X6
|
||||
// A positive zero collapses to XORPS; a negative zero does not.
|
||||
MOVSD $0.0, X7
|
||||
MOVSS $0.0, X8
|
||||
MOVSD $-0.0, X9
|
||||
// The scalar arithmetic reads the pool through r/m (hypot's shape).
|
||||
ADDSD $1.0, X3
|
||||
SUBSD $0.5, X4
|
||||
MULSD $-2.5, X4
|
||||
DIVSD $2.0, X3
|
||||
ADDSS $0.25, X5
|
||||
// Fold everything into one double.
|
||||
ADDSD X5, X3
|
||||
ADDSD X6, X3
|
||||
ADDSD X7, X3
|
||||
ADDSD X8, X3
|
||||
ADDSD X9, X3
|
||||
ADDSD X4, X3
|
||||
ADDSD X0, X3
|
||||
MOVSD X3, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func floatimmfloat32() float32
|
||||
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
|
||||
// The single-width pool constants ride the F3 prefix.
|
||||
MOVSS $1.0, X0
|
||||
MOVSS $-1.0, X1
|
||||
MOVSS $0.0, X2
|
||||
ADDSS $0.5, X0
|
||||
ADDSS X1, X0
|
||||
ADDSS X2, X0
|
||||
MOVSS X0, ret+0(FP)
|
||||
RET
|
||||
Vendored
+44
@@ -0,0 +1,44 @@
|
||||
// The quad-register instructions: the 4FMAPS family (V4FMADDPS,
|
||||
// V4FMADDSS, V4FNMADDPS, V4FNMADDSS) and the 4VNNIW pair (VP4DPWSSD,
|
||||
// VP4DPWSSDS). The bracketed list's low register travels the inverted
|
||||
// V'VVVV field, the memory source keeps r/m, the opmask rides aaa and the
|
||||
// vector length follows the destination (512-bit for the ZMM forms,
|
||||
// 128-bit for the scalar ones) while the disp8xN multiplier stays 16 for
|
||||
// every member. Every result is folded back so no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func quadf4(src *[16]uint32, n int) float32
|
||||
TEXT ·quadf4(SB), NOSPLIT, $0-20
|
||||
MOVQ src+0(FP), SI
|
||||
MOVQ n+8(FP), CX
|
||||
// The packed 4-FMA form over four consecutive ZMM accumulators,
|
||||
// masked with K2, K3 and unmasked alike; the displacements exercise
|
||||
// the disp32 form and the disp8x16 compressed form.
|
||||
V4FMADDPS 17(SI), [Z0-Z3], K2, Z0
|
||||
V4FMADDPS 64(SI), [Z10-Z13], K2, Z1
|
||||
V4FMADDPS (SI), [Z20-Z23], Z2
|
||||
V4FNMADDPS 96(SI), [Z1-Z4], K3, Z5
|
||||
// The scalar form reads XMM lists and takes the 128-bit length; the
|
||||
// displacement compresses by 16.
|
||||
V4FMADDSS 7(AX), [X0-X3], K5, X22
|
||||
V4FMADDSS (DI), [X10-X13], K5, X23
|
||||
V4FNMADDSS 16(SI), [X20-X23], K1, X24
|
||||
// The 4-VNNI dot products, indexed source included.
|
||||
VP4DPWSSD 15(DX)(BX*8), [Z2-Z5], K4, Z17
|
||||
VP4DPWSSDS -7(DI)(R8*1), [Z4-Z7], K1, Z31
|
||||
VP4DPWSSD (SI), [Z12-Z15], Z6
|
||||
// Zeroing keeps the usual rule: a mask register must ride along.
|
||||
V4FMADDPS.Z 128(SI), [Z24-Z27], K4, Z3
|
||||
// Fold every accumulator into one scalar.
|
||||
VPADDD Z0, Z1, Z9
|
||||
VPADDD Z2, Z5, Z10
|
||||
VPADDD Z9, Z17, Z11
|
||||
VPADDD Z10, Z31, Z12
|
||||
VPADDD Z11, Z12, Z13
|
||||
VPADDD Z13, Z14, Z15
|
||||
VADDSS X22, X23, X0
|
||||
VADDSS X24, X0, X1
|
||||
VADDSS X1, X2, X3
|
||||
VMOVSS X3, ret+16(FP)
|
||||
RET
|
||||
@@ -124,10 +124,16 @@ func TestGroundTruthAMD64(t *testing.T) {
|
||||
"../testdata/verify/avx_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/vsib_amd64.s",
|
||||
"../testdata/verify/floatimm_amd64.s",
|
||||
"../testdata/verify/bookkeep_amd64.s",
|
||||
"../testdata/verify/quadreg_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/vsib_amd64.s",
|
||||
"../testdata/verify/floatimm_amd64.s",
|
||||
"../testdata/verify/bookkeep_amd64.s",
|
||||
"../testdata/verify/quadreg_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/doubleshift_amd64.s",
|
||||
|
||||
Reference in New Issue
Block a user