diff --git a/asm/assemble.go b/asm/assemble.go index 6def8ea..12bdcc4 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -5,6 +5,7 @@ package asm import ( "fmt" + "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -28,7 +29,7 @@ import ( // emitted: the bytes match go tool asm only for NOSPLIT functions or // zero-frame leaves, where the toolchain emits no guard either. func Assemble(t *ast.Text) ([]byte, map[string]int, error) { - code, _, labels, _, _, err := assemble(t, nil) + code, _, labels, _, _, _, err := assemble(t, nil) return code, labels, err } @@ -66,9 +67,9 @@ type spadjStep struct { // assemble encodes a TEXT body, returning the machine code, the static-symbol // patch sites (for the file-level layout to resolve), the label table and the // stack-adjustment boundaries. -func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) { +func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) { if err := checkAdjspBalance(t); err != nil { - return nil, nil, nil, nil, nil, err + return nil, nil, nil, nil, nil, nil, err } fi := computeFrame(t) chain := jumpChain(t) @@ -88,6 +89,8 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ offsets := map[string]int{} pcs := make([]int, len(t.Body)) var guardJBlong, guardJBElong, moreJMPlong bool + poolSeen := map[string]bool{} + var poolList []floatPoolEntry for { guard := fi.guardLen(guardJBlong, guardJBElong) pos := guard + len(fi.prologue) @@ -98,7 +101,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ case *ast.Instr: sz, err := instrSize(s, fi, long[i], link) if err != nil { - return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) + return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) } sizes[i] = sz pcs[i] = pos @@ -229,12 +232,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ spadjStep{pos + epi, 0}, ) } - code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) + code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) if err != nil { - return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) + return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) + } + for _, entry := range pool { + if !poolSeen[entry.name] { + poolSeen[entry.name] = true + poolList = append(poolList, entry) + } } if len(code) != sizes[i] { - return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i]) + return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i]) } if strings.ToUpper(s.Mnemonic.Text) == "CALL" { for k := range ps { @@ -272,7 +281,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ pos += len(suffix) } _ = pos - return out, patches, offsets, steps, lines, nil + return out, patches, offsets, steps, lines, poolList, nil } // jumpChain precomputes jump-to-jump folding: a label whose first instruction @@ -593,7 +602,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro } return jumpSize(mnem, long), nil } - code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) + code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) if err != nil { return 0, err } @@ -628,7 +637,7 @@ func jumpSize(mnem string, long bool) int { // (relative to pc, the instruction's own offset). A RET in a frame-pointer // function is prefixed with the epilogue. resolve, when non-nil, redirects a // jump label through the jump-to-jump chain before the offset lookup. -func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) { +func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { mnem := strings.ToUpper(s.Mnemonic.Text) var prefix []byte @@ -638,6 +647,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon var code []byte var ps []sbPatch + var pool []floatPoolEntry var err error if isJumpMnemonic(mnem) { if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { @@ -646,7 +656,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon // or the linker. code, ps, err = encodeSBCall(s, link) if err != nil { - return nil, nil, err + return nil, nil, nil, err } for i := range ps { ps[i].kind = RelCall @@ -656,23 +666,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon ps[i].off += body ps[i].after = body + len(code) } - return append(prefix, code...), ps, nil + return append(prefix, code...), ps, nil, nil } if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) { // JMP/CALL through a register or memory: no relocation and no // label to resolve, the operand fully determines the bytes. code, err = encodeIndirectJump(s, mnem) if err != nil { - return nil, nil, err + return nil, nil, nil, err } - return append(prefix, code...), nil, nil + return append(prefix, code...), nil, nil, nil } code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) } else { - code, ps, err = encodeNormal(s, fi, link) + code, ps, pool, err = encodeNormal(s, fi, link) } if err != nil { - return nil, nil, err + return nil, nil, nil, err } // Anchor the patch fields at function-relative positions: off indexes the // disp32 field, after is the address just past the instruction. @@ -681,31 +691,65 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon ps[i].off += body ps[i].after = body + len(code) } - return append(prefix, code...), ps, nil + return append(prefix, code...), ps, pool, nil } -func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) { - _, size := splitSize(strings.ToUpper(s.Mnemonic.Text)) +func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { + mnemUpper := strings.ToUpper(s.Mnemonic.Text) + if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" { + code, err := encodeBookkeeping(mnemUpper, s) + if err != nil { + return nil, nil, nil, err + } + return code, nil, nil, nil + } + _, size := splitSize(mnemUpper) if size == 0 { size = 8 } ops := make([]Operand, len(s.Operands)) for i, op := range s.Operands { - o, err := operandFromAST(op, size, fi, link) + o, err := operandFromAST(mnemUpper, op, size, fi, link) if err != nil { - return nil, nil, err + return nil, nil, nil, err } ops[i] = o } e := &enc{} if err := e.encode(s.Mnemonic.Text, ops); err != nil { - return nil, nil, err + return nil, nil, nil, err } ps := make([]sbPatch, len(e.patches)) for i, p := range e.patches { ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend} } - return e.out, ps, nil + return e.out, ps, e.floatPoolList(), nil +} + +// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement +// level, before operand conversion: the toolchain's shapes are FUNCDATA +// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the +// function body. The symbol reference must not run through the SB-operand +// path, which demands file-level resolution the statement never needs. +func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) { + if len(s.Operands) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands)) + } + a, b := s.Operands[0], s.Operands[1] + if a.Kind != ast.OpImmediate || !a.Imm.HasVal { + return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper) + } + switch upper { + case "FUNCDATA": + if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" { + return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference") + } + case "PCDATA": + if b.Kind != ast.OpImmediate || !b.Imm.HasVal { + return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate") + } + } + return nil, nil } // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the @@ -756,7 +800,7 @@ func isSBCall(s *ast.Instr) bool { // encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site. func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) { - o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link) + o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link) if err != nil { return nil, nil, err } @@ -813,7 +857,7 @@ func indirectJumpTarget(s *ast.Instr) bool { func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { ops := make([]Operand, len(s.Operands)) for i, op := range s.Operands { - o, err := operandFromAST(op, 8, frameInfo{}, nil) + o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) if err != nil { return nil, err } @@ -830,8 +874,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { var spReg = Reg{idx: 4, size: 8} // operandFromAST converts a parsed operand into an encoder Operand, applying -// the frame translation to FP/SP pseudo-register operands. -func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) { +// the frame translation to FP/SP pseudo-register operands. mnemUpper is the +// instruction's upper-case mnemonic, which the floating-point immediate gate +// needs: only the SSE mnemonics whose encoding takes an XMM/memory source +// accept one. +func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) { switch op.Kind { case ast.OpImmediate: if op.Imm.HasVal { @@ -841,17 +888,43 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op } return Imm(v), nil } + // A floating-point immediate: $1.5, $-1.0 or the parenthesised + // $(-1.0) spelling (the constant-expression folder only folds + // integers, so that shape arrives with an empty Immediate and only + // the raw spelling carries the value). The toolchain rewrites it + // into a pooled-constant read on the SSE scalar paths and rejects + // it everywhere else. + if text, neg, ok := floatImmText(op); ok { + if !sseFloatImm[mnemUpper] { + return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper) + } + return FloatImm{Text: text, Neg: neg}, nil + } return nil, fmt.Errorf("non-integer immediate not supported") case ast.OpAddr: a := op.Addr // A bracketed register range, [Z0-Z3]: the four-register source of - // the 4FMAPS/4VNNIW families. The EVEX quad-register emit path - // needs an encoder operand of its own, so the shape stays a named - // gap rather than an encoding. + // the 4FMAPS/4VNNIW families. The range must span four consecutive + // same-width vector registers, exactly what the toolchain's parser + // takes; the EVEX quad-register emit path reads the low end. if a.Range != nil { - return nil, fmt.Errorf("register range %q needs quad-register encoder support", op.Raw) + lo, ok := ParseReg(a.Range.Lo) + if !ok { + return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo) + } + hi, ok := ParseReg(a.Range.Hi) + if !ok { + return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi) + } + if !lo.isVec() || lo.size != hi.size { + return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw) + } + if hi.idx != lo.idx+3 { + return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw) + } + return RegList{Lo: lo, Hi: hi}, nil } // FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the @@ -921,3 +994,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op } return nil, fmt.Errorf("unsupported operand") } + +// floatImmText recovers a floating-point immediate's magnitude and sign from +// the parsed operand. The ordinary spellings arrive in Imm.Float; the +// parenthesised $(-1.0) leaves the Immediate empty, because the integer +// folder cannot read it, and only the verbatim operand text still carries +// the value. Anything that is not a number a float parser accepts reports +// not-ok, so every other shape keeps its existing diagnostic. +func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) { + if op.Imm.Float != "" { + return op.Imm.Float, op.Imm.Neg, true + } + if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil { + return "", false, false + } + // joinRaw spaced the token texts; the compact spelling is what matters. + compact := strings.ReplaceAll(op.Raw, " ", "") + inner, ok := strings.CutPrefix(compact, "$(") + if !ok || !strings.HasSuffix(inner, ")") { + return "", false, false + } + inner = strings.TrimSuffix(inner, ")") + inner = strings.TrimPrefix(inner, "+") + if s, ok := strings.CutPrefix(inner, "-"); ok { + neg = true + inner = s + } + if inner == "" || !strings.ContainsAny(inner, "0123456789") { + return "", false, false + } + if _, err := strconv.ParseFloat(inner, 64); err != nil { + return "", false, false + } + return inner, neg, true +} diff --git a/asm/assemble_test.go b/asm/assemble_test.go index fba776a..84ee798 100644 --- a/asm/assemble_test.go +++ b/asm/assemble_test.go @@ -568,3 +568,28 @@ TEXT ·framed(SB), $16-8 t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } + +// TestAssembleRegRange pins the bracketed register range at the statement +// level: exactly four consecutive same-width vector registers assemble, the +// toolchain's rejected shapes all report an error. +func TestAssembleRegRange(t *testing.T) { + asm := func(t *testing.T, op string) ([]byte, error) { + t.Helper() + f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse %s: %v", op, errs) + } + code, _, err := Assemble(f.Decls[0].(*ast.Text)) + return code, err + } + for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} { + if _, err := asm(t, op); err != nil { + t.Errorf("%s: %v", op, err) + } + } + for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} { + if _, err := asm(t, op); err == nil { + t.Errorf("%s: assembled, want an error", op) + } + } +} diff --git a/asm/encode.go b/asm/encode.go index 2dc477f..c512342 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -5,6 +5,8 @@ package asm import ( "fmt" + "math" + "strconv" "strings" ) @@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) { type enc struct { out []byte patches []encPatch // disp32 fields awaiting static-symbol resolution + + // FloatPool collects the pooled constants the floating-point + // immediates reference, in first-use order. + floatPool []floatPoolEntry + floatPoolSeen map[string]bool +} + +// floatPoolEntry is one pooled floating-point constant: the symbol name +// the emitted RIP-relative load refers to and its IEEE-754 bytes. +type floatPoolEntry struct { + name string + data []byte +} + +// addFloatPool records a pooled constant, deduplicated by symbol name. +func (e *enc) addFloatPool(name string, bits uint64, width int) { + if e.floatPoolSeen == nil { + e.floatPoolSeen = map[string]bool{} + } + if e.floatPoolSeen[name] { + return + } + e.floatPoolSeen[name] = true + data := make([]byte, width) + for i := range width { + data[i] = byte(bits >> (8 * i)) + } + e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data}) +} + +// floatPoolList returns the pooled constants in first-use order. +func (e *enc) floatPoolList() []floatPoolEntry { + return e.floatPool } // encPatch marks a 4-byte displacement field in enc.out that must receive the @@ -102,6 +137,11 @@ func (e *enc) encode(mnem string, ops []Operand) error { return e.encodeEnd(ops) case "ADJSP": return e.encodeAdjsp(ops) + // The runtime's bookkeeping statements carry no text bytes: go tool asm + // records FUNCDATA and PCDATA in the program list only, so the encoded + // body shows nothing, on every architecture. + case "FUNCDATA", "PCDATA": + return e.encodeFuncdata(upper, ops) } // VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing @@ -141,11 +181,18 @@ func (e *enc) encode(mnem string, ops []Operand) error { } // Legacy SSE packed binaries dispatch on the full name: the packed // integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...), - // which the size split must not eat. + // which the size split must not eat. A floating-point immediate + // rewrites into a pooled-constant read on the scalar members. if m, ok := sseBinTable[upper]; ok { + if f, isFloat := floatImmOperand(ops); isFloat { + return e.encodeSSEFloatBin(upper, m, f, ops) + } return e.encodeSSEBin(m, ops) } if m, ok := sseBinTable[base]; ok { + if f, isFloat := floatImmOperand(ops); isFloat { + return e.encodeSSEFloatBin(upper, m, f, ops) + } return e.encodeSSEBin(m, ops) } // The imm8-controlled legacy instructions, the lane extracts and inserts @@ -222,7 +269,12 @@ func (e *enc) encode(mnem string, ops []Operand) error { return e.encodeCvtInt(base, ops, size) case "FMOVD": return e.encodeFmov(ops) - case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS": + case "MOVSD", "MOVSS": + if f, isFloat := floatImmOperand(ops); isFloat { + return e.encodeSSEFloatMove(upper, f, ops) + } + return e.encodeSSEMove(sseMoveTable[base], ops) + case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD": return e.encodeSSEMove(sseMoveTable[base], ops) } return fmt.Errorf("unsupported instruction %q", mnem) @@ -285,6 +337,33 @@ func (e *enc) encodeData(mnem string, ops []Operand) error { return nil } +// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements: +// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes +// for either (the entries live in the object's ancillary tables, not the +// function body), and the operand shapes it takes are exactly these: an +// integer count first, then a symbol reference for FUNCDATA and an integer +// value for PCDATA. The other architectures accept-and-ignore the same +// statements; amd64 now matches. +func (e *enc) encodeFuncdata(upper string, ops []Operand) error { + if len(ops) != 2 { + return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) + } + if _, ok := ops[0].(Imm); !ok { + return fmt.Errorf("%s: first operand must be an integer immediate", upper) + } + switch upper { + case "FUNCDATA": + if _, ok := ops[1].(sbMem); !ok { + return fmt.Errorf("FUNCDATA: second operand must be a symbol reference") + } + case "PCDATA": + if _, ok := ops[1].(Imm); !ok { + return fmt.Errorf("PCDATA: second operand must be an integer immediate") + } + } + return nil +} + // encodeEnd accepts-and-ignores END. go tool asm drops the statement // entirely: the AEND Prog is skipped when the program list is flushed, so // the statements after an END still belong to the same function and the @@ -319,6 +398,120 @@ func (e *enc) encodeAdjsp(ops []Operand) error { return nil } +// --- floating-point immediates ---------------------------------------------- + +// sseFloatImm lists the mnemonics whose first operand may be a floating-point +// immediate, the set go tool asm rewrites into a pooled-constant read: the +// scalar moves, the four scalar arithmetic pairs and the scalar compares. +// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD) +// reject the immediate in the toolchain and are absent here on purpose. +var sseFloatImm = map[string]bool{ + "MOVSD": true, "MOVSS": true, + "ADDSD": true, "ADDSS": true, + "SUBSD": true, "SUBSS": true, + "MULSD": true, "MULSS": true, + "DIVSD": true, "DIVSS": true, + "COMISD": true, "COMISS": true, + "UCOMISD": true, "UCOMISS": true, +} + +// floatImmOperand reports whether the operand list opens with a +// floating-point immediate in the two-operand spelling (imm, dst). +func floatImmOperand(ops []Operand) (FloatImm, bool) { + if len(ops) != 2 { + return FloatImm{}, false + } + f, ok := ops[0].(FloatImm) + return f, ok +} + +// floatPoolValue evaluates a floating-point immediate at the width its +// mnemonic encodes and names the pool constant the toolchain synthesises: +// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32 +// rounding of the parsed value). The name carries the IEEE-754 bits; the +// section holds them little-endian. +func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) { + v, err := strconv.ParseFloat(f.Text, 64) + if err != nil { + return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text) + } + if f.Neg { + v = -v + } + if strings.HasSuffix(mnem, "D") { + bits = math.Float64bits(v) + return bits, fmt.Sprintf("$f64.%016x", bits), nil + } + bits = uint64(math.Float32bits(float32(v))) + return bits, fmt.Sprintf("$f32.%08x", bits), nil +} + +// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate +// source. A positive zero needs no memory read: the toolchain emits +// XORPS dst, dst. Anything else loads the pooled constant RIP-relative +// ($f64.(SB) / $f32.(SB)), the displacement a patch site the +// file-level layout or the linker resolves. +func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error { + if !sseFloatImm[mnem] { + return fmt.Errorf("%s does not take a floating-point immediate", mnem) + } + dst, ok := ops[1].(Reg) + if !ok || !dst.isVec() { + return fmt.Errorf("%s: destination must be a vector register", mnem) + } + bits, name, err := floatPoolValue(mnem, f) + if err != nil { + return err + } + e.addFloatPool(name, bits, mwidth(mnem)) + if bits == 0 { + i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS + if err := setRM(i, dst, dst, 8); err != nil { + return err + } + return e.emit(i) + } + m := sseMoveTable[mnem] + i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1} + if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil { + return err + } + return e.emit(i) +} + +// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with +// a floating-point immediate source: the constant is read from the pool into +// the instruction's r/m side (reg = destination), the rewrite go tool asm +// performs at the source level. +func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error { + if !sseFloatImm[mnem] { + return fmt.Errorf("%s does not take a floating-point immediate", mnem) + } + dst, ok := ops[1].(Reg) + if !ok || !dst.isVec() { + return fmt.Errorf("%s: destination must be a vector register", mnem) + } + bits, name, err := floatPoolValue(mnem, f) + if err != nil { + return err + } + e.addFloatPool(name, bits, mwidth(mnem)) + i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} + if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil { + return err + } + return e.emit(i) +} + +// mwidth returns the operand width a scalar SSE mnemonic encodes: the double +// spellings end in D, the single spellings in S. +func mwidth(mnem string) int { + if strings.HasSuffix(mnem, "D") { + return 8 + } + return 4 +} + // splitSize separates a trailing B/W/L/Q size suffix from the mnemonic. func splitSize(upper string) (base string, size int) { if upper == "" { diff --git a/asm/encode_test.go b/asm/encode_test.go index 8b76a16..06daa09 100644 --- a/asm/encode_test.go +++ b/asm/encode_test.go @@ -9,6 +9,9 @@ import ( "testing" "golang.org/x/arch/x86/x86asm" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" ) // decode encodes an instruction and decodes it back, returning the decoded @@ -1064,3 +1067,157 @@ func TestAdjsp(t *testing.T) { t.Error("ADJSP AX assembled, want an error") } } + +// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite +// byte for byte against go tool asm: the scalar moves and the scalar +// arithmetic read the constant from a synthesised read-only pool symbol +// ($f64., $f32.) RIP-relative with the displacement left to the +// relocation, and a positive zero on the moves collapses to XORPS dst, dst. +func TestFloatImmediateGroundTruth(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"}, + {"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"}, + {"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"}, + {"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"}, + {"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"}, + {"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"}, + {"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"}, + {"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"}, + {"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"}, + {"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"}, + {"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"}, + {"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: got %s, want %s", c.name, got, c.want) + } + } + + // The pool names carry the IEEE-754 bits, the float32 narrowing for the + // single spellings; negative zero keeps its sign bit and never takes the + // XORPS shortcut. + for _, c := range []struct { + mnem string + imm FloatImm + want string + }{ + {"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"}, + {"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"}, + {"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"}, + {"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"}, + {"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"}, + } { + _, name, err := floatPoolValue(c.mnem, c.imm) + if err != nil { + t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err) + continue + } + if name != c.want { + t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want) + } + } + + // The shapes the toolchain's parser rejects: the packed and uniform + // forms, a non-vector destination, and the integer spellings. + for _, c := range []struct { + name string + mnem string + ops []Operand + }{ + {"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}}, + {"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}}, + {"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}}, + {"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}}, + } { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } +} + +// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore: +// go tool asm emits no text bytes for either, on every architecture. +func TestBookkeepingGroundTruth(t *testing.T) { + for _, c := range []struct { + name string + mnem string + ops []Operand + }{ + {"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}}, + {"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}}, + } { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if len(code) != 0 { + t.Errorf("%s: emitted %x, want no bytes", c.name, code) + } + } + for _, c := range []struct { + name string + mnem string + ops []Operand + }{ + {"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}}, + {"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}}, + {"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}}, + {"PCDATA arity", "PCDATA", []Operand{Imm(1)}}, + {"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}}, + } { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } + + // At the statement level the bookkeeping lines sit between real + // instructions and contribute nothing to the body, symbol reference + // included: the FUNCDATA operand never needs file-level resolution. + f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFile(f) + if err != nil { + t.Fatalf("assemble: %v", err) + } + want := "90c3" + if got := hexCompact(img.Code); got != want { + t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want) + } + if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil { + t.Error("FUNCDATA $1, X0 assembled, want an error") + } + if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil { + t.Error("PCDATA $1, X0 assembled, want an error") + } + + // Encodable mirrors Encode for the names this work touched. + for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} { + if !Encodable(mnem) { + t.Errorf("Encodable(%s) = false, want true", mnem) + } + } +} + +// mustParse parses src or fails the test. +func mustParse(t *testing.T, src string) *ast.File { + t.Helper() + f, errs := parser.Parse("t_amd64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + return f +} diff --git a/asm/evex_test.go b/asm/evex_test.go index f3a31d4..9279f05 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -810,3 +810,127 @@ func TestAvx512CorpusFamilies(t *testing.T) { } } } + +// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the +// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory +// source keeps r/m, the bracketed list's LOW register travels the inverted +// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and +// the vector length follows the destination (L'L=512 for the ZMM forms, +// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every +// member. The x86 decoder has no view of these forms, so no decode check +// runs. +func TestEvexQuadRegisterGroundTruth(t *testing.T) { + sp := vreg(t, "RSP") + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f27f4a9a842411000000"}, + {"V4FMADDPS [Z10-Z13]", "V4FMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f22f4a9a842411000000"}, + {"V4FMADDPS [Z20-Z23]", "V4FMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f25f429a842411000000"}, + {"V4FMADDPS Z8 dst", "V4FMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")}, + "62727f4a9a842411000000"}, + {"V4FMADDPS disp8x16", "V4FMADDPS", + []Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f27f4a9a442404"}, + {"V4FMADDPS unmasked", "V4FMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")}, + "62f27f489a842411000000"}, + {"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")}, + "62e27f0d9bb007000000"}, + {"V4FMADDSS (DI)", "V4FMADDSS", + []Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")}, + "62e27f0d9b37"}, + {"V4FMADDSS [X10-X13]", "V4FMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")}, + "62e22f0d9bb007000000"}, + {"V4FMADDSS [X20-X23]", "V4FMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")}, + "62e25f059bb007000000"}, + {"V4FMADDSS X30 dst", "V4FMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")}, + "62627f0d9bb007000000"}, + {"V4FMADDSS X3 dst", "V4FMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")}, + "62f27f0d9b9807000000"}, + {"V4FMADDSS disp8x16", "V4FMADDSS", + []Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")}, + "62625f059b7001"}, + {"V4FNMADDPS", "V4FNMADDPS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f27f4aaa842411000000"}, + {"V4FNMADDSS", "V4FNMADDSS", + []Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")}, + "62e27f0dabb007000000"}, + {"VP4DPWSSD", "VP4DPWSSD", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")}, + "62f27f4a52842411000000"}, + {"VP4DPWSSDS unmasked", "VP4DPWSSDS", + []Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")}, + "62f27f4853842411000000"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: got %s, want %s", c.name, got, c.want) + } + } +} + +// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects: +// the register class the list and the destination take is fixed per +// instruction, the source is memory only, the opmask slot is positional and +// the list's low register owns V'VVVV. +func TestEvexQuadRegisterErrors(t *testing.T) { + sp := vreg(t, "RSP") + list := func(lo, hi string) RegList { + return RegList{vreg(t, lo), vreg(t, hi)} + } + cases := []struct { + name string + mnem string + ops []Operand + }{ + {"X list on the PS form", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}}, + {"Z list on the SS form", "V4FMADDSS", + []Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}}, + {"Y destination", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}}, + {"register source", "V4FMADDPS", + []Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}}, + {"non-mask third operand", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}}, + {"k0 mask", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}}, + {"K after the destination", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}}, + {"zeroing without a mask", "V4FMADDPS.Z", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}}, + {"SAE suffix", "V4FMADDPS.SAE", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}}, + {"high index source", "VP4DPWSSD", + []Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}}, + {"short operand list", "V4FMADDPS", + []Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}}, + } + for _, c := range cases { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } +} diff --git a/asm/kernels_differential_test.go b/asm/kernels_differential_test.go index dbc48eb..0938783 100644 --- a/asm/kernels_differential_test.go +++ b/asm/kernels_differential_test.go @@ -63,7 +63,10 @@ func toolAsmObject(t *testing.T, path, goarch string) []byte { } // oracleFuncCode extracts the non-package TEXT functions' code bytes from a -// toolchain object, keyed by the name the object records (pkg.name). +// toolchain object, keyed by the name the object records (pkg.name). Each +// function's span is its own symbol size: a toolchain object that follows +// the text with data symbols (the synthesised float-constant pool) would +// otherwise fold them into the last function's bytes. func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte { t.Helper() v := openGoobj(t, obj) @@ -76,18 +79,13 @@ func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte { for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} { preceding += len(v.blk(bi)) / symSize } - total := preceding + len(nps) out := make(map[string][]byte, len(nps)) for i, s := range nps { if s.typ != kindSTEXT { continue } start := le.Uint32(didx[4*(preceding+i):]) - end := uint32(len(data)) - if preceding+i+1 < total { - end = le.Uint32(didx[4*(preceding+i+1):]) - } - out[s.name] = data[start:end] + out[s.name] = data[start : start+s.size] } return out } @@ -129,6 +127,9 @@ func TestDifferentialKernels(t *testing.T) { {filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false}, + {filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false}, + {filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false}, + {filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false}, {filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true}, {filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true}, } { diff --git a/asm/link.go b/asm/link.go index 3a4ab8c..bfbb1d5 100644 --- a/asm/link.go +++ b/asm/link.go @@ -167,6 +167,7 @@ func AssembleFile(f *ast.File) (*Image, error) { known[d.name] = true } link := &linkInfo{symbols: known, allowExternal: true} + poolSeen := map[string]bool{} img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} textOff := map[string]int{} @@ -180,7 +181,26 @@ func AssembleFile(f *ast.File) (*Image, error) { if !ok { continue } - code, patches, labels, steps, lines, err := assemble(t, link) + code, patches, labels, steps, lines, pool, err := assemble(t, link) + if err != nil { + return nil, fmt.Errorf("%s: %w", t.Name.Name, err) + } + // The pooled floating-point constants join the declared data as + // read-only symbols, deduplicated across the file (the toolchain + // synthesises the same symbols into its rodata). + for _, entry := range pool { + if poolSeen[entry.name] { + continue + } + poolSeen[entry.name] = true + dataSyms = append(dataSyms, dataSym{ + name: entry.name, + buf: entry.data, + size: len(entry.data), + rodata: true, + dupok: true, + }) + } if err != nil { return nil, fmt.Errorf("%s: %w", t.Name.Name, err) } diff --git a/testdata/verify/bookkeep_amd64.s b/testdata/verify/bookkeep_amd64.s new file mode 100644 index 0000000..7611b1b --- /dev/null +++ b/testdata/verify/bookkeep_amd64.s @@ -0,0 +1,32 @@ +// The runtime bookkeeping statements: FUNCDATA and PCDATA contribute no +// text bytes on any architecture, and amd64 now matches. They sit between +// real instructions here, with plain, static and offset symbol references +// on the FUNCDATA lines, so the byte counts prove the zero contribution. + +#include "textflag.h" + +// func bookkeep(x int64) int64 +TEXT ·bookkeep(SB), NOSPLIT, $0-16 + PCDATA $0, $-1 + MOVQ x+0(FP), AX + PCDATA $1, $-2 + FUNCDATA $0, args_stackmap(SB) + ADDQ $1, AX + FUNCDATA $5, arginfo0(SB) + PCDATA $1, $3 + MOVQ AX, ret+8(FP) + FUNCDATA $1, externalfuncdata(SB) + PCDATA $0, $0 + RET + +// func bookkeepstatic() int64 +TEXT ·bookkeepstatic(SB), NOSPLIT, $0-8 + // A static symbol and a defined data symbol as the funcdata target. + // (A symbol+offset target the toolchain itself refuses.) + FUNCDATA $2, fdtable<>(SB) + FUNCDATA $3, undefsym(SB) + MOVQ $7, AX + MOVQ AX, ret+0(FP) + RET + +GLOBL fdtable<>(SB), NOPTR, $16 diff --git a/testdata/verify/floatimm_amd64.s b/testdata/verify/floatimm_amd64.s new file mode 100644 index 0000000..6b56412 --- /dev/null +++ b/testdata/verify/floatimm_amd64.s @@ -0,0 +1,55 @@ +// Floating-point immediates on the SSE scalar paths: the constant is +// rewritten into a read from a read-only pool symbol ($f64. or +// $f32., the IEEE-754 bits in the name), RIP-relative with the +// displacement left to the relocation. A positive zero on the moves +// collapses to XORPS dst, dst; a negative zero keeps its sign bit and +// takes the pool. The parenthesised $(-1.0) spelling is the one +// math/floor_amd64.s uses. Every result is folded back so no +// instruction is dead. + +#include "textflag.h" + +// func floatimm(x float64) float64 +TEXT ·floatimm(SB), NOSPLIT, $0-16 + MOVQ x+0(FP), AX + MOVQ AX, X0 + // The floor kernel's sign fold: the parenthesised negative spelling. + MOVSD $ (-1.0), X2 + ANDPD X2, X0 + // Positive and fractional constants on the scalar moves. + MOVSD $1.5, X3 + MOVSD $0.5, X4 + MOVSS $2.5, X5 + MOVSS $-0.5, X6 + // A positive zero collapses to XORPS; a negative zero does not. + MOVSD $0.0, X7 + MOVSS $0.0, X8 + MOVSD $-0.0, X9 + // The scalar arithmetic reads the pool through r/m (hypot's shape). + ADDSD $1.0, X3 + SUBSD $0.5, X4 + MULSD $-2.5, X4 + DIVSD $2.0, X3 + ADDSS $0.25, X5 + // Fold everything into one double. + ADDSD X5, X3 + ADDSD X6, X3 + ADDSD X7, X3 + ADDSD X8, X3 + ADDSD X9, X3 + ADDSD X4, X3 + ADDSD X0, X3 + MOVSD X3, ret+8(FP) + RET + +// func floatimmfloat32() float32 +TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4 + // The single-width pool constants ride the F3 prefix. + MOVSS $1.0, X0 + MOVSS $-1.0, X1 + MOVSS $0.0, X2 + ADDSS $0.5, X0 + ADDSS X1, X0 + ADDSS X2, X0 + MOVSS X0, ret+0(FP) + RET diff --git a/verify/groundtruth_test.go b/verify/groundtruth_test.go index 7112f56..7f7544a 100644 --- a/verify/groundtruth_test.go +++ b/verify/groundtruth_test.go @@ -124,10 +124,16 @@ func TestGroundTruthAMD64(t *testing.T) { "../testdata/verify/avx_amd64.s", "../testdata/verify/pfx_amd64.s", "../testdata/verify/vsib_amd64.s", + "../testdata/verify/floatimm_amd64.s", + "../testdata/verify/bookkeep_amd64.s", + "../testdata/verify/quadreg_amd64.s", "../testdata/verify/rawdata_amd64.s", "../testdata/verify/avx512_amd64.s", "../testdata/verify/pfx_amd64.s", "../testdata/verify/vsib_amd64.s", + "../testdata/verify/floatimm_amd64.s", + "../testdata/verify/bookkeep_amd64.s", + "../testdata/verify/quadreg_amd64.s", "../testdata/verify/rawdata_amd64.s", "../testdata/verify/avx512_amd64.s", "../testdata/verify/doubleshift_amd64.s",