From bafb2fd1304ae9aeb608817fba7efdc5a3c0ff6c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Fri, 2 Oct 2026 00:40:43 +0200 Subject: [PATCH] feat(asm): encode the amd64 and loong64 tails of the corpus testdata Assisted-by: GLM 5.3 --- asm/assemble.go | 239 ++++++++++++++++++++++++++++++--- asm/encodable.go | 13 ++ asm/encode.go | 78 ++++++++++- asm/encode_test.go | 158 +++++++++++++++++++++- asm/evex.go | 91 ++++++++++--- asm/instrs.go | 207 ++++++++++++++++++++++++++++- asm/link_test.go | 14 +- asm/loong64_assemble.go | 263 ++++++++++++++++++++++++++++++++++++- asm/loong64_encode.go | 37 +++++- asm/loong64_encode_test.go | 227 ++++++++++++++++++++++++++++++++ asm/reg.go | 31 ++++- asm/vex.go | 127 +++++++++++++++++- 12 files changed, 1404 insertions(+), 81 deletions(-) diff --git a/asm/assemble.go b/asm/assemble.go index d7bad42..1e879ef 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -808,17 +808,43 @@ func isJumpMnemonic(mnem string) bool { if mnem == "JMP" || mnem == "CALL" { return true } + if isLoopMnemonic(mnem) { + return true + } _, ok := condCode(mnem) return ok } +// isLoopMnemonic reports the LOOP family, rel8 alone (E0-E2). +func isLoopMnemonic(mnem string) bool { + switch mnem { + case "LOOP", "LOOPE", "LOOPNE": + return true + } + return false +} + +// loopOpcode maps the LOOP family to its E0-E2 opcode. +func loopOpcode(mnem string) byte { + switch mnem { + case "LOOPE": + return 0xE1 + case "LOOPNE": + return 0xE0 + } + return 0xE2 +} + // jumpSize returns the length of a jump instruction in the requested form: // short (rel8) where available, otherwise the rel32 form. CALL is always -// rel32. +// rel32; the LOOP family is rel8 alone. func jumpSize(mnem string, long bool) int { if mnem == "CALL" { return 5 // opcode + rel32 } + if isLoopMnemonic(mnem) { + return 2 // opcode + rel8, the only form + } if !long { return 2 // opcode + rel8 } @@ -945,6 +971,16 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) { return encodeTLSBaseLoad(s, fi, link) } + // The old paired-register shift spelling, SHLL CX, R11:AX (a colon + // between the two registers), is the toolchain's SHLD family: SHLDL CL, + // AX, R11 with the count register first, the paired source in the reg + // field and the pair's head in r/m. + if code, ps, err := encodeColonShift(s, mnemUpper, fi, link); code != nil || err != nil { + if err != nil { + return nil, nil, nil, err + } + return code, ps, nil, nil + } _, size := splitSize(mnemUpper) if size == 0 { size = 8 @@ -1051,6 +1087,66 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) { return nil, nil } +// encodeColonShift encodes the paired-register shift spellings, SHLx CX, +// dst:src: the toolchain reads them as the SHLD family (double-precision +// shift by CL), reg = the paired source, r/m = the pair's head. The second +// operand's raw text carries the colon; ok reports the spelling was found. +func encodeColonShift(s *ast.Instr, mnemUpper string, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) { + base, _ := strings.CutPrefix(mnemUpper, "SHL") + if base == mnemUpper || len(s.Operands) != 2 { + return nil, nil, nil + } + _, size := splitSize(mnemUpper) + raw := strings.ReplaceAll(s.Operands[1].Raw, " ", "") + head, tail, ok := strings.Cut(raw, ":") + if !ok || head == "" || tail == "" { + return nil, nil, nil + } + headReg, ok1 := ParseReg(head) + srcReg, ok2 := ParseReg(tail) + if !ok1 || !ok2 { + return nil, nil, fmt.Errorf("%s: invalid paired register %q", mnemUpper, s.Operands[1].Raw) + } + cnt, err := operandFromAST(mnemUpper, s.Operands[0], size, fi, link) + if err != nil { + return nil, nil, err + } + cntReg, ok := cnt.(Reg) + if !ok || cntReg.idx != 1 { + return nil, nil, fmt.Errorf("%s: the paired-register form counts in CL", mnemUpper) + } + // SHLD r/m, reg, CL: 0F A5 (REX.W for the 64-bit width). + e := &enc{} + i := &instr{rexW: size == 8, opcode: []byte{0x0F, 0xA5}, modrm: -1, sib: -1} + if err := setRM(i, srcReg, headReg, size); err != nil { + return nil, nil, err + } + if err := e.emit(i); err != nil { + return nil, nil, err + } + ps := make([]sbPatch, len(e.patches)) + for j, p := range e.patches { + ps[j] = sbPatch{off: p.off, name: p.name, addend: p.addend} + } + return e.out, ps, nil +} + +// trailingIndexGroup recovers a trailing "(index*scale)" or "(index)" group +// from an operand's raw text: the symbol-pseudo parse returns before the +// index group, so foo(SP)(AX*1) keeps its index only in the spelling. +func trailingIndexGroup(raw string) (string, int, bool) { + compact := strings.ReplaceAll(raw, " ", "") + if !strings.HasSuffix(compact, ")") { + return "", 0, false + } + open := strings.LastIndex(compact, "(") + if open < 2 || !strings.Contains(compact[:open], ")") { + return "", 0, false // one group alone: no trailing index + } + name, scale, _, ok := cutParenGroup(compact[open:]) + return name, scale, ok +} + // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // target label or from a numeric ±N(PC) instruction count, in the short // (rel8) or long (rel32) form. numTarget is the resolved byte offset of a @@ -1085,9 +1181,15 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long if mnem == "JMP" { return []byte{0xEB, byte(int8(rel))}, nil } + if isLoopMnemonic(mnem) { + return []byte{loopOpcode(mnem), byte(int8(rel))}, nil + } cc, _ := condCode(mnem) return []byte{0x70 + byte(cc), byte(int8(rel))}, nil } + if isLoopMnemonic(mnem) { + return nil, fmt.Errorf("%s has no long form", mnem) + } switch mnem { case "JMP": return append([]byte{0xE9}, le32(rel)...), nil @@ -1139,6 +1241,72 @@ func labelName(op *ast.Operand) (string, bool) { return "", false } +// jumpOperand returns the branch-target operand of a JMP/CALL, rewriting the +// `*`-prefixed indirect spellings (JMP *(R12), JMP *4(SP)) into their plain +// memory form. The star marks an indirect target and changes no bytes; the +// address parser leaves the operand's address empty because of the leading +// star, so the fields are rebuilt from the raw text onto a copy of the +// operand, never on the shared syntax tree. +func jumpOperand(s *ast.Instr) *ast.Operand { + if len(s.Operands) != 1 { + return nil + } + op := s.Operands[0] + compact := strings.ReplaceAll(op.Raw, " ", "") + inner, ok := strings.CutPrefix(compact, "*") + if !ok { + return op + } + var addr ast.Address + if i := strings.IndexByte(inner, '('); i > 0 { + v, err := strconv.ParseInt(inner[:i], 0, 64) + if err != nil { + return op + } + addr.Offset, addr.HasOff = v, true + inner = inner[i:] + } + base, _, rest, ok := cutParenGroup(inner) + if !ok { + return op + } + if base != "" { + addr.Base = base + } + if rest != "" { + idx, scale, _, ok := cutParenGroup(rest) + if ok && idx != "" { + addr.Index = idx + addr.Scale = scale + } + } + c := *op + c.Addr = addr + return &c +} + +// cutParenGroup splits a leading "(name)" or "(name*n)" off s, returning the +// inner text, the scale it names (1 when the group spells no multiplier) and +// the remainder. +func cutParenGroup(s string) (name string, scale int, rest string, ok bool) { + if !strings.HasPrefix(s, "(") { + return "", 0, "", false + } + i := strings.IndexByte(s, ')') + if i < 0 { + return "", 0, "", false + } + inner, rest := s[1:i], s[i+1:] + if before, after, ok := strings.Cut(inner, "*"); ok { + n, err := strconv.Atoi(after) + if err != nil { + return "", 0, "", false + } + return before, n, rest, true + } + return inner, 1, rest, true +} + // indirectJumpTarget reports whether the JMP/CALL operand addresses a // register or a memory location rather than a label or a static symbol. // A bare identifier is a register when the register table knows the name and @@ -1147,7 +1315,14 @@ func indirectJumpTarget(s *ast.Instr) bool { if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr { return false } - a := s.Operands[0].Addr + op := jumpOperand(s) + if op == nil { + return false + } + if op != s.Operands[0] { + return true // the star marker spells an indirect target + } + a := op.Addr // ±N(PC) is the numeric relative form, the PC counts instructions from // the branch: relative, not indirect. if a.Base == "PC" || a.Index == "PC" { @@ -1165,18 +1340,20 @@ func indirectJumpTarget(s *ast.Instr) bool { } // encodeIndirectJump assembles a JMP/CALL through a register or memory -// operand, which carries no relocation and no label to resolve. +// operand, which carries no relocation and no label to resolve. The +// `*`-prefixed spellings go through jumpOperand first, their star rebuilt +// into a plain memory operand. func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { - ops := make([]Operand, len(s.Operands)) - for i, op := range s.Operands { - o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) - if err != nil { - return nil, err - } - ops[i] = o + op := s.Operands[0] + if cleaned := jumpOperand(s); cleaned != nil { + op = cleaned + } + o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) + if err != nil { + return nil, err } e := &enc{} - if err := e.encodeIndirectBranch(mnem, ops); err != nil { + if err := e.encodeIndirectBranch(mnem, []Operand{o}); err != nil { return nil, err } return e.out, nil @@ -1245,31 +1422,51 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l off := a.Sym.Offset + fi.fpAdjust return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil } - // SP-relative local: x-N(SP) → (spAdjust + offset)(SP). + // SP-relative local: x-N(SP) → (spAdjust + offset)(SP), keeping a scaled + // index beside the virtual stack pointer (foo(SP)(AX*1)). The + // symbol-pseudo parse returns before the index group, so the index + // is recovered from the raw text when the address lacks it. if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" { off := fi.spAdjust + a.Sym.Offset - return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil + m := Mem{Base: spReg, Disp: off, HasBase: true, Size: size} + if name, scale, ok := trailingIndexGroup(op.Raw); ok { + idx, ok := ParseReg(name) + if !ok { + return nil, fmt.Errorf("unknown index register %q", name) + } + m.Index = idx + m.Scale = scale + m.HasIndex = true + } + return m, nil } // SB (global symbol): a symbol defined in the same file (GLOBL) is // encoded RIP-relative and resolved by the file-level layout; - // anything not defined here needs object-file emission. + // anything not defined here needs object-file emission. A static + // (file-local) spelling of an undefined symbol defers the same way + // the toolchain does: the relocation names it and the linker decides. if a.Sym != nil && a.Sym.Pseudo == "SB" { if link == nil || link.symbols == nil { return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name) } - if !link.symbols[a.Sym.Name] { - if a.Sym.Static { - return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name) - } - if !link.allowExternal { - return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name) - } + if !link.symbols[a.Sym.Name] && !link.allowExternal { + return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name) } return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil } // Memory with a real base register: (base), off(base), (base)(index*scale). if a.Base != "" { + // The TLS pseudo-base, off(TLS): the segment-prefixed absolute + // the thread-local access lowers to, 64 8B 04 25 with its + // R_TLS_LE patch site on the disp32. + if a.Base == "TLS" { + seg := byte(0x64) // FS on linux, freebsd, plan9 + if link != nil && link.goos == "windows" { + seg = 0x65 // GS + } + return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil + } // Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS // spellings. The segment override prefixes a disp32 absolute // reference with no relocation. diff --git a/asm/encodable.go b/asm/encodable.go index 3ed5cd7..31843e4 100644 --- a/asm/encodable.go +++ b/asm/encodable.go @@ -20,11 +20,24 @@ func Encodable(mnemonic string) bool { switch upper { case "RET", "NOP", "CALL", "JMP", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", + // The SSE compare family sharing CMPSD's predicate-last shape, the + // far return with its stack pop, the loop family, the bank-crossing + // MMX moves and the one-operand system controls. + "CMPSS", "CMPPS", "CMPPD", "RETFL", + "LOOP", "LOOPE", "LOOPNE", + "MOVDQ2Q", "MOVQ2DQ", + "ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE", // The literal-data pseudo-ops, the accepted-and-ignored END and // bookkeeping statements, and the SP adjust. "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA": return true } + if _, ok := sysUnaryTable[upper]; ok { + return true + } + if _, ok := sseStoreOnly[upper]; ok { + return true + } if _, ok := noOperandTable[upper]; ok { return true } diff --git a/asm/encode.go b/asm/encode.go index 6019cac..f01bd88 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -73,9 +73,23 @@ func (e *enc) encode(mnem string, ops []Operand) error { // Fixed-name instructions (no size suffix). switch { case upper == "RET": + // RET sym(SB), the absolute return: the toolchain encodes it as a + // tail jump, E9 rel32 with a call relocation against the symbol. + if len(ops) == 1 { + if m, ok := ops[0].(sbMem); ok { + return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}) + } + return fmt.Errorf("RET: unsupported operand") + } + if len(ops) != 0 { + return fmt.Errorf("RET expects no operands, got %d", len(ops)) + } return e.encodeRet() case upper == "NOP": - return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1}) + // The toolchain consumes every NOP statement as a pseudo and emits + // nothing for it, operands included (a bare NOP, NOP AX and + // NOP sym(SB) all vanish from the object). + return nil case upper == "CALL" || upper == "JMP": // Through a register or memory: FF /2 (CALL) or FF /4 (JMP). // Anything else is a rel32 against a label resolved by the assembler. @@ -102,6 +116,15 @@ func (e *enc) encode(mnem string, ops []Operand) error { } return e.emit(&instr{opcode: op, modrm: -1, sib: -1}) } + // One-operand system instructions whose reg field is a fixed digit: + // the cache and wait controls under 0F AE/0F 1C and the RDPID read. + if m, ok := sysUnaryTable[upper]; ok { + return e.encodeSysUnary(upper, m, ops) + } + // The store-only SSE moves (the non-temporal store). + if m, ok := sseStoreOnly[upper]; ok { + return e.encodeSSEStoreOnly(upper, m, ops) + } // POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings // are rejected by go tool asm in 64-bit mode, so they stay unsupported. switch upper { @@ -117,14 +140,63 @@ func (e *enc) encode(mnem string, ops []Operand) error { return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1}) case "INT": return e.encodeInt(ops) + // The LOOP family outside the assembler's label settlement: the operand + // is the already-computed rel8 (E0-E2). + case "LOOP", "LOOPE", "LOOPNE": + if len(ops) != 1 { + return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops)) + } + imm, ok := ops[0].(Imm) + if !ok || !fits8(int64(imm)) { + return fmt.Errorf("%s: relative offset must be a signed byte", upper) + } + return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}}) case "LDMXCSR": return e.encodeMxcsr(2, ops) case "STMXCSR": return e.encodeMxcsr(3, ops) // CMPSD is the scalar double compare, whose predicate immediate comes - // LAST in Plan 9 order (src, dst, $imm). + // LAST in Plan 9 order (src, dst, $imm); the family shares the shape. case "CMPSD": - return e.encodeCmpsd(ops) + return e.encodeSSECmp("CMPSD", 0xF2, ops) + case "CMPSS": + return e.encodeSSECmp("CMPSS", 0xF3, ops) + case "CMPPS": + return e.encodeSSECmp("CMPPS", 0x00, ops) + case "CMPPD": + return e.encodeSSECmp("CMPPD", 0x66, ops) + // RETFL pops the immediate's worth of bytes after the far return + // (LRET iw: CA imm16), the toolchain's RETF spelling with a stack + // adjustment. + case "RETFL": + if len(ops) != 1 { + return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops)) + } + imm, ok := ops[0].(Imm) + if !ok { + return fmt.Errorf("RETFL expects an immediate") + } + return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))}) + // MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register + // in the reg field, the other bank's in r/m. + case "MOVDQ2Q", "MOVQ2DQ": + if len(ops) != 2 { + return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) + } + srcReg, ok1 := ops[0].(Reg) + dstReg, ok2 := ops[1].(Reg) + if !ok1 || !ok2 { + return fmt.Errorf("%s takes register operands alone", upper) + } + if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) || + upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) { + return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper) + } + i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} + if err := setRM(i, dstReg, srcReg, 8); err != nil { + return err + } + return e.emit(i) // SHA256RNDS2 carries the round constant in a literal X0 first operand. case "SHA256RNDS2": return e.encodeSha256rnds2(ops) diff --git a/asm/encode_test.go b/asm/encode_test.go index 036d7c1..83d93a2 100644 --- a/asm/encode_test.go +++ b/asm/encode_test.go @@ -265,7 +265,11 @@ func TestImul(t *testing.T) { func TestControl(t *testing.T) { checkSyntax(t, "ret", "RET") - checkSyntax(t, "nop", "NOP") + // NOP contributes nothing on amd64, consumed whole by the toolchain as + // a pseudo; only the bytes pin it (no decodable instruction remains). + if code, err := Encode("NOP"); err != nil || len(code) != 0 { + t.Errorf("NOP: bytes %x (err %v), want empty", code, err) + } checkOp(t, x86asm.JMP, "JMP", Imm(0)) checkOp(t, x86asm.CALL, "CALL", Imm(0)) checkOp(t, x86asm.JGE, "JGE", Imm(0)) @@ -1193,7 +1197,7 @@ func TestBookkeepingGroundTruth(t *testing.T) { if err != nil { t.Fatalf("assemble: %v", err) } - want := "90c3" + want := "c3" if got := hexCompact(img.Code); got != want { t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want) } @@ -1221,3 +1225,153 @@ func mustParse(t *testing.T, src string) *ast.File { } return f } + +// TestCorpusTailSystem pins the system and control forms the toolchain's own +// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare +// family, the far return, the loop, the MMX moves, the CR/DR and segment +// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls. +func TestCorpusTailSystem(t *testing.T) { + regBPT := Reg{idx: 5, size: 8} + X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2") + Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7") + X5, X20 := vreg(t, "X5"), vreg(t, "X20") + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"}, + {"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"}, + {"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"}, + {"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"}, + {"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"}, + {"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"}, + {"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"}, + {"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"}, + {"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"}, + {"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"}, + {"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"}, + {"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"}, + {"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"}, + {"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"}, + {"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"}, + {"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"}, + {"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"}, + {"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"}, + {"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"}, + {"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"}, + {"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"}, + {"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"}, + {"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"}, + {"ENDBR64", "ENDBR64", nil, "f30f1efa"}, + {"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"}, + {"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"}, + {"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"}, + {"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"}, + {"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"}, + {"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"}, + {"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"}, + {"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"}, + {"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"}, + {"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"}, + {"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"}, + {"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"}, + {"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"}, + {"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"}, + {"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"}, + {"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"}, + {"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"}, + {"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"}, + {"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"}, + {"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"}, + {"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"}, + {"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"}, + {"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"}, + {"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"}, + {"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"}, + {"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"}, + {"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"}, + {"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"}, + {"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + } + } +} + +// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64 +// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the +// paired-register shift spelling, the absolute RET and the jump to an +// undefined static symbol (its displacement and the TLS slot offsets are +// relocation sites, zeroed here as the kernel parity suites do). +func TestCorpusTailFileForms(t *testing.T) { + mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 } + cases := []struct { + name string + src string + want string // hex, with X marking a masked 32-bit relocation site + }{ + {"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"}, + {"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"}, + {"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"}, + {"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"}, + {"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"}, + {"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"}, + {"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"}, + } + for _, c := range cases { + f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src) + if len(errs) > 0 { + t.Errorf("%s: parse: %v", c.name, errs) + continue + } + img, err := AssembleFile(f) + if err != nil { + t.Errorf("%s: assemble: %v", c.name, err) + continue + } + fn := img.Funcs[0] + code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...) + if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 { + mask32(code, at/2) // the masked relocation site + } + if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + } + } + // JMP to an undefined static symbol and the absolute RET: their rel32 + // carries a call relocation against the symbol, masked to zero here. + for _, c := range []struct{ name, src string }{ + {"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"}, + {"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"}, + {"absolute ret", "\tRET\n\tRET foo(SB)\n"}, + } { + f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src) + if len(errs) > 0 { + t.Errorf("%s: parse: %v", c.name, errs) + continue + } + img, err := AssembleFile(f) + if err != nil { + t.Errorf("%s: assemble: %v", c.name, err) + continue + } + fn := img.Funcs[0] + code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs) + want := "e900000000c3" + if c.name == "absolute ret" { + want = "c3e900000000" + } + if got := hexCompact(code); got != want { + t.Errorf("%s: bytes %s, want %s", c.name, got, want) + } + } +} diff --git a/asm/evex.go b/asm/evex.go index d38f3ab..53b7a8c 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -856,37 +856,41 @@ type evexMoveSpec struct { vecOK bool // the non-memory operand may be a vector register xmmOnly bool // wider than XMM registers are rejected nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS) + gprOK bool // the r/m side may be a general-purpose register (VMOVQ) } // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. var evexMoveTable = map[string]evexMoveSpec{ // EVEX.128/256/512.F3.0F.W0, unaligned integer move. - "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512.F3.0F.W1, unaligned qword move. - "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // F2 prefix, dword/qword moves F3; the element size only changes the tuple // semantics). - "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // encoding). - "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512.66.0F.W1, unaligned packed double move. - "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false}, + "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512, aligned packed moves. - "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false}, - "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false}, + "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false}, + "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128/256/512.66.0F, aligned integer moves. - "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, - "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, + "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false}, + "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false}, // EVEX.128.F3.0F.W0, scalar single move, memory operands (the // three-operand register form is not supported). - "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true}, + "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false}, // EVEX.128.F2.0F.W1, scalar double move: memory operands and the // three-operand register form (VMOVSD dst, src1, src2). - "VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true}, + "VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false}, // EVEX.128/256/512.0F.W0, unaligned packed single move. - "VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false}, + "VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false}, + // EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20 + // and friends, the EVEX spelling the high registers demand). + "VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. @@ -900,6 +904,9 @@ func isEvex(mnemUpper string) bool { if _, ok := evexMoveTable[mnemUpper]; ok { return true } + if _, ok := evexHptrTable[mnemUpper]; ok { + return true + } return isEvexQuad(mnemUpper) } @@ -911,7 +918,13 @@ func evexRequired(upper string, ops []Operand) bool { _, inVex := vexTable[upper] _, inVexMove := vexMoveTable[upper] if !inVex && !inVexMove { - return true // EVEX-only mnemonic + // The dual-shape moves pick their VEX form by operand count, so + // they are not EVEX-only either. + switch upper { + case "VMOVHPD", "VMOVLPD": + default: + return true // EVEX-only mnemonic + } } // The byte-quad shifts have VEX register forms but EVEX-only memory // forms: a memory count source forces the EVEX encoding. @@ -1018,6 +1031,8 @@ var evexRound = map[string]bool{ "VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true, "VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true, "VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true, + // The scalar compares suppress exceptions on their LIG encoding. + "VCMPSD": true, "VCMPSS": true, } // evexBcstN maps an instruction accepting .BCST to the broadcast element @@ -1082,6 +1097,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error return e.encodeEvexRM(spec, ops, 0, sfx) } spec, inTable := evexTable[mnemUpper] + // A high/low half move that lives in the hptr table alone (the packed + // double twins) reaches the same inTable block below, which completes + // its spec from the hptr entry. + if !inTable { + if _, ok := evexHptrTable[mnemUpper]; ok { + inTable = true + } + } if q, ok := evexQuadTable[mnemUpper]; ok { // The quad-register family carries no rounding, SAE or broadcast; // only masking and zeroing apply. @@ -1446,15 +1469,28 @@ func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx e // assembler. The scalar moves also carry a three-operand register form // (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3 // opens. +// validEvexMoveOther reports whether the non-vector side of an EVEX move may +// take the operand: memory always, a general-purpose register when gprOK. +func validEvexMoveOther(ms evexMoveSpec, op Operand) bool { + if memOperand(op) { + return true + } + if !ms.gprOK { + return false + } + r, ok := op.(Reg) + return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp +} + func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) == 3 { if !ms.nds3 { return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) } - // The masked scalar register form keeps the Go assembler's own - // layout: the store opcode with reg = op0, vvvv = op1 and the - // destination in r/m (op2) — the bytes go tool asm emits, not - // the manual's NDS reading. + // The masked scalar register form keeps the Go assembler's own + // layout: the store opcode with reg = op0, vvvv = op1 and the + // destination in r/m (op2), the bytes go tool asm emits, not + // the manual's NDS reading. src, src1, dst := ops[0], ops[1], ops[2] reg, ok := src.(Reg) if !ok || !reg.isVec() { @@ -1494,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i } reg, rm = srcReg, dst case srcIsVec: - if !memOperand(dst) { + if !validEvexMoveOther(ms, dst) { return fmt.Errorf("%s: invalid destination operand", mnem) } reg, rm = srcReg, dst case dstIsVec: - if !memOperand(src) { + if !validEvexMoveOther(ms, src) { return fmt.Errorf("%s: invalid source operand", mnem) } op = ms.load @@ -1890,6 +1926,15 @@ var evexHptrTable = map[string]evexHptrSpec{ "VMOVLHPS": { insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, }, + // The packed-double twins, 66-prefixed. + "VMOVHPD": { + insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, + store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}}, + }, + "VMOVLPD": { + insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, + store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}}, + }, } // encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib. @@ -1954,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS if !ok || !maskReg.isVec() { return fmt.Errorf("%s: mask must be a vector register", upper) } - vsib, _, err := vsibLen(rest[1], upper) + vsib, idxLen, err := vsibLen(rest[1], upper) if err != nil { return err } @@ -1962,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS if !ok || !dst.isVec() { return fmt.Errorf("%s: destination must be a vector register", upper) } + // The L bit is the wider of the data register and the VSIB index + // lengths (a YMM index under an XMM destination selects 256-bit, the + // bytes go tool asm emits). + ll := max(idxLen, dst.vecLenBit()) spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1} rBit := 0 if dst.idx >= 8 { rBit = 1 } - return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) + return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib) } // encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src, diff --git a/asm/instrs.go b/asm/instrs.go index 4f40c12..e22f030 100644 --- a/asm/instrs.go +++ b/asm/instrs.go @@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{ "LOCK": {0xF0}, "REP": {0xF3}, "REPN": {0xF2}, + "ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA}, +} + +// sysUnaryTable maps the one-operand system instructions to their bytes: +// the prefix, the opcode and the /digit the reg field carries. The operand +// is a register or memory in r/m. +var sysUnaryTable = map[string]struct { + prefix byte + opcode []byte + digit int +}{ + "CLWB": {0x66, []byte{0x0F, 0xAE}, 6}, + "TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6}, + "UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6}, + "UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6}, + "RDPID": {0xF3, []byte{0x0F, 0xC7}, 7}, + "CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0}, +} + +// encodeSysUnary emits a one-operand system instruction: the operand in r/m +// under the fixed /digit, no REX.W. +func (e *enc) encodeSysUnary(mnem string, m struct { + prefix byte + opcode []byte + digit int +}, ops []Operand) error { + if len(ops) != 1 { + return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) + } + i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1} + if err := setRMDigit(i, m.digit, ops[0], 8); err != nil { + return err + } + return e.emit(i) } // --- MOV -------------------------------------------------------------------- @@ -111,6 +145,76 @@ func (e *enc) encodeMov(ops []Operand, size int) error { // silently emit REX.W 8B with the wrong operand meaning. _, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) + + // Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn), + // 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg + // field, the general register r/m; CR8+/DR8+ take REX.R. + if c, ok := src.(Reg); ok && c.ctl != 0 { + g, ok := dst.(Reg) + if !ok || g.isVec() || g.ctl != 0 { + return fmt.Errorf("MOV: control/debug register load needs a general register destination") + } + opc := byte(0x20) + if c.ctl == 2 { + opc = 0x21 + } + return e.emit(&instr{ + opcode: []byte{0x0F, opc}, + modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7), + sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8, + }) + } + if c, ok := dst.(Reg); ok && c.ctl != 0 { + g, ok := src.(Reg) + if !ok || g.isVec() || g.ctl != 0 { + return fmt.Errorf("MOV: control/debug register store needs a general register source") + } + opc := byte(0x22) + if c.ctl == 2 { + opc = 0x23 + } + return e.emit(&instr{ + opcode: []byte{0x0F, opc}, + modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7), + sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8, + }) + } + + // MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX + // load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the + // load opcode. The XMM MOVQ forms follow below. + if m, ok := src.(Reg); ok && m.mmx { + switch d := dst.(type) { + case Reg: + if !d.mmx { + return fmt.Errorf("MOV: MMX register moves stay inside the M bank") + } + i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1} + if err := setRM(i, d, src, 8); err != nil { + return err + } + return e.emit(i) + case Mem: + i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1} + if err := setRM(i, m, d, 8); err != nil { + return err + } + return e.emit(i) + } + return fmt.Errorf("MOV: invalid MMX destination") + } + if m, ok := dst.(Reg); ok && m.mmx { + srcM, ok := src.(Mem) + if !ok { + return fmt.Errorf("MOV: MMX load takes a memory source") + } + i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1} + if err := setRM(i, m, srcM, 8); err != nil { + return err + } + return e.emit(i) + } + if srcVec || dstVec { if dstVec { if g, ok := src.(Reg); ok && !g.isVec() { @@ -518,6 +622,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error { default: return fmt.Errorf("LEA: source must be a memory operand") } + // LEA accepts the full unsigned 32-bit displacement span where the + // loads and stores reject it beyond the signed one; the wide values + // ride the same disp32 bytes as their two's-complement bit pattern. + if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 { + c := m + c.Disp = int64(int32(uint32(m.Disp))) + src = c + } i := newInstr(size, []byte{0x8D}) if err := setRM(i, dstReg, src, size); err != nil { return err @@ -663,6 +775,18 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error { func (e *enc) encodeImul(ops []Operand, size int) error { switch len(ops) { + case 1: + // The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the + // implied destination (the toolchain's one-register shape). + opc := byte(0xF7) + if size == 1 { + opc = 0xF6 + } + i := newInstr(size, []byte{opc}) + if err := setRMDigit(i, 5, ops[0], size); err != nil { + return err + } + return e.emit(i) case 2: // Two shapes. The leading-immediate spelling IMUL $imm, r multiplies // r in place (dst = rm = r): the shape GOROOT's clock code writes. @@ -697,7 +821,7 @@ func (e *enc) encodeImul(ops []Operand, size int) error { // r/m operand (setRM takes registers and memory alike). return e.encodeImulImm(imm, ops[1], dstReg, size) } - return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) + return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops)) } // encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8 @@ -742,6 +866,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error { w16 := size == 2 switch op := ops[0].(type) { case Reg: + // Segment registers: FS and GS carry their own one-byte opcodes + // under 0F (A0/A8 push, A1/A9 pop); the other four spellings are + // not pushable in 64-bit mode. + if n, isSeg := op.segNumber(); isSeg { + switch n { + case 4: // FS + if push { + return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1}) + } + return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1}) + case 5: // GS + if push { + return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1}) + } + return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1}) + } + return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode") + } + if op.mmx || op.isVec() || op.fp || op.ctl != 0 { + return fmt.Errorf("PUSH/POP: invalid register operand") + } base := byte(0x50) // PUSH r; POP is 0x58 if !push { base = 0x58 @@ -1093,6 +1238,38 @@ var sseMoveTable = map[string]sseMove{ "MOVSS": {0xF3, 0x10, 0x11}, // scalar single } +// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register +// rides the reg field and memory r/m (the non-temporal store). +var sseStoreOnly = map[string]struct { + prefix byte + op byte +}{ + "MOVNTDQ": {0x66, 0xE7}, +} + +// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the +// destination memory). +func (e *enc) encodeSSEStoreOnly(mnem string, m struct { + prefix byte + op byte +}, ops []Operand) error { + if len(ops) != 2 { + return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + srcReg, ok := ops[0].(Reg) + if !ok || !srcReg.isVec() { + return fmt.Errorf("%s source must be a vector register", mnem) + } + if !isX86Mem(ops[1]) { + return fmt.Errorf("%s destination must be a memory operand", mnem) + } + i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} + if err := setRM(i, srcReg, ops[1], 8); err != nil { + return err + } + return e.emit(i) +} + // encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the // load form (reg = destination), matching the Go assembler. func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { @@ -1308,14 +1485,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error { } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) - if !ok || !dstReg.isVec() { + if !ok || (!dstReg.isVec() && !dstReg.mmx) { return fmt.Errorf("SSE binary destination must be a vector register") } + // The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix: + // PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE. + prefix := m.prefix + if dstReg.mmx { + if prefix != 0x66 { + return fmt.Errorf("SSE binary: this form takes no MMX register operand") + } + prefix = 0 + } opcode := []byte{0x0F, m.op} if m.map38 { opcode = []byte{0x0F, 0x38, m.op} } - i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} + i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } @@ -1791,12 +1977,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error { // immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family: // F2 0F C2 with reg = dst, rm = src. func (e *enc) encodeCmpsd(ops []Operand) error { + return e.encodeSSECmp("CMPSD", 0xF2, ops) +} + +// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD): +// 0F C2 /r ib with the predicate immediate last in Plan 9 order +// (src, dst, $imm) and the packed forms' prefixes. +func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error { if len(ops) != 3 { - return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops)) + return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops)) } imm, ok := ops[2].(Imm) if !ok { - return fmt.Errorf("CMPSD predicate must be an immediate") + return fmt.Errorf("%s predicate must be an immediate", mnem) } immByte, err := imm8(int64(imm)) if err != nil { @@ -1804,9 +1997,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error { } dstReg, ok2 := ops[1].(Reg) if !ok2 || !dstReg.isVec() { - return fmt.Errorf("CMPSD destination must be a vector register") + return fmt.Errorf("%s destination must be a vector register", mnem) } - i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1} + i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[0], 8); err != nil { return err } diff --git a/asm/link_test.go b/asm/link_test.go index 659c6d0..66f30cf 100644 --- a/asm/link_test.go +++ b/asm/link_test.go @@ -60,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234 } } -// TestAssembleFileErrors checks the static-symbol error paths. +// TestAssembleFileErrors checks the static-symbol error paths. A reference +// to a static symbol no GLOBL defines defers to the linker exactly as the +// toolchain does (an external relocation), so it is not an error here. func TestAssembleFileErrors(t *testing.T) { cases := []struct { name string src string want string // substring of the error }{ - { - "undefined symbol", - ` -#include "textflag.h" -TEXT ·f(SB), NOSPLIT, $0 - VMOVDQU nope<>(SB), X0 - RET -`, - "undefined symbol", - }, { "DATA without GLOBL", ` diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index ece5719..95c17aa 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int { if mnem == "RET" { return len(loong64Return(fi)) } + // BYTE lays down one raw byte per operand, a front-end pseudo-op the + // toolchain spells only on x86 but accepts here the same way the arm64 + // and riscv64 encoders do (a superset spelling, shippable via the goobj + // path). + if mnem == "BYTE" { + return len(ops) + } switch mnem { case "END", "FUNCDATA", "PCDATA": return 0 // bookkeeping statements contribute no bytes @@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } return l64wordLE(uint32(immFromOperand(ops[0]))), nil + case "BYTE": + // BYTE $b lays down one raw byte per operand, the same front-end + // pseudo-op the arm64 and riscv64 encoders accept. + var out []byte + for _, op := range ops { + b := l64Imm64(op) + if b < 0 || b > 0xFF { + return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b) + } + out = append(out, byte(b)) + } + return out, nil case "END", "FUNCDATA", "PCDATA", "GETCALLERPC": // The assembler's bookkeeping statements. END, FUNCDATA and PCDATA // contribute no bytes, the same shapes GOARCH=loong64 go tool asm @@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo } return l64wordLE(l64rr(enc.op, rj, rd)), nil + case l64Fllsc: + // LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both + // 2R encodings op | rj<<5 | rd against a zero-offset memory operand + // (the toolchain's C_ZOREG, which rejects any displacement). + rd, rj, off, _, err := l64MemOperands(ops, fi) + if err != nil { + return nil, fmt.Errorf("%s: %w", mnem, err) + } + if off != 0 { + return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem) + } + return l64wordLE(l64rr(enc.op, rj, rd)), nil + + case l64Fscq: + // SCQ first, middle, (base): op | middle<<10 | base<<5 | first, + // against a zero-offset memory operand as with the LL/SC pair. + if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) { + return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem) + } + first, middle := l64Reg(ops[0]), l64Reg(ops[1]) + rj, off := l64MemWithFrame(ops[2], fi) + if first < 0 || middle < 0 || rj < 0 { + return nil, fmt.Errorf("%s: invalid register operand", mnem) + } + if off != 0 { + return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem) + } + return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil + case l64Firr: // LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd. if len(ops) < 2 || !isImmOperand(ops[0]) { @@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in return rd, rj, off, load, nil } +// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text: +// the shared immediate parse reduces it to the bare number and keeps only +// the text as a witness of the base register. ok reports the form was +// found, with the base's register number (or -1 when the name is not a +// general register). +func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) { + if op.Kind != ast.OpImmediate || !op.Imm.HasVal { + return 0, 0, false + } + raw := strings.ReplaceAll(op.Raw, " ", "") + if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") { + return 0, 0, false + } + open := strings.LastIndexByte(raw, '(') + if open < 2 { + return 0, 0, false + } + base = loong64RegNum(raw[open+1 : len(raw)-1]) + v := op.Imm.Val + if op.Imm.Neg { + v = -v + } + return int32(v), base, base >= 0 +} + // ---- the MOV pseudo-instruction ---- // encodeLOONG64Mov encodes the MOV family, the load/store/immediate @@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs } return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil } + // MOVx $off(rj), rd computes an address: the toolchain's `mov + // $soreg, r` case, a plain addi.d whatever the move's width (both + // MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata). + // A wider offset materialises in R30 first (lu12i.w + ori + add.d, + // its case 10). The immediate's Raw carries the base register, + // which the shared immediate parse reduces to the bare number. + if off, base, ok := l64ImmMem(src); ok { + rd := l64Reg(dst) + if rd < 0 { + return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem) + } + if loong64RegClass(operandRegName(dst)) == l64ClsFP { + return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem) + } + if off >= -2048 && off <= 2047 { + return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil + } + return l64WordsLE( + l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30), + l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30), + l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd), + ), nil + } rd := l64Reg(dst) if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) @@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // pcalau12i + addi.d } + // The $off(rj) address immediate: addi.d in the 12-bit window, + // lu12i.w + ori + add.d beyond it (the toolchain's case 10). + if off, _, ok := l64ImmMem(src); ok { + if off >= -2048 && off <= 2047 { + return 4 + } + return 12 + } if loong64RegClass(operandRegName(dst)) == l64ClsFP { return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid) } @@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) { return l64Mem(op) } +// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's +// vector table falls back to the zero register as the FP-relative base +// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand +// through R3), so the vector moves keep the resolved offset but the zero +// base, exactly as `go tool asm` emits them. +func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) { + rj, off = l64MemWithFrame(op, fi) + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" { + rj = 0 + } + return rj, off +} + // l64MemOffset returns the resolved byte offset of a memory operand. func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 { _, off := l64MemWithFrame(op, fi) @@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string { type l64VecOperand struct { num int // 5-bit register number lasx bool // X bank (LASX) rather than V (LSX) - width byte // suffix width letter (B/H/W/V), 0 on a bare register + width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register lanes int // lane count of a width suffix (B16 → 16) elem int // element index of a .T[i] suffix hasEl bool // the suffix names an element (.T[i]) @@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) { } i++ w := name[i] - if w != 'B' && w != 'H' && w != 'W' && w != 'V' { + if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' { return v, false } v.width, v.hasSuf = w, true @@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([] // VMOVQ rj, vd.T vreplgr2vr (duplicate a general register) // VMOVQ vj.T[i], rd vpickve2gr (extract one element) // VMOVQ rj, vd.T[i] vinsgr2vr (insert one element) +// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX) +// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX) +// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX) +// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX) func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) { enc := l64VmovqTable[lasx] bank := "V" @@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b return loong64RegNum(name), nil } + // Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the + // source element width matching the destination arrangement. An LSX-only + // form: the toolchain's table gives vreplvei no LASX counterpart. + if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl { + if lasx || src.lasx || dst.lasx { + return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank) + } + if src.unsig { + return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix") + } + if src.width != dst.width { + return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw) + } + if _, ok := l64VecSuffixWidth(false, dst); !ok { + return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw) + } + var op uint32 + limit := 0 + switch src.width { + case 'B': + op, limit = enc.rveiB, 15 + case 'H': + op, limit = enc.rveiH, 7 + case 'W': + op, limit = enc.rveiW, 3 + default: + op, limit = enc.rveiD, 1 + } + if src.elem > limit { + return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit) + } + return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil + } + + // Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}), + // a bare X source into an arranged X destination. LASX only. + if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl { + if !lasx || src.lasx != lasx || dst.lasx != lasx { + return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank) + } + var op uint32 + switch dst.width { + case 'B': + if dst.lanes != 32 { + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + op = enc.rve0B + case 'H': + if dst.lanes != 16 { + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + op = enc.rve0H + case 'W': + if dst.lanes != 8 { + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + op = enc.rve0W + case 'V': + if dst.lanes != 4 { + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + op = enc.rve0D + case 'Q': + if dst.lanes != 2 { + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + op = enc.rve0Q + default: + return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw) + } + return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil + } + + // Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X + // source into one word or double-word lane. LASX only. + if srcVec && dstVec && !src.hasSuf && dst.hasEl { + if !lasx || src.lasx != lasx || dst.lasx != lasx { + return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank) + } + op, limit := enc.xinsW, 7 + if dst.width != 'W' { + op, limit = enc.xinsD, 3 + if dst.width != 'V' { + return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw) + } + } + if dst.elem > limit { + return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit) + } + return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil + } + + // Element extract into a vector register: XVMOVQ xj.T[i], xd + // (xvpickve.{w,d}), one word or double-word lane out to a bare X + // register. LASX only. + if srcVec && src.hasEl && dstVec && !dst.hasSuf { + if !lasx || src.lasx != lasx || dst.lasx != lasx { + return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank) + } + op, limit := enc.xpickW, 7 + if src.width != 'W' { + op, limit = enc.xpickD, 3 + if src.width != 'V' { + return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw) + } + } + if src.elem > limit { + return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit) + } + return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil + } + // Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant), // both operands bare registers of the same bank. if srcVec && dstVec { @@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b } return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil } - rj, off := l64MemWithFrame(ops[1], fi) + rj, off := l64VmovqMem(ops[1], fi) if rj < 0 || off < -2048 || off > 2047 { return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]") } @@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b } return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil } - rj, off := l64MemWithFrame(ops[0], fi) - if rj < 0 || off < -2048 || off > 2047 { - return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") + rj, off := l64VmovqMem(ops[0], fi) + if rj < 0 { + return nil, fmt.Errorf("VMOVQ: invalid load operand") } op := enc.ld if dst.hasSuf { @@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b if !ok { return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw) } + // vldrepl keeps the byte offset raw for bytes and scales it by + // the element width for the wider forms, the immediate field + // shrinking a bit per scale exactly as the toolchain encodes it + // (the field mask keeps the two's complement inside its width). + scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047 switch w { case 0: op = enc.replB case 1: op = enc.replH + scale, mask, lo, hi = 2, 0x7FF, -1024, 1023 case 2: op = enc.replW + scale, mask, lo, hi = 4, 0x3FF, -512, 511 default: op = enc.replD + scale, mask, lo, hi = 8, 0x1FF, -256, 255 } + if off%int32(scale) != 0 { + return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale) + } + off /= int32(scale) + if off < int32(lo) || off > int32(hi) { + return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale) + } + off &= mask + } else if off < -2048 || off > 2047 { + return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") } return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil } diff --git a/asm/loong64_encode.go b/asm/loong64_encode.go index 99f13d0..2cd1691 100644 --- a/asm/loong64_encode.go +++ b/asm/loong64_encode.go @@ -279,6 +279,8 @@ const ( l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd + l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand) + l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand ) // l64Enc is one instruction's encoding: its bit layout (format) and the @@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{} // such as vshuf.b). var l64Vec4R = map[string]bool{} -// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit -// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. +// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to +// its exact bit range, read off `go tool objdump` of GOARCH=loong64 +// `go tool asm` kernels and the toolchain's specialLsxMovInst table. type l64VmovqEnc struct { ld, st, ldx, stx uint32 // plain and indexed load/store replB, replH, replW, replD uint32 // vldrepl: load and replicate element @@ -350,6 +353,11 @@ type l64VmovqEnc struct { ins uint32 // vinsgr2vr element insert dup uint32 // vreplgr2vr duplicate (width in [11:10]) move uint32 // vori.b/xvori.b $0 register move + rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX) + rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX) + rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto + xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX) + xpickW, xpickD uint32 // xvpickve: extract element (LASX) } var l64VmovqTable = map[bool]l64VmovqEnc{ @@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{ replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, + rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11, }, true: { // XVMOVQ, the LASX (X) bank ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, + rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10, + rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10, + xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12, + xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12, }, } @@ -373,7 +386,7 @@ func init() { "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SGT": 0x24 << 15, "SGTU": 0x25 << 15, - "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, + "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, "ORN": 0x2c << 15, "ANDN": 0x2d << 15, "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, @@ -467,6 +480,20 @@ func init() { l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} + // Acquire/release LL/SC (2R against a zero-offset memory operand): + // LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding + // op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go + // (ll.acq.{w,d}, sc.rel.{w,d}). + l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10} + l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10} + l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10} + l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10} + + // SCQ (sc.q first, middle, (base)) keeps its own operand order: the + // encoding is op | middle<<10 | base<<5 | first, the memory operand's + // base in the rj field, not the toolchain's generic 3R layout. + l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15} + // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // selected by the operand kind; the shift mnemonics pair the 3R form // with a 5/6-bit shift immediate. @@ -868,7 +895,7 @@ func init() { "VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF}, "XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF}, "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, - "XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, + "XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F}, // vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a // 7-bit field, but the toolchain range-checks it down to si5 as well // (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64). @@ -877,7 +904,7 @@ func init() { "VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F}, "XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F}, "VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F}, - "XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F}, + "XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F}, // vslti compares against a signed (or, in the U spellings, unsigned) // si5/ui5 constant. "VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F}, diff --git a/asm/loong64_encode_test.go b/asm/loong64_encode_test.go index 77ac945..15b6c0b 100644 --- a/asm/loong64_encode_test.go +++ b/asm/loong64_encode_test.go @@ -831,3 +831,230 @@ TEXT ·atoms(SB), NOSPLIT, $0 0x4C000020, ) } + +// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle +// words come from GOARCH=loong64 go tool objdump and the toolchain's own +// loong64enc1.s golden bytes. +func TestLOONG64_llacqScrel(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·llsc(SB), NOSPLIT, $0 + LLACQW (R5), R4 + LLACQV (R5), R4 + SCRELW R4, (R6) + SCRELV R4, (R6) + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x385780A4, // ll.acq.w r4, r5 + 0x385788A4, // ll.acq.d r4, r5 + 0x385784C4, // sc.rel.w r4, r6 + 0x38578CC4, // sc.rel.d r4, r6 + 0x4C000020, + ) + // The toolchain accepts the zero-offset memory form alone. + for i, src := range []string{ + `TEXT ·e(SB), NOSPLIT, $0 + LLACQW 4(R5), R4 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + SCRELV R4, 8(R6) + RET +`, + } { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("case %d: expected an error, got none", i) + } + } +} + +// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move +// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the +// toolchain's loong64enc1.s. +func TestLOONG64_vmovqSuffixed(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·vmovq(SB), NOSPLIT, $0 + VMOVQ V1.B[3], V9.B16 + VMOVQ V2.H[2], V8.H8 + VMOVQ V3.W[1], V7.W4 + VMOVQ V4.V[0], V6.V2 + XVMOVQ X0, X31.B32 + XVMOVQ X1, X30.H16 + XVMOVQ X2, X29.W8 + XVMOVQ X3, X28.V4 + XVMOVQ X3, X27.Q2 + XVMOVQ X0, X31.W[7] + XVMOVQ X1, X29.W[0] + XVMOVQ X3, X28.V[3] + XVMOVQ X4, X27.V[0] + XVMOVQ X31.W[7], X0 + XVMOVQ X29.W[0], X1 + XVMOVQ X28.V[3], X8 + XVMOVQ X27.V[0], X9 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x72F78C29, // vreplvei.b v9, v1, 3 + 0x72F7C848, // vreplvei.h v8, v2, 2 + 0x72F7E467, // vreplvei.w v7, v3, 1 + 0x72F7F086, // vreplvei.d v6, v4, 0 + 0x7707001F, // xvreplve0.b x31, x0 + 0x7707803E, // xvreplve0.h x30, x1 + 0x7707C05D, // xvreplve0.w x29, x2 + 0x7707E07C, // xvreplve0.d x28, x3 + 0x7707F07B, // xvreplve0.q x27, x3 + 0x76FFDC1F, // xvinsve0.w x31, x0, 7 + 0x76FFC03D, // xvinsve0.w x29, x1, 0 + 0x76FFEC7C, // xvinsve0.d x28, x3, 3 + 0x76FFE09B, // xvinsve0.d x27, x4, 0 + 0x7703DFE0, // xvpickve.w x0, x31, 7 + 0x7703C3A1, // xvpickve.w x1, x29, 0 + 0x7703EF88, // xvpickve.d x8, x28, 3 + 0x7703E369, // xvpickve.d x9, x27, 0 + 0x4C000020, + ) + // The rejected shapes: a width mismatch between the element and the + // arrangement, an element index past the lane count, a wrong-bank + // vreplvei and an arrangement the LASX bank does not spell. + for i, src := range []string{ + `TEXT ·e(SB), NOSPLIT, $0 + VMOVQ V1.H[3], V9.B16 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + VMOVQ V1.B[16], V9.B16 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + XVMOVQ X1.B[3], X9.B32 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + XVMOVQ X0, X31.B16 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + XVMOVQ X0, X31.W[8] + RET +`, + } { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("case %d: expected an error, got none", i) + } + } +} + +// TestLOONG64_parityFixes pins the operand forms whose encodings were found +// diverging from the toolchain by the loong64enc1.s differential: the +// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled +// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate +// opcodes and the zero-register base the toolchain gives FP-relative +// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s. +func TestLOONG64_parityFixes(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·parity(SB), NOSPLIT, $0-32 + MOVW $4(R4), R5 + MOVV $4(R4), R5 + MOVW $65536(R4), R5 + MOVW $-4096(R4), R5 + SCQ R4, R5, (R6) + VMOVQ 2(R4), V1.H8 + VMOVQ -6(R4), V1.H8 + VMOVQ -12(R4), V2.W4 + VMOVQ -16(R4), V3.V2 + XVMOVQ -10(R4), X1.H16 + XVSEQB $0, X2, X4 + XVSEQH $3, X2, X4 + XVSEQW $12, X2, X4 + XVSEQV $15, X2, X4 + XVSEQV $-15, X2, X4 + VMOVQ V2, y+16(FP) + VMOVQ y+16(FP), V2 + VMOVQ V2, x+2030(FP) + XVMOVQ X6, y+16(FP) + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x02C01085, // addi.d $4, r4, r5 + 0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d) + 0x1400021E, // lu12i.w $16, r30 + 0x038003DE, // ori $0, r30, r30 + 0x0010F885, // add.d r5, r4, r30 + 0x15FFFFFE, // lu12i.w $-1, r30 + 0x038003DE, // ori $0, r30, r30 + 0x0010F885, // add.d r5, r4, r30 + 0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first + 0x30400481, // vldrepl.h v1, 2(r4) + 0x305FF481, // vldrepl.h v1, -6(r4) + 0x302FF482, // vldrepl.w v2, -12(r4) + 0x3017F883, // vldrepl.d v3, -16(r4) + 0x325FEC81, // xvldrepl.h x1, -10(r4) + 0x76800044, // xvseqi.b x4, x2, 0 + 0x76808C44, // xvseqi.h x4, x2, 3 + 0x76813044, // xvseqi.w x4, x2, 12 + 0x7681BC44, // xvseqi.d x4, x2, 15 + 0x7681C444, // xvseqi.d x4, x2, -15 + 0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base + 0x2C006002, // vld v2, 24(r0) + 0x2C5FD802, // vst v2, 2038(r0) + 0x2CC06006, // xvst x6, 24(r0) + 0x4C000020, + ) + // Misaligned vldrepl offsets are rejected, as the toolchain does. + for i, src := range []string{ + `TEXT ·e(SB), NOSPLIT, $0 + VMOVQ 3(R4), V1.H8 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + MOVW $4(R4), F1 + RET +`, + } { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("case %d: expected an error, got none", i) + } + } +} + +// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the +// loong64 toolchain does not spell but the arm64 and riscv64 encoders of +// this package already accept for byte-exact data layout (a superset +// spelling, shippable via the goobj path). +func TestLOONG64_bytePseudo(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·bytes(SB), NOSPLIT, $0 + BYTE $2 + BYTE $1; BYTE $0 + BYTE $255 + RET +`) + code := assembleLOONG64Helper(t, fn) + // Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad + // the final word the way any sub-word tail does. + want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C} + if !bytes.Equal(code[:len(want)], want) { + t.Errorf("bytes = % x, want % x", code, want) + } + for _, src := range []string{ + `TEXT ·e(SB), NOSPLIT, $0 + BYTE $256 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + BYTE $-1 + RET +`, + } { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("%q: expected an error, got none", src) + } + } +} diff --git a/asm/reg.go b/asm/reg.go index 3fe9b75..437342e 100644 --- a/asm/reg.go +++ b/asm/reg.go @@ -17,13 +17,28 @@ import "strings" // size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which // occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share // those indices but require one. The mask flag marks the AVX-512 opmask -// registers K0-K7, the fp flag the x87 stack registers F0-F7. +// registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the +// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the +// ctl field the control and debug registers, whose number rides an +// instruction's reg field rather than r/m. type Reg struct { idx int size int // informational width implied by the name; the mnemonic decides high bool // AH/CH/DH/BH mask bool // K0-K7 opmask register fp bool // F0-F7 x87 stack register + mmx bool // M0-M7 MMX register + seg int // segment register number plus one (ES=1..GS=6); 0 = not one + ctl byte // 0 none, 1 CRn control register, 2 DRn debug register +} + +// segNumber returns the segment register number (ES=0..GS=5) when r names a +// bare segment register. +func (r Reg) segNumber() (int, bool) { + if r.seg == 0 { + return 0, false + } + return r.seg - 1, true } // Index returns the register number (0-15 for GPRs, 0-31 for vectors). @@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg { for i := 0; i <= 7; i++ { m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true} } + // MMX: M0..M7. + for i := 0; i <= 7; i++ { + m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true} + } + // Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and + // index spellings of FS and GS are handled before register lookup). + for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} { + m[n] = Reg{idx: i, size: 2, seg: i + 1} + } + // Control and debug registers: CR0..CR15, DR0..DR15. + for i := 0; i <= 15; i++ { + m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1} + m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2} + } return m } diff --git a/asm/vex.go b/asm/vex.go index 8fbc7d9..5acf20d 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -76,6 +76,10 @@ const ( // carries a vector length, so the register the L'L field follows is the // XMM source. vexExtractGPR + // vexBlend4 is the four-operand variable blend `OP mask, src2, src1, + // dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2), + // ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0). + vexBlend4 ) // vexSpec describes one VEX instruction's encoding parameters. @@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{ // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8). "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, - // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8). - "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, + // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its + // double twin under op 01; the in-lane permutes under 04/05. + "VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, + "VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM}, + "VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM}, + "VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM}, + // VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding + // pair, the AES key assistant and the string compares. + "VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM}, + "VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM}, + "VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM}, + "VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM}, + "VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM}, + "VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM}, + "VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM}, + // VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory + // (reg = the XMM source, r/m = the destination). + "VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR}, + "VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR}, + // VEX.128.66.0F3A.W0, the four-operand variable blend with its mask + // register in the /is4 byte. + "VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4}, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2, // imm8). @@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool { if _, ok := vexTable[mnemUpper]; ok { return true } - _, ok := vexMoveTable[mnemUpper] - return ok + if _, ok := vexMoveTable[mnemUpper]; ok { + return true + } + // The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand + // count in encodeVex. + switch mnemUpper { + case "VMOVHPD", "VMOVLPD": + return true + } + return false } // encodeVex encodes a VEX instruction with operands in Plan 9 order. @@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops) } } + // The high/low double moves split by operand count: three operands + // load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64, + // the reversed store layout). + if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" { + loadOp, storeOp := byte(0x16), byte(0x17) + if mnemUpper == "VMOVLPD" { + loadOp, storeOp = 0x12, 0x13 + } + switch len(ops) { + case 3: + return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops) + case 2: + return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops) + } + return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops)) + } spec := vexTable[mnemUpper] switch spec.form { case vexNDS3: @@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { return e.encodeVexNDS3Imm(spec, ops) case vexExtract: return e.encodeVexExtract(spec, ops) + case vexExtractGPR: + return e.encodeVexExtractGPR(spec, ops) + case vexBlend4: + return e.encodeVexBlend4(spec, ops) case vexRMSrcLen: return e.encodeVexRMSrcLen(mnemUpper, spec, ops) case vexZero: @@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error { return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1]) } +// encodeVexExtractGPR encodes the lane extract to a general-purpose register +// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in +// ModRM.reg and the destination in r/m, L = 0. +func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error { + if len(ops) != 3 { + return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops)) + } + imm, src, dst := ops[0], ops[1], ops[2] + immVal, ok := imm.(Imm) + if !ok { + return fmt.Errorf("extract lane must be an immediate") + } + srcReg, ok := src.(Reg) + if !ok || !srcReg.isVec() || srcReg.size != 16 { + return fmt.Errorf("extract source must be an XMM register") + } + if _, isReg := dst.(Reg); !isReg && !memOperand(dst) { + return fmt.Errorf("extract destination must be a register or memory") + } + rBit := 0 + if srcReg.idx >= 8 { + rBit = 1 + } + if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil { + return err + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + e.out = append(e.out, immByte) + return nil +} + +// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB): +// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m = +// src2 and the mask XMM register in the trailing /is4 byte. +func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error { + if len(ops) != 4 { + return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops)) + } + mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] + maskReg, ok := mask.(Reg) + if !ok || !maskReg.isVec() || maskReg.size != 16 { + return fmt.Errorf("blend mask must be an XMM register") + } + vvvvReg, ok := src1.(Reg) + if !ok || !vvvvReg.isVec() { + return fmt.Errorf("blend second source must be a vector register") + } + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("blend destination must be a vector register") + } + rBit := 0 + if dstReg.idx >= 8 { + rBit = 1 + } + if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil { + return err + } + // The /is4 byte names the mask register: bits [3:0] its low nibble, + // bit 7 the fourth register bit (X8-X15). + e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4)) + return nil +} + // encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, // VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector // move uses the store-form layout (reg = source, rm = destination), matching