feat(asm): encode the amd64 and loong64 tails of the corpus testdata

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:43 +02:00
parent 2f679326c2
commit bafb2fd130
12 changed files with 1404 additions and 81 deletions
+218 -21
View File
@@ -808,17 +808,43 @@ func isJumpMnemonic(mnem string) bool {
if mnem == "JMP" || mnem == "CALL" { if mnem == "JMP" || mnem == "CALL" {
return true return true
} }
if isLoopMnemonic(mnem) {
return true
}
_, ok := condCode(mnem) _, ok := condCode(mnem)
return ok return ok
} }
// isLoopMnemonic reports the LOOP family, rel8 alone (E0-E2).
func isLoopMnemonic(mnem string) bool {
switch mnem {
case "LOOP", "LOOPE", "LOOPNE":
return true
}
return false
}
// loopOpcode maps the LOOP family to its E0-E2 opcode.
func loopOpcode(mnem string) byte {
switch mnem {
case "LOOPE":
return 0xE1
case "LOOPNE":
return 0xE0
}
return 0xE2
}
// jumpSize returns the length of a jump instruction in the requested form: // jumpSize returns the length of a jump instruction in the requested form:
// short (rel8) where available, otherwise the rel32 form. CALL is always // short (rel8) where available, otherwise the rel32 form. CALL is always
// rel32. // rel32; the LOOP family is rel8 alone.
func jumpSize(mnem string, long bool) int { func jumpSize(mnem string, long bool) int {
if mnem == "CALL" { if mnem == "CALL" {
return 5 // opcode + rel32 return 5 // opcode + rel32
} }
if isLoopMnemonic(mnem) {
return 2 // opcode + rel8, the only form
}
if !long { if !long {
return 2 // opcode + rel8 return 2 // opcode + rel8
} }
@@ -945,6 +971,16 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) { if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
return encodeTLSBaseLoad(s, fi, link) return encodeTLSBaseLoad(s, fi, link)
} }
// The old paired-register shift spelling, SHLL CX, R11:AX (a colon
// between the two registers), is the toolchain's SHLD family: SHLDL CL,
// AX, R11 with the count register first, the paired source in the reg
// field and the pair's head in r/m.
if code, ps, err := encodeColonShift(s, mnemUpper, fi, link); code != nil || err != nil {
if err != nil {
return nil, nil, nil, err
}
return code, ps, nil, nil
}
_, size := splitSize(mnemUpper) _, size := splitSize(mnemUpper)
if size == 0 { if size == 0 {
size = 8 size = 8
@@ -1051,6 +1087,66 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
return nil, nil return nil, nil
} }
// encodeColonShift encodes the paired-register shift spellings, SHLx CX,
// dst:src: the toolchain reads them as the SHLD family (double-precision
// shift by CL), reg = the paired source, r/m = the pair's head. The second
// operand's raw text carries the colon; ok reports the spelling was found.
func encodeColonShift(s *ast.Instr, mnemUpper string, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
base, _ := strings.CutPrefix(mnemUpper, "SHL")
if base == mnemUpper || len(s.Operands) != 2 {
return nil, nil, nil
}
_, size := splitSize(mnemUpper)
raw := strings.ReplaceAll(s.Operands[1].Raw, " ", "")
head, tail, ok := strings.Cut(raw, ":")
if !ok || head == "" || tail == "" {
return nil, nil, nil
}
headReg, ok1 := ParseReg(head)
srcReg, ok2 := ParseReg(tail)
if !ok1 || !ok2 {
return nil, nil, fmt.Errorf("%s: invalid paired register %q", mnemUpper, s.Operands[1].Raw)
}
cnt, err := operandFromAST(mnemUpper, s.Operands[0], size, fi, link)
if err != nil {
return nil, nil, err
}
cntReg, ok := cnt.(Reg)
if !ok || cntReg.idx != 1 {
return nil, nil, fmt.Errorf("%s: the paired-register form counts in CL", mnemUpper)
}
// SHLD r/m, reg, CL: 0F A5 (REX.W for the 64-bit width).
e := &enc{}
i := &instr{rexW: size == 8, opcode: []byte{0x0F, 0xA5}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, headReg, size); err != nil {
return nil, nil, err
}
if err := e.emit(i); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for j, p := range e.patches {
ps[j] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
}
// trailingIndexGroup recovers a trailing "(index*scale)" or "(index)" group
// from an operand's raw text: the symbol-pseudo parse returns before the
// index group, so foo(SP)(AX*1) keeps its index only in the spelling.
func trailingIndexGroup(raw string) (string, int, bool) {
compact := strings.ReplaceAll(raw, " ", "")
if !strings.HasSuffix(compact, ")") {
return "", 0, false
}
open := strings.LastIndex(compact, "(")
if open < 2 || !strings.Contains(compact[:open], ")") {
return "", 0, false // one group alone: no trailing index
}
name, scale, _, ok := cutParenGroup(compact[open:])
return name, scale, ok
}
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label or from a numeric ±N(PC) instruction count, in the short // target label or from a numeric ±N(PC) instruction count, in the short
// (rel8) or long (rel32) form. numTarget is the resolved byte offset of a // (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
@@ -1085,9 +1181,15 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
if mnem == "JMP" { if mnem == "JMP" {
return []byte{0xEB, byte(int8(rel))}, nil return []byte{0xEB, byte(int8(rel))}, nil
} }
if isLoopMnemonic(mnem) {
return []byte{loopOpcode(mnem), byte(int8(rel))}, nil
}
cc, _ := condCode(mnem) cc, _ := condCode(mnem)
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
} }
if isLoopMnemonic(mnem) {
return nil, fmt.Errorf("%s has no long form", mnem)
}
switch mnem { switch mnem {
case "JMP": case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil return append([]byte{0xE9}, le32(rel)...), nil
@@ -1139,6 +1241,72 @@ func labelName(op *ast.Operand) (string, bool) {
return "", false return "", false
} }
// jumpOperand returns the branch-target operand of a JMP/CALL, rewriting the
// `*`-prefixed indirect spellings (JMP *(R12), JMP *4(SP)) into their plain
// memory form. The star marks an indirect target and changes no bytes; the
// address parser leaves the operand's address empty because of the leading
// star, so the fields are rebuilt from the raw text onto a copy of the
// operand, never on the shared syntax tree.
func jumpOperand(s *ast.Instr) *ast.Operand {
if len(s.Operands) != 1 {
return nil
}
op := s.Operands[0]
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "*")
if !ok {
return op
}
var addr ast.Address
if i := strings.IndexByte(inner, '('); i > 0 {
v, err := strconv.ParseInt(inner[:i], 0, 64)
if err != nil {
return op
}
addr.Offset, addr.HasOff = v, true
inner = inner[i:]
}
base, _, rest, ok := cutParenGroup(inner)
if !ok {
return op
}
if base != "" {
addr.Base = base
}
if rest != "" {
idx, scale, _, ok := cutParenGroup(rest)
if ok && idx != "" {
addr.Index = idx
addr.Scale = scale
}
}
c := *op
c.Addr = addr
return &c
}
// cutParenGroup splits a leading "(name)" or "(name*n)" off s, returning the
// inner text, the scale it names (1 when the group spells no multiplier) and
// the remainder.
func cutParenGroup(s string) (name string, scale int, rest string, ok bool) {
if !strings.HasPrefix(s, "(") {
return "", 0, "", false
}
i := strings.IndexByte(s, ')')
if i < 0 {
return "", 0, "", false
}
inner, rest := s[1:i], s[i+1:]
if before, after, ok := strings.Cut(inner, "*"); ok {
n, err := strconv.Atoi(after)
if err != nil {
return "", 0, "", false
}
return before, n, rest, true
}
return inner, 1, rest, true
}
// indirectJumpTarget reports whether the JMP/CALL operand addresses a // indirectJumpTarget reports whether the JMP/CALL operand addresses a
// register or a memory location rather than a label or a static symbol. // register or a memory location rather than a label or a static symbol.
// A bare identifier is a register when the register table knows the name and // A bare identifier is a register when the register table knows the name and
@@ -1147,7 +1315,14 @@ func indirectJumpTarget(s *ast.Instr) bool {
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr { if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
return false return false
} }
a := s.Operands[0].Addr op := jumpOperand(s)
if op == nil {
return false
}
if op != s.Operands[0] {
return true // the star marker spells an indirect target
}
a := op.Addr
// ±N(PC) is the numeric relative form, the PC counts instructions from // ±N(PC) is the numeric relative form, the PC counts instructions from
// the branch: relative, not indirect. // the branch: relative, not indirect.
if a.Base == "PC" || a.Index == "PC" { if a.Base == "PC" || a.Index == "PC" {
@@ -1165,18 +1340,20 @@ func indirectJumpTarget(s *ast.Instr) bool {
} }
// encodeIndirectJump assembles a JMP/CALL through a register or memory // encodeIndirectJump assembles a JMP/CALL through a register or memory
// operand, which carries no relocation and no label to resolve. // operand, which carries no relocation and no label to resolve. The
// `*`-prefixed spellings go through jumpOperand first, their star rebuilt
// into a plain memory operand.
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands)) op := s.Operands[0]
for i, op := range s.Operands { if cleaned := jumpOperand(s); cleaned != nil {
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) op = cleaned
if err != nil { }
return nil, err o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
} if err != nil {
ops[i] = o return nil, err
} }
e := &enc{} e := &enc{}
if err := e.encodeIndirectBranch(mnem, ops); err != nil { if err := e.encodeIndirectBranch(mnem, []Operand{o}); err != nil {
return nil, err return nil, err
} }
return e.out, nil return e.out, nil
@@ -1245,31 +1422,51 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
off := a.Sym.Offset + fi.fpAdjust off := a.Sym.Offset + fi.fpAdjust
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
} }
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP). // SP-relative local: x-N(SP) → (spAdjust + offset)(SP), keeping a scaled
// index beside the virtual stack pointer (foo(SP)(AX*1)). The
// symbol-pseudo parse returns before the index group, so the index
// is recovered from the raw text when the address lacks it.
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" { if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
off := fi.spAdjust + a.Sym.Offset off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil m := Mem{Base: spReg, Disp: off, HasBase: true, Size: size}
if name, scale, ok := trailingIndexGroup(op.Raw); ok {
idx, ok := ParseReg(name)
if !ok {
return nil, fmt.Errorf("unknown index register %q", name)
}
m.Index = idx
m.Scale = scale
m.HasIndex = true
}
return m, nil
} }
// SB (global symbol): a symbol defined in the same file (GLOBL) is // SB (global symbol): a symbol defined in the same file (GLOBL) is
// encoded RIP-relative and resolved by the file-level layout; // encoded RIP-relative and resolved by the file-level layout;
// anything not defined here needs object-file emission. // anything not defined here needs object-file emission. A static
// (file-local) spelling of an undefined symbol defers the same way
// the toolchain does: the relocation names it and the linker decides.
if a.Sym != nil && a.Sym.Pseudo == "SB" { if a.Sym != nil && a.Sym.Pseudo == "SB" {
if link == nil || link.symbols == nil { if link == nil || link.symbols == nil {
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name) return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
} }
if !link.symbols[a.Sym.Name] { if !link.symbols[a.Sym.Name] && !link.allowExternal {
if a.Sym.Static { return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
if !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
} }
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
} }
// Memory with a real base register: (base), off(base), (base)(index*scale). // Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" { if a.Base != "" {
// The TLS pseudo-base, off(TLS): the segment-prefixed absolute
// the thread-local access lowers to, 64 8B 04 25 with its
// R_TLS_LE patch site on the disp32.
if a.Base == "TLS" {
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS // Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
// spellings. The segment override prefixes a disp32 absolute // spellings. The segment override prefixes a disp32 absolute
// reference with no relocation. // reference with no relocation.
+13
View File
@@ -20,11 +20,24 @@ func Encodable(mnemonic string) bool {
switch upper { switch upper {
case "RET", "NOP", "CALL", "JMP", case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2", "POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The SSE compare family sharing CMPSD's predicate-last shape, the
// far return with its stack pop, the loop family, the bank-crossing
// MMX moves and the one-operand system controls.
"CMPSS", "CMPPS", "CMPPD", "RETFL",
"LOOP", "LOOPE", "LOOPNE",
"MOVDQ2Q", "MOVQ2DQ",
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
// The literal-data pseudo-ops, the accepted-and-ignored END and // The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust. // bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA": "BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true return true
} }
if _, ok := sysUnaryTable[upper]; ok {
return true
}
if _, ok := sseStoreOnly[upper]; ok {
return true
}
if _, ok := noOperandTable[upper]; ok { if _, ok := noOperandTable[upper]; ok {
return true return true
} }
+75 -3
View File
@@ -73,9 +73,23 @@ func (e *enc) encode(mnem string, ops []Operand) error {
// Fixed-name instructions (no size suffix). // Fixed-name instructions (no size suffix).
switch { switch {
case upper == "RET": case upper == "RET":
// RET sym(SB), the absolute return: the toolchain encodes it as a
// tail jump, E9 rel32 with a call relocation against the symbol.
if len(ops) == 1 {
if m, ok := ops[0].(sbMem); ok {
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
}
return fmt.Errorf("RET: unsupported operand")
}
if len(ops) != 0 {
return fmt.Errorf("RET expects no operands, got %d", len(ops))
}
return e.encodeRet() return e.encodeRet()
case upper == "NOP": case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1}) // The toolchain consumes every NOP statement as a pseudo and emits
// nothing for it, operands included (a bare NOP, NOP AX and
// NOP sym(SB) all vanish from the object).
return nil
case upper == "CALL" || upper == "JMP": case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP). // Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler. // Anything else is a rel32 against a label resolved by the assembler.
@@ -102,6 +116,15 @@ func (e *enc) encode(mnem string, ops []Operand) error {
} }
return e.emit(&instr{opcode: op, modrm: -1, sib: -1}) return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
} }
// One-operand system instructions whose reg field is a fixed digit:
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
if m, ok := sysUnaryTable[upper]; ok {
return e.encodeSysUnary(upper, m, ops)
}
// The store-only SSE moves (the non-temporal store).
if m, ok := sseStoreOnly[upper]; ok {
return e.encodeSSEStoreOnly(upper, m, ops)
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings // POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported. // are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper { switch upper {
@@ -117,14 +140,63 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1}) return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT": case "INT":
return e.encodeInt(ops) return e.encodeInt(ops)
// The LOOP family outside the assembler's label settlement: the operand
// is the already-computed rel8 (E0-E2).
case "LOOP", "LOOPE", "LOOPNE":
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok || !fits8(int64(imm)) {
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
}
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
case "LDMXCSR": case "LDMXCSR":
return e.encodeMxcsr(2, ops) return e.encodeMxcsr(2, ops)
case "STMXCSR": case "STMXCSR":
return e.encodeMxcsr(3, ops) return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes // CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm). // LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
case "CMPSD": case "CMPSD":
return e.encodeCmpsd(ops) return e.encodeSSECmp("CMPSD", 0xF2, ops)
case "CMPSS":
return e.encodeSSECmp("CMPSS", 0xF3, ops)
case "CMPPS":
return e.encodeSSECmp("CMPPS", 0x00, ops)
case "CMPPD":
return e.encodeSSECmp("CMPPD", 0x66, ops)
// RETFL pops the immediate's worth of bytes after the far return
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
// adjustment.
case "RETFL":
if len(ops) != 1 {
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("RETFL expects an immediate")
}
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks (F2 0F D6), the register
// in the reg field, the other bank's in r/m.
case "MOVDQ2Q", "MOVQ2DQ":
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 {
return fmt.Errorf("%s takes register operands alone", upper)
}
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return err
}
return e.emit(i)
// SHA256RNDS2 carries the round constant in a literal X0 first operand. // SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2": case "SHA256RNDS2":
return e.encodeSha256rnds2(ops) return e.encodeSha256rnds2(ops)
+156 -2
View File
@@ -265,7 +265,11 @@ func TestImul(t *testing.T) {
func TestControl(t *testing.T) { func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET") checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP") // NOP contributes nothing on amd64, consumed whole by the toolchain as
// a pseudo; only the bytes pin it (no decodable instruction remains).
if code, err := Encode("NOP"); err != nil || len(code) != 0 {
t.Errorf("NOP: bytes %x (err %v), want empty", code, err)
}
checkOp(t, x86asm.JMP, "JMP", Imm(0)) checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0)) checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0)) checkOp(t, x86asm.JGE, "JGE", Imm(0))
@@ -1193,7 +1197,7 @@ func TestBookkeepingGroundTruth(t *testing.T) {
if err != nil { if err != nil {
t.Fatalf("assemble: %v", err) t.Fatalf("assemble: %v", err)
} }
want := "90c3" want := "c3"
if got := hexCompact(img.Code); got != want { if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want) t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
} }
@@ -1221,3 +1225,153 @@ func mustParse(t *testing.T, src string) *ast.File {
} }
return f return f
} }
// TestCorpusTailSystem pins the system and control forms the toolchain's own
// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare
// family, the far return, the loop, the MMX moves, the CR/DR and segment
// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls.
func TestCorpusTailSystem(t *testing.T) {
regBPT := Reg{idx: 5, size: 8}
X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")
Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7")
X5, X20 := vreg(t, "X5"), vreg(t, "X20")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"},
{"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"},
{"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"},
{"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"},
{"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"},
{"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"},
{"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"},
{"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"},
{"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"},
{"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"},
{"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"},
{"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"},
{"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"},
{"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"},
{"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"},
{"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"},
{"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"},
{"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"},
{"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"},
{"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"},
{"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"},
{"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"},
{"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"},
{"ENDBR64", "ENDBR64", nil, "f30f1efa"},
{"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"},
{"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"},
{"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"},
{"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"},
{"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"},
{"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"},
{"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"},
{"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"},
{"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"},
{"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"},
{"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"},
{"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"},
{"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"},
{"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"},
{"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"},
{"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"},
{"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"},
{"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"},
{"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"},
{"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"},
{"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"},
{"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"},
{"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"},
{"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"},
{"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"},
{"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"},
{"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"},
{"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"},
{"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64
// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the
// paired-register shift spelling, the absolute RET and the jump to an
// undefined static symbol (its displacement and the TLS slot offsets are
// relocation sites, zeroed here as the kernel parity suites do).
func TestCorpusTailFileForms(t *testing.T) {
mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 }
cases := []struct {
name string
src string
want string // hex, with X marking a masked 32-bit relocation site
}{
{"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"},
{"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"},
{"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"},
{"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"},
{"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"},
{"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"},
{"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"},
}
for _, c := range cases {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 {
mask32(code, at/2) // the masked relocation site
}
if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// JMP to an undefined static symbol and the absolute RET: their rel32
// carries a call relocation against the symbol, masked to zero here.
for _, c := range []struct{ name, src string }{
{"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"},
{"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"},
{"absolute ret", "\tRET\n\tRET foo(SB)\n"},
} {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
want := "e900000000c3"
if c.name == "absolute ret" {
want = "c3e900000000"
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
}
}
}
+70 -21
View File
@@ -856,37 +856,41 @@ type evexMoveSpec struct {
vecOK bool // the non-memory operand may be a vector register vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS) nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
gprOK bool // the r/m side may be a general-purpose register (VMOVQ)
} }
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512, aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the // EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2). // three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true}, "VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false},
// EVEX.128/256/512.0F.W0, unaligned packed single move. // EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false}, "VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20
// and friends, the EVEX spelling the high registers demand).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true},
} }
// isEvex reports whether the mnemonic has an EVEX encoding we handle. // isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -900,6 +904,9 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexMoveTable[mnemUpper]; ok { if _, ok := evexMoveTable[mnemUpper]; ok {
return true return true
} }
if _, ok := evexHptrTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper) return isEvexQuad(mnemUpper)
} }
@@ -911,7 +918,13 @@ func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper] _, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper] _, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove { if !inVex && !inVexMove {
return true // EVEX-only mnemonic // The dual-shape moves pick their VEX form by operand count, so
// they are not EVEX-only either.
switch upper {
case "VMOVHPD", "VMOVLPD":
default:
return true // EVEX-only mnemonic
}
} }
// The byte-quad shifts have VEX register forms but EVEX-only memory // The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding. // forms: a memory count source forces the EVEX encoding.
@@ -1018,6 +1031,8 @@ var evexRound = map[string]bool{
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true, "VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true, "VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true, "VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
// The scalar compares suppress exceptions on their LIG encoding.
"VCMPSD": true, "VCMPSS": true,
} }
// evexBcstN maps an instruction accepting .BCST to the broadcast element // evexBcstN maps an instruction accepting .BCST to the broadcast element
@@ -1082,6 +1097,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx) return e.encodeEvexRM(spec, ops, 0, sfx)
} }
spec, inTable := evexTable[mnemUpper] spec, inTable := evexTable[mnemUpper]
// A high/low half move that lives in the hptr table alone (the packed
// double twins) reaches the same inTable block below, which completes
// its spec from the hptr entry.
if !inTable {
if _, ok := evexHptrTable[mnemUpper]; ok {
inTable = true
}
}
if q, ok := evexQuadTable[mnemUpper]; ok { if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast; // The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply. // only masking and zeroing apply.
@@ -1446,15 +1469,28 @@ func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx e
// assembler. The scalar moves also carry a three-operand register form // assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3 // (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens. // opens.
// validEvexMoveOther reports whether the non-vector side of an EVEX move may
// take the operand: memory always, a general-purpose register when gprOK.
func validEvexMoveOther(ms evexMoveSpec, op Operand) bool {
if memOperand(op) {
return true
}
if !ms.gprOK {
return false
}
r, ok := op.(Reg)
return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp
}
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 { if len(ops) == 3 {
if !ms.nds3 { if !ms.nds3 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
} }
// The masked scalar register form keeps the Go assembler's own // The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the // layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not // destination in r/m (op2), the bytes go tool asm emits, not
// the manual's NDS reading. // the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2] src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg) reg, ok := src.(Reg)
if !ok || !reg.isVec() { if !ok || !reg.isVec() {
@@ -1494,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
} }
reg, rm = srcReg, dst reg, rm = srcReg, dst
case srcIsVec: case srcIsVec:
if !memOperand(dst) { if !validEvexMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem) return fmt.Errorf("%s: invalid destination operand", mnem)
} }
reg, rm = srcReg, dst reg, rm = srcReg, dst
case dstIsVec: case dstIsVec:
if !memOperand(src) { if !validEvexMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem) return fmt.Errorf("%s: invalid source operand", mnem)
} }
op = ms.load op = ms.load
@@ -1890,6 +1926,15 @@ var evexHptrTable = map[string]evexHptrSpec{
"VMOVLHPS": { "VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}}, insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
}, },
// The packed-double twins, 66-prefixed.
"VMOVHPD": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLPD": {
insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
} }
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib. // encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
@@ -1954,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !maskReg.isVec() { if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper) return fmt.Errorf("%s: mask must be a vector register", upper)
} }
vsib, _, err := vsibLen(rest[1], upper) vsib, idxLen, err := vsibLen(rest[1], upper)
if err != nil { if err != nil {
return err return err
} }
@@ -1962,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() { if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper) return fmt.Errorf("%s: destination must be a vector register", upper)
} }
// The L bit is the wider of the data register and the VSIB index
// lengths (a YMM index under an XMM destination selects 256-bit, the
// bytes go tool asm emits).
ll := max(idxLen, dst.vecLenBit())
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1} spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0 rBit := 0
if dst.idx >= 8 { if dst.idx >= 8 {
rBit = 1 rBit = 1
} }
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib)
} }
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src, // encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
+200 -7
View File
@@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{
"LOCK": {0xF0}, "LOCK": {0xF0},
"REP": {0xF3}, "REP": {0xF3},
"REPN": {0xF2}, "REPN": {0xF2},
"ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA},
}
// sysUnaryTable maps the one-operand system instructions to their bytes:
// the prefix, the opcode and the /digit the reg field carries. The operand
// is a register or memory in r/m.
var sysUnaryTable = map[string]struct {
prefix byte
opcode []byte
digit int
}{
"CLWB": {0x66, []byte{0x0F, 0xAE}, 6},
"TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6},
"UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6},
"UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6},
"RDPID": {0xF3, []byte{0x0F, 0xC7}, 7},
"CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0},
}
// encodeSysUnary emits a one-operand system instruction: the operand in r/m
// under the fixed /digit, no REX.W.
func (e *enc) encodeSysUnary(mnem string, m struct {
prefix byte
opcode []byte
digit int
}, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
} }
// --- MOV -------------------------------------------------------------------- // --- MOV --------------------------------------------------------------------
@@ -111,6 +145,76 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// silently emit REX.W 8B with the wrong operand meaning. // silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src) _, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst) dstReg, dstVec := vecReg(dst)
// Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn),
// 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg
// field, the general register r/m; CR8+/DR8+ take REX.R.
if c, ok := src.(Reg); ok && c.ctl != 0 {
g, ok := dst.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register load needs a general register destination")
}
opc := byte(0x20)
if c.ctl == 2 {
opc = 0x21
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
if c, ok := dst.(Reg); ok && c.ctl != 0 {
g, ok := src.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register store needs a general register source")
}
opc := byte(0x22)
if c.ctl == 2 {
opc = 0x23
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
}
return e.emit(i)
case Mem:
i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
srcM, ok := src.(Mem)
if !ok {
return fmt.Errorf("MOV: MMX load takes a memory source")
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, srcM, 8); err != nil {
return err
}
return e.emit(i)
}
if srcVec || dstVec { if srcVec || dstVec {
if dstVec { if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() { if g, ok := src.(Reg); ok && !g.isVec() {
@@ -518,6 +622,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
default: default:
return fmt.Errorf("LEA: source must be a memory operand") return fmt.Errorf("LEA: source must be a memory operand")
} }
// LEA accepts the full unsigned 32-bit displacement span where the
// loads and stores reject it beyond the signed one; the wide values
// ride the same disp32 bytes as their two's-complement bit pattern.
if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 {
c := m
c.Disp = int64(int32(uint32(m.Disp)))
src = c
}
i := newInstr(size, []byte{0x8D}) i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, src, size); err != nil { if err := setRM(i, dstReg, src, size); err != nil {
return err return err
@@ -663,6 +775,18 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error { func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) { switch len(ops) {
case 1:
// The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the
// implied destination (the toolchain's one-register shape).
opc := byte(0xF7)
if size == 1 {
opc = 0xF6
}
i := newInstr(size, []byte{opc})
if err := setRMDigit(i, 5, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 2: case 2:
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies // Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes. // r in place (dst = rm = r): the shape GOROOT's clock code writes.
@@ -697,7 +821,7 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
// r/m operand (setRM takes registers and memory alike). // r/m operand (setRM takes registers and memory alike).
return e.encodeImulImm(imm, ops[1], dstReg, size) return e.encodeImulImm(imm, ops[1], dstReg, size)
} }
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops))
} }
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8 // encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
@@ -742,6 +866,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
w16 := size == 2 w16 := size == 2
switch op := ops[0].(type) { switch op := ops[0].(type) {
case Reg: case Reg:
// Segment registers: FS and GS carry their own one-byte opcodes
// under 0F (A0/A8 push, A1/A9 pop); the other four spellings are
// not pushable in 64-bit mode.
if n, isSeg := op.segNumber(); isSeg {
switch n {
case 4: // FS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1})
case 5: // GS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1})
}
return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode")
}
if op.mmx || op.isVec() || op.fp || op.ctl != 0 {
return fmt.Errorf("PUSH/POP: invalid register operand")
}
base := byte(0x50) // PUSH r; POP is 0x58 base := byte(0x50) // PUSH r; POP is 0x58
if !push { if !push {
base = 0x58 base = 0x58
@@ -1093,6 +1238,38 @@ var sseMoveTable = map[string]sseMove{
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single "MOVSS": {0xF3, 0x10, 0x11}, // scalar single
} }
// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register
// rides the reg field and memory r/m (the non-temporal store).
var sseStoreOnly = map[string]struct {
prefix byte
op byte
}{
"MOVNTDQ": {0x66, 0xE7},
}
// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the
// destination memory).
func (e *enc) encodeSSEStoreOnly(mnem string, m struct {
prefix byte
op byte
}, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("%s source must be a vector register", mnem)
}
if !isX86Mem(ops[1]) {
return fmt.Errorf("%s destination must be a memory operand", mnem)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the // encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler. // load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
@@ -1308,14 +1485,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg) dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() { if !ok || (!dstReg.isVec() && !dstReg.mmx) {
return fmt.Errorf("SSE binary destination must be a vector register") return fmt.Errorf("SSE binary destination must be a vector register")
} }
// The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix:
// PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE.
prefix := m.prefix
if dstReg.mmx {
if prefix != 0x66 {
return fmt.Errorf("SSE binary: this form takes no MMX register operand")
}
prefix = 0
}
opcode := []byte{0x0F, m.op} opcode := []byte{0x0F, m.op}
if m.map38 { if m.map38 {
opcode = []byte{0x0F, 0x38, m.op} opcode = []byte{0x0F, 0x38, m.op}
} }
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil { if err := setRM(i, dstReg, src, 8); err != nil {
return err return err
} }
@@ -1791,12 +1977,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error {
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family: // immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src. // F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error { func (e *enc) encodeCmpsd(ops []Operand) error {
return e.encodeSSECmp("CMPSD", 0xF2, ops)
}
// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD):
// 0F C2 /r ib with the predicate immediate last in Plan 9 order
// (src, dst, $imm) and the packed forms' prefixes.
func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error {
if len(ops) != 3 { if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops)) return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops))
} }
imm, ok := ops[2].(Imm) imm, ok := ops[2].(Imm)
if !ok { if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate") return fmt.Errorf("%s predicate must be an immediate", mnem)
} }
immByte, err := imm8(int64(imm)) immByte, err := imm8(int64(imm))
if err != nil { if err != nil {
@@ -1804,9 +1997,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error {
} }
dstReg, ok2 := ops[1].(Reg) dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() { if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register") return fmt.Errorf("%s destination must be a vector register", mnem)
} }
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1} i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil { if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err return err
} }
+3 -11
View File
@@ -60,23 +60,15 @@ DATA small<>+0(SB)/4, $0x1234
} }
} }
// TestAssembleFileErrors checks the static-symbol error paths. // TestAssembleFileErrors checks the static-symbol error paths. A reference
// to a static symbol no GLOBL defines defers to the linker exactly as the
// toolchain does (an external relocation), so it is not an error here.
func TestAssembleFileErrors(t *testing.T) { func TestAssembleFileErrors(t *testing.T) {
cases := []struct { cases := []struct {
name string name string
src string src string
want string // substring of the error want string // substring of the error
}{ }{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{ {
"DATA without GLOBL", "DATA without GLOBL",
` `
+257 -6
View File
@@ -371,6 +371,13 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
if mnem == "RET" { if mnem == "RET" {
return len(loong64Return(fi)) return len(loong64Return(fi))
} }
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem { switch mnem {
case "END", "FUNCDATA", "PCDATA": case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes return 0 // bookkeeping statements contribute no bytes
@@ -484,6 +491,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
} }
return l64wordLE(uint32(immFromOperand(ops[0]))), nil return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC": case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA // The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm // contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -701,6 +720,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
} }
return l64wordLE(l64rr(enc.op, rj, rd)), nil return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr: case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd. // LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) { if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -1229,6 +1277,31 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
return rd, rj, off, load, nil return rd, rj, off, load, nil
} }
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ---- // ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate // encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1262,6 +1335,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
} }
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
} }
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst) rd := l64Reg(dst)
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1356,6 +1452,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d return 8 // pcalau12i + addi.d
} }
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP { if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid) return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
} }
@@ -1868,6 +1972,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op) return l64Mem(op)
} }
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand. // l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 { func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi) _, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2009,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct { type l64VecOperand struct {
num int // 5-bit register number num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX) lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16) lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i]) hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2049,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
} }
i++ i++
w := name[i] w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' { if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false return v, false
} }
v.width, v.hasSuf = w, true v.width, v.hasSuf = w, true
@@ -2174,6 +2291,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register) // VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element) // VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element) // VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) { func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx] enc := l64VmovqTable[lasx]
bank := "V" bank := "V"
@@ -2200,6 +2321,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil return loong64RegNum(name), nil
} }
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant), // Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank. // both operands bare registers of the same bank.
if srcVec && dstVec { if srcVec && dstVec {
@@ -2224,7 +2457,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
} }
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
} }
rj, off := l64MemWithFrame(ops[1], fi) rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 { if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]") return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
} }
@@ -2247,9 +2480,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
} }
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
} }
rj, off := l64MemWithFrame(ops[0], fi) rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 { if rj < 0 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") return nil, fmt.Errorf("VMOVQ: invalid load operand")
} }
op := enc.ld op := enc.ld
if dst.hasSuf { if dst.hasSuf {
@@ -2257,16 +2490,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok { if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw) return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
} }
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w { switch w {
case 0: case 0:
op = enc.replB op = enc.replB
case 1: case 1:
op = enc.replH op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2: case 2:
op = enc.replW op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default: default:
op = enc.replD op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
} }
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
} }
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
} }
+32 -5
View File
@@ -279,6 +279,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
) )
// l64Enc is one instruction's encoding: its bit layout (format) and the // l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +343,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b). // such as vshuf.b).
var l64Vec4R = map[string]bool{} var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit // l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. // its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct { type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +353,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10]) dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
} }
var l64VmovqTable = map[bool]l64VmovqEnc{ var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +366,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
}, },
true: { // XVMOVQ, the LASX (X) bank true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
}, },
} }
@@ -373,7 +386,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15, "SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15, "ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +480,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form // selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate. // with a 5/6-bit shift immediate.
@@ -868,7 +895,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF}, "VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF}, "XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, "XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a // vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well // 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64). // (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +904,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F}, "VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F}, "XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F}, "VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F}, "XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned) // vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant. // si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F}, "VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
+227
View File
@@ -831,3 +831,230 @@ TEXT ·atoms(SB), NOSPLIT, $0
0x4C000020, 0x4C000020,
) )
} }
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
// loong64enc1.s golden bytes.
func TestLOONG64_llacqScrel(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·llsc(SB), NOSPLIT, $0
LLACQW (R5), R4
LLACQV (R5), R4
SCRELW R4, (R6)
SCRELV R4, (R6)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x385780A4, // ll.acq.w r4, r5
0x385788A4, // ll.acq.d r4, r5
0x385784C4, // sc.rel.w r4, r6
0x38578CC4, // sc.rel.d r4, r6
0x4C000020,
)
// The toolchain accepts the zero-offset memory form alone.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
LLACQW 4(R5), R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
SCRELV R4, 8(R6)
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
// toolchain's loong64enc1.s.
func TestLOONG64_vmovqSuffixed(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·vmovq(SB), NOSPLIT, $0
VMOVQ V1.B[3], V9.B16
VMOVQ V2.H[2], V8.H8
VMOVQ V3.W[1], V7.W4
VMOVQ V4.V[0], V6.V2
XVMOVQ X0, X31.B32
XVMOVQ X1, X30.H16
XVMOVQ X2, X29.W8
XVMOVQ X3, X28.V4
XVMOVQ X3, X27.Q2
XVMOVQ X0, X31.W[7]
XVMOVQ X1, X29.W[0]
XVMOVQ X3, X28.V[3]
XVMOVQ X4, X27.V[0]
XVMOVQ X31.W[7], X0
XVMOVQ X29.W[0], X1
XVMOVQ X28.V[3], X8
XVMOVQ X27.V[0], X9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72F78C29, // vreplvei.b v9, v1, 3
0x72F7C848, // vreplvei.h v8, v2, 2
0x72F7E467, // vreplvei.w v7, v3, 1
0x72F7F086, // vreplvei.d v6, v4, 0
0x7707001F, // xvreplve0.b x31, x0
0x7707803E, // xvreplve0.h x30, x1
0x7707C05D, // xvreplve0.w x29, x2
0x7707E07C, // xvreplve0.d x28, x3
0x7707F07B, // xvreplve0.q x27, x3
0x76FFDC1F, // xvinsve0.w x31, x0, 7
0x76FFC03D, // xvinsve0.w x29, x1, 0
0x76FFEC7C, // xvinsve0.d x28, x3, 3
0x76FFE09B, // xvinsve0.d x27, x4, 0
0x7703DFE0, // xvpickve.w x0, x31, 7
0x7703C3A1, // xvpickve.w x1, x29, 0
0x7703EF88, // xvpickve.d x8, x28, 3
0x7703E369, // xvpickve.d x9, x27, 0
0x4C000020,
)
// The rejected shapes: a width mismatch between the element and the
// arrangement, an element index past the lane count, a wrong-bank
// vreplvei and an arrangement the LASX bank does not spell.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.H[3], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.B[16], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X1.B[3], X9.B32
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.W[8]
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
// diverging from the toolchain by the loong64enc1.s differential: the
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
// opcodes and the zero-register base the toolchain gives FP-relative
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
func TestLOONG64_parityFixes(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·parity(SB), NOSPLIT, $0-32
MOVW $4(R4), R5
MOVV $4(R4), R5
MOVW $65536(R4), R5
MOVW $-4096(R4), R5
SCQ R4, R5, (R6)
VMOVQ 2(R4), V1.H8
VMOVQ -6(R4), V1.H8
VMOVQ -12(R4), V2.W4
VMOVQ -16(R4), V3.V2
XVMOVQ -10(R4), X1.H16
XVSEQB $0, X2, X4
XVSEQH $3, X2, X4
XVSEQW $12, X2, X4
XVSEQV $15, X2, X4
XVSEQV $-15, X2, X4
VMOVQ V2, y+16(FP)
VMOVQ y+16(FP), V2
VMOVQ V2, x+2030(FP)
XVMOVQ X6, y+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C01085, // addi.d $4, r4, r5
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
0x1400021E, // lu12i.w $16, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x15FFFFFE, // lu12i.w $-1, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
0x30400481, // vldrepl.h v1, 2(r4)
0x305FF481, // vldrepl.h v1, -6(r4)
0x302FF482, // vldrepl.w v2, -12(r4)
0x3017F883, // vldrepl.d v3, -16(r4)
0x325FEC81, // xvldrepl.h x1, -10(r4)
0x76800044, // xvseqi.b x4, x2, 0
0x76808C44, // xvseqi.h x4, x2, 3
0x76813044, // xvseqi.w x4, x2, 12
0x7681BC44, // xvseqi.d x4, x2, 15
0x7681C444, // xvseqi.d x4, x2, -15
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
0x2C006002, // vld v2, 24(r0)
0x2C5FD802, // vst v2, 2038(r0)
0x2CC06006, // xvst x6, 24(r0)
0x4C000020,
)
// Misaligned vldrepl offsets are rejected, as the toolchain does.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ 3(R4), V1.H8
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
MOVW $4(R4), F1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
// this package already accept for byte-exact data layout (a superset
// spelling, shippable via the goobj path).
func TestLOONG64_bytePseudo(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bytes(SB), NOSPLIT, $0
BYTE $2
BYTE $1; BYTE $0
BYTE $255
RET
`)
code := assembleLOONG64Helper(t, fn)
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
// the final word the way any sub-word tail does.
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
if !bytes.Equal(code[:len(want)], want) {
t.Errorf("bytes = % x, want % x", code, want)
}
for _, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
BYTE $256
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BYTE $-1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
+30 -1
View File
@@ -17,13 +17,28 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which // size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share // occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask // those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7, the fp flag the x87 stack registers F0-F7. // registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
// ctl field the control and debug registers, whose number rides an
// instruction's reg field rather than r/m.
type Reg struct { type Reg struct {
idx int idx int
size int // informational width implied by the name; the mnemonic decides size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register fp bool // F0-F7 x87 stack register
mmx bool // M0-M7 MMX register
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
}
// segNumber returns the segment register number (ES=0..GS=5) when r names a
// bare segment register.
func (r Reg) segNumber() (int, bool) {
if r.seg == 0 {
return 0, false
}
return r.seg - 1, true
} }
// Index returns the register number (0-15 for GPRs, 0-31 for vectors). // Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ { for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true} m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
} }
// MMX: M0..M7.
for i := 0; i <= 7; i++ {
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
}
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
// index spellings of FS and GS are handled before register lookup).
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
m[n] = Reg{idx: i, size: 2, seg: i + 1}
}
// Control and debug registers: CR0..CR15, DR0..DR15.
for i := 0; i <= 15; i++ {
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
}
return m return m
} }
+123 -4
View File
@@ -76,6 +76,10 @@ const (
// carries a vector length, so the register the L'L field follows is the // carries a vector length, so the register the L'L field follows is the
// XMM source. // XMM source.
vexExtractGPR vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
) )
// vexSpec describes one VEX instruction's encoding parameters. // vexSpec describes one VEX instruction's encoding parameters.
@@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8). // VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM}, "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8). // VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM}, // double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2, // VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8). // imm8).
@@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok { if _, ok := vexTable[mnemUpper]; ok {
return true return true
} }
_, ok := vexMoveTable[mnemUpper] if _, ok := vexMoveTable[mnemUpper]; ok {
return ok return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
return true
}
return false
} }
// encodeVex encodes a VEX instruction with operands in Plan 9 order. // encodeVex encodes a VEX instruction with operands in Plan 9 order.
@@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops) return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
} }
} }
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
loadOp, storeOp = 0x12, 0x13
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper] spec := vexTable[mnemUpper]
switch spec.form { switch spec.form {
case vexNDS3: case vexNDS3:
@@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops) return e.encodeVexNDS3Imm(spec, ops)
case vexExtract: case vexExtract:
return e.encodeVexExtract(spec, ops) return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen: case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops) return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero: case vexZero:
@@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1]) return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
} }
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ, // encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector // VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching // move uses the store-form layout (reg = source, rm = destination), matching