feat(amd64): LOCK and REP prefixes, literal data pseudo-ops and ADJSP
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -67,6 +67,9 @@ type spadjStep struct {
|
|||||||
// patch sites (for the file-level layout to resolve), the label table and the
|
// patch sites (for the file-level layout to resolve), the label table and the
|
||||||
// stack-adjustment boundaries.
|
// stack-adjustment boundaries.
|
||||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
||||||
|
if err := checkAdjspBalance(t); err != nil {
|
||||||
|
return nil, nil, nil, nil, nil, err
|
||||||
|
}
|
||||||
fi := computeFrame(t)
|
fi := computeFrame(t)
|
||||||
chain := jumpChain(t)
|
chain := jumpChain(t)
|
||||||
resolve := func(name string) string {
|
resolve := func(name string) string {
|
||||||
@@ -203,6 +206,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
// frameBase is the SP delta the prologue leaves: 8 for the saved base
|
||||||
|
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
|
||||||
|
// statements' straight-line sum, so a mid-body step's value is the
|
||||||
|
// frame base plus what the body has opened so far.
|
||||||
|
frameBase, bodyDelta := 0, 0
|
||||||
|
if fi.useFP {
|
||||||
|
frameBase = 8 + fi.size
|
||||||
|
}
|
||||||
pos := guardLen + len(fi.prologue)
|
pos := guardLen + len(fi.prologue)
|
||||||
for i, stmt := range t.Body {
|
for i, stmt := range t.Body {
|
||||||
s, ok := stmt.(*ast.Instr)
|
s, ok := stmt.(*ast.Instr)
|
||||||
@@ -230,6 +241,16 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
ps[k].kind = RelCall
|
ps[k].kind = RelCall
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
|
||||||
|
// The statement shifted SP mid-body: record the new running
|
||||||
|
// delta as the value in effect from just past the instruction.
|
||||||
|
v := s.Operands[0].Imm.Val
|
||||||
|
if s.Operands[0].Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
bodyDelta += int(v)
|
||||||
|
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
|
||||||
|
}
|
||||||
patches = append(patches, ps...)
|
patches = append(patches, ps...)
|
||||||
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||||
out = append(out, code...)
|
out = append(out, code...)
|
||||||
@@ -412,6 +433,40 @@ func hasCall(t *ast.Text) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
|
||||||
|
// shifts SP away from the entry state and every RET must see the shifts
|
||||||
|
// closed. The assembler's own prologue and epilogue contribute matching
|
||||||
|
// deltas on both sides, so the statements' straight-line sum must be zero
|
||||||
|
// at each RET; branches do not reset the walk, which runs over the program
|
||||||
|
// list in source order. go tool asm reports an offender as "unbalanced
|
||||||
|
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
|
||||||
|
// $16/$-16 pair, per-RET rather than per-function).
|
||||||
|
func checkAdjspBalance(t *ast.Text) error {
|
||||||
|
delta := 0
|
||||||
|
for _, stmt := range t.Body {
|
||||||
|
in, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||||
|
case "ADJSP":
|
||||||
|
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
|
||||||
|
continue // reported during emission
|
||||||
|
}
|
||||||
|
v := in.Operands[0].Imm.Val
|
||||||
|
if in.Operands[0].Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
delta += int(v)
|
||||||
|
case "RET":
|
||||||
|
if delta != 0 {
|
||||||
|
return fmt.Errorf("unbalanced PUSH/POP")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// guardLen returns the byte length of the stack-split guard prefix. The
|
// guardLen returns the byte length of the stack-split guard prefix. The
|
||||||
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
||||||
// short form and 6 in the long form.
|
// short form and 6 in the long form.
|
||||||
|
|||||||
@@ -439,3 +439,132 @@ func TestSubSPEncodings(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
|
||||||
|
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
|
||||||
|
// asserts the three behaviours the toolchain shows: each prefix statement is
|
||||||
|
// a standalone byte with a PC of its own (so a label placed on the LOCK
|
||||||
|
// points at the F0), the data pseudo-ops write their literal bytes inline,
|
||||||
|
// and END terminates nothing (the statements after it still belong to the
|
||||||
|
// function and carry no trace of it).
|
||||||
|
func TestAssemblePseudoStatements(t *testing.T) {
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·pseudo(SB), NOSPLIT, $0-0
|
||||||
|
pfx:
|
||||||
|
LOCK
|
||||||
|
CMPXCHGQ AX, (BX)
|
||||||
|
REP
|
||||||
|
MOVSQ
|
||||||
|
BYTE $0x0f
|
||||||
|
BYTE $0x1f
|
||||||
|
WORD $0x1234
|
||||||
|
END
|
||||||
|
BYTE $0x02
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, labels, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
|
||||||
|
want := []byte{
|
||||||
|
0xf0,
|
||||||
|
0x48, 0x0f, 0xb1, 0x03,
|
||||||
|
0xf3, 0x48, 0xa5,
|
||||||
|
0x0f, 0x1f, 0x34, 0x12,
|
||||||
|
0x02, 0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
// The label sits on the LOCK byte, exactly where the toolchain's PC
|
||||||
|
// listing puts it.
|
||||||
|
if off := labels["pfx"]; off != 0 {
|
||||||
|
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
|
||||||
|
}
|
||||||
|
// The trailing BYTE lands where the layout says: after the 8 bytes of
|
||||||
|
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
|
||||||
|
// none.
|
||||||
|
if code[12] != 0x02 {
|
||||||
|
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
|
||||||
|
// ADJSP: the straight-line sum of the adjustments must be zero at each
|
||||||
|
// RET, branches in between counting for nothing (verified against go tool
|
||||||
|
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
|
||||||
|
// $16/$-16 pair with a JMP in between assembles).
|
||||||
|
func TestAssembleAdjspBalance(t *testing.T) {
|
||||||
|
// Balanced pair with a branch in between, bytes pinned from go tool asm.
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·adjsp(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
JMP body
|
||||||
|
body:
|
||||||
|
ADJSP $-16
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unbalanced at the RET: the toolchain diagnoses, so must we.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·unbalanced(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The check runs per RET: a closed pair before the first RET does not
|
||||||
|
// excuse an open adjustment before the second.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·tworet(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
mid:
|
||||||
|
ADJSP $8
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framed function: the assembler's own prologue and epilogue
|
||||||
|
// contribute matching deltas, so the pair in the body still balances,
|
||||||
|
// and the bytes match go tool asm end to end.
|
||||||
|
fn = firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·framed(SB), $16-8
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err = Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble framed: %v", err)
|
||||||
|
}
|
||||||
|
want = []byte{
|
||||||
|
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
|
||||||
|
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
|
||||||
|
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
|
||||||
|
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
|
||||||
|
0xC3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+4
-1
@@ -19,7 +19,10 @@ func Encodable(mnemonic string) bool {
|
|||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch upper {
|
switch upper {
|
||||||
case "RET", "NOP", "CALL", "JMP",
|
case "RET", "NOP", "CALL", "JMP",
|
||||||
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
|
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||||
|
// The literal-data pseudo-ops, the accepted-and-ignored END and the
|
||||||
|
// SP adjust.
|
||||||
|
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := noOperandTable[upper]; ok {
|
if _, ok := noOperandTable[upper]; ok {
|
||||||
|
|||||||
@@ -92,6 +92,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||||
case "SHA256RNDS2":
|
case "SHA256RNDS2":
|
||||||
return e.encodeSha256rnds2(ops)
|
return e.encodeSha256rnds2(ops)
|
||||||
|
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
|
||||||
|
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
|
||||||
|
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
|
||||||
|
// the SUBQ and ADDQ forms.
|
||||||
|
case "BYTE", "WORD", "LONG", "QUAD":
|
||||||
|
return e.encodeData(upper, ops)
|
||||||
|
case "END":
|
||||||
|
return e.encodeEnd(ops)
|
||||||
|
case "ADJSP":
|
||||||
|
return e.encodeAdjsp(ops)
|
||||||
}
|
}
|
||||||
|
|
||||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||||
@@ -241,6 +251,73 @@ var prefetchVariant = map[string]int{
|
|||||||
"PREFETCHT2": 3,
|
"PREFETCHT2": 3,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dataWidth is the literal byte count of each data-emission pseudo-op.
|
||||||
|
var dataWidth = map[string]int{
|
||||||
|
"BYTE": 1,
|
||||||
|
"WORD": 2,
|
||||||
|
"LONG": 4,
|
||||||
|
"QUAD": 8,
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
|
||||||
|
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
|
||||||
|
// little-endian, with no opcode lookup. The value is truncated to the
|
||||||
|
// width rather than range-checked, exactly as go tool asm behaves (BYTE
|
||||||
|
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
|
||||||
|
// exactly one immediate is accepted: the toolchain rejects a list such as
|
||||||
|
// BYTE $1, $2, $3.
|
||||||
|
func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s requires an integer immediate", mnem)
|
||||||
|
}
|
||||||
|
width := dataWidth[mnem]
|
||||||
|
out := make([]byte, width)
|
||||||
|
u := uint64(imm)
|
||||||
|
for i := range width {
|
||||||
|
out[i] = byte(u >> (8 * i))
|
||||||
|
}
|
||||||
|
e.out = append(e.out, out...)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||||
|
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||||
|
// the statements after an END still belong to the same function and the
|
||||||
|
// encoded body carries no trace of it, whatever operands follow the name
|
||||||
|
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
|
||||||
|
func (e *enc) encodeEnd(ops []Operand) error {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
|
||||||
|
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
|
||||||
|
// picks (the same selection subSP and addSP make for the frame). go tool
|
||||||
|
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
|
||||||
|
// statement's effect on the SP balance is checked by the function-level
|
||||||
|
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
|
||||||
|
func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("ADJSP requires an integer immediate")
|
||||||
|
}
|
||||||
|
switch v := int(imm); {
|
||||||
|
case v > 0:
|
||||||
|
e.out = append(e.out, subSP(v)...)
|
||||||
|
case v < 0:
|
||||||
|
e.out = append(e.out, addSP(-v)...)
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("ADJSP $0 has no encoding")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||||
func splitSize(upper string) (base string, size int) {
|
func splitSize(upper string) (base string, size int) {
|
||||||
if upper == "" {
|
if upper == "" {
|
||||||
|
|||||||
@@ -925,3 +925,142 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
|
||||||
|
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
|
||||||
|
// prefix field merged into the following instruction, and it validates
|
||||||
|
// nothing about the pairing (LOCK before NOP assembles). The prefixed
|
||||||
|
// atomic and string shapes are the bytes the runtime's own kernels need.
|
||||||
|
func TestPrefixStatements(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"LOCK", "LOCK", nil, "f0"},
|
||||||
|
{"REP", "REP", nil, "f3"},
|
||||||
|
{"REPN", "REPN", nil, "f2"},
|
||||||
|
// LOCK; CMPXCHGQ AX, (BX)
|
||||||
|
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||||
|
// REP; MOVSQ
|
||||||
|
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
|
||||||
|
// REPN; MOVSB
|
||||||
|
{"REPN MOVSB", "MOVSB", nil, "a4"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The prefix statements take no operands, as the toolchain reports for
|
||||||
|
// LOCK AX.
|
||||||
|
if _, err := Encode("LOCK", AX); err == nil {
|
||||||
|
t.Error("LOCK AX assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("REP", Imm(1)); err == nil {
|
||||||
|
t.Error("REP $1 assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
|
||||||
|
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
|
||||||
|
// lookup, truncated to the width rather than range-checked (go tool asm
|
||||||
|
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
|
||||||
|
func TestDataEmission(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
imm Imm
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"BYTE", "BYTE", 0x0f, "0f"},
|
||||||
|
{"BYTE negative", "BYTE", -1, "ff"},
|
||||||
|
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
|
||||||
|
{"WORD", "WORD", 0x1234, "3412"},
|
||||||
|
{"WORD negative", "WORD", -1, "ffff"},
|
||||||
|
{"WORD truncated", "WORD", 0x12345, "4523"},
|
||||||
|
{"LONG", "LONG", 0x11223344, "44332211"},
|
||||||
|
{"LONG negative", "LONG", -1, "ffffffff"},
|
||||||
|
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
|
||||||
|
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.imm)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
|
||||||
|
// register or a missing operand is no immediate at all.
|
||||||
|
if _, err := Encode("BYTE"); err == nil {
|
||||||
|
t.Error("BYTE with no operand assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
|
||||||
|
t.Error("BYTE $1, $2 assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("WORD", AX); err == nil {
|
||||||
|
t.Error("WORD AX assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
|
||||||
|
// encodes to zero bytes and takes any operands without complaint (the
|
||||||
|
// toolchain accepts END $0 and END AX alike).
|
||||||
|
func TestEndIgnored(t *testing.T) {
|
||||||
|
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
|
||||||
|
code, err := Encode("END", ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("END: %v", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(code) != 0 {
|
||||||
|
t.Errorf("END = %x, want no bytes", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
|
||||||
|
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
|
||||||
|
// has no encoding (go tool asm refuses ADJSP $0 outright).
|
||||||
|
func TestAdjsp(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
imm Imm
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"imm8", 112, "4883ec70"},
|
||||||
|
{"imm8 negative", -112, "4883c470"},
|
||||||
|
{"imm32", 200, "4881ecc8000000"},
|
||||||
|
{"imm32 negative", -200, "4881c4c8000000"},
|
||||||
|
{"small", 8, "4883ec08"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode("ADJSP", c.imm)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP", Imm(0)); err == nil {
|
||||||
|
t.Error("ADJSP $0 assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP"); err == nil {
|
||||||
|
t.Error("ADJSP with no operand assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP", AX); err == nil {
|
||||||
|
t.Error("ADJSP AX assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -65,6 +65,15 @@ var bitTestOp = map[string]int{
|
|||||||
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||||
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||||
// full (E8/F0/F8), and PAUSE is F3 90.
|
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||||
|
//
|
||||||
|
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
|
||||||
|
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
|
||||||
|
// respectively), not as a prefix field merged into the next instruction: the
|
||||||
|
// statement that follows is encoded unaware of it, and nothing validates
|
||||||
|
// that the pairing is a legal one (LOCK before NOP assembles without
|
||||||
|
// complaint, each byte pinned against the toolchain). Because the bytes
|
||||||
|
// land in the stream before the following statement anyway, a LOCKed
|
||||||
|
// CMPXCHGQ encodes identically to a prefixed form.
|
||||||
var noOperandTable = map[string][]byte{
|
var noOperandTable = map[string][]byte{
|
||||||
"CPUID": {0x0F, 0xA2},
|
"CPUID": {0x0F, 0xA2},
|
||||||
"RDTSC": {0x0F, 0x31},
|
"RDTSC": {0x0F, 0x31},
|
||||||
@@ -78,6 +87,9 @@ var noOperandTable = map[string][]byte{
|
|||||||
"MFENCE": {0x0F, 0xAE, 0xF0},
|
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||||
"SFENCE": {0x0F, 0xAE, 0xF8},
|
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||||
"UNDEF": {0x0F, 0x0B},
|
"UNDEF": {0x0F, 0x0B},
|
||||||
|
"LOCK": {0xF0},
|
||||||
|
"REP": {0xF3},
|
||||||
|
"REPN": {0xF2},
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- MOV --------------------------------------------------------------------
|
// --- MOV --------------------------------------------------------------------
|
||||||
|
|||||||
Vendored
+103
@@ -0,0 +1,103 @@
|
|||||||
|
// Instruction prefixes: LOCK, REP and REPN. go tool asm encodes each
|
||||||
|
// statement as a standalone one-byte instruction with a PC of its own (F0,
|
||||||
|
// F3 and F2 respectively); the statement that follows is encoded unaware of
|
||||||
|
// it, and nothing validates the pairing. The shapes are the runtime's
|
||||||
|
// atomic read-modify-write family and the string moves, every result folded
|
||||||
|
// back.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func cas64(ptr *uint64, old, new uint64) bool
|
||||||
|
TEXT ·cas64(SB), NOSPLIT, $0-25
|
||||||
|
MOVQ ptr+0(FP), BX
|
||||||
|
MOVQ old+8(FP), AX
|
||||||
|
MOVQ new+16(FP), CX
|
||||||
|
LOCK
|
||||||
|
CMPXCHGQ CX, 0(BX)
|
||||||
|
SETEQ ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func casloop(addr *uint64, v uint64) uint64
|
||||||
|
// The runtime's Or64 shape: a LOCK inside a branch loop, the backward jump
|
||||||
|
// measuring over the prefix statement's own byte.
|
||||||
|
TEXT ·casloop(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ addr+0(FP), BX
|
||||||
|
MOVQ v+8(FP), CX
|
||||||
|
|
||||||
|
loop:
|
||||||
|
MOVQ CX, DX
|
||||||
|
MOVQ (BX), AX
|
||||||
|
ORQ AX, DX
|
||||||
|
LOCK
|
||||||
|
CMPXCHGQ DX, (BX)
|
||||||
|
JNZ loop
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func xadd64(p *uint64, v uint64) uint64
|
||||||
|
TEXT ·xadd64(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
MOVQ v+8(FP), BX
|
||||||
|
LOCK
|
||||||
|
XADDQ BX, (AX)
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func xaddw(p *uint16, v uint16) uint16
|
||||||
|
TEXT ·xaddw(SB), NOSPLIT, $0-12
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
MOVW v+8(FP), BX
|
||||||
|
LOCK
|
||||||
|
XADDW BX, (AX)
|
||||||
|
MOVW AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func lockarith(p *uint64)
|
||||||
|
TEXT ·lockarith(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
LOCK
|
||||||
|
ORQ CX, (AX)
|
||||||
|
LOCK
|
||||||
|
ANDL CX, (AX)
|
||||||
|
LOCK
|
||||||
|
INCQ (AX)
|
||||||
|
LOCK
|
||||||
|
DECQ (AX)
|
||||||
|
LOCK
|
||||||
|
ORB BX, (AX)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func repstring(dst, src *byte, n int)
|
||||||
|
// The memmove shapes: forward copy by quadwords, backward tails.
|
||||||
|
TEXT ·repstring(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ dst+0(FP), DI
|
||||||
|
MOVQ src+8(FP), SI
|
||||||
|
REP
|
||||||
|
MOVSQ
|
||||||
|
REP
|
||||||
|
MOVSB
|
||||||
|
REPN
|
||||||
|
MOVSB
|
||||||
|
REP
|
||||||
|
STOSQ
|
||||||
|
REP
|
||||||
|
STOSB
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func pfxlabel()
|
||||||
|
// Labels pinned on prefix statements' own bytes: pfx: sits on the LOCK,
|
||||||
|
// mid: on the REPN.
|
||||||
|
TEXT ·pfxlabel(SB), NOSPLIT, $0-0
|
||||||
|
pfx:
|
||||||
|
LOCK
|
||||||
|
XCHGL BX, (AX)
|
||||||
|
JMP done
|
||||||
|
|
||||||
|
mid:
|
||||||
|
REPN
|
||||||
|
MOVSB
|
||||||
|
|
||||||
|
done:
|
||||||
|
REP
|
||||||
|
STOSB
|
||||||
|
RET
|
||||||
Vendored
+62
@@ -0,0 +1,62 @@
|
|||||||
|
// Literal data emission: BYTE, WORD, LONG and QUAD write the immediate
|
||||||
|
// into the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
|
||||||
|
// lookup, truncated to the width rather than range-checked; END is
|
||||||
|
// accepted and ignored, contributing no bytes and ending nothing. The
|
||||||
|
// shapes mirror the runtime's hand-laid markers
|
||||||
|
// (crypto/internal/boring/sig/sig_amd64.s) and its syscall stubs
|
||||||
|
// (runtime/sys_linux_amd64.s).
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func marker()
|
||||||
|
// A boring/crypto-style marker: a hand-laid forward branch whose skip
|
||||||
|
// distance is patched at runtime. One BYTE per statement, as the
|
||||||
|
// runtime's own file spells it: the semicolon-separated one-liner the
|
||||||
|
// sys_linux_amd64.s stub uses does not survive gasm fmt, which drops the
|
||||||
|
// statement separators.
|
||||||
|
TEXT ·marker(SB), NOSPLIT, $0-0
|
||||||
|
BYTE $0xEB
|
||||||
|
BYTE $0x1D
|
||||||
|
BYTE $0xF4
|
||||||
|
BYTE $0x48
|
||||||
|
BYTE $0xF4
|
||||||
|
BYTE $0x4B
|
||||||
|
BYTE $0xC3
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func stub()
|
||||||
|
// The sys_linux_amd64.s stub bytes: the sign-extended
|
||||||
|
// "48 c7 c0 0f 00 00 00" form of MOVQ $rt_sigreturn, AX.
|
||||||
|
TEXT ·stub(SB), NOSPLIT, $0-0
|
||||||
|
BYTE $0x48
|
||||||
|
BYTE $0xc7
|
||||||
|
BYTE $0xc0
|
||||||
|
BYTE $0x0f
|
||||||
|
BYTE $0x00
|
||||||
|
BYTE $0x00
|
||||||
|
BYTE $0x00
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func words()
|
||||||
|
// The wider literals, and an END that ends nothing: the WORD after it
|
||||||
|
// still lands in this function.
|
||||||
|
TEXT ·words(SB), NOSPLIT, $0-0
|
||||||
|
WORD $0x1234
|
||||||
|
WORD $-1
|
||||||
|
LONG $0x11223344
|
||||||
|
LONG $-1
|
||||||
|
QUAD $0x1122334455667788
|
||||||
|
QUAD $-2
|
||||||
|
END
|
||||||
|
WORD $0xBEEF
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func trunc()
|
||||||
|
// Truncation, not a range check: each literal keeps its low bytes, exactly
|
||||||
|
// as go tool asm emits them.
|
||||||
|
TEXT ·trunc(SB), NOSPLIT, $0-0
|
||||||
|
BYTE $0x1FF
|
||||||
|
WORD $0x12345
|
||||||
|
LONG $0x123456789
|
||||||
|
QUAD $-2
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user