Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ecb203dcf5 | ||
|
|
6c672567f3 | ||
|
|
cc6e416c59 |
+1
-1
@@ -173,7 +173,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||
return e.encodeUnary(unaryOp[base], ops, size)
|
||||
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||
return e.encodeShift(shiftOp[base], ops, size)
|
||||
return e.encodeShift(base, ops, size)
|
||||
case "BT", "BTS", "BTR", "BTC":
|
||||
return e.encodeBitTest(base, ops, size)
|
||||
case "XCHG":
|
||||
|
||||
@@ -200,10 +200,60 @@ func TestUnary(t *testing.T) {
|
||||
func TestShift(t *testing.T) {
|
||||
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
|
||||
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
|
||||
checkSyntax(t, "shl rdx, cl", "SHLQ", CX, DX)
|
||||
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
|
||||
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
|
||||
}
|
||||
|
||||
// TestDoubleShift pins the three-operand SHL/SHR form, which encodes as
|
||||
// SHLD/SHRD: go tool asm accepts it for SHL/SHR at W/L/Q widths and rejects
|
||||
// it for SAR, SAL, the rotates and the B width. The byte pins mirror the
|
||||
// oracle's objdump output (48 0f a4 fe 0d for the first case, and so on).
|
||||
func TestDoubleShift(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string // hex encoding
|
||||
}{
|
||||
{"SHLQ imm", "SHLQ", []Operand{Imm(0x0d), DI, SI}, "480fa4fe0d"},
|
||||
{"SHLQ CX high regs", "SHLQ", []Operand{CX, Reg{idx: 8, size: 8}, Reg{idx: 9, size: 8}}, "4d0fa5c1"},
|
||||
{"SHRQ imm", "SHRQ", []Operand{Imm(1), AX, CX}, "480facc101"},
|
||||
{"SHLW imm", "SHLW", []Operand{Imm(1), AX, CX}, "660fa4c101"},
|
||||
{"SHRD CL", "SHRQ", []Operand{CL, AX, CX}, "480fadc1"},
|
||||
{"SHLD imm high regs", "SHLQ", []Operand{Imm(2), Reg{idx: 10, size: 8}, Reg{idx: 11, size: 8}}, "4d0fa4d302"},
|
||||
{"SHRD imm max", "SHRQ", []Operand{Imm(63), Reg{idx: 9, size: 8}, Reg{idx: 15, size: 8}}, "4d0faccf3f"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
// Rejected forms: the oracle rejects every one of these.
|
||||
rejected := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"SARQ three operands", "SARQ", []Operand{Imm(1), AX, CX}},
|
||||
{"SALQ three operands", "SALQ", []Operand{Imm(1), AX, CX}},
|
||||
{"ROLQ three operands", "ROLQ", []Operand{Imm(1), AX, CX}},
|
||||
{"SHLB three operands", "SHLB", []Operand{Imm(1), AL, CL}},
|
||||
{"SHRQ memory source", "SHRQ", []Operand{Imm(1), Ptr(AX, 0, 8), CX}},
|
||||
{"SHRQ ECX count", "SHRQ", []Operand{Reg{idx: 1, size: 4}, AX, CX}},
|
||||
}
|
||||
for _, c := range rejected {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: Encode succeeded, want rejection", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestImul(t *testing.T) {
|
||||
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
|
||||
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
|
||||
@@ -277,6 +327,12 @@ func TestSSEMoveGroundTruth(t *testing.T) {
|
||||
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
||||
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
||||
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
||||
// Static-symbol (SB) references: the GOROOT crypto kernels load and
|
||||
// store octa constants by name (MOVOU bswapMask<>+0(SB), X0).
|
||||
{"MOVOU sym,X0", "MOVOU", []Operand{sbMem{size: 16, name: "bswapMask"}, vreg(t, "X0")}, "f30f6f0500000000", "MOVDQU"},
|
||||
{"MOVOU X0,sym+8", "MOVOU", []Operand{vreg(t, "X0"), sbMem{size: 16, name: "bswapMask", addend: 8}}, "f30f7f0500000000", "MOVDQU"},
|
||||
{"MOVO sym,X1", "MOVO", []Operand{sbMem{size: 16, name: "gcmPoly"}, vreg(t, "X1")}, "660f6f0d00000000", "MOVDQA"},
|
||||
{"MOVO X2,sym", "MOVO", []Operand{vreg(t, "X2"), sbMem{size: 16, name: "gcmPoly"}}, "660f7f1500000000", "MOVDQA"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
|
||||
+64
-5
@@ -498,13 +498,34 @@ func (e *enc) encodeUnary(op struct {
|
||||
|
||||
// --- SHL/SHR/SAR ------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
||||
// doubleShiftOp maps the two mnemonics whose three-operand form go tool asm
|
||||
// accepts to the SHLD/SHRD opcode pair (imm8 form, CL form). SAR, SAL and
|
||||
// the rotates have no such form: the oracle rejects SARQ/ROLQ with three
|
||||
// operands, and so do we.
|
||||
var doubleShiftOp = map[string][2]byte{
|
||||
"SHL": {0xA4, 0xA5}, // SHLD
|
||||
"SHR": {0xAC, 0xAD}, // SHRD
|
||||
}
|
||||
|
||||
// isShiftCountCL reports whether a count operand is the CL register or its
|
||||
// CX spelling: go tool asm accepts both (CX names the same low byte) and
|
||||
// rejects ECX/RCX.
|
||||
func isShiftCountCL(o Operand) bool {
|
||||
reg, ok := o.(Reg)
|
||||
return ok && reg.idx == 1 && (reg.size == 1 || reg.size == 2)
|
||||
}
|
||||
|
||||
func (e *enc) encodeShift(base string, ops []Operand, size int) error {
|
||||
digit := shiftOp[base]
|
||||
if len(ops) == 3 {
|
||||
return e.encodeDoubleShift(base, ops, size)
|
||||
}
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
count, dst := ops[0], ops[1]
|
||||
// Count is $1, %CL, or an imm8.
|
||||
if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 {
|
||||
// Count is $1, CL (or its CX spelling), or an imm8.
|
||||
if isShiftCountCL(count) {
|
||||
// CL: 0xD2 (8-bit) / 0xD3.
|
||||
op := byte(0xD3)
|
||||
if size == 1 {
|
||||
@@ -551,6 +572,44 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeDoubleShift emits the three-operand SHL/SHR form, which the Go
|
||||
// assembler spells as a shift but encodes as SHLD/SHRD (0F A4/A5, 0F AC/AD):
|
||||
// the first operand is the count ($imm or CL), the second feeds the vacated
|
||||
// bits (the reg field) and the third is the shifted value (the r/m field),
|
||||
// matching go tool asm byte for byte. The W/L/Q widths exist; the oracle
|
||||
// rejects the three-operand B form and every SAR/rotate one.
|
||||
func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
|
||||
opc, ok := doubleShiftOp[base]
|
||||
if !ok || size == 1 {
|
||||
return fmt.Errorf("%s: shift expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
count, src, dst := ops[0], ops[1], ops[2]
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s: middle operand must be a register, like go tool asm", base)
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, opc[0]})
|
||||
if isShiftCountCL(count) {
|
||||
// CL (or CX) form: 0F A5/AD.
|
||||
i.opcode[1] = opc[1]
|
||||
} else {
|
||||
imm, ok := count.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be $1, CL or an immediate")
|
||||
}
|
||||
// The count is an unsigned imm8: the same range convention as the
|
||||
// two-operand shift above.
|
||||
if imm < 0 || imm > 255 {
|
||||
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||||
}
|
||||
i.imm = []byte{byte(imm)}
|
||||
}
|
||||
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- IMUL -------------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeImul(ops []Operand, size int) error {
|
||||
@@ -986,12 +1045,12 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
case srcVec:
|
||||
if _, ok := dst.(Mem); !ok {
|
||||
if !isX86Mem(dst) {
|
||||
return fmt.Errorf("SSE move: invalid destination operand")
|
||||
}
|
||||
reg, rm = srcReg, dst
|
||||
case dstVec:
|
||||
if _, ok := src.(Mem); !ok {
|
||||
if !isX86Mem(src) {
|
||||
return fmt.Errorf("SSE move: invalid source operand")
|
||||
}
|
||||
op = m.load
|
||||
|
||||
@@ -48,3 +48,16 @@ type sbMem struct {
|
||||
}
|
||||
|
||||
func (sbMem) isOperand() {}
|
||||
|
||||
// isX86Mem reports whether the operand is an amd64 memory reference: a base
|
||||
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
|
||||
// this position" must accept both; the r/m emitters distinguish the two
|
||||
// themselves.
|
||||
func isX86Mem(o Operand) bool {
|
||||
switch o.(type) {
|
||||
case Mem, sbMem:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,6 +31,12 @@ func FuzzFormatIdempotency(f *testing.F) {
|
||||
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
|
||||
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
|
||||
f.Add("garbage ### ???\n")
|
||||
// Line-ending whitespace at the edge of a comment: a CR followed by more
|
||||
// trailing whitespace once survived the first pass and disappeared on
|
||||
// re-lexing, so formatting was not idempotent.
|
||||
f.Add("//\r ")
|
||||
f.Add("// loop \r\t\nMOVQ AX, BX\n")
|
||||
f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n")
|
||||
|
||||
f.Fuzz(func(t *testing.T, src string) {
|
||||
once := Source(src)
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
go test fuzz v1
|
||||
string("//\r ")
|
||||
+7
-3
@@ -186,15 +186,19 @@ func (l *Lexer) Next() token.Token {
|
||||
}
|
||||
|
||||
// lineComment consumes a // comment up to, but not including, the newline. A
|
||||
// trailing \r is part of a CRLF line ending rather than comment content:
|
||||
// dropping it keeps the formatter's output uniformly LF-terminated.
|
||||
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
|
||||
// comment content, so it never enters the token text. Trimming only a \r
|
||||
// directly before the token's end would make the text depend on what follows
|
||||
// the comment (a newline or the end of the input): "//x\r " would carry the
|
||||
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
|
||||
// with \n would then re-lex its own output to a shorter comment.
|
||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for !l.atEnd() && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
|
||||
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
|
||||
}
|
||||
|
||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||
|
||||
@@ -85,6 +85,19 @@ func TestLabelAndComment(t *testing.T) {
|
||||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||||
}
|
||||
|
||||
func TestLineCommentTrailingWhitespace(t *testing.T) {
|
||||
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
|
||||
// comment content. The token text must not depend on what follows the
|
||||
// comment: before the trim covered only a CR directly before the token's
|
||||
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
|
||||
// formatter re-lexed its own output to a shorter comment.
|
||||
eq(t, texts("// loop\r"), []string{"// loop"})
|
||||
eq(t, texts("// loop\r "), []string{"// loop"})
|
||||
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
|
||||
// A CR inside the comment is content and stays.
|
||||
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
|
||||
}
|
||||
|
||||
func TestAVX512Mnemonics(t *testing.T) {
|
||||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||||
|
||||
@@ -21,6 +21,12 @@ import (
|
||||
// The check requires a parseable signature; functions without one, and
|
||||
// functions whose parameters are all covered by frame reads, stay silent.
|
||||
func checkABI0Args(t *ast.Text) []Diagnostic {
|
||||
// An explicit <ABIInternal> TEXT reads its arguments from the register
|
||||
// file by declaration (runtime·memmove<ABIInternal> is the canonical
|
||||
// example), so the ABI0 frame contract does not apply to it.
|
||||
if t.Name != nil && t.Name.ABI != "" {
|
||||
return nil
|
||||
}
|
||||
params, ok := abiParamNames(t.Doc)
|
||||
if !ok || len(params) == 0 {
|
||||
return nil
|
||||
|
||||
@@ -82,6 +82,23 @@ func TestABIArgSizeSkipsRegisterABI(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestABI0ArgsSkipsABIInternal verifies the frame-read check does not fire for
|
||||
// a TEXT declared <ABIInternal>: runtime·memmove<ABIInternal> and friends read
|
||||
// their arguments from the register file by declaration, which is the correct
|
||||
// spelling there, not the register-args port bug the rule hunts.
|
||||
func TestABI0ArgsSkipsABIInternal(t *testing.T) {
|
||||
diags := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"// func memmove(to, from unsafe.Pointer, n uintptr)\n"+
|
||||
"TEXT ·memmove<ABIInternal>(SB), NOSPLIT, $0-24\n"+
|
||||
"\tMOVQ AX, DI\n"+
|
||||
"\tMOVQ BX, SI\n"+
|
||||
"\tMOVQ CX, BX\n"+
|
||||
"\tRET\n")
|
||||
if codes(diags)[CodeABI0RegisterArgs] != 0 {
|
||||
t.Fatalf("ABIInternal TEXT must not be checked against the FP frame: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnreachableCode exercises the dead-code detection and its guard rails.
|
||||
func TestUnreachableCode(t *testing.T) {
|
||||
// Code after a RET is unreachable.
|
||||
|
||||
+110
-7
@@ -338,14 +338,12 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
|
||||
}
|
||||
|
||||
if isJump(cfg.Arch, upper) {
|
||||
for _, op := range st.Operands {
|
||||
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
|
||||
if name, pos, ok := branchTargetRef(cfg.Arch, upper, st.Operands, tab); ok {
|
||||
referenced[name] = pos
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Undefined labels.
|
||||
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
|
||||
@@ -649,17 +647,65 @@ func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
|
||||
return sym.Name, op.Pos, true
|
||||
}
|
||||
|
||||
// branchTargetRef returns the local label a branch transfers control to: the
|
||||
// bare symbol in the destination position, the last operand, since that is
|
||||
// where the Plan 9 branch target sits. A register-named target is a
|
||||
// register-indirect branch (JMP AX, arm64 BR R5, riscv64 JALR X6, loong64
|
||||
// JIRL R1) and yields no reference, unless the encoder reads the target
|
||||
// positionally (positionalBranchTarget): there a label may legitimately
|
||||
// collide with a register alias, riscv64 ZERO being the ABI name of X0, and
|
||||
// a label named zero is ordinary code.
|
||||
func branchTargetRef(a arch.Arch, upper string, ops []*ast.Operand, tab *arch.Table) (string, token.Position, bool) {
|
||||
if len(ops) == 0 {
|
||||
return "", token.Position{}, false
|
||||
}
|
||||
name, pos, ok := localLabelRef(ops[len(ops)-1])
|
||||
if !ok {
|
||||
return "", token.Position{}, false
|
||||
}
|
||||
if !positionalBranchTarget(a, upper) && (tab.IsRegister(name) || arch.IsPseudoReg(name)) {
|
||||
return "", token.Position{}, false
|
||||
}
|
||||
return name, pos, true
|
||||
}
|
||||
|
||||
// positionalBranchTarget reports whether the encoder reads a bare-symbol
|
||||
// operand of the branch as its label target from a fixed position, without
|
||||
// consulting the register file. The riscv64 branch, JMP and JAL encoders do
|
||||
// (labelFromOperand in asm/riscv_assemble.go), as do the loong64 branch,
|
||||
// BFPT/BFPF and jump encoders (l64Label in asm/loong64_assemble.go). amd64
|
||||
// never does, because a bare register operand to JMP/CALL/Jcc is a
|
||||
// register-indirect branch; nor do the register-indirect forms of the RISC
|
||||
// families (arm64 BR/BLR, riscv64 JALR/JR, loong64 JIRL).
|
||||
func positionalBranchTarget(a arch.Arch, upper string) bool {
|
||||
switch a {
|
||||
case arch.RISCV:
|
||||
return riscvBranches[upper] || upper == "JMP" || upper == "JAL"
|
||||
case arch.LOONG64:
|
||||
return loong64Branches[upper] || upper == "JMP" || upper == "B" ||
|
||||
upper == "JAL" || upper == "BL"
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
|
||||
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
|
||||
// instructions (BCLR, BSET, …) are never mistaken for branches.
|
||||
// instructions (BCLR, BSET, …) are never mistaken for branches. The sets
|
||||
// mirror the encoder's own branch cases: the B-type table entries
|
||||
// (riscv_encode.go), the branch-zero pseudos and the reversed branches
|
||||
// BGT/BGTU/BLE/BLEU (riscv_assemble.go), and for loong64 the 16-bit branch
|
||||
// table plus the single-register forms of l64branch21Table (BEQZ/BNEZ and the
|
||||
// floating-point branches BFPT/BFPF).
|
||||
var riscvBranches = map[string]bool{
|
||||
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
||||
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
|
||||
"BGT": true, "BGTU": true, "BLE": true, "BLEU": true,
|
||||
}
|
||||
|
||||
var loong64Branches = map[string]bool{
|
||||
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
||||
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
|
||||
"BEQZ": true, "BNEZ": true, "BFPT": true, "BFPF": true,
|
||||
}
|
||||
|
||||
// isJump reports whether the mnemonic is any branch.
|
||||
@@ -676,7 +722,8 @@ func isJump(a arch.Arch, upper string) bool {
|
||||
upper == "JR" || upper == "BR"
|
||||
case arch.LOONG64:
|
||||
return upper == "CALL" || loong64Branches[upper] ||
|
||||
upper == "JIRL" || upper == "JMP" || upper == "BR"
|
||||
upper == "JIRL" || upper == "JMP" || upper == "BR" ||
|
||||
upper == "B" || upper == "JAL" || upper == "BL"
|
||||
default: // amd64
|
||||
return upper == "CALL" || strings.HasPrefix(upper, "J")
|
||||
}
|
||||
@@ -692,7 +739,8 @@ func isUnconditionalJump(a arch.Arch, upper string) bool {
|
||||
return upper == "JMP" || upper == "J" || upper == "JAL" ||
|
||||
upper == "JALR" || upper == "JR" || upper == "BR"
|
||||
case arch.LOONG64:
|
||||
return upper == "JMP" || upper == "JIRL" || upper == "BR"
|
||||
return upper == "JMP" || upper == "JIRL" || upper == "BR" || upper == "B" ||
|
||||
upper == "JAL" || upper == "BL"
|
||||
default:
|
||||
return upper == "JMP"
|
||||
}
|
||||
@@ -792,6 +840,43 @@ func isSPReg(op *ast.Operand, a arch.Arch) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// shiftRotateBases are the shift and rotate mnemonics without their width
|
||||
// suffix. These are the instructions whose encoder path (encodeShift) reads
|
||||
// the count from the first operand.
|
||||
var shiftRotateBases = map[string]bool{
|
||||
"SHL": true, "SHR": true, "SAR": true, "SAL": true,
|
||||
"ROL": true, "ROR": true, "RCL": true, "RCR": true,
|
||||
}
|
||||
|
||||
// isShiftCountOperand reports whether operand i of mnem is the shift count.
|
||||
// The ISA fixes the shift/rotate count register at CL: the D2/D3 group (and
|
||||
// C0/C1 for immediates) encode the count outside the ModRM register field,
|
||||
// so the count operand is 8-bit by definition no matter how wide the data is.
|
||||
// The count arrives as the first of the two operands; the one-operand form
|
||||
// does not exist.
|
||||
func isShiftCountOperand(mnem string, i, nops int) bool {
|
||||
if nops != 2 || i != 0 {
|
||||
return false
|
||||
}
|
||||
if shiftRotateBases[mnem] {
|
||||
return true
|
||||
}
|
||||
if len(mnem) > 1 {
|
||||
switch mnem[len(mnem)-1] {
|
||||
case 'Q', 'L', 'W', 'B':
|
||||
return shiftRotateBases[mnem[:len(mnem)-1]]
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isSetcc reports whether the mnemonic is a SETcc: SET plus a condition code.
|
||||
// The membership test is the encoder's own SET dispatch, which asm.Encodable
|
||||
// mirrors.
|
||||
func isSetcc(mnem string) bool {
|
||||
return strings.HasPrefix(mnem, "SET") && asm.Encodable(mnem)
|
||||
}
|
||||
|
||||
// checkRegisterWidth detects amd64 register-width mismatches. The naming
|
||||
// truth of the Go assembler governs: AX, BX, CX, DX, SI, DI, BP, SP and
|
||||
// R8-R15 ARE the 64-bit register names (there are no separate EAX/RAX
|
||||
@@ -802,6 +887,14 @@ func isSPReg(op *ast.Operand, a arch.Arch) bool {
|
||||
// register (EAX under the gasm alias extension, or a byte form), and byte
|
||||
// registers in L/W operations.
|
||||
func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
|
||||
// A SETcc stores one byte: the destination is an 8-bit register or an
|
||||
// 8-bit memory location by definition (0F 90+cc), whichever condition it
|
||||
// tests. The trailing letter of spellings like SETPL or SETEQ is part of
|
||||
// the condition code, not an operand width, so the whole family is
|
||||
// exempt from the suffix logic.
|
||||
if isSetcc(mnem) {
|
||||
return ""
|
||||
}
|
||||
// Determine expected width from mnemonic suffix.
|
||||
var expected int // 0=unknown, 8/4/2/1=bytes
|
||||
switch {
|
||||
@@ -816,10 +909,20 @@ func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
|
||||
default:
|
||||
return "" // no suffix, can't determine width
|
||||
}
|
||||
for _, op := range ops {
|
||||
for i, op := range ops {
|
||||
if op.Kind != ast.OpAddr || op.Addr.Sym == nil {
|
||||
continue
|
||||
}
|
||||
// Only a bare register carries a width to compare: frame and static
|
||||
// symbol references (ch+8(FP), foo(SB)) and memory operands are not
|
||||
// registers even when their name collides with one.
|
||||
if op.Addr.Sym.Pseudo != "" || op.Addr.Base != "" || op.Addr.Index != "" {
|
||||
continue
|
||||
}
|
||||
// The shift/rotate count is exempt: fixed at 8 bits by the ISA.
|
||||
if isShiftCountOperand(mnem, i, len(ops)) {
|
||||
continue
|
||||
}
|
||||
name := strings.ToLower(op.Addr.Sym.Name)
|
||||
regWidth := amd64RegWidth(name)
|
||||
if regWidth == 0 {
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
@@ -21,6 +22,21 @@ func lintSrc(t *testing.T, src string) []Diagnostic {
|
||||
return File(f, Config{Arch: arch.AMD64})
|
||||
}
|
||||
|
||||
// lintArchFile parses and lints src under a, then hands the same file to
|
||||
// assemble so the assertion is pinned against the encoder: a kernel the
|
||||
// linter reasons about must also be one the encoder accepts.
|
||||
func lintArchFile(t *testing.T, filename, src string, a arch.Arch, assemble func(*ast.File) (*asm.Image, error)) []Diagnostic {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse(filename, src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if _, err := assemble(f); err != nil {
|
||||
t.Fatalf("encoder rejects the kernel: %v", err)
|
||||
}
|
||||
return File(f, Config{Arch: a})
|
||||
}
|
||||
|
||||
// lintSrcArch lints src under the architecture inferred from filename.
|
||||
func lintSrcArch(t *testing.T, filename, src string) []Diagnostic {
|
||||
t.Helper()
|
||||
@@ -262,6 +278,138 @@ loop:
|
||||
}
|
||||
}
|
||||
|
||||
func TestRiscvBranchFamilyRegistersLabels(t *testing.T) {
|
||||
// Every riscv64 pseudo-branch that references a label must register that
|
||||
// reference: the reversed branches BGT/BGTU/BLE/BLEU (GOROOT's
|
||||
// memmove_riscv64 branches with BGTU) and a label named like the ZERO
|
||||
// register alias (GOROOT's memclr_riscv64 carries a label named zero;
|
||||
// ZERO is the ABI name of X0) must not be reported unused.
|
||||
diags := lintSrcArch(t, "f_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
BGTU X10, X11, backward
|
||||
BGT X10, X11, zero
|
||||
BLE X10, X11, one
|
||||
BLEU X10, X11, two
|
||||
BEQZ X10, zero
|
||||
BNEZ X10, one
|
||||
JMP two
|
||||
backward:
|
||||
RET
|
||||
zero:
|
||||
RET
|
||||
one:
|
||||
RET
|
||||
two:
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnusedLabel] != 0 {
|
||||
t.Fatalf("branch-referenced labels must not be flagged unused: %+v", diags)
|
||||
}
|
||||
if codes(diags)[CodeUndefinedLabel] != 0 {
|
||||
t.Fatalf("defined labels must resolve: %+v", diags)
|
||||
}
|
||||
|
||||
// A branch to a truly undefined label still reports.
|
||||
diags = lintSrcArch(t, "f_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
BGT X10, X11, nowhere
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUndefinedLabel] != 1 {
|
||||
t.Fatalf("undefined branch target must be flagged: %+v", diags)
|
||||
}
|
||||
|
||||
// A register-indirect JALR is not a label reference.
|
||||
diags = lintSrcArch(t, "f_riscv64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
JALR X1
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUndefinedLabel] != 0 {
|
||||
t.Fatalf("register operand of JALR is not a label: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoong64BranchFamilyRegistersLabels(t *testing.T) {
|
||||
// The loong64 jumps and single-register branches (JAL, B, BL, BEQZ/BNEZ,
|
||||
// BFPT/BFPF) all reference their label from the last operand; GOROOT's
|
||||
// own basic kernels tail-call with JAL, so the reference must register.
|
||||
diags := lintSrcArch(t, "f_loong64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
BEQZ R4, fin
|
||||
BNEZ R4, fin
|
||||
BLTZ R4, fin
|
||||
JAL fin
|
||||
BL fin
|
||||
B fin
|
||||
RET
|
||||
fin:
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnusedLabel] != 0 {
|
||||
t.Fatalf("branch-referenced labels must not be flagged unused: %+v", diags)
|
||||
}
|
||||
if codes(diags)[CodeUndefinedLabel] != 0 {
|
||||
t.Fatalf("defined labels must resolve: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBranchFamiliesAssemble pins the lint branch sets to the encoder: every
|
||||
// mnemonic the linter classifies as a riscv64 or loong64 label branch must be
|
||||
// a branch the encoder actually assembles, with the label in the last
|
||||
// operand. If the encoder gains or renames a branch, this test fails and the
|
||||
// set follows it.
|
||||
func TestBranchFamiliesAssemble(t *testing.T) {
|
||||
riscvForms := map[string]string{}
|
||||
for m := range riscvBranches {
|
||||
riscvForms[m] = m + " X10, X11, tgt"
|
||||
}
|
||||
for _, m := range []string{"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ"} {
|
||||
riscvForms[m] = m + " X10, tgt"
|
||||
}
|
||||
riscvForms["JMP"] = "JMP tgt"
|
||||
riscvForms["JAL"] = "JAL tgt"
|
||||
|
||||
loongForms := map[string]string{}
|
||||
for _, m := range []string{"BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU"} {
|
||||
loongForms[m] = m + " R4, R5, tgt"
|
||||
}
|
||||
for _, m := range []string{"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ", "BFPT", "BFPF"} {
|
||||
loongForms[m] = m + " R4, tgt"
|
||||
}
|
||||
loongForms["JMP"] = "JMP tgt"
|
||||
loongForms["B"] = "B tgt"
|
||||
loongForms["JAL"] = "JAL tgt"
|
||||
loongForms["BL"] = "BL tgt"
|
||||
|
||||
for m, form := range riscvForms {
|
||||
src := "#include \"textflag.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\t" + form + "\n" +
|
||||
"tgt:\n" +
|
||||
"\tRET\n"
|
||||
diags := lintArchFile(t, "f_riscv64.s", src, arch.RISCV, asm.AssembleFileRISCV)
|
||||
if codes(diags)[CodeUnusedLabel] != 0 || codes(diags)[CodeUndefinedLabel] != 0 {
|
||||
t.Errorf("riscv64 %s: label reference not registered: %+v", m, diags)
|
||||
}
|
||||
}
|
||||
for m, form := range loongForms {
|
||||
src := "#include \"textflag.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\t" + form + "\n" +
|
||||
"tgt:\n" +
|
||||
"\tRET\n"
|
||||
diags := lintArchFile(t, "f_loong64.s", src, arch.LOONG64, asm.AssembleFileLOONG64)
|
||||
if codes(diags)[CodeUnusedLabel] != 0 || codes(diags)[CodeUndefinedLabel] != 0 {
|
||||
t.Errorf("loong64 %s: label reference not registered: %+v", m, diags)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestInvalidTextflag(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
@@ -413,6 +561,72 @@ TEXT ·f(SB), NOSPLIT, $0
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterWidthShiftCount(t *testing.T) {
|
||||
// The shift and rotate count lives in CL by ISA definition (the D2/D3
|
||||
// group encodes the count outside the ModRM register field), so the count
|
||||
// operand is 8-bit no matter how wide the data is: SHLQ CL, AX is the
|
||||
// normal spelling of a 64-bit shift. The data operand keeps its check.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
SHLQ CL, AX
|
||||
SHRL CL, BX
|
||||
SARQ CL, CX
|
||||
ROLL CL, DX
|
||||
RORQ CL, R8
|
||||
RCLL CL, R9
|
||||
RCRQ CL, R10
|
||||
MOVQ CL, R10
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeRegisterWidthMismatch] != 1 {
|
||||
t.Fatalf("only the MOVQ CL data move must be flagged, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterWidthSetcc(t *testing.T) {
|
||||
// A SETcc stores one byte whichever condition it tests (0F 90+cc), so
|
||||
// SETNE AL is always right and the trailing letters of SETEQ, SETPL and
|
||||
// SETLS are condition codes, not width suffixes.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
CMPQ AX, BX
|
||||
SETNE AL
|
||||
SETEQ AL
|
||||
SETPL AL
|
||||
SETLS AL
|
||||
SETCC (BX)
|
||||
SETGE (R8)
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeRegisterWidthMismatch] != 0 {
|
||||
t.Fatalf("SETcc destinations are 8-bit by definition: %+v", diags)
|
||||
}
|
||||
if codes(diags)[CodeUnknownInstr] != 0 {
|
||||
t.Fatalf("every SETcc spelling must be known: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterWidthFrameNames(t *testing.T) {
|
||||
// GOROOT's BSD syscall stubs carry frame parameters whose names collide
|
||||
// with byte register names (kevent's ch and nch): MOVQ ch+8(FP), SI is a
|
||||
// frame reference, not the CH register.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·kevent(SB), NOSPLIT, $0-36
|
||||
MOVL kq+0(FP), DI
|
||||
MOVQ ch+8(FP), SI
|
||||
MOVL nch+16(FP), DX
|
||||
MOVQ ev+24(FP), R10
|
||||
MOVQ AX, ret+32(FP)
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeRegisterWidthMismatch] != 0 {
|
||||
t.Fatalf("frame and static symbol names are not registers: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonportableRegisterName(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
|
||||
Vendored
+33
@@ -0,0 +1,33 @@
|
||||
// The three-operand SHL/SHR forms, which go tool asm encodes as SHLD/SHRD:
|
||||
// immediate and CL (or its CX spelling) counts at the Q and W widths, next
|
||||
// to the two-operand CX-count spelling GOROOT's bignum kernels use. Every
|
||||
// result is folded back so no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func dblshift(x, y uint64) uint64
|
||||
TEXT ·dblshift(SB), NOSPLIT, $0-24
|
||||
MOVQ x+0(FP), SI
|
||||
MOVQ y+8(FP), DI
|
||||
MOVQ $12, CX
|
||||
SHLQ $13, SI, DI
|
||||
SHRQ $7, DI, SI
|
||||
SHLQ CX, SI, DI
|
||||
SHRQ CX, DI, SI
|
||||
SHLQ CX, SI
|
||||
SHLQ $9, DI
|
||||
SHLW $1, SI, DI
|
||||
SHRW $3, DI, SI
|
||||
XORQ DI, SI
|
||||
MOVQ SI, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func dblshift32(a, b uint32) uint32
|
||||
TEXT ·dblshift32(SB), NOSPLIT, $0-12
|
||||
MOVL a+0(FP), SI
|
||||
MOVL b+4(FP), DI
|
||||
SHLL $5, SI, DI
|
||||
SHRL $2, DI, SI
|
||||
XORL SI, DI
|
||||
MOVL DI, ret+8(FP)
|
||||
RET
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
// Legacy SSE octa moves against static (SB) symbols: the load and store
|
||||
// shapes GOROOT's AES-CTR, AES-GCM and P-256 kernels spell (MOVOU
|
||||
// bswapMask<>+0(SB), X0 and the reverse), including offsets into the symbol
|
||||
// and the aligned MOVO pair. Every result is folded back so no instruction
|
||||
// is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func ssestatic() uint64
|
||||
TEXT ·ssestatic(SB), NOSPLIT, $0-8
|
||||
MOVOU bswapMask<>+0(SB), X0
|
||||
MOVOU bswapMask<>+8(SB), X1
|
||||
MOVO rodataMask<>+0(SB), X2
|
||||
PXOR X1, X0
|
||||
PXOR X2, X0
|
||||
MOVOU X0, sink<>+0(SB)
|
||||
MOVOU sink<>+0(SB), X3
|
||||
PXOR X3, X0
|
||||
MOVQ X0, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL bswapMask<>(SB), RODATA|NOPTR, $16
|
||||
|
||||
GLOBL rodataMask<>(SB), RODATA|NOPTR, $16
|
||||
|
||||
GLOBL sink<>(SB), NOPTR, $16
|
||||
@@ -122,6 +122,8 @@ func TestGroundTruthAMD64(t *testing.T) {
|
||||
"../testdata/verify/crypto_amd64.s",
|
||||
"../testdata/verify/sse_amd64.s",
|
||||
"../testdata/verify/avx_amd64.s",
|
||||
"../testdata/verify/doubleshift_amd64.s",
|
||||
"../testdata/verify/ssestatic_amd64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
f, errs := parser.Parse(path, mustRead(t, path))
|
||||
|
||||
Reference in New Issue
Block a user