feat(parser): bracket register ranges, index-only VSIB and bare trailing immediates
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+60
-6
@@ -304,7 +304,7 @@ func (p *state) parseGlobl(line []token.Token) *ast.Globl {
|
||||
rest = rest[1:]
|
||||
}
|
||||
if len(rest) > 0 && rest[0].Kind == token.Dollar {
|
||||
g.Size = parseOperand(rest)
|
||||
g.Size = parseOperand(rest, false)
|
||||
}
|
||||
return g
|
||||
}
|
||||
@@ -321,7 +321,7 @@ func (p *state) parseData(line []token.Token) *ast.Data {
|
||||
d.Name = sym
|
||||
d.Width = width
|
||||
if len(valuePart) > 0 {
|
||||
d.Value = parseOperand(stripComment(valuePart))
|
||||
d.Value = parseOperand(stripComment(valuePart), false)
|
||||
}
|
||||
return d
|
||||
}
|
||||
@@ -332,8 +332,13 @@ func (p *state) parseInstr(line []token.Token) {
|
||||
return
|
||||
}
|
||||
instr := &ast.Instr{Mnemonic: body[0], Comment: comment}
|
||||
for _, grp := range splitOperands(body[1:]) {
|
||||
if op := parseOperand(grp); op != nil {
|
||||
grps := splitOperands(body[1:])
|
||||
for i, grp := range grps {
|
||||
// Only the final operand slot may carry a bare constant: the
|
||||
// toolchain reads the trailing 1 of CMPSD X1, X0, 1 as $1
|
||||
// (math/floor_amd64.s), while an earlier bare number names an
|
||||
// absolute address, a form this parser keeps out of the tree.
|
||||
if op := parseOperand(grp, i == len(grps)-1); op != nil {
|
||||
instr.Operands = append(instr.Operands, op)
|
||||
}
|
||||
}
|
||||
@@ -417,8 +422,10 @@ func setName(raw string, sym *ast.Symbol) {
|
||||
|
||||
// --- operand parsing --------------------------------------------------------
|
||||
|
||||
// parseOperand parses one operand group into an Operand.
|
||||
func parseOperand(g []token.Token) *ast.Operand {
|
||||
// parseOperand parses one operand group into an Operand. allowBare marks
|
||||
// the final operand slot of an instruction, where the toolchain reads a
|
||||
// bare constant expression as an immediate.
|
||||
func parseOperand(g []token.Token, allowBare bool) *ast.Operand {
|
||||
g = stripComment(g)
|
||||
if len(g) == 0 {
|
||||
return nil
|
||||
@@ -431,9 +438,25 @@ func parseOperand(g []token.Token) *ast.Operand {
|
||||
}
|
||||
op.Kind = ast.OpAddr
|
||||
op.Addr = parseAddress(g)
|
||||
// A trailing bare constant leaves every address field empty: the
|
||||
// grammar sees no register, memory reference or symbol, and the closed
|
||||
// constant expression is the whole group. Read it as the immediate it
|
||||
// names, exactly what the $ spelling would produce.
|
||||
if allowBare && isEmptyAddress(op.Addr) {
|
||||
if v, rest, ok := foldExpr(g); ok && len(rest) == 0 {
|
||||
op.Kind = ast.OpImmediate
|
||||
op.Imm = ast.Immediate{Val: v, HasVal: true}
|
||||
}
|
||||
}
|
||||
return op
|
||||
}
|
||||
|
||||
// isEmptyAddress reports whether parseAddress populated nothing, its sign
|
||||
// that the group is no register, memory reference, symbol or register range.
|
||||
func isEmptyAddress(a ast.Address) bool {
|
||||
return a.Sym == nil && a.Base == "" && a.Index == "" && a.Range == nil && a.Shift == ""
|
||||
}
|
||||
|
||||
// parseImmediate parses the tokens following a '$'.
|
||||
func parseImmediate(g []token.Token) ast.Immediate {
|
||||
var imm ast.Immediate
|
||||
@@ -497,6 +520,14 @@ func parseAddress(g []token.Token) ast.Address {
|
||||
if len(g) == 0 {
|
||||
return addr
|
||||
}
|
||||
// A bracketed register range, [Z0-Z3]: the amd64 4FMAPS/4VNNIW
|
||||
// multi-source operand. The bracket runes arrive as Illegal tokens
|
||||
// (the lexer has no bracket kind), so the shape matches on their text.
|
||||
if isBracket(g[0], "[") && len(g) == 5 && g[1].Kind == token.Ident &&
|
||||
g[2].Kind == token.Minus && g[3].Kind == token.Ident && isBracket(g[4], "]") {
|
||||
addr.Range = &ast.RegRange{Lo: g[1].Text, Hi: g[3].Text, Pos: g[0].Pos}
|
||||
return addr
|
||||
}
|
||||
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
||||
// When the prefix is not a valid symbol name (e.g. a bare number like
|
||||
// 0(SP) in RISC-V), sym is nil, and we fall through to regular memory
|
||||
@@ -577,6 +608,16 @@ func parseAddress(g []token.Token) ast.Address {
|
||||
}
|
||||
}
|
||||
}
|
||||
// A lone (index*scale) group is the VSIB index-only form: the
|
||||
// gather/scatter families address memory through a scaled vector index
|
||||
// with no base register, 8(X4*1). The two-group grammar below reads
|
||||
// (base)(index*scale), so a first group whose member carries a scale
|
||||
// factor can only be an index.
|
||||
if isIndexGroup(g[i:]) {
|
||||
addr.Index = g[i+1].Text
|
||||
addr.Scale = int(parseInt(g[i+3].Text))
|
||||
i += 5
|
||||
}
|
||||
// First parenthesised group: the base register.
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
i++
|
||||
@@ -633,6 +674,19 @@ func findPseudoParen(g []token.Token) int {
|
||||
return -1
|
||||
}
|
||||
|
||||
// isBracket reports whether t is a square bracket. The lexer has no bracket
|
||||
// kind, so '[' and ']' arrive as Illegal tokens.
|
||||
func isBracket(t token.Token, text string) bool {
|
||||
return t.Kind == token.Illegal && t.Text == text
|
||||
}
|
||||
|
||||
// isIndexGroup reports whether g begins with a complete (index*scale) group:
|
||||
// one identifier followed by a scale factor, all inside a single parenthesis.
|
||||
func isIndexGroup(g []token.Token) bool {
|
||||
return len(g) >= 5 && g[0].Kind == token.LParen && g[1].Kind == token.Ident &&
|
||||
g[2].Kind == token.Star && g[3].Kind == token.Number && g[4].Kind == token.RParen
|
||||
}
|
||||
|
||||
// --- token helpers ----------------------------------------------------------
|
||||
|
||||
// splitOperands splits a token slice on top-level commas (commas outside any
|
||||
|
||||
Reference in New Issue
Block a user