feat(arch): add the scaled index to the amd64 memory operands

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:06:11 +02:00
1 parent ccb155437e
commit 11cac26508
4 files changed
+183 -44

No files matched your search

+106 -44
View File
@@ -22,9 +22,9 @@
// memory forms beside them, the scalar ones the manual spells m16 and the
// packed ones with the {1toN} broadcast, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
// broadcast laying EVEX.b over the same displacement semantics. A scaled
// index, write masking ({k1}{z}) and embedded rounding still arrive with a
// later slice.
// scaled index and the broadcast laying the SIB byte and EVEX.b over the
// same displacement semantics. Write masking ({k1}{z}) and embedded
// rounding still arrive with a later slice.
package arch
@@ -198,6 +198,94 @@ func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte {
return out
}
// amd64EncodeScaledMemory returns the register-form template with a
// base-plus-scaled-index memory operand filled in: the SIB byte follows the
// ModR/M and carries the scale field, the index and the base, whose number
// rides the r/m field as 100. In a SIB form EVEX.B keeps carrying base bit
// three, as amd64Encode laid it from the base, and EVEX.X changes meaning
// from the register's bit four to the index's bit three, so it clears when
// the index sits above 7. The ModR/M and displacement choices stay the
// canonical ones amd64EncodeMemory makes, with the RBP and R13 bases
// keeping their forced displacement: with a SIB byte present, mod 00 with
// base 101 addresses baseless disp32, never through the base.
func amd64EncodeScaledMemory(b []byte, dest, vvvv, base, index, scale int, disp int64) []byte {
out := amd64Encode(b, dest, vvvv, base)
if index&8 != 0 {
out[1] &^= 0x40
}
rm := base & 7
var tail []byte
mod := byte(0)
switch {
case rm == 5 || disp != 0:
if disp >= -128 && disp <= 127 {
mod = 1
tail = []byte{byte(disp)}
} else {
mod = 2
tail = []byte{byte(disp), byte(disp >> 8), byte(disp >> 16), byte(disp >> 24)}
}
}
sib := byte(rm)
sib |= byte(index&7) << 3
switch scale {
case 2:
sib |= 1 << 6
case 4:
sib |= 2 << 6
case 8:
sib |= 3 << 6
}
tail = append([]byte{sib}, tail...)
out[5] = out[5]&0x38 | mod<<6 | 4
return append(out, tail...)
}
// amd64Index validates the scaled index of a memory operand: a general
// register inside 0-15 and never RSP, whose SIB encoding 100 means no index,
// and a scale the byte multipliers carry.
func (in ExtInstr) amd64Index(op ExtOperand, pos int) (index, scale int, err error) {
if op.Index < 0 || op.Index > 15 {
return 0, 0, fmt.Errorf("%s: operand %d names index register %d, outside 0-15", in.Name, pos, op.Index)
}
if op.Index == 4 {
return 0, 0, fmt.Errorf("%s: operand %d names RSP as the index, which the SIB byte cannot encode", in.Name, pos)
}
switch op.Scale {
case 1, 2, 4, 8:
default:
return 0, 0, fmt.Errorf("%s: operand %d carries a scale of %d, outside the byte multipliers 1, 2, 4 and 8", in.Name, pos, op.Scale)
}
return op.Index, op.Scale, nil
}
// amd64MemBytes encodes one validated memory position: the plain
// base-plus-displacement form, the scaled index over it, and the broadcast
// bit over either, each an additive layer on the same displacement
// semantics. The operand must have passed amd64Memory's kind gate, which
// the encode paths reach only at the entry's Mem position.
func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos int) ([]byte, error) {
base, disp, err := in.amd64Memory(op, pos)
if err != nil {
return nil, err
}
if op.HasIndex {
index, scale, err := in.amd64Index(op, pos)
if err != nil {
return nil, err
}
out := amd64EncodeScaledMemory(b, dest, vvvv, base, index, scale, disp)
if op.Broadcast {
out[3] |= 0x10
}
return out, nil
}
if op.Broadcast {
return amd64EncodeBroadcast(b, dest, vvvv, base, disp), nil
}
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
}
// amd64PlainReg checks the invariants every amd64 register operand carries:
// no arm64 arrangement, no predicate qualifier, and a register number inside
// the class the instruction encodes. A broadcast spelling names a memory
@@ -294,17 +382,10 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if ops[1].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2)
}
for i, op := range ops[1:] {
if err := in.amd64Vector(op, class, i+2); err != nil {
@@ -320,14 +401,10 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
// unused, which the encoding spells as vvvv 1111.
func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
base, disp, err := in.amd64Memory(ops[0], 1)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
}
// encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9),
@@ -339,11 +416,7 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2)
}
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
@@ -360,27 +433,22 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
destClass = amd64HalfClass(class)
}
if in.Mem == 1 && ops[0].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[0], 1)
if err != nil {
// The memory spelling is validated before the destination: on the
// half form the destination's narrowed class is the likelier
// rejection, but a miswritten source spelling names itself first.
if _, _, err := in.amd64Memory(ops[0], 1); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
return nil, err
}
if ops[0].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2)
}
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
@@ -449,14 +517,13 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
return nil, err
}
if in.Mem == 3 && ops[2].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[2], 3)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[3], class, 4); err != nil {
return nil, err
}
out := amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3)
if err != nil {
return nil, err
}
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
@@ -489,11 +556,10 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
}
var out []byte
if in.Mem == 3 && ops[2].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[2], 3)
out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3)
if err != nil {
return nil, err
}
out = amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
@@ -543,14 +609,10 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2)
}
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
+37
View File
@@ -141,6 +141,43 @@ func TestAmd64ExtBroadcastEncoding(t *testing.T) {
}
}
// TestAmd64ExtScaledMemoryEncoding pins the SIB layer over the memory
// encoding: the scale field, the index and the base in one byte, the r/m
// field 100, EVEX.X clearing on an index above 7, and the ModR/M and
// displacement choices keeping the plain semantics, RBP's forced
// displacement included.
func TestAmd64ExtScaledMemoryEncoding(t *testing.T) {
add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}
for _, tt := range []struct {
name string
base int
index int
scale int
disp int64
want string
}{
{"scale 1 encodes the scale field zero", 1, 2, 1, 0,
"62651440583411"},
{"scale 2", 1, 2, 2, 0,
"62651440583451"},
{"scale 4", 1, 2, 4, 0,
"62651440583491"},
{"scale 8", 1, 2, 8, 0,
"626514405834d1"},
{"an index above 7 clears EVEX.X", 1, 12, 2, 0,
"62251440583461"},
{"RBP base keeps the forced displacement", 5, 14, 8, 0,
"622514405874f500"},
{"RSP base takes the SIB byte with the index", 12, 3, 4, 0,
"6245144058349c"},
} {
got := amd64EncodeScaledMemory(add, 30, 29, tt.base, tt.index, tt.scale, tt.disp)
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the
// memory operand and its two forms.
func TestAmd64ExtMemoryVocabulary(t *testing.T) {
+24
View File
@@ -429,6 +429,21 @@ var amd64GoldenRows = []amd64GoldenRow{
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"},
// The scaled index: the SIB byte over the same displacement semantics,
// where EVEX.X carries the index's bit three. The compare row quotes
// the listing's indexed row outright; the arithmetic rows pin the bytes
// of {k7}-masked GNU rows, their mask bits riding the bits the layer
// leaves clear.
{"vcomish memory source over a scaled index", "VCOMISH",
[]ExtOperand{ExtScaledMemory(5, 14, 8, 0x10000000), ExtXmm(30)},
"62257c082fb4f500000010", "62 25 7c 08 2f b4 f5 00 00 00 10 vcomish 0x10000000(%rbp,%r14,8),%xmm30"},
{"vaddph memory source over a scaled index", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtScaledMemory(5, 14, 8, 0x10000000), ExtZmm(30)},
"6225144058b4f500000010", "62 25 14 47 58 b4 f5 00 00 00 10 vaddph 0x10000000(%rbp,%r14,8),%zmm29,%zmm30{%k7} (the GNU row adds {k7})"},
{"vsqrtph memory source over a scaled index", "VSQRTPH",
[]ExtOperand{ExtScaledMemory(5, 14, 8, 0x10000000), ExtZmm(30)},
"62257c4851b4f500000010", "62 25 7c 4f 51 b4 f5 00 00 00 10 vsqrtph 0x10000000(%rbp,%r14,8),%zmm30{%k7} (the GNU row adds {k7})"},
// The BF16 memory forms: the dot product reads its second source and
// the narrow convert its full-width source from memory.
{"vdpbf16ps memory source", "VDPBF16PS",
@@ -754,6 +769,15 @@ func TestAmd64ExtRejects(t *testing.T) {
{"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16",
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"carries a broadcast, the entry's memory operand takes none"},
{"scaled index beyond r15", "VCOMISH",
[]ExtOperand{ExtScaledMemory(5, 16, 8, 0x10000000), ExtXmm(30)},
"index register 16, outside 0-15"},
{"RSP as the scaled index", "VCOMISH",
[]ExtOperand{ExtScaledMemory(5, 4, 8, 0x10000000), ExtXmm(30)},
"cannot encode"},
{"a scale the multipliers do not carry", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtScaledMemory(1, 14, 3, 0), ExtZmm(30)},
"outside the byte multipliers"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
+16
View File
@@ -175,6 +175,14 @@ type ExtOperand struct {
// every lane of the destination, which the encoder lays down as EVEX.b.
// Only the memory positions of the entries that carry Bcast accept it.
Broadcast bool
// Index and Scale spell the scaled index of an amd64 memory operand,
// the SIB byte's shape: base plus index times scale. Scale carries the
// byte multiplier 1, 2, 4 or 8, and HasIndex separates a spelled index
// from the plain base-plus-displacement operand. The index is a
// general register 0..15, and RSP is no index.
Index int
Scale int
HasIndex bool
}
// ExtVector builds a scalable vector operand, ADD Z1.S style.
@@ -213,6 +221,14 @@ func ExtBroadcast(base int, disp int64) ExtOperand {
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Broadcast: true}
}
// ExtScaledMemory builds the base-plus-scaled-index memory operand, the amd64
// SIB shape: the base and the index are 64-bit general register numbers,
// 0..15, the scale is the byte multiplier 1, 2, 4 or 8, and the displacement
// keeps the plain ModR/M disp8 or disp32 semantics.
func ExtScaledMemory(base, index, scale int, disp int64) ExtOperand {
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Index: index, Scale: scale, HasIndex: true}
}
// ExtField is one named field of the 32-bit encoding word: a bit offset from
// the least significant end and the field's width.
type ExtField struct {