diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 8c30562..f65fe25 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -22,9 +22,9 @@ // memory forms beside them, the scalar ones the manual spells m16 and the // packed ones with the {1toN} broadcast, base-relative operands with the // ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the -// broadcast laying EVEX.b over the same displacement semantics. A scaled -// index, write masking ({k1}{z}) and embedded rounding still arrive with a -// later slice. +// scaled index and the broadcast laying the SIB byte and EVEX.b over the +// same displacement semantics. Write masking ({k1}{z}) and embedded +// rounding still arrive with a later slice. package arch @@ -198,6 +198,94 @@ func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte { return out } +// amd64EncodeScaledMemory returns the register-form template with a +// base-plus-scaled-index memory operand filled in: the SIB byte follows the +// ModR/M and carries the scale field, the index and the base, whose number +// rides the r/m field as 100. In a SIB form EVEX.B keeps carrying base bit +// three, as amd64Encode laid it from the base, and EVEX.X changes meaning +// from the register's bit four to the index's bit three, so it clears when +// the index sits above 7. The ModR/M and displacement choices stay the +// canonical ones amd64EncodeMemory makes, with the RBP and R13 bases +// keeping their forced displacement: with a SIB byte present, mod 00 with +// base 101 addresses baseless disp32, never through the base. +func amd64EncodeScaledMemory(b []byte, dest, vvvv, base, index, scale int, disp int64) []byte { + out := amd64Encode(b, dest, vvvv, base) + if index&8 != 0 { + out[1] &^= 0x40 + } + rm := base & 7 + var tail []byte + mod := byte(0) + switch { + case rm == 5 || disp != 0: + if disp >= -128 && disp <= 127 { + mod = 1 + tail = []byte{byte(disp)} + } else { + mod = 2 + tail = []byte{byte(disp), byte(disp >> 8), byte(disp >> 16), byte(disp >> 24)} + } + } + sib := byte(rm) + sib |= byte(index&7) << 3 + switch scale { + case 2: + sib |= 1 << 6 + case 4: + sib |= 2 << 6 + case 8: + sib |= 3 << 6 + } + tail = append([]byte{sib}, tail...) + out[5] = out[5]&0x38 | mod<<6 | 4 + return append(out, tail...) +} + +// amd64Index validates the scaled index of a memory operand: a general +// register inside 0-15 and never RSP, whose SIB encoding 100 means no index, +// and a scale the byte multipliers carry. +func (in ExtInstr) amd64Index(op ExtOperand, pos int) (index, scale int, err error) { + if op.Index < 0 || op.Index > 15 { + return 0, 0, fmt.Errorf("%s: operand %d names index register %d, outside 0-15", in.Name, pos, op.Index) + } + if op.Index == 4 { + return 0, 0, fmt.Errorf("%s: operand %d names RSP as the index, which the SIB byte cannot encode", in.Name, pos) + } + switch op.Scale { + case 1, 2, 4, 8: + default: + return 0, 0, fmt.Errorf("%s: operand %d carries a scale of %d, outside the byte multipliers 1, 2, 4 and 8", in.Name, pos, op.Scale) + } + return op.Index, op.Scale, nil +} + +// amd64MemBytes encodes one validated memory position: the plain +// base-plus-displacement form, the scaled index over it, and the broadcast +// bit over either, each an additive layer on the same displacement +// semantics. The operand must have passed amd64Memory's kind gate, which +// the encode paths reach only at the entry's Mem position. +func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos int) ([]byte, error) { + base, disp, err := in.amd64Memory(op, pos) + if err != nil { + return nil, err + } + if op.HasIndex { + index, scale, err := in.amd64Index(op, pos) + if err != nil { + return nil, err + } + out := amd64EncodeScaledMemory(b, dest, vvvv, base, index, scale, disp) + if op.Broadcast { + out[3] |= 0x10 + } + return out, nil + } + if op.Broadcast { + return amd64EncodeBroadcast(b, dest, vvvv, base, disp), nil + } + return amd64EncodeMemory(b, dest, vvvv, base, disp), nil +} + // amd64PlainReg checks the invariants every amd64 register operand carries: // no arm64 arrangement, no predicate qualifier, and a register number inside // the class the instruction encodes. A broadcast spelling names a memory @@ -294,17 +382,10 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { return nil, err } if in.Mem == 2 && ops[1].Kind == ExtMem { - base, disp, err := in.amd64Memory(ops[1], 2) - if err != nil { - return nil, err - } if err := in.amd64Vector(ops[2], class, 3); err != nil { return nil, err } - if ops[1].Broadcast { - return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil - } - return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2) } for i, op := range ops[1:] { if err := in.amd64Vector(op, class, i+2); err != nil { @@ -320,14 +401,10 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { // unused, which the encoding spells as vvvv 1111. func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) - base, disp, err := in.amd64Memory(ops[0], 1) - if err != nil { - return nil, err - } if err := in.amd64Vector(ops[1], class, 2); err != nil { return nil, err } - return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1) } // encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9), @@ -339,11 +416,7 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) { if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } - base, disp, err := in.amd64Memory(ops[1], 2) - if err != nil { - return nil, err - } - return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2) } // encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows @@ -360,27 +433,22 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { destClass = amd64HalfClass(class) } if in.Mem == 1 && ops[0].Kind == ExtMem { - base, disp, err := in.amd64Memory(ops[0], 1) - if err != nil { + // The memory spelling is validated before the destination: on the + // half form the destination's narrowed class is the likelier + // rejection, but a miswritten source spelling names itself first. + if _, _, err := in.amd64Memory(ops[0], 1); err != nil { return nil, err } if err := in.amd64Vector(ops[1], destClass, 2); err != nil { return nil, err } - if ops[0].Broadcast { - return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil - } - return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1) } if in.Mem == 2 && ops[1].Kind == ExtMem { if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } - base, disp, err := in.amd64Memory(ops[1], 2) - if err != nil { - return nil, err - } - return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2) } if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err @@ -449,14 +517,13 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { return nil, err } if in.Mem == 3 && ops[2].Kind == ExtMem { - base, disp, err := in.amd64Memory(ops[2], 3) - if err != nil { - return nil, err - } if err := in.amd64Vector(ops[3], class, 4); err != nil { return nil, err } - out := amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp) + out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3) + if err != nil { + return nil, err + } return append(out, imm), nil } if err := in.amd64Vector(ops[2], class, 3); err != nil { @@ -489,11 +556,10 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) { } var out []byte if in.Mem == 3 && ops[2].Kind == ExtMem { - base, disp, err := in.amd64Memory(ops[2], 3) + out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3) if err != nil { return nil, err } - out = amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp) return append(out, imm), nil } if err := in.amd64Vector(ops[2], class, 3); err != nil { @@ -543,14 +609,10 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { return nil, err } if in.Mem == 2 && ops[1].Kind == ExtMem { - base, disp, err := in.amd64Memory(ops[1], 2) - if err != nil { - return nil, err - } if err := in.amd64Vector(ops[2], class, 3); err != nil { return nil, err } - return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil + return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2) } if err := in.amd64Gpr(ops[1], 2); err != nil { return nil, err diff --git a/arch/amd64_ext_mem_test.go b/arch/amd64_ext_mem_test.go index 11d982c..07c9378 100644 --- a/arch/amd64_ext_mem_test.go +++ b/arch/amd64_ext_mem_test.go @@ -141,6 +141,43 @@ func TestAmd64ExtBroadcastEncoding(t *testing.T) { } } +// TestAmd64ExtScaledMemoryEncoding pins the SIB layer over the memory +// encoding: the scale field, the index and the base in one byte, the r/m +// field 100, EVEX.X clearing on an index above 7, and the ModR/M and +// displacement choices keeping the plain semantics, RBP's forced +// displacement included. +func TestAmd64ExtScaledMemoryEncoding(t *testing.T) { + add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0} + for _, tt := range []struct { + name string + base int + index int + scale int + disp int64 + want string + }{ + {"scale 1 encodes the scale field zero", 1, 2, 1, 0, + "62651440583411"}, + {"scale 2", 1, 2, 2, 0, + "62651440583451"}, + {"scale 4", 1, 2, 4, 0, + "62651440583491"}, + {"scale 8", 1, 2, 8, 0, + "626514405834d1"}, + {"an index above 7 clears EVEX.X", 1, 12, 2, 0, + "62251440583461"}, + {"RBP base keeps the forced displacement", 5, 14, 8, 0, + "622514405874f500"}, + {"RSP base takes the SIB byte with the index", 12, 3, 4, 0, + "6245144058349c"}, + } { + got := amd64EncodeScaledMemory(add, 30, 29, tt.base, tt.index, tt.scale, tt.disp) + if hex.EncodeToString(got) != tt.want { + t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want) + } + } +} + // TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the // memory operand and its two forms. func TestAmd64ExtMemoryVocabulary(t *testing.T) { diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 27cad27..1f3379b 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -429,6 +429,21 @@ var amd64GoldenRows = []amd64GoldenRow{ []ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)}, "62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"}, + // The scaled index: the SIB byte over the same displacement semantics, + // where EVEX.X carries the index's bit three. The compare row quotes + // the listing's indexed row outright; the arithmetic rows pin the bytes + // of {k7}-masked GNU rows, their mask bits riding the bits the layer + // leaves clear. + {"vcomish memory source over a scaled index", "VCOMISH", + []ExtOperand{ExtScaledMemory(5, 14, 8, 0x10000000), ExtXmm(30)}, + "62257c082fb4f500000010", "62 25 7c 08 2f b4 f5 00 00 00 10 vcomish 0x10000000(%rbp,%r14,8),%xmm30"}, + {"vaddph memory source over a scaled index", "VADDPH", + []ExtOperand{ExtZmm(29), ExtScaledMemory(5, 14, 8, 0x10000000), ExtZmm(30)}, + "6225144058b4f500000010", "62 25 14 47 58 b4 f5 00 00 00 10 vaddph 0x10000000(%rbp,%r14,8),%zmm29,%zmm30{%k7} (the GNU row adds {k7})"}, + {"vsqrtph memory source over a scaled index", "VSQRTPH", + []ExtOperand{ExtScaledMemory(5, 14, 8, 0x10000000), ExtZmm(30)}, + "62257c4851b4f500000010", "62 25 7c 4f 51 b4 f5 00 00 00 10 vsqrtph 0x10000000(%rbp,%r14,8),%zmm30{%k7} (the GNU row adds {k7})"}, + // The BF16 memory forms: the dot product reads its second source and // the narrow convert its full-width source from memory. {"vdpbf16ps memory source", "VDPBF16PS", @@ -754,6 +769,15 @@ func TestAmd64ExtRejects(t *testing.T) { {"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16", []ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)}, "carries a broadcast, the entry's memory operand takes none"}, + {"scaled index beyond r15", "VCOMISH", + []ExtOperand{ExtScaledMemory(5, 16, 8, 0x10000000), ExtXmm(30)}, + "index register 16, outside 0-15"}, + {"RSP as the scaled index", "VCOMISH", + []ExtOperand{ExtScaledMemory(5, 4, 8, 0x10000000), ExtXmm(30)}, + "cannot encode"}, + {"a scale the multipliers do not carry", "VADDPH", + []ExtOperand{ExtZmm(29), ExtScaledMemory(1, 14, 3, 0), ExtZmm(30)}, + "outside the byte multipliers"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index df82f3c..800c689 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -175,6 +175,14 @@ type ExtOperand struct { // every lane of the destination, which the encoder lays down as EVEX.b. // Only the memory positions of the entries that carry Bcast accept it. Broadcast bool + // Index and Scale spell the scaled index of an amd64 memory operand, + // the SIB byte's shape: base plus index times scale. Scale carries the + // byte multiplier 1, 2, 4 or 8, and HasIndex separates a spelled index + // from the plain base-plus-displacement operand. The index is a + // general register 0..15, and RSP is no index. + Index int + Scale int + HasIndex bool } // ExtVector builds a scalable vector operand, ADD Z1.S style. @@ -213,6 +221,14 @@ func ExtBroadcast(base int, disp int64) ExtOperand { return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Broadcast: true} } +// ExtScaledMemory builds the base-plus-scaled-index memory operand, the amd64 +// SIB shape: the base and the index are 64-bit general register numbers, +// 0..15, the scale is the byte multiplier 1, 2, 4 or 8, and the displacement +// keeps the plain ModR/M disp8 or disp32 semantics. +func ExtScaledMemory(base, index, scale int, disp int64) ExtOperand { + return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Index: index, Scale: scale, HasIndex: true} +} + // ExtField is one named field of the 32-bit encoding word: a bit offset from // the least significant end and the field's width. type ExtField struct {