Compare commits

...
4 Commits
Author SHA1 Message Date
petrbalvin 81d4bd81e4 test(verify): register the wave kernels
Test / test (push) Failing after 2m20s
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:31 +02:00
petrbalvin 687678a2ea feat(elf): emit data relocations on arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin b0f9071bf5 feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin 81e2673923 feat(amd64): encode the AVX-512 and BMI corpus families
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
20 changed files with 1939 additions and 91 deletions
+1
View File
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
for i := 0; i <= 30; i++ { for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register") add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
} }
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
add("ZR", Special, "zero register (reads as 0)") add("ZR", Special, "zero register (reads as 0)")
add("SP", Special, "stack pointer") add("SP", Special, "stack pointer")
add("LR", Special, "link register (alias of R30)") add("LR", Special, "link register (alias of R30)")
+130 -30
View File
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
return 4 return 4
} }
} }
// The funcdata pseudo-statements contribute no bytes. // The funcdata pseudo-statements contribute no bytes, the expanded
// FUNCDATA/PCDATA forms included.
switch mnem { switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
return 0 return 0
} }
switch mnem { switch mnem {
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS, // The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine // GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
// code: the toolchain emits zero instruction bytes for them, and so does // code: the toolchain emits zero instruction bytes for them, and so does
// the encoder here. // the encoder here. Files that include funcdata.h spell them after
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
// bookkeeping too (the same treatment the loong64 encoder applies).
switch mnem { switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
return nil, nil return nil, nil
case "END":
if len(ops) != 0 {
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
}
return nil, nil
case "FUNCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $n")
}
return nil, nil
} }
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// SIMD element moves (VDUP, VMOV with lane indices) take precedence // SIMD element moves (VDUP, VMOV with lane indices) take precedence
// over the plain arrangement paths, which carry no index. // over the plain arrangement paths, which carry no index.
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) { if mnem == "VDUP" || mnem == "VMOV" {
return encodeARM64Dup(mnem, ops) if arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
}
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
// vector (asm7.go case 82, shared by both mnemonics). The element
// paths above only run when a lane index is spelled, so this is the
// whole-vector shape's only route.
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
return b, err
}
} }
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
} }
// Integer → integer: ORR Rd, ZR, Rs. // Integer → integer. Every truncating register move lowers to an
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
// ORR Xd, XZR, Xm.
if rs != 31 {
switch mnem {
case "MOVB":
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVH":
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVW":
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVBU":
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVHU":
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
}
}
sf := uint32(1) // 64-bit sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" || if mnem == "MOVWU" {
mnem == "MOVH" || mnem == "MOVHU" { sf = 0
}
// A narrow move out of the zero register loses its width: the
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
// WZR. MOVD and MOV keep the 64-bit form.
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
sf = 0 sf = 0
} }
op := uint32(1<<29 | 0x0a<<24) // ORR op := uint32(1<<29 | 0x0a<<24) // ORR
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
} }
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
} }
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
// for bare V registers alone: the arranged forms and every other
// three-register mnemonic are rejected outright.
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
vm, ok1 := arm64VecOf(ops[0])
vn, ok2 := arm64VecOf(ops[1])
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
}
base := uint32(0x5ee08400) // VADD
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
}
if len(ops) != 3 { if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
} }
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
} }
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
// general register into an arranged vector, the spelling asm7.go's case 82
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
// shape, so the caller falls through to the arrangement and element paths;
// the toolchain rejects the bare spellings outright, and the reverse
// Vd.<T>, Rs with them.
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
if len(ops) != 2 {
return nil, false, nil
}
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
return nil, false, nil
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, false, nil
}
dst, ok := arm64VecOf(ops[1])
if !ok || dst.hasIdx || dst.arr == "" {
return nil, false, nil
}
b, err := a64GPVecWhole(mnem, rs, dst)
return b, true, err
}
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
// lane width and Q the vector length. Both VMOV and VDUP take this form
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D2":
imm5, q = 8, 1<<30
default:
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
// spelling for this form, so the encoder refuses it too.
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with // encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
// lane indices: // lane indices:
// //
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: source must be a general register", mnem) return nil, fmt.Errorf("%s: source must be a general register", mnem)
} }
if !dst.hasIdx { if !dst.hasIdx {
var imm5, q uint32 // Duplicates the register across every lane (DUP Vd.T, Rn).
switch dst.arr { return a64GPVecWhole(mnem, rs, dst)
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D1":
imm5, q = 8, 0
case "D2":
imm5, q = 8, 1<<30
default:
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
} }
f, ok := a64ElemField(dst.arr, dst.idx) f, ok := a64ElemField(dst.arr, dst.idx)
if !ok { if !ok {
+5
View File
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
return 17 return 17
case "R18": case "R18":
return 18 return 18
case "R18_PLATFORM":
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
// cmd/asm/internal/arch so assembly cannot use it by accident, and
// sys_windows_arm64.s references it only through this name.
return 18
case "R19": case "R19":
return 19 return 19
case "R20": case "R20":
+94
View File
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
}{ }{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31}, {"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31}, {"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"R18_PLATFORM", 18},
{"F0", 0}, {"F4", 4}, {"F31", 31}, {"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1}, {"INVALID", -1}, {"X0", -1}, {"", -1},
} }
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
} }
} }
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
// rejected, as the toolchain rejects it.
func TestArm64GPIntoVector(t *testing.T) {
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
want := []uint32{
0x4e010ca5, // VMOV R5, V5.B16
0x0e010c22, // VMOV R1, V2.B8
0x0e020c64, // VMOV R3, V4.H4
0x4e040d2a, // VMOV R9, V10.S4
0x4e020cff, // VMOV R7, V31.H8
0x4e080d6c, // VMOV R11, V12.D2
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
0x4e020d2a, // VDUP R9, V10.H8
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
}
}
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
func TestArm64SimdTwoOperand(t *testing.T) {
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
want := []uint32{
0x5ee78508, // VADD V7, V8
0x7ee78508, // VSUB V7, V8
0x5ee18442, // VADD V1, V2
0x4e208422, // VADD arranged: the ordinary three-register path
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64TruncMove pins the truncating register moves against
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
// drops to the W ORR too (asm7.go case 45).
func TestArm64TruncMove(t *testing.T) {
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
want := []uint32{
0x93401c64, // MOVB = SXTB
0x93403ca6, // MOVH = SXTH
0x93407d2a, // MOVW = SXTW
0xd3401c64, // MOVBU = UXTB
0xd3403c64, // MOVHU = UXTH
0x2a0303e4, // MOVWU = ORR W
0xaa0303e4, // MOVD = ORR X
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
0x2a1f03e5, // MOVWU ZR, R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores. // TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) { func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+ got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
+68 -11
View File
@@ -18,12 +18,16 @@ const (
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset) rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction) rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset) rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
// lives with the DWARF fixup constants as rAARCH64Abs64.
rArm64Abs32 = 258
) )
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for // ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the // AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an // amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text. // optional .rela.text and an optional .rela.data.
func (img *Image) ELFAARCH64Object() ([]byte, error) { func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian le := binary.LittleEndian
@@ -133,6 +137,50 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
} }
} }
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rAARCH64Abs64
case 4:
typ = rArm64Abs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables. // String tables.
stNames := newElfStrtab() stNames := newElfStrtab()
for _, s := range syms { for _, s := range syms {
@@ -142,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames { for _, n := range dwarfSectionNames {
stSections.add(n) stSections.add(n)
} }
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout. // Layout.
var out []byte var out []byte
out = append(out, make([]byte, 64)...) out = append(out, make([]byte, 64)...)
@@ -188,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
strtabOff := len(out) strtabOff := len(out)
out = append(out, stNames.bytes()...) out = append(out, stNames.bytes()...)
var relaOff int var relaOff, relaDataOff int
if hasRela { if hasRela {
align(8) align(8)
relaOff = len(out) relaOff = len(out)
@@ -200,6 +243,17 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
out = append(out, b[:]...) out = append(out, b[:]...)
} }
} }
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
@@ -257,6 +311,9 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
if hasRela { if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order. // DWARF section headers; their indices follow the write order.
if dw != nil { if dw != nil {
+115
View File
@@ -197,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Error("unexpected .rela.text section when there are no relocations") t.Error("unexpected .rela.text section when there are no relocations")
} }
} }
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+68 -11
View File
@@ -23,12 +23,16 @@ const (
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i) rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st) rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping) rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
// with the DWARF fixup constants as rLarchAbs64.
rLarchAbs32 = 1
) )
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for // ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the // LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an // amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text. // optional .rela.text and an optional .rela.data.
func (img *Image) ELFLOONG64Object() ([]byte, error) { func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian le := binary.LittleEndian
@@ -117,6 +121,50 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
} }
} }
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rLarchAbs64
case 4:
typ = rLarchAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables. // String tables.
stNames := newElfStrtab() stNames := newElfStrtab()
for _, s := range syms { for _, s := range syms {
@@ -126,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames { for _, n := range dwarfSectionNames {
stSections.add(n) stSections.add(n)
} }
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout. // Layout.
var out []byte var out []byte
out = append(out, make([]byte, 64)...) out = append(out, make([]byte, 64)...)
@@ -172,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
strtabOff := len(out) strtabOff := len(out)
out = append(out, stNames.bytes()...) out = append(out, stNames.bytes()...)
var relaOff int var relaOff, relaDataOff int
if hasRela { if hasRela {
align(8) align(8)
relaOff = len(out) relaOff = len(out)
@@ -184,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
out = append(out, b[:]...) out = append(out, b[:]...)
} }
} }
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
@@ -239,6 +293,9 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
if hasRela { if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order. // DWARF section headers; their indices follow the write order.
if dw != nil { if dw != nil {
+115
View File
@@ -245,3 +245,118 @@ func TestELFLOONG64BranchRelocation(t *testing.T) {
} }
} }
} }
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+68 -10
View File
@@ -24,11 +24,16 @@ const (
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20 rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
// with the DWARF fixup constants as rRISCVAbs64.
rRISVCAbs32 = 1
) )
// ELFRISCVObject returns the image as an ELF64 relocatable object file for // ELFRISCVObject returns the image as an ELF64 relocatable object file for
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64 // RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text. // ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
// an optional .rela.data.
func (img *Image) ELFRISCVObject() ([]byte, error) { func (img *Image) ELFRISCVObject() ([]byte, error) {
le := binary.LittleEndian le := binary.LittleEndian
@@ -129,6 +134,50 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
} }
} }
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rRISCVAbs64
case 4:
typ = rRISVCAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables. // String tables.
stNames := newElfStrtab() stNames := newElfStrtab()
for _, s := range syms { for _, s := range syms {
@@ -138,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n) stSections.add(n)
} }
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames { for _, n := range dwarfSectionNames {
stSections.add(n) stSections.add(n)
} }
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout. // Layout.
var out []byte var out []byte
out = append(out, make([]byte, 64)...) out = append(out, make([]byte, 64)...)
@@ -184,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
strtabOff := len(out) strtabOff := len(out)
out = append(out, stNames.bytes()...) out = append(out, stNames.bytes()...)
var relaOff int var relaOff, relaDataOff int
if hasRela { if hasRela {
align(8) align(8)
relaOff = len(out) relaOff = len(out)
@@ -196,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
out = append(out, b[:]...) out = append(out, b[:]...)
} }
} }
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out) shstrOff := len(out)
out = append(out, stSections.bytes()...) out = append(out, stSections.bytes()...)
@@ -251,6 +306,9 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if hasRela { if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
} }
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order. // DWARF section headers; their indices follow the write order.
if dw != nil { if dw != nil {
+128
View File
@@ -0,0 +1,128 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFRISCVObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+1
View File
@@ -113,6 +113,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return err return err
} }
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
isEvexPrefGather(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" { base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx) return e.encodeVec(base, ops, sfx)
} }
+496 -22
View File
@@ -5,6 +5,7 @@ package asm
import ( import (
"fmt" "fmt"
"slices"
"strings" "strings"
) )
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ; // EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form). // the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst, // EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv). // rm=src, no vvvv).
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle. // EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, "VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst). // EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, "VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, "VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, "VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38, half-precision convert (half-width source). // EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src, // EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
@@ -494,6 +495,246 @@ var evexTable = map[string]evexSpec{
// destination (VPMOVDW dword→word, VPMOVQD qword→dword). // destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
// --- the AVX-512 families the avx512enc corpus exercises, read off
// the toolchain opcodetables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
// immediate; there is no register-count twin).
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
// whose EVEX form drops the legacy prefix entirely.
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
// split the high/low lane spellings.
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
// forms live in evexRegFormTable.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
// three-operand insert form (rm = m64 source, vvvv = preserved,
// reg = dst) and the two-operand store (reg = source, rm = m64);
// the encoder splits on the operand count. VMOVLHPS is the
// three-operand form alone.
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
} }
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
@@ -529,32 +770,38 @@ type evexMoveSpec struct {
n [3]int n [3]int
vecOK bool // the non-memory operand may be a vector register vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
} }
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512, aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
} }
// isEvex reports whether the mnemonic has an EVEX encoding we handle. // isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -579,6 +826,13 @@ func evexRequired(upper string, ops []Operand) bool {
if !inVex && !inVexMove { if !inVex && !inVexMove {
return true // EVEX-only mnemonic return true // EVEX-only mnemonic
} }
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
if slices.ContainsFunc(ops, memOperand) {
return true
}
}
for _, op := range ops { for _, op := range ops {
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) { if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
return true return true
@@ -752,6 +1006,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
} }
spec.n = [3]int{n, n, n} spec.n = [3]int{n, n, n}
} }
// A mnemonic with an immediate and a register spelling (the
// variable-count shifts, the permutes) encodes the register one
// when the first operand is not an immediate.
if len(ops) > 0 {
if _, isImm := ops[0].(Imm); !isImm {
if alt, ok := evexRegFormTable[mnemUpper]; ok {
spec, inTable = alt, true
}
}
}
// The high/low half moves split by operand count: three operands
// insert, two store (VMOVHPS m64, X1).
if hs, ok := evexHptrTable[mnemUpper]; ok {
if len(ops) == 2 {
if hs.store.opcode == 0 {
return fmt.Errorf("%s has no two-operand form", mnemUpper)
}
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
}
spec = hs.insert
}
} else if sfx.evexOnly() { } else if sfx.evexOnly() {
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper) return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
} }
@@ -811,6 +1086,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
} }
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx) return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
} }
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
}
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
}
if !inTable { if !inTable {
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper) return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
} }
@@ -829,6 +1110,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx) return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
case vexExtract: case vexExtract:
return e.encodeEvexExtract(spec, ops, mask, sfx) return e.encodeEvexExtract(spec, ops, mask, sfx)
case vexExtractGPR:
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
case vexRMSrcLen: case vexRMSrcLen:
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx) return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
} }
@@ -908,6 +1191,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
if dstReg.mask { if dstReg.mask {
if r, ok := src.(Reg); ok && r.isVec() { if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit() ll = r.vecLenBit()
} else if l, err := soleLen(spec.n); err == nil {
// A memory source with a length-fixed mnemonic
// (VFPCLASSPDX/Y/Z): the length comes from the table's
// single valid slot, not from the operand.
ll = l
} }
} else if r, ok := src.(Reg); ok && r.isVec() { } else if r, ok := src.(Reg); ok && r.isVec() {
ll = r.vecLenBit() ll = r.vecLenBit()
@@ -934,9 +1222,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if !ok { if !ok {
return fmt.Errorf("shift count must be an immediate") return fmt.Errorf("shift count must be an immediate")
} }
srcReg, ok := src.(Reg) // The count source is a vector register or memory; the length the L'L
if !ok || !srcReg.isVec() { // field and the disp8×N multiplier follow is the destination's either
return fmt.Errorf("shift source must be a vector register") // way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
} }
dstReg, ok := dst.(Reg) dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() { if !ok || !dstReg.isVec() {
@@ -946,7 +1236,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
if err != nil { if err != nil {
return err return err
} }
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil { if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
return err return err
} }
e.out = append(e.out, immByte) e.out = append(e.out, immByte)
@@ -1018,10 +1308,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
return nil return nil
} }
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
// destination, imm8). The encoding is 128-bit regardless of register
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
// element size the table carries.
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("extract source must be a vector register")
}
switch dst.(type) {
case Reg:
if dst.(Reg).isVec() {
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
case Mem, sbMem:
default:
return fmt.Errorf("extract destination must be a general-purpose register or memory")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses // encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
// the store-form opcode (reg = source, rm = destination), matching the Go // the store-form opcode (reg = source, rm = destination), matching the Go
// assembler. // assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
if !ok || !reg.isVec() {
return fmt.Errorf("%s: first operand must be a vector register", mnem)
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("%s: second operand must be a vector register", mnem)
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
}
if len(ops) != 2 { if len(ops) != 2 {
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
} }
@@ -1141,12 +1498,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
return fmt.Errorf("broadcast destination must be a vector register") return fmt.Errorf("broadcast destination must be a vector register")
} }
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1} spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
switch src.(type) { switch r := src.(type) {
case Mem, sbMem: case Mem, sbMem:
spec.opcode = bs.opMem spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n} spec.n = [3]int{bs.n, bs.n, bs.n}
case Reg: case Reg:
spec.opcode = bs.opReg // A GPR source uses the register broadcast opcode; a vector
// source shares the xmm/mem one (the low byte is copied from
// the lane or from the memory operand).
if r.isVec() {
spec.opcode = bs.opMem
spec.n = [3]int{bs.n, bs.n, bs.n}
} else {
spec.opcode = bs.opReg
}
default: default:
return fmt.Errorf("broadcast source must be a register or memory") return fmt.Errorf("broadcast source must be a register or memory")
} }
@@ -1349,6 +1714,102 @@ func isScatter(upper string) bool {
return ok return ok
} }
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
// prefetch hint.
func isEvexPrefGather(upper string) bool {
_, ok := evexPrefGatherTable[upper]
return ok
}
// evexRegFormTable holds the register-count twin of the immediate-form
// entries in evexTable. Several mnemonics name two encodings: an immediate
// count or control ($imm, src, dst …) and a register-count one whose second
// operand is a vector register or memory (count, src2, src1, dst). The
// immediate spelling lives in evexTable, this table carries the register
// spelling, and encodeEvex picks by whether the first operand is an
// immediate, the way vexVarShift does on the VEX side.
var evexRegFormTable = map[string]evexSpec{
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
// controls live in evexTable under 0F3A).
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.NDS.0F38, the register-count permil shuffles.
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
}
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
// operand with a VSIB index and an opmask register, no destination. The
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
// the mask register is the instruction's only register operand.
type evexPrefGatherSpec struct {
mapSel int
opcode byte
w int
pp int
opdigit int
n int
}
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
}
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
// three-operand insert shares an opcode with a two-operand store whose
// source is the vector register and whose destination is m64.
type evexHptrSpec struct {
insert evexSpec
store evexSpec // store.opcode == 0 when the mnemonic has no store form
}
var evexHptrTable = map[string]evexHptrSpec{
"VMOVHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
}
m, ok := ops[0].(Mem)
if !ok || !m.HasIndex || !m.Index.isVec() {
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
}
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
}
// vsibLen validates a VSIB memory operand (the index must be a vector // vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects, the // register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register. // EVEX L'L field follows the index register, not the data register.
@@ -1370,7 +1831,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return err return err
} }
if mask != 0 || sfx.any() { if mask != 0 || sfx.any() {
// EVEX form: OP vsib, K, dst. // EVEX form: OP vsib, K, dst. The L'L field is the wider of the
// index and the data register lengths (the Go assembler's
// layout); the disp8×N multiplier stays the index element size.
if len(rest) != 2 { if len(rest) != 2 {
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops)) return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
} }
@@ -1382,6 +1845,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() { if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper) return fmt.Errorf("%s: destination must be a vector register", upper)
} }
if d := dst.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}} evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx) return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
} }
@@ -1431,6 +1897,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
if err != nil { if err != nil {
return err return err
} }
// The L'L field is the wider of the data register and the VSIB index
// lengths, the bytes go tool asm emits.
if d := src.vecLenBit(); d > ll {
ll = d
}
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}} evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx) return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
} }
@@ -1441,6 +1912,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
var evexKOperand = map[string]bool{ var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true, "VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true, "VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
// The K-to-vector broadcast reads its opmask source from r/m.
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
} }
// kmovSpec describes a KMOV width: the opcode depends on the operand // kmovSpec describes a KMOV width: the opcode depends on the operand
@@ -1541,6 +2014,7 @@ var kOpsTable = map[string]kOpSpec{
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3}, "KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3}, "KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
+89
View File
@@ -721,3 +721,92 @@ func hexCompact(b []byte) string {
} }
return string(out) return string(out)
} }
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
// families the toolchain's avx512enc corpus exercises: the bytes are the
// go tool asm output for exactly these operands, and the same families are
// covered end to end by the avx512_amd64.s differential kernel.
func TestAvx512CorpusFamilies(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// AES rounds (EVEX NDS, VEX twin routed by operand width).
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
// Integer VNNI and the bit algorithm group.
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
// Permutations: immediate and register counts.
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
// Shifts: immediate, register-count and memory-count forms; the
// count source carries its own XMM tuple width.
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
// Conversions and shuffles with the F2 prefix and no prefix.
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
// Gather and scatter prefetch hints (memory-only, /digit in reg).
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
// Opmask broadcasts and the K logic.
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
// Lane extracts to general registers (EVEX and VEX routes).
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
// Moves: masked unaligned, masked scalar register form, half moves
// and non-temporal stores.
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
// Scalar compares with and without the 66 prefix.
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
// Floating point helpers.
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
+221 -7
View File
@@ -61,6 +61,21 @@ const (
// vexImmRMGPR is the immediate form over general-purpose registers // vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0. // (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR vexImmRMGPR
// vexRMOpGPR is the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
// (op0), VEX.vvvv = dst (op1), L = 0.
vexRMOpGPR
// vexCountGPR is the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
// family, with reg = dst (op2), L = 0.
vexCountGPR
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
) )
// vexSpec describes one VEX instruction's encoding parameters. // vexSpec describes one VEX instruction's encoding parameters.
@@ -230,8 +245,34 @@ var vexTable = map[string]vexSpec{
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR}, "ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR}, "MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR}, "MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR}, // VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR}, // share the F7/F5 opcodes across W, the variable shifts carry their
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
// in F2/F3.
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
// rm = src, vvvv = dst).
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src). // VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM}, "KTESTW": {1, 0x99, 0, 0, -1, vexRM},
@@ -290,6 +331,128 @@ var vexTable = map[string]vexSpec{
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, "VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
// spellings, read off the toolchain opcode tables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
// spellings (no 66 prefix; WIG, so W = 0).
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
// VEX.128.66.0F, the scalar and packed compare forms.
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
} }
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of // vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
@@ -420,6 +583,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3GPR(spec, ops) return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR: case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops) return e.encodeVexImmRMGPR(spec, ops)
case vexRMOpGPR:
return e.encodeVexRMOpGPR(spec, ops)
case vexCountGPR:
return e.encodeVexCountGPR(spec, ops)
case vexRMRev: case vexRMRev:
return e.encodeVexRMRev(spec, ops) return e.encodeVexRMRev(spec, ops)
} }
@@ -523,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if !ok { if !ok {
return fmt.Errorf("shift count must be an immediate") return fmt.Errorf("shift count must be an immediate")
} }
srcReg, ok := src.(Reg) // The count source is a vector register or memory; the VEX length
if !ok || !srcReg.isVec() { // follows the destination register either way.
return fmt.Errorf("shift source must be a vector register") if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
} }
dstReg, ok := dst.(Reg) dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() { if !ok || !dstReg.isVec() {
@@ -533,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
} }
vvvvBar := 15 - (dstReg.idx & 15) vvvvBar := 15 - (dstReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil { if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
return err return err
} }
immByte, err := imm8(int64(immVal)) immByte, err := imm8(int64(immVal))
@@ -700,7 +868,11 @@ func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if !ok || vvvvReg.isVec() { if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register") return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
} }
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2) rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
} }
// encodeVexImmRMGPR encodes the immediate form over general-purpose // encodeVexImmRMGPR encodes the immediate form over general-purpose
@@ -729,6 +901,48 @@ func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
return nil return nil
} }
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
// ModRM.rm = src and VEX.vvvv = dst.
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
}
// encodeVexCountGPR encodes the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
}
src, count, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
countReg, ok := count.(Reg)
if !ok || countReg.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register")
}
srcReg, ok := src.(Reg)
if !ok || srcReg.isVec() {
return fmt.Errorf("VEX count source must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the // encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ, // vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form). // a store with no register-destination form).
+57
View File
@@ -31,6 +31,51 @@ var x86asmUnrecognised = map[string]bool{
"RORXQ": true, "RORXQ": true,
"VFMADD213SD": true, "VFMADD213SD": true,
"VFNMADD231SD": true, "VFNMADD231SD": true,
// The scalar FMA spellings the decoder's tables lack entirely.
"VFMADD132SD": true,
"VFMADD132SS": true,
"VFMADD213SS": true,
"VFMADD231SD": true,
"VFMADD231SS": true,
"VFMSUB132SD": true,
"VFMSUB132SS": true,
"VFMSUB213SD": true,
"VFMSUB213SS": true,
"VFMSUB231SD": true,
"VFMSUB231SS": true,
"VFNMADD132SD": true,
"VFNMADD132SS": true,
"VFNMADD213SD": true,
"VFNMADD213SS": true,
"VFNMADD231SS": true,
"VFNMSUB132SD": true,
"VFNMSUB132SS": true,
"VFNMSUB213SD": true,
"VFNMSUB213SS": true,
"VFNMSUB231SD": true,
"VFNMSUB231SS": true,
// The BMI1 unary bit ops the decoder's AVX tables lack.
"BLSIL": true,
"BLSIQ": true,
"BLSMSKL": true,
"BLSMSKQ": true,
"BLSRL": true,
"BLSRQ": true,
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
"BEXTRL": true,
"BEXTRQ": true,
"BZHIL": true,
"BZHIQ": true,
"PDEPL": true,
"PDEPQ": true,
"PEXTL": true,
"PEXTQ": true,
"SARXL": true,
"SARXQ": true,
"SHLXL": true,
"SHLXQ": true,
"SHRXL": true,
"SHRXQ": true,
} }
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS // TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
@@ -237,6 +282,18 @@ func TestVexGroundTruth(t *testing.T) {
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""}, {"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""}, {"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""}, {"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// BMI2 variable shifts and bit ops (three general registers).
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
// Two-operand reg/rm form (v̄vvv must be 1111). // Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""}, {"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""}, {"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
+191
View File
@@ -0,0 +1,191 @@
// The AVX-512 families behind the avx512enc gap: AES round ops, integer
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
// register count, lane broadcasts and extracts, gather and scatter prefetch
// hints, opmask broadcasts, the high/low half moves and the non-temporal
// stores. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func avx512int(p *byte, n int) uint64
TEXT ·avx512int(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ n+16(FP), CX
// AES rounds through the EVEX spellings, masks included.
VAESENC Z20, Z21, Z22
VAESENCLAST Z23, Z24, Z25
VAESDEC (SI), Z26, Z27
VAESDECLAST Z28, Z29, Z30
// Integer VNNI and the bit algorithm group.
VPDPBUSD Z1, Z2, K2, Z3
VPDPBUSDS Z4, Z5, K2, Z6
VPDPWSSD Z7, Z8, Z9
VPDPWSSDS Z10, Z11, K2, Z12
VPOPCNTW Z12, K3, Z13
VPOPCNTB Z14, Z15
VGF2P8MULB Z16, Z17, K4, Z18
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
// Byte/word arithmetic with saturation and masks.
VPADDSB Z1, Z2, K1, Z3
VPADDUSW Z3, Z4, K1, Z5
VPSUBSW Z5, Z6, K1, Z7
VPSUBUSB Z7, Z8, K1, Z9
VPSADBW Z9, Z10, Z11
VPMULHRSW Z11, Z12, Z13
VPMULHW Z13, Z14, Z15
VPUNPCKLBW Z15, Z16, K2, Z17
VPUNPCKHBW Z17, Z18, K2, Z19
VPUNPCKLWD Z19, Z20, K2, Z21
VPUNPCKHWD Z21, Z22, K2, Z23
VPCMPEQB Z23, Z24, K2, K3
VPCMPGTW Z25, Z26, K2, K3
VPCMPEQQ Z27, Z28, K2
VPMULTISHIFTQB Z29, Z30, K3, Z31
VDBPSADBW $3, Z1, Z2, K3, Z3
MOVQ CX, ret+16(FP)
RET
// func avx512perm(p *byte) uint64
TEXT ·avx512perm(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Permutations: immediate and register counts, ternary logic.
VALIGNQ $3, Z1, Z2, K1, Z3
VPERMT2B Z3, Z4, K1, Z5
VPERMT2W Z5, Z6, K1, Z7
VPERMT2PS Z7, Z8, K1, Z9
VPERMI2W Z9, Z10, K1, Z11
VPERMI2PS Z11, Z12, K1, Z13
VPERMI2PD Z13, Z14, K1, Z15
VPERMB Z15, Z16, K1, Z17
VPERMW Z17, Z18, K1, Z19
VPERMPS Z19, Z20, Z21
VPERMD Z20, Z21, Z22
VPERMQ $1, Z1, K2, Z2
VPERMQ Z3, Z4, K2, Z5
VPERMPD $1, Z5, K2, Z6
VPERMPD Z7, Z8, K2, Z9
VPERMILPS $5, Z9, K2, Z10
VPERMILPS Z11, Z12, K2, Z13
VPERMILPD $1, Z13, K2, Z14
VPERMILPD Z15, Z16, K2, Z17
VPTERNLOGD $6, Z17, Z18, K2, Z19
VPTERNLOGQ $9, Z19, Z20, K2, Z21
// Lane shuffle and blend families.
VSHUFPD $1, Z1, Z2, K1, Z3
VSHUFPS $2, Z4, Z5, K1, Z6
VBLENDMPD Z7, Z8, K1, Z9
VBLENDMPS Z9, Z10, K1, Z11
VPBLENDMB Z11, Z12, K1, Z13
VPBLENDMW Z13, Z14, K1, Z15
VPBLENDMD Z15, Z16, K1, Z17
VPBLENDMQ Z17, Z18, K1, Z19
// Conflicts and leading zero counts.
VPCONFLICTD Z1, K1, Z2
VPCONFLICTQ Z3, K1, Z4
VPLZCNTD Z5, K1, Z6
VPLZCNTQ Z7, K1, Z8
// Compress and expand, byte and word widths.
VPCOMPRESSB Z1, K1, (SI)
VPCOMPRESSW Z2, K1, (SI)
VPEXPANDB (SI), K1, Z3
VPEXPANDW (SI), K1, Z4
MOVQ SI, ret+8(FP)
RET
// func avx512shift(p *byte) uint64
TEXT ·avx512shift(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Variable shifts and shuffles with masks.
VPSLLVW Z1, Z2, K1, Z3
VPSRLVW Z3, Z4, K1, Z5
VPSRAVW Z5, Z6, K1, Z7
VPSHLDVW Z7, Z8, K1, Z9
VPSHRDVW Z9, Z10, K1, Z11
VPSHLDVD Z11, Z12, K1, Z13
VPSHLDVQ Z13, Z14, K1, Z15
VPSHRDVD Z15, Z16, K1, Z17
VPSHRDVQ Z17, Z18, K1, Z19
// Immediate shifts, the word/byte-quad widths and masks.
VPSLLW $3, Z1, K2, Z2
VPSRLW $5, Z3, K2, Z4
VPSRAW $7, Z5, K2, Z6
VPSLLDQ $9, Z7, Z8
VPSRLDQ $11, Z9, Z10
// Register-count shifts and their memory-count forms.
VPSLLD X1, Z2, K1, Z3
VPSRLD 16(SI), Z4, K1, Z5
VPSLLQ X6, Z7, K1, Z8
VPSRLQ X9, Z10, K1, Z11
VPSLLW X12, Z13, K1, Z14
VPSRAW X15, Z16, K1, Z17
VPSRAQ $13, Z12, K1, Z13
VPSRAD X14, Z15, K1, Z16
// Lane shuffles in and out.
VPSHLDW $2, Z1, Z2, K1, Z3
VPSHLDQ $4, Z3, Z4, K1, Z5
VPSHRDW $6, Z5, Z6, K1, Z7
VPSHRDQ $8, Z7, Z8, K1, Z9
VPSHUFBITQMB Z9, Z10, K3
VPTESTMB Z11, Z12, K4
VPTESTNMQ Z13, Z14, K5
MOVQ SI, ret+8(FP)
RET
// func avx512float(x float64) float64
TEXT ·avx512float(SB), NOSPLIT, $0-16
// Square roots, compares and the EXP2/RCP28 helpers.
MOVQ x+0(FP), AX
VSQRTPD Z1, K1, Z2
VSQRTPS Z3, K1, Z4
VSQRTSD X1, X2, K1, X3
VSQRTSS X3, X4, X5
VCOMISD X5, X6
VUCOMISS X7, X8
VEXP2PD Z5, K1, Z6
VRCP28PD Z7, K1, Z8
VRCP28SD X9, X8, K1, X10
VRSQRT28PS Z11, K1, Z12
VRSQRT28SS X11, X10, K1, X12
VCVTSD2SS X1, X2, X3
VCVTSS2SD X3, X2, K1, X4
VFMADD132PD Z1, Z2, K1, Z3
VFMADD231SD X1, X2, K1, X3
VFMSUBADD213PS Z3, Z4, K1, Z5
VFNMSUB231PD Z5, Z6, K1, Z7
// Broadcasts and masked moves.
VBROADCASTF32X2 X1, K1, Z2
VBROADCASTI64X2 (SI), K1, Z3
VMOVUPS Z1, K2, Z3
VMOVSD X14, X5, K3, X22
VMOVSS X18, X3, K2, X25
VMOVHPS (SI), X18, X19
VMOVHPS X20, 8(SI)
VMOVLHPS X16, X5, X17
VMOVNTDQ Z7, (SI)
VMOVNTDQA 64(SI), Z8
VMOVNTPD Z9, (SI)
MOVQ SI, ret+8(FP)
RET
// func avx512mask(p *byte) uint64
TEXT ·avx512mask(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
// Omask broadcasts and the K register logic.
VPBROADCASTMB2Q K1, Z2
VPBROADCASTMW2D K3, Z4
KUNPCKWD K6, K4, K1
KADDB K2, K3, K5
KORW K1, K2, K7
// Gather and scatter prefetch hints.
VGATHERPF0DPD K5, (SI)(Y29*8)
VSCATTERPF1DPS K2, (SI)(Z28*4)
// Masked gathers ride the EVEX spelling; the data length wins L'L.
VGATHERDPD (SI)(X10*4), K7, Y22
VPSCATTERDQ Y6, K2, (SI)(X4*1)
// Lane extracts to general registers.
VPEXTRB $3, X1, AX
VPEXTRD $1, X2, DI
VPINSRQ $1, SI, X3, X4
VEXTRACTI32X4 $1, Z1, X5
VINSERTI64X2 $1, X6, Z7, K2, Z8
MOVQ SI, ret+8(FP)
RET
+31
View File
@@ -0,0 +1,31 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 bookkeeping statements: the funcdata.h
// pseudo-directives (GO_ARGS, NO_LOCAL_POINTERS, FUNCDATA, PCDATA) contribute
// no instruction bytes, and every function is byte-compared against
// go tool asm.
#include "textflag.h"
#include "funcdata.h"
// func bookkeep()
TEXT ·bookkeep(SB), NOSPLIT, $8-0
GO_ARGS
FUNCDATA $3, inline_tree(SB)
PCDATA $1, $2
MOVD R1, 0(RSP)
RET
// func bookkeepNoLocals()
TEXT ·bookkeepNoLocals(SB), NOSPLIT, $16-0
NO_LOCAL_POINTERS
PCDATA $0, $0
PCDATA $1, $1
MOVD R2, 8(RSP)
RET
// func bookkeepPlain()
TEXT ·bookkeepPlain(SB), NOSPLIT, $0-0
MOVD R3, R4
RET
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 whole-vector moves between a general
// register and an arranged vector (VMOV/VDUP Rs, Vd.<T>), the two-operand
// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved
// R18_PLATFORM register name. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func gpIntoVector()
TEXT ·gpIntoVector(SB), NOSPLIT, $0-0
VMOV R1, V2.B8
VMOV R3, V4.B16
VMOV R5, V6.H4
VMOV R7, V8.H8
VMOV R9, V10.S2
VMOV R11, V12.S4
VMOV R13, V14.D2
VDUP R15, V16.B8
VDUP R17, V18.B16
VDUP R19, V20.H8
VDUP R21, V22.S4
VDUP R23, V24.D2
RET
// func simdAccumulate()
TEXT ·simdAccumulate(SB), NOSPLIT, $0-0
VADD V7, V8
VSUB V7, V8
VADD V1, V2
VSUB V30, V31
VADD V0.B16, V1.B16, V2.B16
VSUB V0.S4, V1.S4, V2.S4
RET
// func truncMove()
TEXT ·truncMove(SB), NOSPLIT, $0-0
MOVB R3, R4
MOVH R5, R6
MOVW R9, R10
MOVBU R3, R4
MOVHU R3, R4
MOVWU R3, R4
MOVD R3, R4
RET
// func platformRegister()
TEXT ·platformRegister(SB), NOSPLIT, $0-0
MOVD R18_PLATFORM, R3
MOVW R18_PLATFORM, R4
MOVD R3, R18_PLATFORM
MOVD 0x68(R18_PLATFORM), R5
MOVD R5, 0x68(R18_PLATFORM)
MOVW 8(R18_PLATFORM), R6
RET
+2
View File
@@ -34,6 +34,8 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/simd_arm64.s", "../testdata/verify/simd_arm64.s",
"../testdata/verify/widenimm_arm64.s", "../testdata/verify/widenimm_arm64.s",
"../testdata/verify/carryshift_arm64.s", "../testdata/verify/carryshift_arm64.s",
"../testdata/verify/simdmove_arm64.s",
"../testdata/verify/bookkeep_arm64.s",
"../testdata/verify/system_arm64.s", "../testdata/verify/system_arm64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {
+2
View File
@@ -124,8 +124,10 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/avx_amd64.s", "../testdata/verify/avx_amd64.s",
"../testdata/verify/pfx_amd64.s", "../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s", "../testdata/verify/rawdata_amd64.s",
"../testdata/verify/avx512_amd64.s",
"../testdata/verify/pfx_amd64.s", "../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s", "../testdata/verify/rawdata_amd64.s",
"../testdata/verify/avx512_amd64.s",
"../testdata/verify/doubleshift_amd64.s", "../testdata/verify/doubleshift_amd64.s",
"../testdata/verify/ssestatic_amd64.s", "../testdata/verify/ssestatic_amd64.s",
} { } {