Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
81d4bd81e4 | ||
|
|
687678a2ea | ||
|
|
b0f9071bf5 | ||
|
|
81e2673923 |
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
|
||||
for i := 0; i <= 30; i++ {
|
||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||
}
|
||||
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
|
||||
add("ZR", Special, "zero register (reads as 0)")
|
||||
add("SP", Special, "stack pointer")
|
||||
add("LR", Special, "link register (alias of R30)")
|
||||
|
||||
+130
-30
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
return 4
|
||||
}
|
||||
}
|
||||
// The funcdata pseudo-statements contribute no bytes.
|
||||
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||
// FUNCDATA/PCDATA forms included.
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
|
||||
return 0
|
||||
}
|
||||
switch mnem {
|
||||
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
||||
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
||||
// code: the toolchain emits zero instruction bytes for them, and so does
|
||||
// the encoder here.
|
||||
// the encoder here. Files that include funcdata.h spell them after
|
||||
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
|
||||
// bookkeeping too (the same treatment the loong64 encoder applies).
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
return nil, nil
|
||||
case "END":
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
|
||||
}
|
||||
return nil, nil
|
||||
case "FUNCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
|
||||
}
|
||||
return nil, nil
|
||||
case "PCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
||||
return nil, fmt.Errorf("PCDATA expects $n, $n")
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
||||
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
||||
// over the plain arrangement paths, which carry no index.
|
||||
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
if mnem == "VDUP" || mnem == "VMOV" {
|
||||
if arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
}
|
||||
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
|
||||
// vector (asm7.go case 82, shared by both mnemonics). The element
|
||||
// paths above only run when a lane index is spelled, so this is the
|
||||
// whole-vector shape's only route.
|
||||
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
|
||||
return b, err
|
||||
}
|
||||
}
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// Integer → integer: ORR Rd, ZR, Rs.
|
||||
// Integer → integer. Every truncating register move lowers to an
|
||||
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
|
||||
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
|
||||
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
|
||||
// ORR Xd, XZR, Xm.
|
||||
if rs != 31 {
|
||||
switch mnem {
|
||||
case "MOVB":
|
||||
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVH":
|
||||
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVW":
|
||||
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVBU":
|
||||
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVHU":
|
||||
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
}
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
|
||||
mnem == "MOVH" || mnem == "MOVHU" {
|
||||
if mnem == "MOVWU" {
|
||||
sf = 0
|
||||
}
|
||||
// A narrow move out of the zero register loses its width: the
|
||||
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
|
||||
// WZR. MOVD and MOV keep the 64-bit form.
|
||||
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
|
||||
sf = 0
|
||||
}
|
||||
op := uint32(1<<29 | 0x0a<<24) // ORR
|
||||
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
|
||||
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
|
||||
// for bare V registers alone: the arranged forms and every other
|
||||
// three-register mnemonic are rejected outright.
|
||||
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
|
||||
vm, ok1 := arm64VecOf(ops[0])
|
||||
vn, ok2 := arm64VecOf(ops[1])
|
||||
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
|
||||
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
|
||||
}
|
||||
base := uint32(0x5ee08400) // VADD
|
||||
if mnem == "VSUB" {
|
||||
base = 0x7ee08400
|
||||
}
|
||||
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
|
||||
// general register into an arranged vector, the spelling asm7.go's case 82
|
||||
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
|
||||
// shape, so the caller falls through to the arrangement and element paths;
|
||||
// the toolchain rejects the bare spellings outright, and the reverse
|
||||
// Vd.<T>, Rs with them.
|
||||
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, false, nil
|
||||
}
|
||||
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
|
||||
return nil, false, nil
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, false, nil
|
||||
}
|
||||
dst, ok := arm64VecOf(ops[1])
|
||||
if !ok || dst.hasIdx || dst.arr == "" {
|
||||
return nil, false, nil
|
||||
}
|
||||
b, err := a64GPVecWhole(mnem, rs, dst)
|
||||
return b, true, err
|
||||
}
|
||||
|
||||
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
|
||||
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
|
||||
// lane width and Q the vector length. Both VMOV and VDUP take this form
|
||||
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
|
||||
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
|
||||
// spelling for this form, so the encoder refuses it too.
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
// lane indices:
|
||||
//
|
||||
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
||||
}
|
||||
if !dst.hasIdx {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D1":
|
||||
imm5, q = 8, 0
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
// Duplicates the register across every lane (DUP Vd.T, Rn).
|
||||
return a64GPVecWhole(mnem, rs, dst)
|
||||
}
|
||||
f, ok := a64ElemField(dst.arr, dst.idx)
|
||||
if !ok {
|
||||
|
||||
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
|
||||
return 17
|
||||
case "R18":
|
||||
return 18
|
||||
case "R18_PLATFORM":
|
||||
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
|
||||
// cmd/asm/internal/arch so assembly cannot use it by accident, and
|
||||
// sys_windows_arm64.s references it only through this name.
|
||||
return 18
|
||||
case "R19":
|
||||
return 19
|
||||
case "R20":
|
||||
|
||||
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
|
||||
}{
|
||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||
{"R18_PLATFORM", 18},
|
||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||
}
|
||||
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
|
||||
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
|
||||
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
|
||||
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
|
||||
// rejected, as the toolchain rejects it.
|
||||
func TestArm64GPIntoVector(t *testing.T) {
|
||||
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
|
||||
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
|
||||
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
|
||||
want := []uint32{
|
||||
0x4e010ca5, // VMOV R5, V5.B16
|
||||
0x0e010c22, // VMOV R1, V2.B8
|
||||
0x0e020c64, // VMOV R3, V4.H4
|
||||
0x4e040d2a, // VMOV R9, V10.S4
|
||||
0x4e020cff, // VMOV R7, V31.H8
|
||||
0x4e080d6c, // VMOV R11, V12.D2
|
||||
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
|
||||
0x4e020d2a, // VDUP R9, V10.H8
|
||||
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if _, err := AssembleFileARM64(f); err == nil {
|
||||
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
|
||||
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
|
||||
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
|
||||
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
|
||||
func TestArm64SimdTwoOperand(t *testing.T) {
|
||||
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
|
||||
want := []uint32{
|
||||
0x5ee78508, // VADD V7, V8
|
||||
0x7ee78508, // VSUB V7, V8
|
||||
0x5ee18442, // VADD V1, V2
|
||||
0x4e208422, // VADD arranged: the ordinary three-register path
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64TruncMove pins the truncating register moves against
|
||||
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
|
||||
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
|
||||
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
|
||||
// drops to the W ORR too (asm7.go case 45).
|
||||
func TestArm64TruncMove(t *testing.T) {
|
||||
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
|
||||
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
|
||||
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
|
||||
want := []uint32{
|
||||
0x93401c64, // MOVB = SXTB
|
||||
0x93403ca6, // MOVH = SXTH
|
||||
0x93407d2a, // MOVW = SXTW
|
||||
0xd3401c64, // MOVBU = UXTB
|
||||
0xd3403c64, // MOVHU = UXTH
|
||||
0x2a0303e4, // MOVWU = ORR W
|
||||
0xaa0303e4, // MOVD = ORR X
|
||||
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
|
||||
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
|
||||
0x2a1f03e5, // MOVWU ZR, R5
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
||||
func TestArm64SIMDLoadStore(t *testing.T) {
|
||||
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
||||
|
||||
+68
-11
@@ -18,12 +18,16 @@ const (
|
||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
||||
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
||||
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
||||
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
|
||||
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
|
||||
// lives with the DWARF fixup constants as rAARCH64Abs64.
|
||||
rArm64Abs32 = 258
|
||||
)
|
||||
|
||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||
// optional .rela.text and an optional .rela.data.
|
||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -133,6 +137,50 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rAARCH64Abs64
|
||||
case 4:
|
||||
typ = rArm64Abs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -142,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
@@ -188,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -200,6 +243,17 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
@@ -257,6 +311,9 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
|
||||
@@ -197,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
t.Error("unexpected .rela.text section when there are no relocations")
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
|
||||
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFAARCH64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+68
-11
@@ -23,12 +23,16 @@ const (
|
||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
||||
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
|
||||
// with the DWARF fixup constants as rLarchAbs64.
|
||||
rLarchAbs32 = 1
|
||||
)
|
||||
|
||||
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
||||
// optional .rela.text.
|
||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||
// optional .rela.text and an optional .rela.data.
|
||||
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -117,6 +121,50 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rLarchAbs64
|
||||
case 4:
|
||||
typ = rLarchAbs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -126,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
@@ -172,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -184,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
@@ -239,6 +293,9 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
|
||||
@@ -245,3 +245,118 @@ func TestELFLOONG64BranchRelocation(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
|
||||
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileLOONG64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||
}
|
||||
obj, err := img.ELFLOONG64Object()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+68
-10
@@ -24,11 +24,16 @@ const (
|
||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
|
||||
// with the DWARF fixup constants as rRISCVAbs64.
|
||||
rRISVCAbs32 = 1
|
||||
)
|
||||
|
||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
||||
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
|
||||
// an optional .rela.data.
|
||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
le := binary.LittleEndian
|
||||
|
||||
@@ -129,6 +134,50 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||
// DATA line's width at the field's data-section offset, S + A with no
|
||||
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||
// cannot hold an address, so they are refused rather than truncated.
|
||||
var dataRelas []elfRela
|
||||
for _, d := range img.DataSyms {
|
||||
for _, r := range d.Relocs {
|
||||
idx, ok := symIdx[r.Name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||
}
|
||||
var typ uint32
|
||||
switch r.Siz {
|
||||
case 8:
|
||||
typ = rRISCVAbs64
|
||||
case 4:
|
||||
typ = rRISVCAbs32
|
||||
default:
|
||||
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||
}
|
||||
dataRelas = append(dataRelas, elfRela{
|
||||
off: uint64(d.Offset + r.Off),
|
||||
sym: idx,
|
||||
typ: typ,
|
||||
addend: r.Addend,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Section presence: .rela.text only when there are code relocations,
|
||||
// .rela.data only when a DATA line holds a symbol value.
|
||||
hasRela := len(relas) > 0
|
||||
hasDataRela := len(dataRelas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections++
|
||||
}
|
||||
if hasDataRela {
|
||||
nSections++
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// String tables.
|
||||
stNames := newElfStrtab()
|
||||
for _, s := range syms {
|
||||
@@ -138,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||
stSections.add(n)
|
||||
}
|
||||
if hasDataRela {
|
||||
stSections.add(".rela.data")
|
||||
}
|
||||
for _, n := range dwarfSectionNames {
|
||||
stSections.add(n)
|
||||
}
|
||||
|
||||
hasRela := len(relas) > 0
|
||||
nSections := 6
|
||||
if hasRela {
|
||||
nSections = 7
|
||||
}
|
||||
secSymtab, secStrtab := 3, 4
|
||||
secShstr := nSections - 1
|
||||
|
||||
// Layout.
|
||||
var out []byte
|
||||
out = append(out, make([]byte, 64)...)
|
||||
@@ -184,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
strtabOff := len(out)
|
||||
out = append(out, stNames.bytes()...)
|
||||
|
||||
var relaOff int
|
||||
var relaOff, relaDataOff int
|
||||
if hasRela {
|
||||
align(8)
|
||||
relaOff = len(out)
|
||||
@@ -196,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
if hasDataRela {
|
||||
align(8)
|
||||
relaDataOff = len(out)
|
||||
for _, r := range dataRelas {
|
||||
var b [24]byte
|
||||
le.PutUint64(b[0:], r.off)
|
||||
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||
le.PutUint64(b[16:], uint64(r.addend))
|
||||
out = append(out, b[:]...)
|
||||
}
|
||||
}
|
||||
|
||||
shstrOff := len(out)
|
||||
out = append(out, stSections.bytes()...)
|
||||
@@ -251,6 +306,9 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||
if hasRela {
|
||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||
}
|
||||
if hasDataRela {
|
||||
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||
}
|
||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||
// DWARF section headers; their indices follow the write order.
|
||||
if dw != nil {
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"debug/elf"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
|
||||
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
|
||||
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
|
||||
// field's offset within .data, against the named symbol, external targets
|
||||
// included.
|
||||
func TestELFRISCVObjectDataRelocation(t *testing.T) {
|
||||
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
|
||||
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
GLOBL holder(SB), NOPTR, $32
|
||||
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||
DATA holder+8(SB)/8, $holder(SB)
|
||||
DATA holder+16(SB)/8, $extvar(SB)
|
||||
DATA holder+24(SB)/4, $Keep(SB)
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileRISCV(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||
}
|
||||
obj, err := img.ELFRISCVObject()
|
||||
if err != nil {
|
||||
t.Fatalf("ELFRISCVObject: %v", err)
|
||||
}
|
||||
checkELFSectionAccounting(t, obj)
|
||||
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||
if err != nil {
|
||||
t.Fatalf("parse emitted object: %v", err)
|
||||
}
|
||||
defer ef.Close()
|
||||
relaData := ef.Section(".rela.data")
|
||||
if relaData == nil {
|
||||
t.Fatal("missing .rela.data section")
|
||||
}
|
||||
if relaData.Type != elf.SHT_RELA {
|
||||
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||
}
|
||||
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||
}
|
||||
if ef.Sections[relaData.Info].Name != ".data" {
|
||||
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||
}
|
||||
relas, err := relaData.Data()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}
|
||||
for i := 0; i+24 <= len(relas); i += 24 {
|
||||
got = append(got, struct {
|
||||
off uint64
|
||||
sym uint32
|
||||
typ uint32
|
||||
addend int64
|
||||
}{
|
||||
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||
// r_info packs the type in the low dword and the symbol index
|
||||
// in the high dword.
|
||||
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||
})
|
||||
}
|
||||
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||
syms, err := ef.Symbols()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := func(idx uint32) string {
|
||||
if idx >= 1 && int(idx) <= len(syms) {
|
||||
return syms[idx-1].Name
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// The offsets are data-section-relative: the field's DATA offset plus
|
||||
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||
base := uint64(0)
|
||||
for _, d := range img.DataSyms {
|
||||
if d.Name == "holder" {
|
||||
base = uint64(d.Offset)
|
||||
}
|
||||
}
|
||||
want := []struct {
|
||||
off uint64
|
||||
typ uint32
|
||||
addend int64
|
||||
target string
|
||||
}{
|
||||
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
|
||||
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
|
||||
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
|
||||
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
g := got[i]
|
||||
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||
}
|
||||
if n := name(g.sym); n != w.target {
|
||||
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -113,6 +113,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return err
|
||||
}
|
||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||
isEvexPrefGather(base) ||
|
||||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||
return e.encodeVec(base, ops, sfx)
|
||||
}
|
||||
|
||||
+496
-22
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
|
||||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
||||
// the W bit distinguishes it from VPSRAD's E2 form).
|
||||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
||||
// rm=src, no vvvv).
|
||||
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
|
||||
|
||||
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
||||
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
||||
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||
@@ -202,7 +203,7 @@ var evexTable = map[string]evexSpec{
|
||||
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
@@ -324,7 +325,7 @@ var evexTable = map[string]evexSpec{
|
||||
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.66.0F38, half-precision convert (half-width source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
||||
@@ -494,6 +495,246 @@ var evexTable = map[string]evexSpec{
|
||||
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
||||
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||
|
||||
// --- the AVX-512 families the avx512enc corpus exercises, read off
|
||||
// the toolchain opcodetables ---
|
||||
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
|
||||
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
|
||||
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
|
||||
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
||||
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
|
||||
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
|
||||
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
|
||||
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
|
||||
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
|
||||
// immediate; there is no register-count twin).
|
||||
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
|
||||
// whose EVEX form drops the legacy prefix entirely.
|
||||
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
|
||||
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
|
||||
// split the high/low lane spellings.
|
||||
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
|
||||
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
|
||||
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
|
||||
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
|
||||
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
|
||||
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
|
||||
|
||||
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
|
||||
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
|
||||
// forms live in evexRegFormTable.
|
||||
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
|
||||
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
|
||||
// three-operand insert form (rm = m64 source, vvvv = preserved,
|
||||
// reg = dst) and the two-operand store (reg = source, rm = m64);
|
||||
// the encoder splits on the operand count. VMOVLHPS is the
|
||||
// three-operand form alone.
|
||||
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||
}
|
||||
|
||||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||||
@@ -529,32 +770,38 @@ type evexMoveSpec struct {
|
||||
n [3]int
|
||||
vecOK bool // the non-memory operand may be a vector register
|
||||
xmmOnly bool // wider than XMM registers are rejected
|
||||
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
|
||||
}
|
||||
|
||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||
var evexMoveTable = map[string]evexMoveSpec{
|
||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||
// semantics).
|
||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||
// encoding).
|
||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512, aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||
// three-operand register form is not supported).
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
|
||||
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
|
||||
// three-operand register form (VMOVSD dst, src1, src2).
|
||||
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
|
||||
// EVEX.128/256/512.0F.W0, unaligned packed single move.
|
||||
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
|
||||
}
|
||||
|
||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||
@@ -579,6 +826,13 @@ func evexRequired(upper string, ops []Operand) bool {
|
||||
if !inVex && !inVexMove {
|
||||
return true // EVEX-only mnemonic
|
||||
}
|
||||
// The byte-quad shifts have VEX register forms but EVEX-only memory
|
||||
// forms: a memory count source forces the EVEX encoding.
|
||||
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
|
||||
if slices.ContainsFunc(ops, memOperand) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
for _, op := range ops {
|
||||
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
||||
return true
|
||||
@@ -752,6 +1006,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
}
|
||||
spec.n = [3]int{n, n, n}
|
||||
}
|
||||
// A mnemonic with an immediate and a register spelling (the
|
||||
// variable-count shifts, the permutes) encodes the register one
|
||||
// when the first operand is not an immediate.
|
||||
if len(ops) > 0 {
|
||||
if _, isImm := ops[0].(Imm); !isImm {
|
||||
if alt, ok := evexRegFormTable[mnemUpper]; ok {
|
||||
spec, inTable = alt, true
|
||||
}
|
||||
}
|
||||
}
|
||||
// The high/low half moves split by operand count: three operands
|
||||
// insert, two store (VMOVHPS m64, X1).
|
||||
if hs, ok := evexHptrTable[mnemUpper]; ok {
|
||||
if len(ops) == 2 {
|
||||
if hs.store.opcode == 0 {
|
||||
return fmt.Errorf("%s has no two-operand form", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
|
||||
}
|
||||
spec = hs.insert
|
||||
}
|
||||
} else if sfx.evexOnly() {
|
||||
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
||||
}
|
||||
@@ -811,6 +1086,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
}
|
||||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
||||
}
|
||||
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
|
||||
if sfx.any() {
|
||||
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
|
||||
}
|
||||
if !inTable {
|
||||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||||
}
|
||||
@@ -829,6 +1110,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
||||
case vexExtract:
|
||||
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
||||
case vexExtractGPR:
|
||||
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
||||
}
|
||||
@@ -908,6 +1191,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
||||
if dstReg.mask {
|
||||
if r, ok := src.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
} else if l, err := soleLen(spec.n); err == nil {
|
||||
// A memory source with a length-fixed mnemonic
|
||||
// (VFPCLASSPDX/Y/Z): the length comes from the table's
|
||||
// single valid slot, not from the operand.
|
||||
ll = l
|
||||
}
|
||||
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
@@ -934,9 +1222,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("shift source must be a vector register")
|
||||
// The count source is a vector register or memory; the length the L'L
|
||||
// field and the disp8×N multiplier follow is the destination's either
|
||||
// way.
|
||||
if !vecOrMem(src) {
|
||||
return fmt.Errorf("shift source must be a vector register or memory")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
@@ -946,7 +1236,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
|
||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
@@ -1018,10 +1308,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
|
||||
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
|
||||
// destination, imm8). The encoding is 128-bit regardless of register
|
||||
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
|
||||
// element size the table carries.
|
||||
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
||||
}
|
||||
imm, src, dst := ops[0], ops[1], ops[2]
|
||||
immVal, ok := imm.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("extract lane must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("extract source must be a vector register")
|
||||
}
|
||||
switch dst.(type) {
|
||||
case Reg:
|
||||
if dst.(Reg).isVec() {
|
||||
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||
}
|
||||
case Mem, sbMem:
|
||||
default:
|
||||
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||||
// assembler.
|
||||
// assembler. The scalar moves also carry a three-operand register form
|
||||
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
|
||||
// opens.
|
||||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) == 3 {
|
||||
if !ms.nds3 {
|
||||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
// The masked scalar register form keeps the Go assembler's own
|
||||
// layout: the store opcode with reg = op0, vvvv = op1 and the
|
||||
// destination in r/m (op2) — the bytes go tool asm emits, not
|
||||
// the manual's NDS reading.
|
||||
src, src1, dst := ops[0], ops[1], ops[2]
|
||||
reg, ok := src.(Reg)
|
||||
if !ok || !reg.isVec() {
|
||||
return fmt.Errorf("%s: first operand must be a vector register", mnem)
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || !vvvvReg.isVec() {
|
||||
return fmt.Errorf("%s: second operand must be a vector register", mnem)
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||
}
|
||||
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
|
||||
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||
}
|
||||
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
|
||||
}
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -1141,12 +1498,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
|
||||
return fmt.Errorf("broadcast destination must be a vector register")
|
||||
}
|
||||
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
||||
switch src.(type) {
|
||||
switch r := src.(type) {
|
||||
case Mem, sbMem:
|
||||
spec.opcode = bs.opMem
|
||||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||
case Reg:
|
||||
spec.opcode = bs.opReg
|
||||
// A GPR source uses the register broadcast opcode; a vector
|
||||
// source shares the xmm/mem one (the low byte is copied from
|
||||
// the lane or from the memory operand).
|
||||
if r.isVec() {
|
||||
spec.opcode = bs.opMem
|
||||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||
} else {
|
||||
spec.opcode = bs.opReg
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("broadcast source must be a register or memory")
|
||||
}
|
||||
@@ -1349,6 +1714,102 @@ func isScatter(upper string) bool {
|
||||
return ok
|
||||
}
|
||||
|
||||
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
|
||||
// prefetch hint.
|
||||
func isEvexPrefGather(upper string) bool {
|
||||
_, ok := evexPrefGatherTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// evexRegFormTable holds the register-count twin of the immediate-form
|
||||
// entries in evexTable. Several mnemonics name two encodings: an immediate
|
||||
// count or control ($imm, src, dst …) and a register-count one whose second
|
||||
// operand is a vector register or memory (count, src2, src1, dst). The
|
||||
// immediate spelling lives in evexTable, this table carries the register
|
||||
// spelling, and encodeEvex picks by whether the first operand is an
|
||||
// immediate, the way vexVarShift does on the VEX side.
|
||||
var evexRegFormTable = map[string]evexSpec{
|
||||
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
|
||||
// controls live in evexTable under 0F3A).
|
||||
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
// EVEX.NDS.0F38, the register-count permil shuffles.
|
||||
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
}
|
||||
|
||||
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
|
||||
// operand with a VSIB index and an opmask register, no destination. The
|
||||
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
|
||||
// the mask register is the instruction's only register operand.
|
||||
type evexPrefGatherSpec struct {
|
||||
mapSel int
|
||||
opcode byte
|
||||
w int
|
||||
pp int
|
||||
opdigit int
|
||||
n int
|
||||
}
|
||||
|
||||
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
|
||||
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
|
||||
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
|
||||
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
|
||||
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
|
||||
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
|
||||
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
|
||||
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
|
||||
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
|
||||
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
|
||||
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
|
||||
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
|
||||
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
|
||||
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
|
||||
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
|
||||
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
|
||||
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
|
||||
}
|
||||
|
||||
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
|
||||
// three-operand insert shares an opcode with a two-operand store whose
|
||||
// source is the vector register and whose destination is m64.
|
||||
type evexHptrSpec struct {
|
||||
insert evexSpec
|
||||
store evexSpec // store.opcode == 0 when the mnemonic has no store form
|
||||
}
|
||||
|
||||
var evexHptrTable = map[string]evexHptrSpec{
|
||||
"VMOVHPS": {
|
||||
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
|
||||
},
|
||||
"VMOVLHPS": {
|
||||
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||
},
|
||||
}
|
||||
|
||||
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
|
||||
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
|
||||
}
|
||||
m, ok := ops[0].(Mem)
|
||||
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
|
||||
}
|
||||
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
|
||||
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
|
||||
}
|
||||
|
||||
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||
// register) and returns it with the vector length the index selects, the
|
||||
// EVEX L'L field follows the index register, not the data register.
|
||||
@@ -1370,7 +1831,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
||||
return err
|
||||
}
|
||||
if mask != 0 || sfx.any() {
|
||||
// EVEX form: OP vsib, K, dst.
|
||||
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
|
||||
// index and the data register lengths (the Go assembler's
|
||||
// layout); the disp8×N multiplier stays the index element size.
|
||||
if len(rest) != 2 {
|
||||
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||
}
|
||||
@@ -1382,6 +1845,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||
}
|
||||
if d := dst.vecLenBit(); d > ll {
|
||||
ll = d
|
||||
}
|
||||
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||
}
|
||||
@@ -1431,6 +1897,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// The L'L field is the wider of the data register and the VSIB index
|
||||
// lengths, the bytes go tool asm emits.
|
||||
if d := src.vecLenBit(); d > ll {
|
||||
ll = d
|
||||
}
|
||||
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||
}
|
||||
@@ -1441,6 +1912,8 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
||||
var evexKOperand = map[string]bool{
|
||||
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||
// The K-to-vector broadcast reads its opmask source from r/m.
|
||||
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
|
||||
}
|
||||
|
||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||
@@ -1541,6 +2014,7 @@ var kOpsTable = map[string]kOpSpec{
|
||||
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
||||
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
||||
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
||||
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
|
||||
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
||||
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
||||
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
||||
|
||||
@@ -721,3 +721,92 @@ func hexCompact(b []byte) string {
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
|
||||
// families the toolchain's avx512enc corpus exercises: the bytes are the
|
||||
// go tool asm output for exactly these operands, and the same families are
|
||||
// covered end to end by the avx512_amd64.s differential kernel.
|
||||
func TestAvx512CorpusFamilies(t *testing.T) {
|
||||
vsib := func(base, idx string, scale int) Operand {
|
||||
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
// AES rounds (EVEX NDS, VEX twin routed by operand width).
|
||||
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
|
||||
// Integer VNNI and the bit algorithm group.
|
||||
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
|
||||
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
|
||||
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
|
||||
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
|
||||
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
|
||||
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
|
||||
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
|
||||
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
|
||||
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
|
||||
// Permutations: immediate and register counts.
|
||||
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
|
||||
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
|
||||
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
|
||||
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
|
||||
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
|
||||
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
|
||||
// Shifts: immediate, register-count and memory-count forms; the
|
||||
// count source carries its own XMM tuple width.
|
||||
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
|
||||
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
|
||||
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
|
||||
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
|
||||
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
|
||||
// Conversions and shuffles with the F2 prefix and no prefix.
|
||||
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
|
||||
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
|
||||
// Gather and scatter prefetch hints (memory-only, /digit in reg).
|
||||
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
|
||||
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
|
||||
// Opmask broadcasts and the K logic.
|
||||
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
|
||||
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
|
||||
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
|
||||
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
|
||||
// Lane extracts to general registers (EVEX and VEX routes).
|
||||
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
|
||||
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
|
||||
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
|
||||
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
|
||||
// Moves: masked unaligned, masked scalar register form, half moves
|
||||
// and non-temporal stores.
|
||||
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
|
||||
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
|
||||
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
|
||||
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
|
||||
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
|
||||
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
|
||||
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
|
||||
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
|
||||
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
|
||||
// Scalar compares with and without the 66 prefix.
|
||||
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
|
||||
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
|
||||
// Floating point helpers.
|
||||
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
|
||||
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
|
||||
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
|
||||
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
|
||||
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+221
-7
@@ -61,6 +61,21 @@ const (
|
||||
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||
vexImmRMGPR
|
||||
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
||||
// (op0), VEX.vvvv = dst (op1), L = 0.
|
||||
vexRMOpGPR
|
||||
// vexCountGPR is the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
||||
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
||||
// family, with reg = dst (op2), L = 0.
|
||||
vexCountGPR
|
||||
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
||||
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
||||
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
||||
// carries a vector length, so the register the L'L field follows is the
|
||||
// XMM source.
|
||||
vexExtractGPR
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
@@ -230,8 +245,34 @@ var vexTable = map[string]vexSpec{
|
||||
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
||||
// share the F7/F5 opcodes across W, the variable shifts carry their
|
||||
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
||||
// in F2/F3.
|
||||
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
||||
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
||||
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
||||
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
||||
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
||||
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
||||
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
||||
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
||||
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
||||
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
||||
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
||||
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
||||
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
||||
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
||||
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
||||
// rm = src, vvvv = dst).
|
||||
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
||||
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
||||
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
||||
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
||||
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
||||
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
|
||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
@@ -290,6 +331,128 @@ var vexTable = map[string]vexSpec{
|
||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
||||
// spellings, read off the toolchain opcode tables ---
|
||||
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
||||
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
||||
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
||||
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
||||
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
||||
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
||||
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
||||
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
||||
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
||||
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
||||
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
||||
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
||||
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
||||
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
||||
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
||||
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
||||
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
||||
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
||||
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
||||
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
||||
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
||||
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
||||
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
||||
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
||||
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
||||
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
||||
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
||||
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
||||
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
||||
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
||||
|
||||
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
||||
// spellings (no 66 prefix; WIG, so W = 0).
|
||||
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
||||
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
||||
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
||||
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
||||
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
||||
// VEX.128.66.0F, the scalar and packed compare forms.
|
||||
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
||||
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
||||
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
||||
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
||||
}
|
||||
|
||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||
@@ -420,6 +583,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3GPR(spec, ops)
|
||||
case vexImmRMGPR:
|
||||
return e.encodeVexImmRMGPR(spec, ops)
|
||||
case vexRMOpGPR:
|
||||
return e.encodeVexRMOpGPR(spec, ops)
|
||||
case vexCountGPR:
|
||||
return e.encodeVexCountGPR(spec, ops)
|
||||
case vexRMRev:
|
||||
return e.encodeVexRMRev(spec, ops)
|
||||
}
|
||||
@@ -523,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("shift source must be a vector register")
|
||||
// The count source is a vector register or memory; the VEX length
|
||||
// follows the destination register either way.
|
||||
if !vecOrMem(src) {
|
||||
return fmt.Errorf("shift source must be a vector register or memory")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
@@ -533,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
}
|
||||
|
||||
vvvvBar := 15 - (dstReg.idx & 15)
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
||||
return err
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
@@ -700,7 +868,11 @@ func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||
if !ok || vvvvReg.isVec() {
|
||||
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
||||
}
|
||||
|
||||
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||
@@ -729,6 +901,48 @@ func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
||||
// ModRM.rm = src and VEX.vvvv = dst.
|
||||
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
||||
}
|
||||
|
||||
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
||||
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
|
||||
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src, count, dst := ops[0], ops[1], ops[2]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
countReg, ok := count.(Reg)
|
||||
if !ok || countReg.isVec() {
|
||||
return fmt.Errorf("VEX count operand must be a general-purpose register")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || srcReg.isVec() {
|
||||
return fmt.Errorf("VEX count source must be a general-purpose register")
|
||||
}
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
||||
}
|
||||
|
||||
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||
// a store with no register-destination form).
|
||||
|
||||
@@ -31,6 +31,51 @@ var x86asmUnrecognised = map[string]bool{
|
||||
"RORXQ": true,
|
||||
"VFMADD213SD": true,
|
||||
"VFNMADD231SD": true,
|
||||
// The scalar FMA spellings the decoder's tables lack entirely.
|
||||
"VFMADD132SD": true,
|
||||
"VFMADD132SS": true,
|
||||
"VFMADD213SS": true,
|
||||
"VFMADD231SD": true,
|
||||
"VFMADD231SS": true,
|
||||
"VFMSUB132SD": true,
|
||||
"VFMSUB132SS": true,
|
||||
"VFMSUB213SD": true,
|
||||
"VFMSUB213SS": true,
|
||||
"VFMSUB231SD": true,
|
||||
"VFMSUB231SS": true,
|
||||
"VFNMADD132SD": true,
|
||||
"VFNMADD132SS": true,
|
||||
"VFNMADD213SD": true,
|
||||
"VFNMADD213SS": true,
|
||||
"VFNMADD231SS": true,
|
||||
"VFNMSUB132SD": true,
|
||||
"VFNMSUB132SS": true,
|
||||
"VFNMSUB213SD": true,
|
||||
"VFNMSUB213SS": true,
|
||||
"VFNMSUB231SD": true,
|
||||
"VFNMSUB231SS": true,
|
||||
// The BMI1 unary bit ops the decoder's AVX tables lack.
|
||||
"BLSIL": true,
|
||||
"BLSIQ": true,
|
||||
"BLSMSKL": true,
|
||||
"BLSMSKQ": true,
|
||||
"BLSRL": true,
|
||||
"BLSRQ": true,
|
||||
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
|
||||
"BEXTRL": true,
|
||||
"BEXTRQ": true,
|
||||
"BZHIL": true,
|
||||
"BZHIQ": true,
|
||||
"PDEPL": true,
|
||||
"PDEPQ": true,
|
||||
"PEXTL": true,
|
||||
"PEXTQ": true,
|
||||
"SARXL": true,
|
||||
"SARXQ": true,
|
||||
"SHLXL": true,
|
||||
"SHLXQ": true,
|
||||
"SHRXL": true,
|
||||
"SHRXQ": true,
|
||||
}
|
||||
|
||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||
@@ -237,6 +282,18 @@ func TestVexGroundTruth(t *testing.T) {
|
||||
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||
// BMI2 variable shifts and bit ops (three general registers).
|
||||
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
|
||||
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
|
||||
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
|
||||
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
|
||||
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
|
||||
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
|
||||
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
|
||||
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
|
||||
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
|
||||
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
|
||||
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
|
||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||
|
||||
Vendored
+191
@@ -0,0 +1,191 @@
|
||||
// The AVX-512 families behind the avx512enc gap: AES round ops, integer
|
||||
// VNNI and bit algorithms, word shifts and permutes with an immediate or a
|
||||
// register count, lane broadcasts and extracts, gather and scatter prefetch
|
||||
// hints, opmask broadcasts, the high/low half moves and the non-temporal
|
||||
// stores. Every result is folded back so no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func avx512int(p *byte, n int) uint64
|
||||
TEXT ·avx512int(SB), NOSPLIT, $0-24
|
||||
MOVQ p+0(FP), SI
|
||||
MOVQ n+16(FP), CX
|
||||
// AES rounds through the EVEX spellings, masks included.
|
||||
VAESENC Z20, Z21, Z22
|
||||
VAESENCLAST Z23, Z24, Z25
|
||||
VAESDEC (SI), Z26, Z27
|
||||
VAESDECLAST Z28, Z29, Z30
|
||||
// Integer VNNI and the bit algorithm group.
|
||||
VPDPBUSD Z1, Z2, K2, Z3
|
||||
VPDPBUSDS Z4, Z5, K2, Z6
|
||||
VPDPWSSD Z7, Z8, Z9
|
||||
VPDPWSSDS Z10, Z11, K2, Z12
|
||||
VPOPCNTW Z12, K3, Z13
|
||||
VPOPCNTB Z14, Z15
|
||||
VGF2P8MULB Z16, Z17, K4, Z18
|
||||
VGF2P8AFFINEQB $7, Z18, Z19, K5, Z20
|
||||
// Byte/word arithmetic with saturation and masks.
|
||||
VPADDSB Z1, Z2, K1, Z3
|
||||
VPADDUSW Z3, Z4, K1, Z5
|
||||
VPSUBSW Z5, Z6, K1, Z7
|
||||
VPSUBUSB Z7, Z8, K1, Z9
|
||||
VPSADBW Z9, Z10, Z11
|
||||
VPMULHRSW Z11, Z12, Z13
|
||||
VPMULHW Z13, Z14, Z15
|
||||
VPUNPCKLBW Z15, Z16, K2, Z17
|
||||
VPUNPCKHBW Z17, Z18, K2, Z19
|
||||
VPUNPCKLWD Z19, Z20, K2, Z21
|
||||
VPUNPCKHWD Z21, Z22, K2, Z23
|
||||
VPCMPEQB Z23, Z24, K2, K3
|
||||
VPCMPGTW Z25, Z26, K2, K3
|
||||
VPCMPEQQ Z27, Z28, K2
|
||||
VPMULTISHIFTQB Z29, Z30, K3, Z31
|
||||
VDBPSADBW $3, Z1, Z2, K3, Z3
|
||||
MOVQ CX, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func avx512perm(p *byte) uint64
|
||||
TEXT ·avx512perm(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Permutations: immediate and register counts, ternary logic.
|
||||
VALIGNQ $3, Z1, Z2, K1, Z3
|
||||
VPERMT2B Z3, Z4, K1, Z5
|
||||
VPERMT2W Z5, Z6, K1, Z7
|
||||
VPERMT2PS Z7, Z8, K1, Z9
|
||||
VPERMI2W Z9, Z10, K1, Z11
|
||||
VPERMI2PS Z11, Z12, K1, Z13
|
||||
VPERMI2PD Z13, Z14, K1, Z15
|
||||
VPERMB Z15, Z16, K1, Z17
|
||||
VPERMW Z17, Z18, K1, Z19
|
||||
VPERMPS Z19, Z20, Z21
|
||||
VPERMD Z20, Z21, Z22
|
||||
VPERMQ $1, Z1, K2, Z2
|
||||
VPERMQ Z3, Z4, K2, Z5
|
||||
VPERMPD $1, Z5, K2, Z6
|
||||
VPERMPD Z7, Z8, K2, Z9
|
||||
VPERMILPS $5, Z9, K2, Z10
|
||||
VPERMILPS Z11, Z12, K2, Z13
|
||||
VPERMILPD $1, Z13, K2, Z14
|
||||
VPERMILPD Z15, Z16, K2, Z17
|
||||
VPTERNLOGD $6, Z17, Z18, K2, Z19
|
||||
VPTERNLOGQ $9, Z19, Z20, K2, Z21
|
||||
// Lane shuffle and blend families.
|
||||
VSHUFPD $1, Z1, Z2, K1, Z3
|
||||
VSHUFPS $2, Z4, Z5, K1, Z6
|
||||
VBLENDMPD Z7, Z8, K1, Z9
|
||||
VBLENDMPS Z9, Z10, K1, Z11
|
||||
VPBLENDMB Z11, Z12, K1, Z13
|
||||
VPBLENDMW Z13, Z14, K1, Z15
|
||||
VPBLENDMD Z15, Z16, K1, Z17
|
||||
VPBLENDMQ Z17, Z18, K1, Z19
|
||||
// Conflicts and leading zero counts.
|
||||
VPCONFLICTD Z1, K1, Z2
|
||||
VPCONFLICTQ Z3, K1, Z4
|
||||
VPLZCNTD Z5, K1, Z6
|
||||
VPLZCNTQ Z7, K1, Z8
|
||||
// Compress and expand, byte and word widths.
|
||||
VPCOMPRESSB Z1, K1, (SI)
|
||||
VPCOMPRESSW Z2, K1, (SI)
|
||||
VPEXPANDB (SI), K1, Z3
|
||||
VPEXPANDW (SI), K1, Z4
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512shift(p *byte) uint64
|
||||
TEXT ·avx512shift(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Variable shifts and shuffles with masks.
|
||||
VPSLLVW Z1, Z2, K1, Z3
|
||||
VPSRLVW Z3, Z4, K1, Z5
|
||||
VPSRAVW Z5, Z6, K1, Z7
|
||||
VPSHLDVW Z7, Z8, K1, Z9
|
||||
VPSHRDVW Z9, Z10, K1, Z11
|
||||
VPSHLDVD Z11, Z12, K1, Z13
|
||||
VPSHLDVQ Z13, Z14, K1, Z15
|
||||
VPSHRDVD Z15, Z16, K1, Z17
|
||||
VPSHRDVQ Z17, Z18, K1, Z19
|
||||
// Immediate shifts, the word/byte-quad widths and masks.
|
||||
VPSLLW $3, Z1, K2, Z2
|
||||
VPSRLW $5, Z3, K2, Z4
|
||||
VPSRAW $7, Z5, K2, Z6
|
||||
VPSLLDQ $9, Z7, Z8
|
||||
VPSRLDQ $11, Z9, Z10
|
||||
// Register-count shifts and their memory-count forms.
|
||||
VPSLLD X1, Z2, K1, Z3
|
||||
VPSRLD 16(SI), Z4, K1, Z5
|
||||
VPSLLQ X6, Z7, K1, Z8
|
||||
VPSRLQ X9, Z10, K1, Z11
|
||||
VPSLLW X12, Z13, K1, Z14
|
||||
VPSRAW X15, Z16, K1, Z17
|
||||
VPSRAQ $13, Z12, K1, Z13
|
||||
VPSRAD X14, Z15, K1, Z16
|
||||
// Lane shuffles in and out.
|
||||
VPSHLDW $2, Z1, Z2, K1, Z3
|
||||
VPSHLDQ $4, Z3, Z4, K1, Z5
|
||||
VPSHRDW $6, Z5, Z6, K1, Z7
|
||||
VPSHRDQ $8, Z7, Z8, K1, Z9
|
||||
VPSHUFBITQMB Z9, Z10, K3
|
||||
VPTESTMB Z11, Z12, K4
|
||||
VPTESTNMQ Z13, Z14, K5
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512float(x float64) float64
|
||||
TEXT ·avx512float(SB), NOSPLIT, $0-16
|
||||
// Square roots, compares and the EXP2/RCP28 helpers.
|
||||
MOVQ x+0(FP), AX
|
||||
VSQRTPD Z1, K1, Z2
|
||||
VSQRTPS Z3, K1, Z4
|
||||
VSQRTSD X1, X2, K1, X3
|
||||
VSQRTSS X3, X4, X5
|
||||
VCOMISD X5, X6
|
||||
VUCOMISS X7, X8
|
||||
VEXP2PD Z5, K1, Z6
|
||||
VRCP28PD Z7, K1, Z8
|
||||
VRCP28SD X9, X8, K1, X10
|
||||
VRSQRT28PS Z11, K1, Z12
|
||||
VRSQRT28SS X11, X10, K1, X12
|
||||
VCVTSD2SS X1, X2, X3
|
||||
VCVTSS2SD X3, X2, K1, X4
|
||||
VFMADD132PD Z1, Z2, K1, Z3
|
||||
VFMADD231SD X1, X2, K1, X3
|
||||
VFMSUBADD213PS Z3, Z4, K1, Z5
|
||||
VFNMSUB231PD Z5, Z6, K1, Z7
|
||||
// Broadcasts and masked moves.
|
||||
VBROADCASTF32X2 X1, K1, Z2
|
||||
VBROADCASTI64X2 (SI), K1, Z3
|
||||
VMOVUPS Z1, K2, Z3
|
||||
VMOVSD X14, X5, K3, X22
|
||||
VMOVSS X18, X3, K2, X25
|
||||
VMOVHPS (SI), X18, X19
|
||||
VMOVHPS X20, 8(SI)
|
||||
VMOVLHPS X16, X5, X17
|
||||
VMOVNTDQ Z7, (SI)
|
||||
VMOVNTDQA 64(SI), Z8
|
||||
VMOVNTPD Z9, (SI)
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func avx512mask(p *byte) uint64
|
||||
TEXT ·avx512mask(SB), NOSPLIT, $0-16
|
||||
MOVQ p+0(FP), SI
|
||||
// Omask broadcasts and the K register logic.
|
||||
VPBROADCASTMB2Q K1, Z2
|
||||
VPBROADCASTMW2D K3, Z4
|
||||
KUNPCKWD K6, K4, K1
|
||||
KADDB K2, K3, K5
|
||||
KORW K1, K2, K7
|
||||
// Gather and scatter prefetch hints.
|
||||
VGATHERPF0DPD K5, (SI)(Y29*8)
|
||||
VSCATTERPF1DPS K2, (SI)(Z28*4)
|
||||
// Masked gathers ride the EVEX spelling; the data length wins L'L.
|
||||
VGATHERDPD (SI)(X10*4), K7, Y22
|
||||
VPSCATTERDQ Y6, K2, (SI)(X4*1)
|
||||
// Lane extracts to general registers.
|
||||
VPEXTRB $3, X1, AX
|
||||
VPEXTRD $1, X2, DI
|
||||
VPINSRQ $1, SI, X3, X4
|
||||
VEXTRACTI32X4 $1, Z1, X5
|
||||
VINSERTI64X2 $1, X6, Z7, K2, Z8
|
||||
MOVQ SI, ret+8(FP)
|
||||
RET
|
||||
Vendored
+31
@@ -0,0 +1,31 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 bookkeeping statements: the funcdata.h
|
||||
// pseudo-directives (GO_ARGS, NO_LOCAL_POINTERS, FUNCDATA, PCDATA) contribute
|
||||
// no instruction bytes, and every function is byte-compared against
|
||||
// go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
#include "funcdata.h"
|
||||
|
||||
// func bookkeep()
|
||||
TEXT ·bookkeep(SB), NOSPLIT, $8-0
|
||||
GO_ARGS
|
||||
FUNCDATA $3, inline_tree(SB)
|
||||
PCDATA $1, $2
|
||||
MOVD R1, 0(RSP)
|
||||
RET
|
||||
|
||||
// func bookkeepNoLocals()
|
||||
TEXT ·bookkeepNoLocals(SB), NOSPLIT, $16-0
|
||||
NO_LOCAL_POINTERS
|
||||
PCDATA $0, $0
|
||||
PCDATA $1, $1
|
||||
MOVD R2, 8(RSP)
|
||||
RET
|
||||
|
||||
// func bookkeepPlain()
|
||||
TEXT ·bookkeepPlain(SB), NOSPLIT, $0-0
|
||||
MOVD R3, R4
|
||||
RET
|
||||
Vendored
+57
@@ -0,0 +1,57 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 whole-vector moves between a general
|
||||
// register and an arranged vector (VMOV/VDUP Rs, Vd.<T>), the two-operand
|
||||
// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved
|
||||
// R18_PLATFORM register name. Every function is byte-compared against
|
||||
// go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func gpIntoVector()
|
||||
TEXT ·gpIntoVector(SB), NOSPLIT, $0-0
|
||||
VMOV R1, V2.B8
|
||||
VMOV R3, V4.B16
|
||||
VMOV R5, V6.H4
|
||||
VMOV R7, V8.H8
|
||||
VMOV R9, V10.S2
|
||||
VMOV R11, V12.S4
|
||||
VMOV R13, V14.D2
|
||||
VDUP R15, V16.B8
|
||||
VDUP R17, V18.B16
|
||||
VDUP R19, V20.H8
|
||||
VDUP R21, V22.S4
|
||||
VDUP R23, V24.D2
|
||||
RET
|
||||
|
||||
// func simdAccumulate()
|
||||
TEXT ·simdAccumulate(SB), NOSPLIT, $0-0
|
||||
VADD V7, V8
|
||||
VSUB V7, V8
|
||||
VADD V1, V2
|
||||
VSUB V30, V31
|
||||
VADD V0.B16, V1.B16, V2.B16
|
||||
VSUB V0.S4, V1.S4, V2.S4
|
||||
RET
|
||||
|
||||
// func truncMove()
|
||||
TEXT ·truncMove(SB), NOSPLIT, $0-0
|
||||
MOVB R3, R4
|
||||
MOVH R5, R6
|
||||
MOVW R9, R10
|
||||
MOVBU R3, R4
|
||||
MOVHU R3, R4
|
||||
MOVWU R3, R4
|
||||
MOVD R3, R4
|
||||
RET
|
||||
|
||||
// func platformRegister()
|
||||
TEXT ·platformRegister(SB), NOSPLIT, $0-0
|
||||
MOVD R18_PLATFORM, R3
|
||||
MOVW R18_PLATFORM, R4
|
||||
MOVD R3, R18_PLATFORM
|
||||
MOVD 0x68(R18_PLATFORM), R5
|
||||
MOVD R5, 0x68(R18_PLATFORM)
|
||||
MOVW 8(R18_PLATFORM), R6
|
||||
RET
|
||||
@@ -34,6 +34,8 @@ func TestGroundTruthARM64(t *testing.T) {
|
||||
"../testdata/verify/simd_arm64.s",
|
||||
"../testdata/verify/widenimm_arm64.s",
|
||||
"../testdata/verify/carryshift_arm64.s",
|
||||
"../testdata/verify/simdmove_arm64.s",
|
||||
"../testdata/verify/bookkeep_arm64.s",
|
||||
"../testdata/verify/system_arm64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
|
||||
@@ -124,8 +124,10 @@ func TestGroundTruthAMD64(t *testing.T) {
|
||||
"../testdata/verify/avx_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/pfx_amd64.s",
|
||||
"../testdata/verify/rawdata_amd64.s",
|
||||
"../testdata/verify/avx512_amd64.s",
|
||||
"../testdata/verify/doubleshift_amd64.s",
|
||||
"../testdata/verify/ssestatic_amd64.s",
|
||||
} {
|
||||
|
||||
Reference in New Issue
Block a user