Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f52e23f1bc | ||
|
|
c9775c2b95 | ||
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 |
+148
-4
@@ -308,6 +308,106 @@ var evexTable = map[string]evexSpec{
|
|||||||
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
||||||
// rm=dst, imm8 — the extract layout).
|
// rm=dst, imm8 — the extract layout).
|
||||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX — unsigned and truncating conversions. The PD sources are the
|
||||||
|
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
|
||||||
|
// the length); the PS/UQQ destinations are wide and follow the
|
||||||
|
// destination.
|
||||||
|
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
|
||||||
|
// source; disp8×N follows its size).
|
||||||
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
|
||||||
|
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
|
||||||
|
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
|
||||||
|
// into a vector (rm = K source, reg = vector destination), *2M does the
|
||||||
|
// reverse (reg = K destination, rm = vector source, the length follows
|
||||||
|
// the vector).
|
||||||
|
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX — scalar conversions between vector and general-purpose
|
||||||
|
// registers. Vector to GPR (two operands: vec/mem source, GPR
|
||||||
|
// destination, vvvv unused): the signed and truncated pair, and the
|
||||||
|
// unsigned forms (EVEX only).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||||
// the xmm/ymm/zmm destination lengths).
|
// the xmm/ymm/zmm destination lengths).
|
||||||
@@ -543,6 +643,15 @@ var evexRound = map[string]bool{
|
|||||||
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||||||
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||||||
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||||||
|
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||||||
|
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||||||
|
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||||||
|
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
|
||||||
|
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
|
||||||
|
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
|
||||||
|
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
|
||||||
|
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
|
||||||
|
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||||
@@ -562,6 +671,9 @@ var evexBcstN = map[string]int{
|
|||||||
"VRANGEPD": 8, "VRANGEPS": 4,
|
"VRANGEPD": 8, "VRANGEPS": 4,
|
||||||
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||||||
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||||||
|
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
|
||||||
|
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
|
||||||
|
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||||
@@ -591,6 +703,18 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
|
|||||||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||||||
// broadcast.
|
// broadcast.
|
||||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
// The mask/vector conversions take the K register as a genuine operand
|
||||||
|
// (source or destination), not as a mask, and accept no suffixes.
|
||||||
|
if evexKOperand[mnemUpper] {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
spec, ok := evexTable[mnemUpper]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("unsupported instruction %q", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||||
|
}
|
||||||
spec, inTable := evexTable[mnemUpper]
|
spec, inTable := evexTable[mnemUpper]
|
||||||
if inTable {
|
if inTable {
|
||||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||||
@@ -713,17 +837,29 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf
|
|||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||||
// no vvvv), e.g. VCVTQQ2PD.
|
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||||||
|
// *2M mask conversions) or a general-purpose register (the scalar
|
||||||
|
// vector-to-GPR conversions); in both cases the vector length comes from
|
||||||
|
// the source.
|
||||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
src, dst := ops[0], ops[1]
|
src, dst := ops[0], ops[1]
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok {
|
||||||
return fmt.Errorf("EVEX destination must be a vector register")
|
return fmt.Errorf("EVEX destination must be a register")
|
||||||
}
|
}
|
||||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
ll := dstReg.vecLenBit()
|
||||||
|
if !dstReg.isVec() {
|
||||||
|
// Mask or GPR destination: the length follows the vector source
|
||||||
|
// (128 for a memory source).
|
||||||
|
ll = 0
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||||
@@ -1260,6 +1396,14 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// evexKOperand lists the instructions whose K register is a genuine operand
|
||||||
|
// (the source or destination of a mask/vector conversion) rather than a
|
||||||
|
// mask modifier — the M2 and 2M conversions. They take no masking.
|
||||||
|
var evexKOperand = map[string]bool{
|
||||||
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||||
|
|||||||
@@ -467,6 +467,164 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||||
|
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||||
|
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||||
|
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||||
|
// for byte against the Go assembler, including memory sources and extended
|
||||||
|
// GPRs.
|
||||||
|
func TestEvexGprGroundTruth(t *testing.T) {
|
||||||
|
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||||
|
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||||
|
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||||
|
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||||
|
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||||
|
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||||
|
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||||
|
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||||
|
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||||
|
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||||
|
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||||
|
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||||
|
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||||
|
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||||
|
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||||
|
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||||
|
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||||
|
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||||
|
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||||
|
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||||
|
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||||
|
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||||
|
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||||
|
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||||
|
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||||
|
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||||
|
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||||
|
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||||
|
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||||
|
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||||
|
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||||
|
// the Go assembler.
|
||||||
|
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Unsigned and truncating conversions.
|
||||||
|
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||||
|
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||||
|
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||||
|
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||||
|
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||||
|
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||||
|
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||||
|
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||||
|
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||||
|
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||||
|
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||||
|
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||||
|
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||||
|
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||||
|
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||||
|
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||||
|
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||||
|
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||||
|
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||||
|
// The remaining sign/zero-extending moves.
|
||||||
|
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||||
|
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||||
|
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||||
|
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||||
|
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||||
|
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||||
|
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||||
|
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||||
|
// Signed narrowing stores.
|
||||||
|
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||||
|
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||||
|
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||||
|
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||||
|
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||||
|
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||||
|
// Unsigned narrowing stores.
|
||||||
|
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||||
|
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||||
|
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||||
|
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||||
|
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||||
|
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||||
|
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||||
|
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||||
|
// Mask/vector conversions (the K register is an operand, not a
|
||||||
|
// mask).
|
||||||
|
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||||
|
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||||
|
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||||
|
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||||
|
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||||
|
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||||
|
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||||
|
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestEvexErrors checks the EVEX-specific error paths.
|
// TestEvexErrors checks the EVEX-specific error paths.
|
||||||
func TestEvexErrors(t *testing.T) {
|
func TestEvexErrors(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
|
|||||||
+32
-1
@@ -130,9 +130,15 @@ var vexTable = map[string]vexSpec{
|
|||||||
// no vvvv).
|
// no vvvv).
|
||||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM},
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||||
@@ -198,6 +204,29 @@ var vexTable = map[string]vexSpec{
|
|||||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||||
|
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||||
|
// spellings: the destination is always XMM and the spelling fixes the
|
||||||
|
// source length (X = 128, Y = 256).
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
|
// VEX scalar conversions between vector and general-purpose registers.
|
||||||
|
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||||
|
// unused; the length follows the source).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||||
@@ -222,6 +251,8 @@ var vexSrcLen = map[string]int{
|
|||||||
"VCVTPD2DQY": 1,
|
"VCVTPD2DQY": 1,
|
||||||
"VCVTTPD2DQX": 0,
|
"VCVTTPD2DQX": 0,
|
||||||
"VCVTTPD2DQY": 1,
|
"VCVTTPD2DQY": 1,
|
||||||
|
"VCVTPD2PSX": 0,
|
||||||
|
"VCVTPD2PSY": 1,
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||||
|
|||||||
+6
-3
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if inst.Op.String() != mnem {
|
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+59
-1
@@ -24,11 +24,12 @@ import (
|
|||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version is the release version, stamped at build time via
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||||
var version = "0.14.0"
|
var version = "0.18.0"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
@@ -46,6 +47,8 @@ func main() {
|
|||||||
os.Exit(cmdLint(os.Args[2:]))
|
os.Exit(cmdLint(os.Args[2:]))
|
||||||
case "asm":
|
case "asm":
|
||||||
os.Exit(cmdAsm(os.Args[2:]))
|
os.Exit(cmdAsm(os.Args[2:]))
|
||||||
|
case "verify":
|
||||||
|
os.Exit(cmdVerify(os.Args[2:]))
|
||||||
case "lsp":
|
case "lsp":
|
||||||
os.Exit(cmdLSP(os.Args[2:]))
|
os.Exit(cmdLSP(os.Args[2:]))
|
||||||
case "version", "--version", "-V":
|
case "version", "--version", "-V":
|
||||||
@@ -80,6 +83,7 @@ Commands:
|
|||||||
fmt canonicalise formatting (gofmt for assembly)
|
fmt canonicalise formatting (gofmt for assembly)
|
||||||
lint run static checks
|
lint run static checks
|
||||||
asm assemble .s files to machine code (amd64)
|
asm assemble .s files to machine code (amd64)
|
||||||
|
verify JIT-assemble and run dynamic checks (amd64)
|
||||||
lsp run the language server over stdio
|
lsp run the language server over stdio
|
||||||
version print the version (same as --version)
|
version print the version (same as --version)
|
||||||
|
|
||||||
@@ -466,3 +470,57 @@ requires -p, the package path, and the installed Go toolchain).
|
|||||||
}
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func cmdVerify(args []string) int {
|
||||||
|
fs := newCommand("verify", "gasm verify <file.s>", `
|
||||||
|
Assemble FILE (amd64), map it into executable memory and report the available
|
||||||
|
functions. This confirms the assembled image is self-consistent (no
|
||||||
|
unresolved external symbols) and executable — the prerequisite for dynamic
|
||||||
|
testing.
|
||||||
|
|
||||||
|
With -smoke, each NOSPLIT function is called with a zeroed argument block to
|
||||||
|
confirm the JIT trampoline works end-to-end. This is safe only for functions
|
||||||
|
that tolerate nil pointers and zero lengths in their arguments.
|
||||||
|
`)
|
||||||
|
smoke := fs.Bool("smoke", false, "call each NOSPLIT function with zeroed args")
|
||||||
|
fs.Parse(args)
|
||||||
|
if fs.NArg() != 1 {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: gasm verify [-smoke] <file.s>")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
path := fs.Arg(0)
|
||||||
|
if arch.FromFilename(path) != arch.AMD64 {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm verify: only amd64 is supported")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
k, err := verify.Load(path)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer k.Close()
|
||||||
|
|
||||||
|
names := k.FuncNames()
|
||||||
|
fmt.Printf("%s: %d functions JIT-loaded\n", path, len(names))
|
||||||
|
rc := 0
|
||||||
|
for _, name := range names {
|
||||||
|
fl, _ := k.Func(name)
|
||||||
|
flags := ""
|
||||||
|
if fl.NoSplit {
|
||||||
|
flags = " NOSPLIT"
|
||||||
|
}
|
||||||
|
fmt.Printf(" %s: %d bytes, args=%d, frame=%d%s\n", name, fl.Size, fl.Args, fl.Frame, flags)
|
||||||
|
if *smoke && fl.NoSplit {
|
||||||
|
args := make([]byte, fl.Args)
|
||||||
|
_, err := k.CallFunc(name, args)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Printf(" smoke: FAIL — %v\n", err)
|
||||||
|
rc = 1
|
||||||
|
} else {
|
||||||
|
fmt.Printf(" smoke: OK\n")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return rc
|
||||||
|
}
|
||||||
|
|||||||
+30
-2
@@ -240,8 +240,11 @@ expand/compress family, the broadcasts, the opmask-register instructions
|
|||||||
moves and the remaining extending/narrowing moves, the floating-point
|
moves and the remaining extending/narrowing moves, the floating-point
|
||||||
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||||
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||||
opmask destination, and the remaining VCVT* conversions including the
|
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||||
half-precision pair), and gather/scatter with VSIB addressing — both the
|
truncating, including the length-suffixed X/Y spellings and the
|
||||||
|
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
|
||||||
|
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
|
||||||
|
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
|
||||||
VEX spelling with a vector mask register and the EVEX spelling with an
|
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||||
explicit K mask, where the EVEX length follows the VSIB index register,
|
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||||
not the data register. The EVEX mnemonic
|
not the data register. The EVEX mnemonic
|
||||||
@@ -281,6 +284,31 @@ references and the implicit funcdata/DWARF symbols remain future work (the
|
|||||||
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
linker fills the latter's defaults); the rest of Phase 2 is those, the
|
||||||
remaining EVEX forms and the other architectures.
|
remaining EVEX forms and the other architectures.
|
||||||
|
|
||||||
|
### `verify`
|
||||||
|
|
||||||
|
The dynamic-analysis substrate (Phase 3). It JIT-loads assembled images into
|
||||||
|
executable memory and invokes them directly, enabling differential testing,
|
||||||
|
runtime ABI checks and coverage profiling.
|
||||||
|
|
||||||
|
The execution model is pure Go (stdlib only). `Map` copies machine code into
|
||||||
|
an anonymous `syscall.Mmap` mapping and enforces W^X (write the bytes, then
|
||||||
|
`mprotect` to read-execute). `Call` prepares a stack whose first word is the
|
||||||
|
address of an assembly trampoline (`leaveJIT`), lays the ABI0 argument
|
||||||
|
block after it, switches to that stack via `enterJIT` (which saves the Go
|
||||||
|
stack pointer in a package global and jumps to the target), and recovers
|
||||||
|
control when the function RETs into `leaveJIT` (which restores the Go stack
|
||||||
|
and returns). A 64-byte pad below the return address accommodates the
|
||||||
|
ABIInternal wrapper that the Go runtime interposes on assembly functions.
|
||||||
|
|
||||||
|
`Load` / `LoadSource` / `LoadAST` parse, assemble and map a `.s` file in one
|
||||||
|
step, returning a `Kernel` whose `CallFunc` method marshals the argument block
|
||||||
|
by name. The image must be self-contained (no external relocations); the
|
||||||
|
assembler’s `Image.Bytes()` provides the code-and-data concatenation.
|
||||||
|
|
||||||
|
The `gasm verify` CLI subcommand exposes this: it loads a file, reports the
|
||||||
|
available functions and (with `-smoke`) calls each NOSPLIT function with zeroed
|
||||||
|
arguments to confirm the trampoline round-trips.
|
||||||
|
|
||||||
## Extension points
|
## Extension points
|
||||||
|
|
||||||
- **New architecture:** add an entry to the generator in `_gen`, run
|
- **New architecture:** add an entry to the generator in `_gen`, run
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# Deferred decisions
|
||||||
|
|
||||||
|
Design decisions deliberately postponed, with enough context to pick them up
|
||||||
|
again without re-deriving the analysis. Each entry records what is deferred,
|
||||||
|
why, the options on the table, and the trigger that should reopen it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## GOOBJ external (cross-package) symbol references
|
||||||
|
|
||||||
|
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
||||||
|
symbols defined in the file being assembled; a reference to any other symbol
|
||||||
|
is rejected.
|
||||||
|
|
||||||
|
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
||||||
|
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
||||||
|
symbol in the *referenced package's* symbol-definition table. That ordering
|
||||||
|
is not derivable from the reference site — it lives in the referenced
|
||||||
|
package's gc export data (the iexport binary format, which evolves with the
|
||||||
|
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
||||||
|
import, so emitting external references means either parsing export data
|
||||||
|
ourselves or taking a dependency that does.
|
||||||
|
|
||||||
|
**What works today.** Single-package objects: every symbol the file defines
|
||||||
|
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
||||||
|
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
||||||
|
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
||||||
|
references the assembly symbols, never the reverse. Such a package builds
|
||||||
|
with its assembly object replaced by a gasm-emitted one.
|
||||||
|
|
||||||
|
**The options, when we return.**
|
||||||
|
|
||||||
|
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
||||||
|
The straightforward path: read each imported package's export file
|
||||||
|
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
||||||
|
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
||||||
|
the export files' build IDs) and positional references. Robust across
|
||||||
|
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
||||||
|
production dependency beyond the standard library, an explicit deviation
|
||||||
|
from the "production code depends only on the standard library"
|
||||||
|
principle in the README. Requires the user's explicit agreement.
|
||||||
|
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
||||||
|
Substantial effort and inherently fragile: the format is an internal
|
||||||
|
contract that changes with Go releases, so the parser needs a
|
||||||
|
version-gated fallback and regression tests against several toolchains.
|
||||||
|
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
||||||
|
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
||||||
|
toolchain command exposes a package's symbols *in definition-index
|
||||||
|
order* — `go tool nm` sorts differently — so this does not solve the
|
||||||
|
core problem on its own; it would only feed option 1 or 2.
|
||||||
|
|
||||||
|
**Trigger to reopen.** An assembly file that needs a cross-package
|
||||||
|
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
||||||
|
metadata written in assembly), or any kernel that calls into another
|
||||||
|
package directly. Until then, option 3's limitation is moot and the
|
||||||
|
single-package emitter suffices.
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||||
|
|
||||||
version := "0.14.0"
|
version := "0.18.0"
|
||||||
|
|
||||||
default:
|
default:
|
||||||
@just --list
|
@just --list
|
||||||
|
|||||||
Vendored
+67
@@ -0,0 +1,67 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func add(a, b int64) int64
|
||||||
|
TEXT ·add(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
ADDQ b+8(FP), AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func sum(data []int64) int64
|
||||||
|
// Sums all elements of the slice.
|
||||||
|
TEXT ·sum(SB), NOSPLIT, $0-32
|
||||||
|
MOVQ data_base+0(FP), SI
|
||||||
|
MOVQ data_len+8(FP), CX
|
||||||
|
XORQ AX, AX
|
||||||
|
TESTQ CX, CX
|
||||||
|
JZ sum_done
|
||||||
|
|
||||||
|
sum_loop:
|
||||||
|
ADDQ (SI), AX
|
||||||
|
ADDQ $8, SI
|
||||||
|
DECQ CX
|
||||||
|
JNZ sum_loop
|
||||||
|
|
||||||
|
sum_done:
|
||||||
|
MOVQ AX, ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func wideCopy(dst, src []byte)
|
||||||
|
// Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves.
|
||||||
|
TEXT ·wideCopy(SB), NOSPLIT, $0-48
|
||||||
|
MOVQ dst_base+0(FP), DI
|
||||||
|
MOVQ dst_len+8(FP), BX
|
||||||
|
MOVQ src_base+24(FP), SI
|
||||||
|
MOVQ src_len+32(FP), R8
|
||||||
|
CMPQ BX, R8
|
||||||
|
JLE wc_have_n
|
||||||
|
MOVQ R8, BX
|
||||||
|
|
||||||
|
wc_have_n:
|
||||||
|
CMPQ BX, $32
|
||||||
|
JB wc_small
|
||||||
|
|
||||||
|
VMOVDQU (SI), Y0
|
||||||
|
VMOVDQU Y0, (DI)
|
||||||
|
VMOVDQU -32(SI)(BX*1), Y0
|
||||||
|
VMOVDQU Y0, -32(DI)(BX*1)
|
||||||
|
VZEROUPPER
|
||||||
|
RET
|
||||||
|
|
||||||
|
wc_small:
|
||||||
|
TESTQ BX, BX
|
||||||
|
JZ wc_done
|
||||||
|
|
||||||
|
wc_byte:
|
||||||
|
MOVB (SI), R8B
|
||||||
|
MOVB R8B, (DI)
|
||||||
|
INCQ SI
|
||||||
|
INCQ DI
|
||||||
|
DECQ BX
|
||||||
|
JNZ wc_byte
|
||||||
|
|
||||||
|
wc_done:
|
||||||
|
RET
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"reflect"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// savedSP holds the Go stack pointer while a JIT call is in flight.
|
||||||
|
// Referenced by the assembly trampoline (trampoline_amd64.s).
|
||||||
|
var savedSP uintptr
|
||||||
|
|
||||||
|
// enterJIT switches to the prepared stack and jumps to fn.
|
||||||
|
// It does not return normally; the JIT function's RET transfers control
|
||||||
|
// to leaveJIT, which restores the Go stack.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func enterJIT(fn uintptr, stack uintptr)
|
||||||
|
|
||||||
|
// leaveJIT restores the Go stack after a JIT function returns.
|
||||||
|
// Its address is placed as the return address on the prepared stack.
|
||||||
|
//
|
||||||
|
//go:nosplit
|
||||||
|
func leaveJIT()
|
||||||
|
|
||||||
|
// leaveJITAddr is the machine address of leaveJIT, resolved once at init.
|
||||||
|
var leaveJITAddr uintptr
|
||||||
|
|
||||||
|
func init() {
|
||||||
|
leaveJITAddr = reflect.ValueOf(leaveJIT).Pointer()
|
||||||
|
}
|
||||||
|
|
||||||
|
// stackPad is padding below the return address on the prepared stack.
|
||||||
|
// The ABIInternal wrapper that leaveJIT's address resolves to executes
|
||||||
|
// PUSHQ BP and CALL before reaching the raw assembly, writing up to 16
|
||||||
|
// bytes below the return-address slot. 64 bytes of headroom is ample.
|
||||||
|
const stackPad = 64
|
||||||
|
|
||||||
|
// Call invokes the assembled function at fnAddr with the given ABI0 argument
|
||||||
|
// block (the raw bytes that would appear at FP+0). It returns the argument
|
||||||
|
// block after the call, which contains any results the function wrote back
|
||||||
|
// (the ABI0 convention shares the argument area for inputs and outputs).
|
||||||
|
//
|
||||||
|
// The function must be NOSPLIT (no stack growth) and must not reference
|
||||||
|
// external symbols — the image is self-contained.
|
||||||
|
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||||
|
// Prepare the stack: [padding][leaveJIT addr][args...]
|
||||||
|
stackSize := stackPad + 8 + len(args) + 64 // padding + ret + args + safety
|
||||||
|
stackMem, err := syscall.Mmap(-1, 0, stackSize,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: stack mmap: %w", err)
|
||||||
|
}
|
||||||
|
defer syscall.Munmap(stackMem)
|
||||||
|
|
||||||
|
// The return address sits after the padding; the function's SP will
|
||||||
|
// point here, leaving stackPad bytes below for the wrapper's pushes.
|
||||||
|
retOff := stackPad
|
||||||
|
binary.LittleEndian.PutUint64(stackMem[retOff:retOff+8], uint64(leaveJITAddr))
|
||||||
|
// The ABI0 argument area follows the return address.
|
||||||
|
copy(stackMem[retOff+8:], args)
|
||||||
|
|
||||||
|
stackBase := uintptr(unsafe.Pointer(&stackMem[retOff]))
|
||||||
|
enterJIT(fnAddr, stackBase)
|
||||||
|
|
||||||
|
// Copy out the (possibly modified) argument area.
|
||||||
|
out := make([]byte, len(args))
|
||||||
|
copy(out, stackMem[retOff+8:retOff+8+len(args)])
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build !amd64
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
// Call is unavailable on non-amd64 architectures.
|
||||||
|
func Call(fnAddr uintptr, args []byte) ([]byte, error) {
|
||||||
|
return nil, fmt.Errorf("verify: JIT execution requires amd64")
|
||||||
|
}
|
||||||
@@ -0,0 +1,295 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"math/rand"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// decodeBlockGo is a minimal portable LZ4 block decoder used as the
|
||||||
|
// differential-testing oracle. It mirrors the contract of
|
||||||
|
// go-lz4's decodeBlockGo: (bytesWritten, code) where code is
|
||||||
|
// 0 = ok, 1 = malformed, 2 = zero offset.
|
||||||
|
func decodeBlockGo(src, dst []byte) (int, int) {
|
||||||
|
if len(src) == 0 {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
si, di := 0, 0
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1 // truncated: no token
|
||||||
|
}
|
||||||
|
token := int(src[si])
|
||||||
|
si++
|
||||||
|
|
||||||
|
// Literals.
|
||||||
|
lLen := token >> 4
|
||||||
|
if lLen == 15 {
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
b := int(src[si])
|
||||||
|
si++
|
||||||
|
lLen += b
|
||||||
|
if b != 255 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if si+lLen > len(src) {
|
||||||
|
return 0, 1 // truncated literals
|
||||||
|
}
|
||||||
|
if di+lLen > len(dst) {
|
||||||
|
return 0, 1 // destination overflow
|
||||||
|
}
|
||||||
|
copy(dst[di:di+lLen], src[si:si+lLen])
|
||||||
|
di += lLen
|
||||||
|
si += lLen
|
||||||
|
|
||||||
|
// End of block.
|
||||||
|
if si >= len(src) {
|
||||||
|
return di, 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match offset.
|
||||||
|
if si+2 > len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
offset := int(src[si]) | int(src[si+1])<<8
|
||||||
|
si += 2
|
||||||
|
if offset == 0 {
|
||||||
|
return 0, 2
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match length.
|
||||||
|
mLen := token & 15
|
||||||
|
if mLen == 15 {
|
||||||
|
for {
|
||||||
|
if si >= len(src) {
|
||||||
|
return 0, 1
|
||||||
|
}
|
||||||
|
b := int(src[si])
|
||||||
|
si++
|
||||||
|
mLen += b
|
||||||
|
if b != 255 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
mLen += 4
|
||||||
|
|
||||||
|
// Copy match (overlapping-safe).
|
||||||
|
if di-offset < 0 {
|
||||||
|
return 0, 1 // offset reaches before dst start
|
||||||
|
}
|
||||||
|
if di+mLen > len(dst) {
|
||||||
|
return 0, 1 // destination overflow
|
||||||
|
}
|
||||||
|
for i := 0; i < mLen; i++ {
|
||||||
|
dst[di+i] = dst[di-offset+i]
|
||||||
|
}
|
||||||
|
di += mLen
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// genLZ4Block generates a random valid LZ4 block that decompresses into
|
||||||
|
// approximately wantSize bytes. The block is always well-formed (ends with
|
||||||
|
// a literals-only sequence).
|
||||||
|
func genLZ4Block(rng *rand.Rand, wantSize int) []byte {
|
||||||
|
var block []byte
|
||||||
|
produced := 0
|
||||||
|
for produced < wantSize {
|
||||||
|
remaining := wantSize - produced
|
||||||
|
|
||||||
|
// Decide: emit a literals+match sequence or the final literals.
|
||||||
|
if remaining <= 8 || rng.Intn(4) == 0 {
|
||||||
|
// Final literals-only sequence.
|
||||||
|
lLen := remaining
|
||||||
|
if lLen > 60 {
|
||||||
|
lLen = 1 + rng.Intn(60)
|
||||||
|
}
|
||||||
|
block = appendToken(block, lLen, 0)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
produced += lLen
|
||||||
|
break
|
||||||
|
}
|
||||||
|
|
||||||
|
// Literals + match.
|
||||||
|
lLen := rng.Intn(min(16, remaining))
|
||||||
|
if produced+lLen == 0 {
|
||||||
|
lLen = 1 // must have at least 1 literal before the first match
|
||||||
|
}
|
||||||
|
mLenRaw := rng.Intn(12) // match length = mLenRaw + 4
|
||||||
|
mLen := mLenRaw + 4
|
||||||
|
if produced+mLen > remaining {
|
||||||
|
mLen = remaining - produced
|
||||||
|
if mLen < 4 {
|
||||||
|
// Not enough room for a match; emit final literals.
|
||||||
|
lLen = remaining
|
||||||
|
block = appendToken(block, lLen, 0)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
mLenRaw = mLen - 4
|
||||||
|
}
|
||||||
|
|
||||||
|
block = appendToken(block, lLen, mLenRaw)
|
||||||
|
for i := 0; i < lLen; i++ {
|
||||||
|
block = append(block, byte(rng.Intn(256)))
|
||||||
|
}
|
||||||
|
produced += lLen
|
||||||
|
|
||||||
|
// Offset: must be <= produced (can't reference before start).
|
||||||
|
maxOff := produced
|
||||||
|
if maxOff > 65535 {
|
||||||
|
maxOff = 65535
|
||||||
|
}
|
||||||
|
offset := 1 + rng.Intn(maxOff)
|
||||||
|
block = append(block, byte(offset), byte(offset>>8))
|
||||||
|
produced += mLen
|
||||||
|
}
|
||||||
|
return block
|
||||||
|
}
|
||||||
|
|
||||||
|
// appendToken appends a token (and extension bytes if needed) for the given
|
||||||
|
// literal and match lengths.
|
||||||
|
func appendToken(block []byte, lLen, mLenRaw int) []byte {
|
||||||
|
lit4 := lLen
|
||||||
|
if lit4 > 15 {
|
||||||
|
lit4 = 15
|
||||||
|
}
|
||||||
|
ml4 := mLenRaw
|
||||||
|
if ml4 > 15 {
|
||||||
|
ml4 = 15
|
||||||
|
}
|
||||||
|
block = append(block, byte(lit4<<4|ml4))
|
||||||
|
// Literal extension bytes.
|
||||||
|
rem := lLen - 15
|
||||||
|
for rem >= 255 {
|
||||||
|
block = append(block, 255)
|
||||||
|
rem -= 255
|
||||||
|
}
|
||||||
|
if lLen >= 15 {
|
||||||
|
block = append(block, byte(rem))
|
||||||
|
}
|
||||||
|
// Match extension bytes.
|
||||||
|
rem = mLenRaw - 15
|
||||||
|
for rem >= 255 {
|
||||||
|
block = append(block, 255)
|
||||||
|
rem -= 255
|
||||||
|
}
|
||||||
|
if mLenRaw >= 15 {
|
||||||
|
block = append(block, byte(rem))
|
||||||
|
}
|
||||||
|
return block
|
||||||
|
}
|
||||||
|
|
||||||
|
func min(a, b int) int {
|
||||||
|
if a < b {
|
||||||
|
return a
|
||||||
|
}
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialLZ4Fuzz drives the JIT-assembled decodeBlockAVX2 with
|
||||||
|
// random valid LZ4 blocks and compares the output bit-for-bit against the
|
||||||
|
// portable Go reference.
|
||||||
|
func TestDifferentialLZ4Fuzz(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
const iterations = 5000
|
||||||
|
rng := rand.New(rand.NewSource(42))
|
||||||
|
|
||||||
|
for i := 0; i < iterations; i++ {
|
||||||
|
wantSize := 1 + rng.Intn(4096)
|
||||||
|
src := genLZ4Block(rng, wantSize)
|
||||||
|
dstSize := wantSize + 64 // generous destination
|
||||||
|
|
||||||
|
// Go reference.
|
||||||
|
goDst := make([]byte, dstSize)
|
||||||
|
goN, goCode := decodeBlockGo(src, goDst)
|
||||||
|
|
||||||
|
// JIT kernel.
|
||||||
|
jitDst := make([]byte, dstSize)
|
||||||
|
jitN, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||||
|
|
||||||
|
if jitCode != goCode {
|
||||||
|
t.Fatalf("iter %d: code mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||||
|
i, jitCode, goCode, len(src))
|
||||||
|
}
|
||||||
|
if jitCode != 0 {
|
||||||
|
continue // both agree it's malformed/zero-offset
|
||||||
|
}
|
||||||
|
if jitN != goN {
|
||||||
|
t.Fatalf("iter %d: n mismatch: JIT=%d, Go=%d (src len=%d)",
|
||||||
|
i, jitN, goN, len(src))
|
||||||
|
}
|
||||||
|
if !bytes.Equal(jitDst[:jitN], goDst[:goN]) {
|
||||||
|
t.Fatalf("iter %d: output mismatch (n=%d, src len=%d)", i, jitN, len(src))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialLZ4Hostile drives the kernel with random garbage to check
|
||||||
|
// that error codes agree with the Go reference (no crashes, same classification).
|
||||||
|
func TestDifferentialLZ4Hostile(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
const iterations = 2000
|
||||||
|
rng := rand.New(rand.NewSource(99))
|
||||||
|
|
||||||
|
for i := 0; i < iterations; i++ {
|
||||||
|
srcLen := rng.Intn(128)
|
||||||
|
src := make([]byte, srcLen)
|
||||||
|
rng.Read(src)
|
||||||
|
dstSize := rng.Intn(512)
|
||||||
|
dst := make([]byte, dstSize)
|
||||||
|
|
||||||
|
// Go reference.
|
||||||
|
goDst := make([]byte, dstSize)
|
||||||
|
copy(goDst, dst)
|
||||||
|
_, goCode := decodeBlockGo(src, goDst)
|
||||||
|
|
||||||
|
// JIT kernel.
|
||||||
|
jitDst := make([]byte, dstSize)
|
||||||
|
copy(jitDst, dst)
|
||||||
|
_, jitCode := callDecodeBlockAVX2(t, k, src, jitDst)
|
||||||
|
|
||||||
|
if jitCode != goCode {
|
||||||
|
t.Fatalf("iter %d: hostile code mismatch: JIT=%d, Go=%d (srcLen=%d, dstSize=%d)",
|
||||||
|
i, jitCode, goCode, srcLen, dstSize)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// callDecodeBlockAVX2Raw is like callDecodeBlockAVX2 but accepts explicit
|
||||||
|
// dst size (for hostile tests where dst may be smaller than the output).
|
||||||
|
func callDecodeBlockAVX2Raw(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||||
|
t.Helper()
|
||||||
|
args := make([]byte, 64)
|
||||||
|
if len(src) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(src)))
|
||||||
|
PutUint64(args, 16, uint64(cap(src)))
|
||||||
|
if len(dst) > 0 {
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 32, uint64(len(dst)))
|
||||||
|
PutUint64(args, 40, uint64(cap(dst)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||||
|
}
|
||||||
|
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Package verify provides the dynamic-analysis substrate for gasm: it
|
||||||
|
// JIT-assembles Plan 9 amd64 kernels into executable memory and calls them
|
||||||
|
// directly, enabling differential testing against portable Go references,
|
||||||
|
// runtime ABI checks and basic-block coverage profiling.
|
||||||
|
//
|
||||||
|
// The execution model is pure Go (stdlib only): machine code is mapped with
|
||||||
|
// syscall.Mmap and invoked through an assembly trampoline that switches to a
|
||||||
|
// prepared ABI0 stack. No cgo, no external toolchain.
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Executable maps a copy of code into a read-execute memory region suitable
|
||||||
|
// for direct invocation. The mapping is anonymous and private; the original
|
||||||
|
// slice is not retained. Call Unmap to release the region.
|
||||||
|
type Executable struct {
|
||||||
|
addr uintptr // base address of the mapping
|
||||||
|
size int
|
||||||
|
mem []byte // the mmap'd slice (for Unmap)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map copies code into a freshly allocated RX region and returns it.
|
||||||
|
// The mapping is PROT_READ|PROT_EXEC; writes are not permitted after the
|
||||||
|
// copy, matching W^X policy.
|
||||||
|
func Map(code []byte) (*Executable, error) {
|
||||||
|
size := len(code)
|
||||||
|
if size == 0 {
|
||||||
|
return nil, fmt.Errorf("verify: cannot map zero-length code")
|
||||||
|
}
|
||||||
|
// Round up to the page size.
|
||||||
|
const pageSize = 4096
|
||||||
|
mapSize := (size + pageSize - 1) &^ (pageSize - 1)
|
||||||
|
|
||||||
|
mem, err := syscall.Mmap(-1, 0, mapSize,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE, syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: mmap: %w", err)
|
||||||
|
}
|
||||||
|
copy(mem, code)
|
||||||
|
|
||||||
|
// Remove write permission (W^X).
|
||||||
|
if err := syscall.Mprotect(mem, syscall.PROT_READ|syscall.PROT_EXEC); err != nil {
|
||||||
|
syscall.Munmap(mem)
|
||||||
|
return nil, fmt.Errorf("verify: mprotect: %w", err)
|
||||||
|
}
|
||||||
|
return &Executable{
|
||||||
|
addr: uintptr(unsafe.Pointer(&mem[0])),
|
||||||
|
size: size,
|
||||||
|
mem: mem,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unmap releases the executable region.
|
||||||
|
func (e *Executable) Unmap() {
|
||||||
|
if e.mem != nil {
|
||||||
|
syscall.Munmap(e.mem)
|
||||||
|
e.mem = nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// FuncAddr returns the absolute address of a function at the given offset
|
||||||
|
// within the mapped image.
|
||||||
|
func (e *Executable) FuncAddr(offset int) uintptr {
|
||||||
|
return e.addr + uintptr(offset)
|
||||||
|
}
|
||||||
|
|
||||||
|
// PutUint64 writes v into buf at byte offset off (little-endian).
|
||||||
|
func PutUint64(buf []byte, off int, v uint64) {
|
||||||
|
binary.LittleEndian.PutUint64(buf[off:off+8], v)
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetUint64 reads a little-endian uint64 from buf at byte offset off.
|
||||||
|
func GetUint64(buf []byte, off int) uint64 {
|
||||||
|
return binary.LittleEndian.Uint64(buf[off : off+8])
|
||||||
|
}
|
||||||
|
|
||||||
|
// PutPtr writes a pointer value into buf at byte offset off.
|
||||||
|
func PutPtr(buf []byte, off int, p unsafe.Pointer) {
|
||||||
|
binary.LittleEndian.PutUint64(buf[off:off+8], uint64(uintptr(p)))
|
||||||
|
}
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
func loadBasic(t *testing.T) *Kernel {
|
||||||
|
t.Helper()
|
||||||
|
k, err := Load("../testdata/verify/basic_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
return k
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITAdd(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
a, b, want int64
|
||||||
|
}{
|
||||||
|
{0, 0, 0},
|
||||||
|
{1, 2, 3},
|
||||||
|
{-1, 1, 0},
|
||||||
|
{1 << 62, 1 << 62, -9223372036854775808}, // overflow wraps (MinInt64)
|
||||||
|
{-100, -200, -300},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
args := make([]byte, 24)
|
||||||
|
PutUint64(args, 0, uint64(tt.a))
|
||||||
|
PutUint64(args, 8, uint64(tt.b))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("add", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(add, %d, %d): %v", tt.a, tt.b, err)
|
||||||
|
}
|
||||||
|
got := int64(GetUint64(out, 16))
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("add(%d, %d) = %d, want %d", tt.a, tt.b, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITSum(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
data []int64
|
||||||
|
want int64
|
||||||
|
}{
|
||||||
|
{nil, 0},
|
||||||
|
{[]int64{1}, 1},
|
||||||
|
{[]int64{1, 2, 3, 4, 5}, 15},
|
||||||
|
{[]int64{-10, 20, -30, 40}, 20},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
args := make([]byte, 32)
|
||||||
|
if len(tt.data) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&tt.data[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(tt.data)))
|
||||||
|
PutUint64(args, 16, uint64(cap(tt.data)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("sum", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(sum, %v): %v", tt.data, err)
|
||||||
|
}
|
||||||
|
got := int64(GetUint64(out, 24))
|
||||||
|
if got != tt.want {
|
||||||
|
t.Errorf("sum(%v) = %d, want %d", tt.data, got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJITWideCopy(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
n int
|
||||||
|
}{
|
||||||
|
{"empty", 0},
|
||||||
|
{"tiny", 7},
|
||||||
|
{"exact32", 32},
|
||||||
|
{"overlap_range", 48},
|
||||||
|
{"exact64", 64},
|
||||||
|
{"unaligned", 45},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
src := make([]byte, tt.n)
|
||||||
|
for i := range src {
|
||||||
|
src[i] = byte(i * 7)
|
||||||
|
}
|
||||||
|
dst := make([]byte, tt.n)
|
||||||
|
|
||||||
|
args := make([]byte, 48)
|
||||||
|
if tt.n > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(tt.n)) // dst_len
|
||||||
|
PutUint64(args, 16, uint64(tt.n)) // dst_cap
|
||||||
|
PutUint64(args, 32, uint64(tt.n)) // src_len
|
||||||
|
PutUint64(args, 40, uint64(tt.n)) // src_cap
|
||||||
|
|
||||||
|
_, err := k.CallFunc("wideCopy", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(wideCopy): %v", err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(dst, src) {
|
||||||
|
t.Errorf("wideCopy: dst ≠ src\n got %x\n want %x", dst, src)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelFuncNames(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
names := k.FuncNames()
|
||||||
|
want := []string{"add", "sum", "wideCopy"}
|
||||||
|
if len(names) != len(want) {
|
||||||
|
t.Fatalf("FuncNames() = %v, want %v", names, want)
|
||||||
|
}
|
||||||
|
for i, n := range names {
|
||||||
|
if n != want[i] {
|
||||||
|
t.Errorf("FuncNames()[%d] = %q, want %q", i, n, want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelFuncNotFound(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
_, err := k.CallFunc("nonexistent", make([]byte, 8))
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for nonexistent function")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestKernelArgTooSmall(t *testing.T) {
|
||||||
|
k := loadBasic(t)
|
||||||
|
_, err := k.CallFunc("add", make([]byte, 8)) // needs 24
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for too-small arg block")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMapZeroLength(t *testing.T) {
|
||||||
|
_, err := Map(nil)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for zero-length code")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,160 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
"unsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// lz4KernelPath is the sibling repository's AVX2 kernel, used for
|
||||||
|
// integration testing. The test is skipped when the file is absent
|
||||||
|
// (e.g. in CI without the sibling checkout).
|
||||||
|
const lz4KernelPath = "../../go-libraries/go-lz4/avx2_amd64.s"
|
||||||
|
|
||||||
|
func loadLZ4Kernel(t *testing.T) *Kernel {
|
||||||
|
t.Helper()
|
||||||
|
if _, err := os.Stat(lz4KernelPath); err != nil {
|
||||||
|
t.Skipf("sibling kernel not available: %v", err)
|
||||||
|
}
|
||||||
|
k, err := Load(lz4KernelPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load(%s): %v", lz4KernelPath, err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
return k
|
||||||
|
}
|
||||||
|
|
||||||
|
// callDecodeBlockAVX2 invokes the JIT-assembled decodeBlockAVX2 with the
|
||||||
|
// given src and dst buffers, returning (n, code).
|
||||||
|
func callDecodeBlockAVX2(t *testing.T, k *Kernel, src, dst []byte) (int, int) {
|
||||||
|
t.Helper()
|
||||||
|
args := make([]byte, 64)
|
||||||
|
if len(src) > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(len(src)))
|
||||||
|
PutUint64(args, 16, uint64(cap(src)))
|
||||||
|
if len(dst) > 0 {
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&dst[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 32, uint64(len(dst)))
|
||||||
|
PutUint64(args, 40, uint64(cap(dst)))
|
||||||
|
|
||||||
|
out, err := k.CallFunc("decodeBlockAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CallFunc(decodeBlockAVX2): %v", err)
|
||||||
|
}
|
||||||
|
return int(GetUint64(out, 48)), int(GetUint64(out, 56))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLZ4DecodeKnownAnswers(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
src []byte
|
||||||
|
dstSize int
|
||||||
|
wantDst []byte
|
||||||
|
wantN int
|
||||||
|
wantCode int
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "literals_only",
|
||||||
|
src: []byte{0x50, 'H', 'e', 'l', 'l', 'o'},
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: []byte("Hello"),
|
||||||
|
wantN: 5,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "literals_and_match",
|
||||||
|
src: []byte{0x54, 'A', 'A', 'A', 'A', 'A', 0x05, 0x00, 0x30, 'B', 'B', 'B'},
|
||||||
|
dstSize: 32,
|
||||||
|
wantDst: []byte("AAAAAAAAAAAAABBB"),
|
||||||
|
wantN: 16,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "overlapping_match",
|
||||||
|
// 1 literal 'X', then match offset=1 length=4+4=8 → "XXXXXXXXX",
|
||||||
|
// then final 1 literal 'Y'.
|
||||||
|
src: []byte{0x14, 'X', 0x01, 0x00, 0x10, 'Y'},
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: []byte("XXXXXXXXXY"),
|
||||||
|
wantN: 10,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "malformed_truncated",
|
||||||
|
src: []byte{0x50, 'H', 'e'}, // claims 5 literals, has 2
|
||||||
|
dstSize: 16,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 1,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "zero_offset",
|
||||||
|
src: []byte{0x14, 'X', 0x00, 0x00},
|
||||||
|
dstSize: 16,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 2,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty_token",
|
||||||
|
src: []byte{0x00}, // 0 literals, end of block
|
||||||
|
dstSize: 16,
|
||||||
|
wantDst: nil,
|
||||||
|
wantN: 0,
|
||||||
|
wantCode: 0,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
dst := make([]byte, tt.dstSize)
|
||||||
|
n, code := callDecodeBlockAVX2(t, k, tt.src, dst)
|
||||||
|
if n != tt.wantN || code != tt.wantCode {
|
||||||
|
t.Fatalf("decodeBlockAVX2: got (n=%d, code=%d), want (n=%d, code=%d)",
|
||||||
|
n, code, tt.wantN, tt.wantCode)
|
||||||
|
}
|
||||||
|
if tt.wantCode == 0 && tt.wantDst != nil {
|
||||||
|
if !bytes.Equal(dst[:n], tt.wantDst) {
|
||||||
|
t.Errorf("output mismatch:\n got %q\n want %q", dst[:n], tt.wantDst)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLZ4WideCopyAVX2(t *testing.T) {
|
||||||
|
k := loadLZ4Kernel(t)
|
||||||
|
|
||||||
|
sizes := []int{0, 1, 15, 16, 31, 32, 33, 63, 64, 100, 256, 1024}
|
||||||
|
for _, n := range sizes {
|
||||||
|
src := make([]byte, n)
|
||||||
|
for i := range src {
|
||||||
|
src[i] = byte(i*13 + 7)
|
||||||
|
}
|
||||||
|
dst := make([]byte, n)
|
||||||
|
|
||||||
|
args := make([]byte, 48)
|
||||||
|
if n > 0 {
|
||||||
|
PutPtr(args, 0, unsafe.Pointer(&dst[0]))
|
||||||
|
PutPtr(args, 24, unsafe.Pointer(&src[0]))
|
||||||
|
}
|
||||||
|
PutUint64(args, 8, uint64(n))
|
||||||
|
PutUint64(args, 16, uint64(n))
|
||||||
|
PutUint64(args, 32, uint64(n))
|
||||||
|
PutUint64(args, 40, uint64(n))
|
||||||
|
|
||||||
|
_, err := k.CallFunc("wideCopyAVX2", args)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("wideCopyAVX2(n=%d): %v", n, err)
|
||||||
|
}
|
||||||
|
if !bytes.Equal(dst, src) {
|
||||||
|
t.Errorf("wideCopyAVX2(n=%d): output mismatch", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// ABI0 JIT trampoline. enterJIT switches from the Go stack to a prepared
|
||||||
|
// stack and jumps to the assembled function; when the function RETs, control
|
||||||
|
// lands in leaveJIT, which restores the Go stack and returns to the Go caller.
|
||||||
|
//
|
||||||
|
// The prepared stack must begin with the address of leaveJIT (the return
|
||||||
|
// address the JIT function will pop), followed by the function's ABI0
|
||||||
|
// argument area.
|
||||||
|
//
|
||||||
|
// Single-threaded: savedSP is a package global, so only one JIT call may be
|
||||||
|
// in flight at a time. gasm verify runs sequentially.
|
||||||
|
|
||||||
|
// func enterJIT(fn uintptr, stack uintptr)
|
||||||
|
// Switches to the prepared stack and jumps to fn. Does not return normally;
|
||||||
|
// the JIT function's RET transfers control to leaveJIT.
|
||||||
|
TEXT ·enterJIT(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ fn+0(FP), AX // target function address (before SP switch)
|
||||||
|
MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer
|
||||||
|
MOVQ stack+8(FP), SP // switch to the prepared stack
|
||||||
|
JMP AX
|
||||||
|
|
||||||
|
// func leaveJIT()
|
||||||
|
// Restores the Go stack pointer and returns to enterJIT's caller.
|
||||||
|
TEXT ·leaveJIT(SB), NOSPLIT, $0-0
|
||||||
|
MOVQ ·savedSP(SB), SP
|
||||||
|
RET
|
||||||
@@ -0,0 +1,106 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Kernel is a JIT-loaded assembly image ready for direct invocation.
|
||||||
|
// It wraps an executable memory mapping and the function layout metadata
|
||||||
|
// needed to marshal ABI0 calls.
|
||||||
|
type Kernel struct {
|
||||||
|
exec *Executable
|
||||||
|
img *asm.Image
|
||||||
|
funcs map[string]int // function name → index into img.Funcs
|
||||||
|
}
|
||||||
|
|
||||||
|
// Load parses, assembles and maps a .s file into executable memory.
|
||||||
|
// The returned Kernel is ready for Call. The caller must call Close to
|
||||||
|
// release the mapping.
|
||||||
|
func Load(path string) (*Kernel, error) {
|
||||||
|
src, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: %w", err)
|
||||||
|
}
|
||||||
|
return LoadSource(path, string(src))
|
||||||
|
}
|
||||||
|
|
||||||
|
// LoadSource parses, assembles and maps assembly source into executable memory.
|
||||||
|
func LoadSource(filename, src string) (*Kernel, error) {
|
||||||
|
file, errs := parser.Parse(filename, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil, fmt.Errorf("verify: parse %s: %v", filename, errs[0])
|
||||||
|
}
|
||||||
|
return LoadAST(file)
|
||||||
|
}
|
||||||
|
|
||||||
|
// LoadAST assembles a parsed AST file and maps the result into executable
|
||||||
|
// memory.
|
||||||
|
func LoadAST(file *ast.File) (*Kernel, error) {
|
||||||
|
img, err := asm.AssembleFile(file)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("verify: assemble: %w", err)
|
||||||
|
}
|
||||||
|
if len(img.Externals) > 0 {
|
||||||
|
return nil, fmt.Errorf("verify: unresolved external symbols: %v", img.Externals)
|
||||||
|
}
|
||||||
|
code := img.Bytes()
|
||||||
|
exec, err := Map(code)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
funcs := make(map[string]int, len(img.Funcs))
|
||||||
|
for i, f := range img.Funcs {
|
||||||
|
funcs[f.Name] = i
|
||||||
|
}
|
||||||
|
return &Kernel{exec: exec, img: img, funcs: funcs}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Func returns the layout metadata for the named function.
|
||||||
|
func (k *Kernel) Func(name string) (asm.FuncLayout, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return asm.FuncLayout{}, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
return k.img.Funcs[idx], nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// FuncNames returns the names of all functions in the kernel, in source order.
|
||||||
|
func (k *Kernel) FuncNames() []string {
|
||||||
|
names := make([]string, len(k.img.Funcs))
|
||||||
|
for i, f := range k.img.Funcs {
|
||||||
|
names[i] = f.Name
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
// CallFunc invokes the named function with the given ABI0 argument block.
|
||||||
|
// The arg block is the raw bytes of the function's argument/result area
|
||||||
|
// (as declared by the TEXT $frame-args suffix). Returns the arg block
|
||||||
|
// after the call (with any results written back by the function).
|
||||||
|
func (k *Kernel) CallFunc(name string, args []byte) ([]byte, error) {
|
||||||
|
idx, ok := k.funcs[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("verify: function %q not found", name)
|
||||||
|
}
|
||||||
|
fl := k.img.Funcs[idx]
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
return nil, fmt.Errorf("verify: %s: arg block too small: got %d, need %d", name, len(args), fl.Args)
|
||||||
|
}
|
||||||
|
fnAddr := k.exec.FuncAddr(fl.Offset)
|
||||||
|
return Call(fnAddr, args)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Close releases the executable mapping.
|
||||||
|
func (k *Kernel) Close() {
|
||||||
|
if k.exec != nil {
|
||||||
|
k.exec.Unmap()
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user