From 187e4856d3e38d4c8a3096f5b3da41efd800edbf Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 00:38:24 +0200 Subject: [PATCH] feat(amd64): encode the mixed-width extend family and PMOVMSKB Assisted-by: GLM 5.3 --- asm/encodable.go | 7 +++++ asm/encode.go | 8 +++++- asm/encode_test.go | 12 ++++++++ asm/instrs.go | 53 ++++++++++++++++++++++++++--------- testdata/verify/widen_amd64.s | 39 ++++++++++++++++++++++++++ 5 files changed, 105 insertions(+), 14 deletions(-) create mode 100644 testdata/verify/widen_amd64.s diff --git a/asm/encodable.go b/asm/encodable.go index 903f5df..14694aa 100644 --- a/asm/encodable.go +++ b/asm/encodable.go @@ -86,9 +86,16 @@ func Encodable(mnemonic string) bool { "BSWAP", "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2", "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX", + "MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX", "CVTSL2SD", "CVTSQ2SD", "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS": return true } + // Full-name dispatches the size split would eat (a trailing width + // letter that is part of the mnemonic). + switch upper { + case "PMOVMSKB": + return true + } return false } diff --git a/asm/encode.go b/asm/encode.go index 2a6e487..1ee3861 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -101,6 +101,11 @@ func (e *enc) encode(mnem string, ops []Operand) error { if m, ok := sseBinTable[base]; ok { return e.encodeSSEBin(m, ops) } + // PMOVMSKB ends in a width letter the size split would eat, so it + // dispatches on the full name like the packed binaries above. + if upper == "PMOVMSKB" { + return e.encodePmovmskb(upper, ops) + } switch base { case "MOV": return e.encodeMov(ops, size) @@ -126,7 +131,8 @@ func (e *enc) encode(mnem string, ops []Operand) error { return e.encodeBswap(ops, size) case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2": return e.encodePrefetch(base, ops) - case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX": + case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX", + "MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX": return e.encodeMovExtend(base, ops) case "CVTSL2SD", "CVTSQ2SD": return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops) diff --git a/asm/encode_test.go b/asm/encode_test.go index 214c978..a77208f 100644 --- a/asm/encode_test.go +++ b/asm/encode_test.go @@ -357,6 +357,18 @@ func TestScalarGroundTruth(t *testing.T) { {"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"}, {"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"}, {"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"}, + // The width pairs the toolchain accepts and GOROOT uses; bytes + // pinned from go tool asm (see testdata/verify/widen_amd64.s). + {"MOVBWZX (BX),R11W", "MOVBWZX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fb61b", "MOVZX"}, + {"MOVBWSX (BX),R11W", "MOVBWSX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fbe1b", "MOVSX"}, + {"MOVBLSX (BX),AX", "MOVBLSX", []Operand{Ptr(BX, 0, 1), AX}, "0fbe03", "MOVSX"}, + {"MOVBQSX (BX),R8", "MOVBQSX", []Operand{Ptr(BX, 0, 1), r8}, "4c0fbe03", "MOVSX"}, + {"MOVWQSX (BX),R9", "MOVWQSX", []Operand{Ptr(BX, 0, 2), r9}, "4c0fbf0b", "MOVSX"}, + // A long to quad zero-extend is a plain 32-bit move. + {"MOVLQZX (BX),DX", "MOVLQZX", []Operand{Ptr(BX, 0, 4), DX}, "8b13", "MOV"}, + {"MOVLQZX AX,DX", "MOVLQZX", []Operand{AX, DX}, "8bd0", "MOV"}, + {"PMOVMSKB X1,AX", "PMOVMSKB", []Operand{vreg(t, "X1"), AX}, "660fd7c1", "PMOVMSKB"}, + {"PMOVMSKB X11,CX", "PMOVMSKB", []Operand{vreg(t, "X11"), CX}, "66410fd7cb", "PMOVMSKB"}, {"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"}, {"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"}, {"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"}, diff --git a/asm/instrs.go b/asm/instrs.go index 5072724..9718021 100644 --- a/asm/instrs.go +++ b/asm/instrs.go @@ -838,15 +838,24 @@ func (e *enc) encodeBswap(ops []Operand, size int) error { // width. The source is narrower than the destination, so the plain size-suffix // convention does not apply to these names. var movExtendOp = map[string]struct { - op []byte - dst64 bool + op []byte + dstSize int }{ - "MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend - "MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend - "MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend - "MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend - "MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend - "MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD) + "MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend + "MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend + "MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend + "MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend + "MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend + "MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD) + "MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend + "MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend + "MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend + "MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend + "MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend + // A long → quad zero-extend is a plain 32-bit move: every 32-bit + // operation zero-extends its result into the full register, so the + // toolchain lowers MOVLQZX to the plain MOVL encoding. + "MOVLQZX": {[]byte{0x8B}, 4}, } // encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider @@ -860,12 +869,30 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error { if !ok { return fmt.Errorf("%s destination must be a register", base) } - size := 4 - if spec.dst64 { - size = 8 + i := newInstr(spec.dstSize, spec.op) + if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil { + return err } - i := newInstr(size, spec.op) - if err := setRM(i, dstReg, ops[0], size); err != nil { + return e.emit(i) +} + +// encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the +// XMM source's sign bytes pack into a GP destination, 66 0F D7 /r. +func (e *enc) encodePmovmskb(base string, ops []Operand) error { + if len(ops) != 2 { + return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) + } + srcReg, srcVec := vecReg(ops[0]) + if !srcVec { + return fmt.Errorf("%s source must be an XMM register", base) + } + dstReg, ok := ops[1].(Reg) + if !ok { + return fmt.Errorf("%s destination must be a register", base) + } + i := newInstr(4, []byte{0x0F, 0xD7}) + i.prefix = 0x66 + if err := setRM(i, dstReg, srcReg, 4); err != nil { return err } return e.emit(i) diff --git a/testdata/verify/widen_amd64.s b/testdata/verify/widen_amd64.s new file mode 100644 index 0000000..61b6932 --- /dev/null +++ b/testdata/verify/widen_amd64.s @@ -0,0 +1,39 @@ +// Mixed-width sign- and zero-extending moves plus PMOVMSKB, the spellings +// GOROOT's runtime and bytealg kernels use. Every result is folded back so +// no instruction is dead. + +#include "textflag.h" + +// func widen(p *byte) uint64 +TEXT ·widen(SB), NOSPLIT, $0-16 + MOVBQZX 0(DI), AX + MOVWQZX 2(DI), CX + ADDQ CX, AX + MOVLQZX 4(DI), DX + ADDQ DX, AX + MOVBQSX 8(DI), R8 + ADDQ R8, AX + MOVWQSX 12(DI), R9 + ADDQ R9, AX + MOVBLSX 16(DI), R10 + ADDL R10, AX + MOVLQSX 20(DI), R11 + ADDQ R11, AX + MOVQ AX, ret+8(FP) + RET + +// func widenw(p *byte) int32 +TEXT ·widenw(SB), NOSPLIT, $0-16 + MOVBWZX 0(DI), AX + MOVBWSX 1(DI), CX + ADDL CX, AX + MOVLQZX AX, DX + MOVL DX, ret+8(FP) + RET + +// func mask(x *XMM) int +TEXT ·mask(SB), NOSPLIT, $0-16 + MOVOU 0(DI), X1 + PMOVMSKB X1, AX + MOVQ AX, ret+8(FP) + RET