diff --git a/testdata/add_riscv64.s b/testdata/add_riscv64.s index b77b64e..e2c4075 100644 --- a/testdata/add_riscv64.s +++ b/testdata/add_riscv64.s @@ -5,8 +5,8 @@ // func add(a, b int64) int64 TEXT ·add(SB), NOSPLIT, $0-24 - MOV a+0(FP), X10 - MOV b+8(FP), X11 - ADD X11, X10, X10 - MOV X10, ret+16(FP) - RET + MOV a+0(FP), X10 + MOV b+8(FP), X11 + ADD X11, X10, X10 + MOV X10, ret+16(FP) + RET diff --git a/testdata/amo_fp_riscv64.s b/testdata/amo_fp_riscv64.s index 2b52d1a..fdce934 100644 --- a/testdata/amo_fp_riscv64.s +++ b/testdata/amo_fp_riscv64.s @@ -5,16 +5,16 @@ // func atomicAdd(ptr *int64, val int64) int64 TEXT ·atomicAdd(SB), NOSPLIT, $0-24 - MOV a+0(FP), X10 - MOV b+8(FP), X11 - AMOADDD X11, (X10), X12 - MOV X12, ret+16(FP) - RET + MOV a+0(FP), X10 + MOV b+8(FP), X11 + AMOADDD X11, (X10), X12 + MOV X12, ret+16(FP) + RET // func fpAdd(a, b float64) float64 TEXT ·fpAdd(SB), NOSPLIT, $0-24 - FLD a+0(FP), F10 - FLD b+8(FP), F11 - FADDD F10, F11, F12 - FSD F12, ret+16(FP) - RET + FLD a+0(FP), F10 + FLD b+8(FP), F11 + FADDD F10, F11, F12 + FSD F12, ret+16(FP) + RET diff --git a/testdata/csr_riscv64.s b/testdata/csr_riscv64.s index 0bb6643..fc1f345 100644 --- a/testdata/csr_riscv64.s +++ b/testdata/csr_riscv64.s @@ -5,22 +5,22 @@ // func readCSR(csr int64) int64 TEXT ·readCSR(SB), NOSPLIT, $0-16 - MOV a+0(FP), X10 - CSRRS $0x300, X0, X11 - MOV X11, ret+8(FP) - RET + MOV a+0(FP), X10 + CSRRS $0x300, X0, X11 + MOV X11, ret+8(FP) + RET // func setCSRBit(csr, bit int64) int64 TEXT ·setCSRBit(SB), NOSPLIT, $0-24 - MOV a+0(FP), X10 - MOV b+8(FP), X11 - CSRRS $0x304, X11, X12 - MOV X12, ret+16(FP) - RET + MOV a+0(FP), X10 + MOV b+8(FP), X11 + CSRRS $0x304, X11, X12 + MOV X12, ret+16(FP) + RET // func writeCSR(val int64) int64 TEXT ·writeCSR(SB), NOSPLIT, $0-16 - MOV a+0(FP), X10 - CSRRW $0x305, X10, X11 - MOV X11, ret+8(FP) - RET + MOV a+0(FP), X10 + CSRRW $0x305, X10, X11 + MOV X11, ret+8(FP) + RET diff --git a/testdata/fma_riscv64.s b/testdata/fma_riscv64.s index 2361496..f7cd530 100644 --- a/testdata/fma_riscv64.s +++ b/testdata/fma_riscv64.s @@ -5,18 +5,18 @@ // func fma(a, b, c float64) float64 TEXT ·fma(SB), NOSPLIT, $0-32 - FLD a+0(FP), F10 - FLD b+8(FP), F11 - FLD c+16(FP), F12 - FMADDD F10, F11, F12, F13 - FSD F13, ret+24(FP) - RET + FLD a+0(FP), F10 + FLD b+8(FP), F11 + FLD c+16(FP), F12 + FMADDD F10, F11, F12, F13 + FSD F13, ret+24(FP) + RET // func fms(a, b, c float64) float64 TEXT ·fms(SB), NOSPLIT, $0-32 - FLD a+0(FP), F10 - FLD b+8(FP), F11 - FLD c+16(FP), F12 - FMSUBD F10, F11, F12, F13 - FSD F13, ret+24(FP) - RET + FLD a+0(FP), F10 + FLD b+8(FP), F11 + FLD c+16(FP), F12 + FMSUBD F10, F11, F12, F13 + FSD F13, ret+24(FP) + RET diff --git a/testdata/lrsc_cvt_riscv64.s b/testdata/lrsc_cvt_riscv64.s index 3f53b37..900a0a4 100644 --- a/testdata/lrsc_cvt_riscv64.s +++ b/testdata/lrsc_cvt_riscv64.s @@ -6,31 +6,32 @@ // func casLoop(ptr *int64, old, new int64) bool TEXT ·casLoop(SB), NOSPLIT, $0-32 cas_retry: - MOV a+0(FP), X10 - LRD (X10), X11 - MOV b+8(FP), X12 - BNE X11, X12, cas_fail - MOV c+16(FP), X13 - SCD X13, (X10), X14 - BNE X14, X0, cas_retry - ADDI X0, $1, X15 - MOV X15, ret+24(FP) - RET + MOV a+0(FP), X10 + LRD (X10), X11 + MOV b+8(FP), X12 + BNE X11, X12, cas_fail + MOV c+16(FP), X13 + SCD X13, (X10), X14 + BNE X14, X0, cas_retry + ADDI X0, $1, X15 + MOV X15, ret+24(FP) + RET + cas_fail: - MOV X0, ret+24(FP) - RET + MOV X0, ret+24(FP) + RET // func intToFloat(x int64) float64 TEXT ·intToFloat(SB), NOSPLIT, $0-16 - MOV a+0(FP), X10 - FCVTDL X10, F10 - FSD F10, ret+8(FP) - RET + MOV a+0(FP), X10 + FCVTDL X10, F10 + FSD F10, ret+8(FP) + RET // func compare(a, b float64) bool TEXT ·compare(SB), NOSPLIT, $0-24 - FLD a+0(FP), F10 - FLD b+8(FP), F11 - FLTD F10, F11, X10 - MOV X10, ret+16(FP) - RET + FLD a+0(FP), F10 + FLD b+8(FP), F11 + FLTD F10, F11, X10 + MOV X10, ret+16(FP) + RET diff --git a/testdata/sample_amd64.s b/testdata/sample_amd64.s index 90e8a2c..43cdca4 100644 --- a/testdata/sample_amd64.s +++ b/testdata/sample_amd64.s @@ -15,43 +15,44 @@ DATA mask24<>+4(SB)/4, $0x80050403 // func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool) TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65 - MOVQ swin_base+0(FP), SI - MOVQ dstP_base+24(FP), DI - MOVQ dstP_len+32(FP), BX - MOVQ hist+48(FP), R13 + MOVQ swin_base+0(FP), SI + MOVQ dstP_base+24(FP), DI + MOVQ dstP_len+32(FP), BX + MOVQ hist+48(FP), R13 - VPCMPEQD Y0, Y0, Y0 - VPSLLD $31, Y0, Y0 + VPCMPEQD Y0, Y0, Y0 + VPSLLD $31, Y0, Y0 - LEAQ (SI)(BX*4), R9 - MOVQ BX, R10 - ANDQ $-8, R10 + LEAQ (SI)(BX*4), R9 + MOVQ BX, R10 + ANDQ $-8, R10 vec1: - CMPQ SI, R10 - JGE vec1done - VMOVDQU (SI), Y1 - VMOVDQU 4(SI), Y2 - VPSUBD Y1, Y2, Y3 - ADDQ $32, SI - JMP vec1 + CMPQ SI, R10 + JGE vec1done + VMOVDQU (SI), Y1 + VMOVDQU 4(SI), Y2 + VPSUBD Y1, Y2, Y3 + ADDQ $32, SI + JMP vec1 + vec1done: - MOVQ AX, partSum+56(FP) - MOVB AL, overflow+64(FP) + MOVQ AX, partSum+56(FP) + MOVB AL, overflow+64(FP) VZEROUPPER RET // func decodeFixedO1AVX512(samples []int32, residual []int32) TEXT ·decodeFixedO1AVX512(SB), NOSPLIT, $0-48 - MOVQ samples_base+0(FP), SI - MOVQ residual_base+16(FP), DI + MOVQ samples_base+0(FP), SI + MOVQ residual_base+16(FP), DI VPBROADCASTD AX, Z15 - VMOVDQU32 (DI)(AX*1), Z0 - VALIGND $15, Z9, Z0, Z1 - VFMADD231PD Z14, Z12, Z10 - VPCMPEQD Z0, Z3, K1 - KTESTW K1, K1 - VPSRAQ X31, Z8, Z8 - VMOVDQU32 Z0, 4(SI)(AX*1) + VMOVDQU32 (DI)(AX*1), Z0 + VALIGND $15, Z9, Z0, Z1 + VFMADD231PD Z14, Z12, Z10 + VPCMPEQD Z0, Z3, K1 + KTESTW K1, K1 + VPSRAQ X31, Z8, Z8 + VMOVDQU32 Z0, 4(SI)(AX*1) RET diff --git a/testdata/verify/basic_amd64.s b/testdata/verify/basic_amd64.s index 455e424..0384079 100644 --- a/testdata/verify/basic_amd64.s +++ b/testdata/verify/basic_amd64.s @@ -13,55 +13,55 @@ TEXT ·add(SB), NOSPLIT, $0-24 // func sum(data []int64) int64 // Sums all elements of the slice. TEXT ·sum(SB), NOSPLIT, $0-32 - MOVQ data_base+0(FP), SI - MOVQ data_len+8(FP), CX - XORQ AX, AX + MOVQ data_base+0(FP), SI + MOVQ data_len+8(FP), CX + XORQ AX, AX TESTQ CX, CX - JZ sum_done + JZ sum_done sum_loop: - ADDQ (SI), AX - ADDQ $8, SI - DECQ CX - JNZ sum_loop + ADDQ (SI), AX + ADDQ $8, SI + DECQ CX + JNZ sum_loop sum_done: - MOVQ AX, ret+24(FP) + MOVQ AX, ret+24(FP) RET // func wideCopy(dst, src []byte) // Non-overlapping copy of min(len(dst), len(src)) bytes using 32-byte moves. TEXT ·wideCopy(SB), NOSPLIT, $0-48 - MOVQ dst_base+0(FP), DI - MOVQ dst_len+8(FP), BX - MOVQ src_base+24(FP), SI - MOVQ src_len+32(FP), R8 - CMPQ BX, R8 - JLE wc_have_n - MOVQ R8, BX + MOVQ dst_base+0(FP), DI + MOVQ dst_len+8(FP), BX + MOVQ src_base+24(FP), SI + MOVQ src_len+32(FP), R8 + CMPQ BX, R8 + JLE wc_have_n + MOVQ R8, BX wc_have_n: - CMPQ BX, $32 - JB wc_small + CMPQ BX, $32 + JB wc_small - VMOVDQU (SI), Y0 - VMOVDQU Y0, (DI) - VMOVDQU -32(SI)(BX*1), Y0 - VMOVDQU Y0, -32(DI)(BX*1) + VMOVDQU (SI), Y0 + VMOVDQU Y0, (DI) + VMOVDQU -32(SI)(BX*1), Y0 + VMOVDQU Y0,-32(DI)(BX*1) VZEROUPPER RET wc_small: - TESTQ BX, BX - JZ wc_done + TESTQ BX, BX + JZ wc_done wc_byte: - MOVB (SI), R8B - MOVB R8B, (DI) - INCQ SI - INCQ DI - DECQ BX - JNZ wc_byte + MOVB (SI), R8B + MOVB R8B, (DI) + INCQ SI + INCQ DI + DECQ BX + JNZ wc_byte wc_done: RET diff --git a/testdata/verify/basic_arm64.s b/testdata/verify/basic_arm64.s index ef49f9c..0f5d2b5 100644 --- a/testdata/verify/basic_arm64.s +++ b/testdata/verify/basic_arm64.s @@ -5,53 +5,56 @@ // add returns a + b. TEXT ·add(SB), NOSPLIT, $0-24 - MOVD a+0(FP), R4 - MOVD b+8(FP), R5 - ADD R5, R4, R4 - MOVD R4, ret+16(FP) + MOVD a+0(FP), R4 + MOVD b+8(FP), R5 + ADD R5, R4, R4 + MOVD R4, ret+16(FP) RET // arith exercises the register-register integer set. TEXT ·arith(SB), NOSPLIT, $0-0 - ADD R4, R5, R6 - SUB R7, R8, R9 - AND R10, R11, R12 - ORR R12, R13, R14 - EOR R14, R15, R16 - CMP R16, R17 - ADD R4, R5 - SUB R6, R7 + ADD R4, R5, R6 + SUB R7, R8, R9 + AND R10, R11, R12 + ORR R12, R13, R14 + EOR R14, R15, R16 + CMP R16, R17 + ADD R4, R5 + SUB R6, R7 RET // branch exercises conditional and unconditional control flow. TEXT ·branch(SB), NOSPLIT, $0-0 - BEQ done - BNE skip - BGE done - BLT done - BGT done - BLE done + BEQ done + BNE skip + BGE done + BLT done + BGT done + BLE done + skip: - B loop + B loop + loop: - ADD R4, R5 + ADD R4, R5 RET + done: RET // mov exercises the MOV pseudo-instruction. TEXT ·mov(SB), NOSPLIT, $0-16 - MOVD $0, R4 - MOVD $1, R5 - MOVD $42, R6 - MOVD a+0(FP), R7 - MOVD R7, ret+0(FP) - MOVW $100, R8 + MOVD $0, R4 + MOVD $1, R5 + MOVD $42, R6 + MOVD a+0(FP), R7 + MOVD R7, ret+0(FP) + MOVW $100, R8 RET // frame exercises the prologue/epilogue of a function with a real frame. TEXT ·frame(SB), NOSPLIT, $32-8 - MOVD arg+0(FP), R4 - ADD $1, R4, R4 - MOVD R4, ret+0(FP) + MOVD arg+0(FP), R4 + ADD $1, R4, R4 + MOVD R4, ret+0(FP) RET diff --git a/testdata/verify/basic_loong64.s b/testdata/verify/basic_loong64.s index a6bf0b8..fc3488d 100644 --- a/testdata/verify/basic_loong64.s +++ b/testdata/verify/basic_loong64.s @@ -5,50 +5,53 @@ // add returns a + b. TEXT ·add(SB), NOSPLIT, $0-24 - MOVV a+0(FP), R4 - MOVV b+8(FP), R5 - ADDV R5, R4, R4 - MOVV R4, ret+16(FP) + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) RET // arith exercises the 3R integer and FP set. TEXT ·arith(SB), NOSPLIT, $0-0 - ADDV R4, R5, R6 - SUBV R7, R8, R9 - MULV R10, R11, R12 - DIVV R13, R14, R15 - AND R16, R17, R18 - OR R18, R19, R20 - XOR R20, R21, R2 - SLLV R2, R23, R24 - SRLV R24, R25, R26 - SRAV R26, R27, R28 + ADDV R4, R5, R6 + SUBV R7, R8, R9 + MULV R10, R11, R12 + DIVV R13, R14, R15 + AND R16, R17, R18 + OR R18, R19, R20 + XOR R20, R21, R2 + SLLV R2, R23, R24 + SRLV R24, R25, R26 + SRAV R26, R27, R28 RET // imm exercises the immediate forms. TEXT ·imm(SB), NOSPLIT, $0-0 - ADDV $42, R4, R5 - ADDV $-8, R6 - AND $0xff, R7, R8 - OR $1, R9, R10 - XOR $0, R11, R12 - SGT $100, R13, R14 - SLLV $4, R15, R16 - MOVV $0x12345, R17 + ADDV $42, R4, R5 + ADDV $-8, R6 + AND $0xff, R7, R8 + OR $1, R9, R10 + XOR $0, R11, R12 + SGT $100, R13, R14 + SLLV $4, R15, R16 + MOVV $0x12345, R17 RET // branch exercises conditional and unconditional control flow. TEXT ·branch(SB), NOSPLIT, $0-0 - BEQ R4, R5, done - BNE R6, R7, skip - BLT R8, R9, done - BGE R10, R11, done - BLTU R12, R13, done - BGEU R14, R15, done + BEQ R4, R5, done + BNE R6, R7, skip + BLT R8, R9, done + BGE R10, R11, done + BLTU R12, R13, done + BGEU R14, R15, done + skip: - JMP loop + JMP loop + loop: - JAL skip + JAL skip RET + done: RET diff --git a/testdata/verify/branch_arm64.s b/testdata/verify/branch_arm64.s index 3b73c18..93e599f 100644 --- a/testdata/verify/branch_arm64.s +++ b/testdata/verify/branch_arm64.s @@ -5,23 +5,26 @@ // branch exercises all conditional branch forms and jump chain folding. TEXT ·branch(SB), NOSPLIT, $0-0 - BEQ done - BNE skip - BGE done - BLT done - BGT done - BLE done - BCS done - BCC done - BMI done - BPL done - BVS done - BVC done - BHI done - BLS done + BEQ done + BNE skip + BGE done + BLT done + BGT done + BLE done + BCS done + BCC done + BMI done + BPL done + BVS done + BVC done + BHI done + BLS done + skip: - B loop + B loop + loop: - ADD R4, R5 + ADD R4, R5 + done: RET diff --git a/testdata/verify/branch_riscv64.s b/testdata/verify/branch_riscv64.s index d31231b..df76ce8 100644 --- a/testdata/verify/branch_riscv64.s +++ b/testdata/verify/branch_riscv64.s @@ -5,31 +5,39 @@ TEXT ·branches(SB), NOSPLIT, $0 ADDI $1, X10, X10 - BEQ X10, X11, beq_done + BEQ X10, X11, beq_done ADDI $2, X10, X10 + beq_done: - BNE X10, X11, bne_done + BNE X10, X11, bne_done ADDI $3, X10, X10 + bne_done: - BLT X10, X11, blt_done + BLT X10, X11, blt_done ADDI $4, X10, X10 + blt_done: - BGE X10, X11, bge_done + BGE X10, X11, bge_done ADDI $5, X10, X10 + bge_done: BLTU X10, X11, bltu_done ADDI $6, X10, X10 + bltu_done: BGEU X10, X11, bgeu_done ADDI $7, X10, X10 + bgeu_done: RET TEXT ·jumps(SB), NOSPLIT, $0 - JMP done + JMP done ADDI $1, X10, X10 + done: - JAL X11, skip + JAL X11, skip ADDI $2, X10, X10 + skip: RET diff --git a/testdata/verify/call_arm64.s b/testdata/verify/call_arm64.s index 426dd19..26aa7ae 100644 --- a/testdata/verify/call_arm64.s +++ b/testdata/verify/call_arm64.s @@ -5,6 +5,6 @@ // caller exercises BL to an external symbol (produces a relocation). TEXT ·caller(SB), NOSPLIT, $0-0 - BL other(SB) - ADD R4, R5 + BL other(SB) + ADD R4, R5 RET diff --git a/testdata/verify/fp_arm64.s b/testdata/verify/fp_arm64.s index fd2be60..3505fbd 100644 --- a/testdata/verify/fp_arm64.s +++ b/testdata/verify/fp_arm64.s @@ -5,115 +5,115 @@ // fparith exercises the FP arithmetic set. TEXT ·fparith(SB), NOSPLIT, $0-0 - FADDD F0, F1, F2 - FSUBD F3, F4, F5 - FMULD F6, F7, F8 - FDIVD F9, F10, F11 - FADDS F12, F13, F14 - FSUBS F15, F16, F17 - FMULS F18, F19, F20 - FDIVS F21, F22, F23 - FSQRTD F24, F25 - FSQRTS F26, F27 - FNEGD F28, F29 - FNEGS F30, F31 - FABSD F0, F1 - FABSS F2, F3 - FNMULD F4, F5, F6 - FNMULS F7, F8, F9 - FMIND F10, F11, F12 - FMAXD F13, F14, F15 - FMINS F16, F17, F18 - FMAXS F19, F20, F21 + FADDD F0, F1, F2 + FSUBD F3, F4, F5 + FMULD F6, F7, F8 + FDIVD F9, F10, F11 + FADDS F12, F13, F14 + FSUBS F15, F16, F17 + FMULS F18, F19, F20 + FDIVS F21, F22, F23 + FSQRTD F24, F25 + FSQRTS F26, F27 + FNEGD F28, F29 + FNEGS F30, F31 + FABSD F0, F1 + FABSS F2, F3 + FNMULD F4, F5, F6 + FNMULS F7, F8, F9 + FMIND F10, F11, F12 + FMAXD F13, F14, F15 + FMINS F16, F17, F18 + FMAXS F19, F20, F21 RET // fpfma exercises fused multiply-add. TEXT ·fpfma(SB), NOSPLIT, $0-0 - FMADDD F0, F1, F2, F3 - FMSUBD F4, F5, F6, F7 - FNMADDD F8, F9, F10, F11 - FNMSUBD F12, F13, F14, F15 - FMADDS F16, F17, F18, F19 - FMSUBS F20, F21, F22, F23 - FNMADDS F24, F25, F26, F27 - FNMSUBS F28, F29, F30, F0 + FMADDD F0, F1, F2, F3 + FMSUBD F4, F5, F6, F7 + FNMADDD F8, F9, F10, F11 + FNMSUBD F12, F13, F14, F15 + FMADDS F16, F17, F18, F19 + FMSUBS F20, F21, F22, F23 + FNMADDS F24, F25, F26, F27 + FNMSUBS F28, F29, F30, F0 RET // fpconv exercises FP↔integer conversion and cross-precision. // Syntax: FCVTZSD Fd, Rn (float→int: FP source first, int dest second) // SCVTFD Rn, Fd (int→float: int source first, FP dest second) TEXT ·fpconv(SB), NOSPLIT, $0-0 - FCVTSD F0, F1 - FCVTDS F2, F3 - FCVTZSD F4, R0 - FCVTZSS F5, R1 - FCVTZUD F6, R2 - FCVTZUS F7, R3 - SCVTFD R4, F8 - SCVTFS R5, F9 - UCVTFD R6, F10 - UCVTFS R7, F11 - SCVTFWD R0, F12 - SCVTFWS R1, F13 - UCVTFWD R2, F14 - UCVTFWS R3, F15 - FMOVS F14, R20 - FMOVS R21, F15 - FMOVD F16, R22 - FMOVD R23, F17 + FCVTSD F0, F1 + FCVTDS F2, F3 + FCVTZSD F4, R0 + FCVTZSS F5, R1 + FCVTZUD F6, R2 + FCVTZUS F7, R3 + SCVTFD R4, F8 + SCVTFS R5, F9 + UCVTFD R6, F10 + UCVTFS R7, F11 + SCVTFWD R0, F12 + SCVTFWS R1, F13 + UCVTFWD R2, F14 + UCVTFWS R3, F15 + FMOVS F14, R20 + FMOVS R21, F15 + FMOVD F16, R22 + FMOVD R23, F17 RET // fpcmp exercises FP compare and conditional compare. // FCCMP syntax: FCCMP cond, Fn, Fm, $nzcv // FCSEL syntax: FCSEL cond, Fn, Fm, Fd TEXT ·fpcmp(SB), NOSPLIT, $0-0 - FCMPS F0, F1 - FCMPD F2, F3 - FCMPS $0.0, F4 - FCMPD $0.0, F5 - FCCMPS EQ, F6, F7, $0 - FCCMPD NE, F8, F9, $0 - FCSELS GE, F10, F11, F12 - FCSELD LT, F13, F14, F15 + FCMPS F0, F1 + FCMPD F2, F3 + FCMPS $0.0, F4 + FCMPD $0.0, F5 + FCCMPS EQ, F6, F7, $0 + FCCMPD NE, F8, F9, $0 + FCSELS GE, F10, F11, F12 + FCSELD LT, F13, F14, F15 RET // frint exercises FP rounding. TEXT ·frint(SB), NOSPLIT, $0-0 - FRINTND F0, F1 - FRINTNS F2, F3 - FRINTPD F4, F5 - FRINTPS F6, F7 - FRINTMD F8, F9 - FRINTMS F10, F11 - FRINTZD F12, F13 - FRINTZS F14, F15 - FRINTAD F16, F17 - FRINTAS F18, F19 - FRINTXD F20, F21 - FRINTXS F22, F23 - FRINTID F24, F25 - FRINTIS F26, F27 - FMOVD F0, F1 - FMOVS F2, F3 + FRINTND F0, F1 + FRINTNS F2, F3 + FRINTPD F4, F5 + FRINTPS F6, F7 + FRINTMD F8, F9 + FRINTMS F10, F11 + FRINTZD F12, F13 + FRINTZS F14, F15 + FRINTAD F16, F17 + FRINTAS F18, F19 + FRINTXD F20, F21 + FRINTXS F22, F23 + FRINTID F24, F25 + FRINTIS F26, F27 + FMOVD F0, F1 + FMOVS F2, F3 RET // condsel exercises conditional select and CRC32. TEXT ·condsel(SB), NOSPLIT, $0-0 - CSEL EQ, R0, R1, R2 - CSINC NE, R3, R4, R5 - CSINV GE, R6, R7, R8 - CSNEG LT, R9, R10, R11 - CSET EQ, R12 - CSETM NE, R13 - CINC EQ, R14, R15 - CINV NE, R16, R17 - CNEG GE, R19, R20 - CRC32B R0, R2 - CRC32H R3, R5 - CRC32W R6, R8 - CRC32X R9, R11 - CRC32CB R12, R14 - CRC32CH R15, R0 - CRC32CW R1, R3 - CRC32CX R4, R6 + CSEL EQ, R0, R1, R2 + CSINC NE, R3, R4, R5 + CSINV GE, R6, R7, R8 + CSNEG LT, R9, R10, R11 + CSET EQ, R12 + CSETM NE, R13 + CINC EQ, R14, R15 + CINV NE, R16, R17 + CNEG GE, R19, R20 + CRC32B R0, R2 + CRC32H R3, R5 + CRC32W R6, R8 + CRC32X R9, R11 + CRC32CB R12, R14 + CRC32CH R15, R0 + CRC32CW R1, R3 + CRC32CX R4, R6 RET diff --git a/testdata/verify/fp_loong64.s b/testdata/verify/fp_loong64.s index 51da362..41706b9 100644 --- a/testdata/verify/fp_loong64.s +++ b/testdata/verify/fp_loong64.s @@ -6,138 +6,154 @@ // fp exercises the floating-point set: 3R arithmetic, 2R unary, compares // into FCC, fused multiply-add and the register moves. TEXT ·fp(SB), NOSPLIT, $0-0 - ADDD F4, F5, F6 - SUBD F7, F8, F9 - MULD F9, F10, F11 - DIVD F11, F12, F13 - MULF F13, F14, F15 - ADDF F15, F16, F17 - SQRTD F17, F18 - SQRTF F18, F19 - ABSD F19, F20 - NEGD F20, F21 - MOVD F21, F22 - CMPEQD F22, F23, FCC0 - CMPGTF F23, F24, FCC1 - CMPGED F24, F25, FCC2 - FMADDD F0, F1, F2, F3 - FMSUBF F3, F4, F5, F6 - FNMADDD F6, F7, F8, F9 - FNMSUBF F9, F10, F11, F12 - FMAXD F12, F13, F14 - FMINF F14, F15, F16 - FMAXAD F16, F17, F18 - FMINAF F18, F19, F20 - FSCALEBF F20, F21, F22 - FCOPYSGD F22, F23, F24 - MOVV F25, R25 - MOVV R26, F27 - MOVW R28, F29 - MOVW F30, R31 + ADDD F4, F5, F6 + SUBD F7, F8, F9 + MULD F9, F10, F11 + DIVD F11, F12, F13 + MULF F13, F14, F15 + ADDF F15, F16, F17 + SQRTD F17, F18 + SQRTF F18, F19 + ABSD F19, F20 + NEGD F20, F21 + MOVD F21, F22 + CMPEQD F22, F23, FCC0 + CMPGTF F23, F24, FCC1 + CMPGED F24, F25, FCC2 + FMADDD F0, F1, F2, F3 + FMSUBF F3, F4, F5, F6 + FNMADDD F6, F7, F8, F9 + FNMSUBF F9, F10, F11, F12 + FMAXD F12, F13, F14 + FMINF F14, F15, F16 + FMAXAD F16, F17, F18 + FMINAF F18, F19, F20 + FSCALEBF F20, F21, F22 + FCOPYSGD F22, F23, F24 + MOVV F25, R25 + MOVV R26, F27 + MOVW R28, F29 + MOVW F30, R31 RET // mov forms: register moves, immediates (12/32/64-bit), memory with FP/SP // pseudo-registers and the register-indexed forms. TEXT ·mov(SB), NOSPLIT, $0-16 - MOVV R4, R5 - MOVW R6, R7 - MOVB R8, R9 - MOVBU R10, R11 - MOVHU R12, R13 - MOVWU R14, R15 - MOVV $42, R16 - MOVV $0x12345, R17 - MOVV $0x100000, R18 - MOVW $-100, R19 - MOVV $0x123456789, R20 - MOVV a+0(FP), R21 - MOVV R23, b+8(FP) - MOVW c+16(FP), R24 - MOVV (R24)(R25), R26 - MOVV R27, (R28)(R29) + MOVV R4, R5 + MOVW R6, R7 + MOVB R8, R9 + MOVBU R10, R11 + MOVHU R12, R13 + MOVWU R14, R15 + MOVV $42, R16 + MOVV $0x12345, R17 + MOVV $0x100000, R18 + MOVW $-100, R19 + MOVV $0x123456789, R20 + MOVV a+0(FP), R21 + MOVV R23, b+8(FP) + MOVW c+16(FP), R24 + MOVV (R24)(R25), R26 + MOVV R27, (R28)(R29) RET // frame exercises the prologue/epilogue of a function with a real frame. TEXT ·frame(SB), NOSPLIT, $32-8 - MOVV R4, R5 - MOVV arg+0(FP), R6 - MOVV R7, local-8(SP) - MOVV local-8(SP), R8 - MOVV R9, ret+0(FP) + MOVV R4, R5 + MOVV arg+0(FP), R6 + MOVV R7, local-8(SP) + MOVV local-8(SP), R8 + MOVV R9, ret+0(FP) RET // branches21 exercises the single-register and zero-register branch forms // with 21-bit offsets. TEXT ·branches21(SB), NOSPLIT, $0-0 - BEQ R0, R4, l1 - BEQ R5, R0, l2 - BNE R0, R6, l3 - BNE R7, R0, l4 - BLTZ R8, l5 - BGEZ R9, l6 - BLEZ R10, l7 - BGTZ R11, l8 - JMP l9 + BEQ R0, R4, l1 + BEQ R5, R0, l2 + BNE R0, R6, l3 + BNE R7, R0, l4 + BLTZ R8, l5 + BGEZ R9, l6 + BLEZ R10, l7 + BGTZ R11, l8 + JMP l9 + l1: - JMP l10 + JMP l10 + l2: - JMP l11 + JMP l11 + l3: - JMP l12 + JMP l12 + l4: - JMP l13 + JMP l13 + l5: - JMP l14 + JMP l14 + l6: - JMP l15 + JMP l15 + l7: - JMP l16 + JMP l16 + l8: - JMP l16 + JMP l16 + l9: - MOVV R1, R2 + MOVV R1, R2 + l10: - LL (R12), R13 - LLV (R14), R15 - SC R16, (R17) - SCV R18, (R19) - RDTIMED R20, R21 + LL (R12), R13 + LLV (R14), R15 + SC R16, (R17) + SCV R18, (R19) + RDTIMED R20, R21 SYSCALL DBAR RET + l11: - JAL (R30) + JAL (R30) RET + l12: - BSTRINSV $7, R4, $0, R5 - BSTRPICKV $63, R6, $32, R7 - ALSLV $2, R8, R9, R10 - ADDV16 $65536, R11, R12 + BSTRINSV $7, R4, $0, R5 + BSTRPICKV $63, R6, $32, R7 + ALSLV $2, R8, R9, R10 + ADDV16 $65536, R11, R12 RET + l13: - MOVV $0xffffffffffffffff, R13 + MOVV $0xffffffffffffffff, R13 RET + l14: - CPUCFG R14, R14 + CPUCFG R14, R14 RET + l15: - NOR R15, R16, R17 - ORN R18, R19, R20 - ANDN R21, R24, R25 + NOR R15, R16, R17 + ORN R18, R19, R20 + ANDN R21, R24, R25 RET + l16: - MOVB R26, (R27) - MOVB (R28), R29 + MOVB R26, (R27) + MOVB (R28), R29 RET // sbdata loads and stores a static symbol with relocations (the relocation // fields are masked before comparison). -GLOBL ·table(SB), RODATA, $16 -DATA ·table+0(SB)/8, $0x1122334455667788 -DATA ·table+8(SB)/8, $0x8877665544332211 +GLOBL ·table(SB), RODATA, $16 +DATA ·table+0(SB)/8, $0x1122334455667788 +DATA ·table+8(SB)/8, $0x8877665544332211 TEXT ·sbdata(SB), NOSPLIT, $0-0 - MOVV $·table(SB), R4 - MOVV ·table(SB), R5 - MOVV R6, ·table+8(SB) + MOVV $·table(SB), R4 + MOVV ·table(SB), R5 + MOVV R6, ·table+8(SB) RET diff --git a/testdata/verify/largeimm_riscv64.s b/testdata/verify/largeimm_riscv64.s index fffa627..ccf5928 100644 --- a/testdata/verify/largeimm_riscv64.s +++ b/testdata/verify/largeimm_riscv64.s @@ -7,6 +7,6 @@ TEXT ·largeimm(SB), NOSPLIT, $0 ADDI $2048, X5 ADDI $4095, X5, X6 ANDI $4095, X5, X6 - ORI $-4096, X5, X6 + ORI $-4096, X5, X6 XORI $0x12345, X5, X6 RET diff --git a/testdata/verify/loadstore_riscv64.s b/testdata/verify/loadstore_riscv64.s index 4633a3b..1af7e6a 100644 --- a/testdata/verify/loadstore_riscv64.s +++ b/testdata/verify/loadstore_riscv64.s @@ -4,14 +4,14 @@ #include "textflag.h" TEXT ·ldst(SB), NOSPLIT, $0 - LD (X8), X9 - SD X9, (X8) - LW (X8), X9 - SW X9, (X8) - LD 8(X2), X10 - SD X10, 16(X2) - LW 4(X2), X11 - SW X11, 8(X2) + LD (X8), X9 + SD X9, (X8) + LW (X8), X9 + SW X9, (X8) + LD 8(X2), X10 + SD X10, 16(X2) + LW 4(X2), X11 + SW X11, 8(X2) RET TEXT ·addi4spn(SB), NOSPLIT, $0 diff --git a/testdata/verify/movimm_arm64.s b/testdata/verify/movimm_arm64.s index 1080cae..178f485 100644 --- a/testdata/verify/movimm_arm64.s +++ b/testdata/verify/movimm_arm64.s @@ -5,17 +5,17 @@ // movimm exercises MOV with various immediate values. TEXT ·movimm(SB), NOSPLIT, $0-0 - MOVD $0, R0 - MOVD $1, R1 - MOVD $42, R2 - MOVD $255, R3 - MOVD $256, R4 - MOVD $0xFFFF, R5 - MOVD $0x12345678, R6 - MOVD $0x123456789ABCDEF0, R7 - MOVD $-1, R8 - MOVD $-2, R9 - MOVW $0, R10 - MOVW $100, R11 - MOVW $0x12345, R12 + MOVD $0, R0 + MOVD $1, R1 + MOVD $42, R2 + MOVD $255, R3 + MOVD $256, R4 + MOVD $0xFFFF, R5 + MOVD $0x12345678, R6 + MOVD $0x123456789ABCDEF0, R7 + MOVD $-1, R8 + MOVD $-2, R9 + MOVW $0, R10 + MOVW $100, R11 + MOVW $0x12345, R12 RET diff --git a/testdata/verify/rvc_riscv64.s b/testdata/verify/rvc_riscv64.s index 33b9ccb..c901b04 100644 --- a/testdata/verify/rvc_riscv64.s +++ b/testdata/verify/rvc_riscv64.s @@ -10,9 +10,9 @@ TEXT ·shifts(SB), NOSPLIT, $0 RET TEXT ·logic(SB), NOSPLIT, $0 - AND X11, X10, X10 - OR X11, X10, X10 - XOR X11, X10, X10 + AND X11, X10, X10 + OR X11, X10, X10 + XOR X11, X10, X10 ANDI $7, X10, X10 RET diff --git a/verify/abi_amd64.s b/verify/abi_amd64.s index e7f4ba2..9c89300 100644 --- a/verify/abi_amd64.s +++ b/verify/abi_amd64.s @@ -17,32 +17,32 @@ // - R14 holds the goroutine pointer and must survive across any call. // Sentinel values chosen to be unlikely in normal execution. -#define SENTINEL_BP 0xDEADBEEFCAFEF00D +#define SENTINEL_BP 0xDEADBEEFCAFEF00D #define SENTINEL_R14 0x0BADF00DDEADBEEF // GLOBL holding the raw address of the leave trampoline, read by Go. GLOBL ·leaveCheckedPtr(SB), NOPTR, $8 -DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) +DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // func enterJITChecked(fn uintptr, stack uintptr) // Sets sentinels in BP and R14, switches to the prepared stack and jumps // to fn. The prepared stack's return address must be leaveJITCheckedRaw // (read from leaveCheckedPtr). TEXT ·enterJITChecked(SB), NOSPLIT, $0-16 - MOVQ fn+0(FP), AX // target (before SP switch) - MOVQ SP, ·savedSP(SB) // preserve Go stack - MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber) - MOVQ R14, ·savedR14(SB) // preserve the goroutine pointer - MOVQ $SENTINEL_BP, BP // sentinel in BP - MOVQ $SENTINEL_R14, R14 // sentinel in R14 - MOVQ stack+8(FP), SP // switch to prepared stack + MOVQ fn+0(FP), AX // target (before SP switch) + MOVQ SP, ·savedSP(SB) // preserve Go stack + MOVQ BP, ·savedBP(SB) // preserve frame pointer (vet requires save before clobber) + MOVQ R14, ·savedR14(SB) // preserve the goroutine pointer + MOVQ $SENTINEL_BP, BP // sentinel in BP + MOVQ $SENTINEL_R14, R14 // sentinel in R14 + MOVQ stack+8(FP), SP // switch to prepared stack JMP AX -// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function -// declaration, so no ABIInternal wrapper is generated; the JIT function's -// RET lands here directly, seeing BP and R14 exactly as the function left -// them. It checks the sentinels, records violations in abiResult, then -// restores the Go stack and returns. + // leaveJITCheckedRaw is the raw return trampoline. It has NO Go function + // declaration, so no ABIInternal wrapper is generated; the JIT function's + // RET lands here directly, seeing BP and R14 exactly as the function left + // them. It checks the sentinels, records violations in abiResult, then + // restores the Go stack and returns. TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 // Check BP against the sentinel. MOVQ $SENTINEL_BP, CX @@ -58,10 +58,10 @@ bp_ok: ORQ $2, ·abiResult(SB) r14_ok: - MOVQ ·savedR14(SB), R14 // restore the goroutine pointer: the runtime - // needs it the moment Go code resumes, whether - // or not the kernel violated it (the violation - // is already recorded in abiResult) - MOVQ ·savedBP(SB), BP // restore the frame pointer + MOVQ ·savedR14(SB), R14 // restore the goroutine pointer: the runtime + // needs it the moment Go code resumes, whether + // or not the kernel violated it (the violation + // is already recorded in abiResult) + MOVQ ·savedBP(SB), BP // restore the frame pointer MOVQ ·savedSP(SB), SP RET diff --git a/verify/abi_arm64.s b/verify/abi_arm64.s index 707bfcf..e3bdfd1 100644 --- a/verify/abi_arm64.s +++ b/verify/abi_arm64.s @@ -21,12 +21,12 @@ // cannot cover it. // Sentinel values chosen to be unlikely in normal execution. -#define SENTINEL_FP 0xDEADBEEFCAFEF00D -#define SENTINEL_G 0x0BADF00DDEADBEEF +#define SENTINEL_FP 0xDEADBEEFCAFEF00D +#define SENTINEL_G 0x0BADF00DDEADBEEF // GLOBL holding the raw address of the leave trampoline, read by Go. GLOBL ·leaveCheckedPtr(SB), NOPTR, $8 -DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) +DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // func enterJITChecked(fn uintptr, stack uintptr) // Sets sentinels in R29, R28 and R18, switches to the prepared stack and @@ -34,26 +34,26 @@ DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // leaveJITCheckedRaw (read from leaveCheckedPtr). Only R0 and R3 are used // as scratch: caller-saved, and not among the checked registers. TEXT ·enterJITChecked(SB), NOSPLIT, $0-16 - MOVD fn+0(FP), R0 // target (before SP switch) - MOVD R30, savedLR(SB) // save link register - MOVD R3, savedSP(SB) // save Go stack pointer - MOVD R29, savedFP(SB) // save frame pointer (vet requires save before clobber) - MOVD g, savedG(SB) // save g - MOVD $SENTINEL_FP, R29 // sentinel in the frame pointer - MOVD $SENTINEL_G, g // sentinel in g - MOVD stack+8(FP), R3 // load prepared stack pointer - MOVD 0(R3), R30 // load leaveJITCheckedRaw into LR - MOVD R3, RSP // SP stays on the leave slot: the kernel - // reads its first argument at SP+8 - JMP (R0) // branch to JIT function + MOVD fn+0(FP), R0 // target (before SP switch) + MOVD R30, savedLR(SB) // save link register + MOVD R3, savedSP(SB) // save Go stack pointer + MOVD R29, savedFP(SB) // save frame pointer (vet requires save before clobber) + MOVD g, savedG(SB) // save g + MOVD $SENTINEL_FP, R29 // sentinel in the frame pointer + MOVD $SENTINEL_G, g // sentinel in g + MOVD stack+8(FP), R3 // load prepared stack pointer + MOVD 0(R3), R30 // load leaveJITCheckedRaw into LR + MOVD R3, RSP // SP stays on the leave slot: the kernel + // reads its first argument at SP+8 + JMP (R0) // branch to JIT function -// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function -// declaration, so no ABIInternal wrapper is generated; the JIT function's -// RET lands here directly, seeing R29 and g exactly as the function left -// them. It checks the sentinels, records violations in abiResult, then -// restores the Go stack and returns. + // leaveJITCheckedRaw is the raw return trampoline. It has NO Go function + // declaration, so no ABIInternal wrapper is generated; the JIT function's + // RET lands here directly, seeing R29 and g exactly as the function left + // them. It checks the sentinels, records violations in abiResult, then + // restores the Go stack and returns. TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 - MOVD $0, R4 // accumulated violation bits + MOVD $0, R4 // accumulated violation bits // Check the frame pointer against the sentinel. MOVD $SENTINEL_FP, R3 @@ -61,6 +61,7 @@ TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 BEQ fp_ok MOVD $1, R5 ORR R5, R4, R4 + fp_ok: // Check g against the sentinel. MOVD $SENTINEL_G, R3 @@ -68,22 +69,24 @@ fp_ok: BEQ g_ok MOVD $2, R5 ORR R5, R4, R4 + g_ok: CBZ R4, restore MOVD R4, ·abiResult(SB) restore: - MOVD savedSP(SB), R3 // restore Go stack pointer + MOVD savedSP(SB), R3 // restore Go stack pointer MOVD R3, RSP - MOVD savedLR(SB), R30 // restore link register - MOVD savedFP(SB), R29 // restore frame pointer: Go code needs it the - // moment it resumes, violation or not - MOVD savedG(SB), g // restore g - RET // return to Go caller + MOVD savedLR(SB), R30 // restore link register + MOVD savedFP(SB), R29 // restore frame pointer: Go code needs it the + // moment it resumes, violation or not + MOVD savedG(SB), g // restore g + RET // return to Go caller // Package-level storage for the saved frame pointer. Like savedSP and // savedLR in trampoline_arm64.s, this is assembly-side state: the amd64 // checked trampoline saves the caller's frame pointer for vet's sake and // never restores it, and this file mirrors that. GLOBL savedFP(SB), NOPTR, $8 + GLOBL savedG(SB), NOPTR, $8 diff --git a/verify/abi_loong64.s b/verify/abi_loong64.s index b2a112e..211dc08 100644 --- a/verify/abi_loong64.s +++ b/verify/abi_loong64.s @@ -18,11 +18,11 @@ // spells this register "g"; R22 is not accepted. // Sentinel value chosen to be unlikely in normal execution. -#define SENTINEL_G 0x0BADF00DDEADBEEF +#define SENTINEL_G 0x0BADF00DDEADBEEF // GLOBL holding the raw address of the leave trampoline, read by Go. GLOBL ·leaveCheckedPtr(SB), NOPTR, $8 -DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) +DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // func enterJITChecked(fn uintptr, stack uintptr) // Sets a sentinel in g (R22), switches to the prepared stack and jumps to @@ -30,22 +30,22 @@ DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // leaveJITCheckedRaw (read from leaveCheckedPtr). Only R4 and R5 are used // as scratch: caller-saved, and R22 is not among them. TEXT ·enterJITChecked(SB), NOSPLIT, $0-16 - MOVV fn+0(FP), R4 // target function address (A0) - MOVV R1, savedRA(SB) // save return address (RA) - MOVV R3, savedSP(SB) // save Go stack pointer (SP) - MOVV g, savedG(SB) // save g - MOVV $SENTINEL_G, g // sentinel in g - MOVV stack+8(FP), R5 // load prepared stack pointer (A1) - MOVV 0(R5), R1 // load leaveJITCheckedRaw into RA - MOVV R5, R3 // SP stays on the leave slot: the kernel - // reads its first argument at SP+8 - JIRL R0, R4, 0 // jump to JIT function + MOVV fn+0(FP), R4 // target function address (A0) + MOVV R1, savedRA(SB) // save return address (RA) + MOVV R3, savedSP(SB) // save Go stack pointer (SP) + MOVV g, savedG(SB) // save g + MOVV $SENTINEL_G, g // sentinel in g + MOVV stack+8(FP), R5 // load prepared stack pointer (A1) + MOVV 0(R5), R1 // load leaveJITCheckedRaw into RA + MOVV R5, R3 // SP stays on the leave slot: the kernel + // reads its first argument at SP+8 + JIRL R0, R4, 0 // jump to JIT function -// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function -// declaration, so no ABIInternal wrapper is generated; the JIT function's -// RET lands here directly, seeing g exactly as the function left it. It -// checks the sentinel, records violations in abiResult, then restores the -// Go stack and returns. + // leaveJITCheckedRaw is the raw return trampoline. It has NO Go function + // declaration, so no ABIInternal wrapper is generated; the JIT function's + // RET lands here directly, seeing g exactly as the function left it. It + // checks the sentinel, records violations in abiResult, then restores the + // Go stack and returns. TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 // Check g against the sentinel. MOVV $SENTINEL_G, R5 @@ -56,11 +56,11 @@ TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 MOVV R4, ·abiResult(SB) g_ok: - MOVV savedSP(SB), R5 // restore Go stack pointer + MOVV savedSP(SB), R5 // restore Go stack pointer MOVV R5, R3 - MOVV savedRA(SB), R1 // restore return address - MOVV savedG(SB), g // restore g: Go code needs it the moment it - // resumes, violation or not - JIRL R0, R1, 0 // return to Go caller + MOVV savedRA(SB), R1 // restore return address + MOVV savedG(SB), g // restore g: Go code needs it the moment it + // resumes, violation or not + JIRL R0, R1, 0 // return to Go caller GLOBL savedG(SB), NOPTR, $8 diff --git a/verify/abi_riscv64.s b/verify/abi_riscv64.s index 98ff480..2cd7afd 100644 --- a/verify/abi_riscv64.s +++ b/verify/abi_riscv64.s @@ -18,11 +18,11 @@ // spells this register "g"; X27 is not accepted. // Sentinel value chosen to be unlikely in normal execution. -#define SENTINEL_G 0x0BADF00DDEADBEEF +#define SENTINEL_G 0x0BADF00DDEADBEEF // GLOBL holding the raw address of the leave trampoline, read by Go. GLOBL ·leaveCheckedPtr(SB), NOPTR, $8 -DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) +DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // func enterJITChecked(fn uintptr, stack uintptr) // Sets a sentinel in g (X27), switches to the prepared stack and jumps to @@ -30,22 +30,22 @@ DATA ·leaveCheckedPtr(SB)/8, $·leaveJITCheckedRaw(SB) // leaveJITCheckedRaw (read from leaveCheckedPtr). Only X5 and X6 are used // as scratch: caller-saved, and X27 is not among them. TEXT ·enterJITChecked(SB), NOSPLIT, $0-16 - MOV fn+0(FP), X5 // target function address (T0) - MOV X1, savedRA(SB) // save return address - MOV X2, savedSP(SB) // save Go stack pointer - MOV g, savedG(SB) // save g - MOV $SENTINEL_G, g // sentinel in g - MOV stack+8(FP), X6 // load prepared stack pointer (T1) - LD 0(X6), X1 // load leaveJITCheckedRaw into RA - MOV X6, X2 // SP stays on the leave slot: the kernel - // reads its first argument at SP+8 - JALR X0, 0(X5) // jump to JIT function + MOV fn+0(FP), X5 // target function address (T0) + MOV X1, savedRA(SB) // save return address + MOV X2, savedSP(SB) // save Go stack pointer + MOV g, savedG(SB) // save g + MOV $SENTINEL_G, g // sentinel in g + MOV stack+8(FP), X6 // load prepared stack pointer (T1) + LD 0(X6), X1 // load leaveJITCheckedRaw into RA + MOV X6, X2 // SP stays on the leave slot: the kernel + // reads its first argument at SP+8 + JALR X0, 0(X5) // jump to JIT function -// leaveJITCheckedRaw is the raw return trampoline. It has NO Go function -// declaration, so no ABIInternal wrapper is generated; the JIT function's -// RET lands here directly, seeing g exactly as the function left it. It -// checks the sentinel, records violations in abiResult, then restores the -// Go stack and returns. + // leaveJITCheckedRaw is the raw return trampoline. It has NO Go function + // declaration, so no ABIInternal wrapper is generated; the JIT function's + // RET lands here directly, seeing g exactly as the function left it. It + // checks the sentinel, records violations in abiResult, then restores the + // Go stack and returns. TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 // Check g against the sentinel. MOV $SENTINEL_G, X6 @@ -56,11 +56,11 @@ TEXT ·leaveJITCheckedRaw(SB), NOSPLIT, $0-0 MOV X7, ·abiResult(SB) g_ok: - MOV savedSP(SB), X6 // restore Go stack pointer + MOV savedSP(SB), X6 // restore Go stack pointer MOV X6, X2 - MOV savedRA(SB), X1 // restore return address - MOV savedG(SB), g // restore g: Go code needs it the moment it - // resumes, violation or not - JALR X0, 0(X1) // return to Go caller + MOV savedRA(SB), X1 // restore return address + MOV savedG(SB), g // restore g: Go code needs it the moment it + // resumes, violation or not + JALR X0, 0(X1) // return to Go caller GLOBL savedG(SB), NOPTR, $8 diff --git a/verify/trampoline_amd64.s b/verify/trampoline_amd64.s index 7b3d7f6..7ae9cd5 100644 --- a/verify/trampoline_amd64.s +++ b/verify/trampoline_amd64.s @@ -18,13 +18,13 @@ // Switches to the prepared stack and jumps to fn. Does not return normally; // the JIT function's RET transfers control to leaveJIT. TEXT ·enterJIT(SB), NOSPLIT, $0-16 - MOVQ fn+0(FP), AX // target function address (before SP switch) - MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer - MOVQ stack+8(FP), SP // switch to the prepared stack + MOVQ fn+0(FP), AX // target function address (before SP switch) + MOVQ SP, ·savedSP(SB) // preserve the Go stack pointer + MOVQ stack+8(FP), SP // switch to the prepared stack JMP AX -// func leaveJIT() -// Restores the Go stack pointer and returns to enterJIT's caller. + // func leaveJIT() + // Restores the Go stack pointer and returns to enterJIT's caller. TEXT ·leaveJIT(SB), NOSPLIT, $0-0 MOVQ ·savedSP(SB), SP RET diff --git a/verify/trampoline_arm64.s b/verify/trampoline_arm64.s index c0bad91..1e3aeef 100644 --- a/verify/trampoline_arm64.s +++ b/verify/trampoline_arm64.s @@ -15,32 +15,33 @@ // func enterJIT(fn uintptr, stack uintptr) TEXT ·enterJIT(SB), NOSPLIT, $0-16 - MOVD fn+0(FP), R0 // target function address - MOVD R30, savedLR(SB) // save link register - MOVD R3, savedSP(SB) // save Go stack pointer - MOVD stack+8(FP), R3 // load prepared stack pointer - MOVD 0(R3), R30 // load leaveJIT address into LR - MOVD R3, RSP // switch to the prepared stack: SP stays on - // the leave slot, so the kernel reads its - // first argument at SP+8 per the frame - // convention (amd64 lays the stack out the - // same way) - JMP (R0) // branch to JIT function + MOVD fn+0(FP), R0 // target function address + MOVD R30, savedLR(SB) // save link register + MOVD R3, savedSP(SB) // save Go stack pointer + MOVD stack+8(FP), R3 // load prepared stack pointer + MOVD 0(R3), R30 // load leaveJIT address into LR + MOVD R3, RSP // switch to the prepared stack: SP stays on + // the leave slot, so the kernel reads its + // first argument at SP+8 per the frame + // convention (amd64 lays the stack out the + // same way) + JMP (R0) // branch to JIT function -// func leaveJIT() + // func leaveJIT() TEXT ·leaveJIT(SB), NOSPLIT, $0-0 - MOVD savedSP(SB), R3 // restore Go stack pointer + MOVD savedSP(SB), R3 // restore Go stack pointer MOVD R3, RSP - MOVD savedLR(SB), R30 // restore link register - RET // return to Go caller + MOVD savedLR(SB), R30 // restore link register + RET // return to Go caller // leaveRawAddr holds the raw .abi0 address of leaveJIT, read by call_arm64.go // in preference to reflect.ValueOf(leaveJIT), which returns the address of the // ABIInternal wrapper the linker interposes: the wrapper's prologue clobbers // the saved-register window the JIT call depends on. GLOBL ·leaveRawAddr(SB), NOPTR, $8 -DATA ·leaveRawAddr(SB)/8, $·leaveJIT(SB) +DATA ·leaveRawAddr(SB)/8, $·leaveJIT(SB) // Package-level storage for saved registers. GLOBL savedLR(SB), NOPTR, $8 + GLOBL savedSP(SB), NOPTR, $8 diff --git a/verify/trampoline_loong64.s b/verify/trampoline_loong64.s index 64e0f89..e3d9283 100644 --- a/verify/trampoline_loong64.s +++ b/verify/trampoline_loong64.s @@ -11,22 +11,23 @@ // func enterJIT(fn uintptr, stack uintptr) TEXT ·enterJIT(SB), NOSPLIT, $0-16 - MOVV fn+0(FP), R4 // target function address (A0) - MOVV R1, savedRA(SB) // save return address (RA) - MOVV R3, savedSP(SB) // save Go stack pointer (SP) - MOVV stack+8(FP), R5 // load prepared stack pointer (A1) - MOVV 0(R5), R1 // load leaveJIT address into RA - MOVV R5, R3 // switch to the prepared stack: SP stays on - // the leave slot, so the kernel reads its - // first argument at SP+8 - JIRL R0, R4, 0 // jump to JIT function + MOVV fn+0(FP), R4 // target function address (A0) + MOVV R1, savedRA(SB) // save return address (RA) + MOVV R3, savedSP(SB) // save Go stack pointer (SP) + MOVV stack+8(FP), R5 // load prepared stack pointer (A1) + MOVV 0(R5), R1 // load leaveJIT address into RA + MOVV R5, R3 // switch to the prepared stack: SP stays on + // the leave slot, so the kernel reads its + // first argument at SP+8 + JIRL R0, R4, 0 // jump to JIT function -// func leaveJIT() + // func leaveJIT() TEXT ·leaveJIT(SB), NOSPLIT, $0-0 - MOVV savedSP(SB), R5 // restore Go stack pointer - MOVV R5, R3 // restore SP - MOVV savedRA(SB), R1 // restore return address - JIRL R0, R1, 0 // return to Go caller + MOVV savedSP(SB), R5 // restore Go stack pointer + MOVV R5, R3 // restore SP + MOVV savedRA(SB), R1 // restore return address + JIRL R0, R1, 0 // return to Go caller GLOBL savedRA(SB), NOPTR, $8 + GLOBL savedSP(SB), NOPTR, $8 diff --git a/verify/trampoline_riscv64.s b/verify/trampoline_riscv64.s index 041a64f..17e143f 100644 --- a/verify/trampoline_riscv64.s +++ b/verify/trampoline_riscv64.s @@ -11,22 +11,23 @@ // func enterJIT(fn uintptr, stack uintptr) TEXT ·enterJIT(SB), NOSPLIT, $0-16 - MOV fn+0(FP), X5 // target function address (T0) - MOV X1, savedRA(SB) // save return address - MOV X2, savedSP(SB) // save Go stack pointer - MOV stack+8(FP), X6 // load prepared stack pointer (T1) - LD 0(X6), X1 // load leaveJIT address into RA - MOV X6, X2 // switch to the prepared stack: SP stays on - // the leave slot, so the kernel reads its - // first argument at SP+8 - JALR X0, 0(X5) // jump to JIT function + MOV fn+0(FP), X5 // target function address (T0) + MOV X1, savedRA(SB) // save return address + MOV X2, savedSP(SB) // save Go stack pointer + MOV stack+8(FP), X6 // load prepared stack pointer (T1) + LD 0(X6), X1 // load leaveJIT address into RA + MOV X6, X2 // switch to the prepared stack: SP stays on + // the leave slot, so the kernel reads its + // first argument at SP+8 + JALR X0, 0(X5) // jump to JIT function -// func leaveJIT() + // func leaveJIT() TEXT ·leaveJIT(SB), NOSPLIT, $0-0 - MOV savedSP(SB), X6 // restore Go stack pointer - MOV X6, X2 // restore SP - MOV savedRA(SB), X1 // restore return address - JALR X0, 0(X1) // return to Go caller + MOV savedSP(SB), X6 // restore Go stack pointer + MOV X6, X2 // restore SP + MOV savedRA(SB), X1 // restore return address + JALR X0, 0(X1) // return to Go caller GLOBL savedRA(SB), NOPTR, $8 + GLOBL savedSP(SB), NOPTR, $8