diff options
| author | Sunil K Pandey <skpgkp2@gmail.com> | 2022-03-07 10:47:08 -0800 |
|---|---|---|
| committer | Sunil K Pandey <skpgkp2@gmail.com> | 2022-03-07 21:14:09 -0800 |
| commit | e71f7abba687b1d39ae83e0a8c2435f5c2e2d14b (patch) | |
| tree | e7911824d6bd515d043a1e8178191d30c42ee048 | |
| parent | 92127a8f41020f893057cc19cf74ef987d578b7b (diff) | |
| download | glibc-e71f7abba687b1d39ae83e0a8c2435f5c2e2d14b.tar.xz glibc-e71f7abba687b1d39ae83e0a8c2435f5c2e2d14b.zip | |
x86_64: Fix svml_d_acosh4_core_avx2.S code formatting
This commit contains following formatting changes
1. Instructions proceeded by a tab.
2. Instruction less than 8 characters in length have a tab
between it and the first operand.
3. Instruction greater than 7 characters in length have a
space between it and the first operand.
4. Tabs after `#define`d names and their value.
5. 8 space at the beginning of line replaced by tab.
6. Indent comments with code.
7. Remove redundent .text section.
8. 1 space between line content and line comment.
9. Space after all commas.
Reviewed-by: Noah Goldstein <goldstein.w.n@gmail.com>
| -rw-r--r-- | sysdeps/x86_64/fpu/multiarch/svml_d_acosh4_core_avx2.S | 2871 |
1 files changed, 1435 insertions, 1436 deletions
diff --git a/sysdeps/x86_64/fpu/multiarch/svml_d_acosh4_core_avx2.S b/sysdeps/x86_64/fpu/multiarch/svml_d_acosh4_core_avx2.S index b1b6a80f04..5d0b23b72c 100644 --- a/sysdeps/x86_64/fpu/multiarch/svml_d_acosh4_core_avx2.S +++ b/sysdeps/x86_64/fpu/multiarch/svml_d_acosh4_core_avx2.S @@ -33,1504 +33,1503 @@ /* Offsets for data table __svml_dacosh_data_internal */ -#define Log_HA_table 0 -#define Log_LA_table 8224 -#define poly_coeff 12352 -#define ExpMask 12480 -#define Two10 12512 -#define MinLog1p 12544 -#define MaxLog1p 12576 -#define One 12608 -#define SgnMask 12640 -#define XThreshold 12672 -#define XhMask 12704 -#define Threshold 12736 -#define Bias 12768 -#define Bias1 12800 -#define ExpMask0 12832 -#define ExpMask2 12864 -#define L2 12896 -#define dBigThreshold 12928 -#define dC1 12960 -#define dC2 12992 -#define dC3 13024 -#define dC4 13056 -#define dC5 13088 -#define dLargestFinite 13120 -#define dThirtyOne 13152 -#define dTopMask12 13184 -#define dTopMask29 13216 -#define XScale 13248 +#define Log_HA_table 0 +#define Log_LA_table 8224 +#define poly_coeff 12352 +#define ExpMask 12480 +#define Two10 12512 +#define MinLog1p 12544 +#define MaxLog1p 12576 +#define One 12608 +#define SgnMask 12640 +#define XThreshold 12672 +#define XhMask 12704 +#define Threshold 12736 +#define Bias 12768 +#define Bias1 12800 +#define ExpMask0 12832 +#define ExpMask2 12864 +#define L2 12896 +#define dBigThreshold 12928 +#define dC1 12960 +#define dC2 12992 +#define dC3 13024 +#define dC4 13056 +#define dC5 13088 +#define dLargestFinite 13120 +#define dThirtyOne 13152 +#define dTopMask12 13184 +#define dTopMask29 13216 +#define XScale 13248 /* Lookup bias for data table __svml_dacosh_data_internal. */ -#define Table_Lookup_Bias -0x405fe0 +#define Table_Lookup_Bias -0x405fe0 #include <sysdep.h> - .text - .section .text.avx2,"ax",@progbits + .section .text.avx2, "ax", @progbits ENTRY(_ZGVdN4v_acosh_avx2) - pushq %rbp - cfi_def_cfa_offset(16) - movq %rsp, %rbp - cfi_def_cfa(6, 16) - cfi_offset(6, -16) - andq $-32, %rsp - subq $96, %rsp - lea Table_Lookup_Bias+__svml_dacosh_data_internal(%rip), %r8 + pushq %rbp + cfi_def_cfa_offset(16) + movq %rsp, %rbp + cfi_def_cfa(6, 16) + cfi_offset(6, -16) + andq $-32, %rsp + subq $96, %rsp + lea Table_Lookup_Bias+__svml_dacosh_data_internal(%rip), %r8 -/* Load the constant 1 and possibly other stuff */ - vmovupd One+__svml_dacosh_data_internal(%rip), %ymm8 + /* Load the constant 1 and possibly other stuff */ + vmovupd One+__svml_dacosh_data_internal(%rip), %ymm8 -/* - * Now 1 / (1 + d) - * = 1 / (1 + (sqrt(1 - e) - 1)) - * = 1 / sqrt(1 - e) - * = 1 + 1/2 * e + 3/8 * e^2 + 5/16 * e^3 + 35/128 * e^4 + - * 63/256 * e^5 + 231/1024 * e^6 + .... - * So compute the first five nonconstant terms of that, so that - * we have a relative correction (1 + Corr) to apply to S etc. - * C1 = 1/2 - * C2 = 3/8 - * C3 = 5/16 - * C4 = 35/128 - * C5 = 63/256 - */ - vmovupd dC5+__svml_dacosh_data_internal(%rip), %ymm3 - vmovapd %ymm0, %ymm9 - vmovapd %ymm8, %ymm13 - vfmsub231pd %ymm9, %ymm9, %ymm13 + /* + * Now 1 / (1 + d) + * = 1 / (1 + (sqrt(1 - e) - 1)) + * = 1 / sqrt(1 - e) + * = 1 + 1/2 * e + 3/8 * e^2 + 5/16 * e^3 + 35/128 * e^4 + + * 63/256 * e^5 + 231/1024 * e^6 + .... + * So compute the first five nonconstant terms of that, so that + * we have a relative correction (1 + Corr) to apply to S etc. + * C1 = 1/2 + * C2 = 3/8 + * C3 = 5/16 + * C4 = 35/128 + * C5 = 63/256 + */ + vmovupd dC5+__svml_dacosh_data_internal(%rip), %ymm3 + vmovapd %ymm0, %ymm9 + vmovapd %ymm8, %ymm13 + vfmsub231pd %ymm9, %ymm9, %ymm13 -/* - * Check that 1 < X < +inf; otherwise go to the callout function. - * We need the callout for X = 1 to avoid division by zero below. - * This test ensures that callout handles NaN and either infinity. - */ - vcmpnle_uqpd dLargestFinite+__svml_dacosh_data_internal(%rip), %ymm9, %ymm10 - vcmpngt_uqpd %ymm8, %ymm9, %ymm11 + /* + * Check that 1 < X < +inf; otherwise go to the callout function. + * We need the callout for X = 1 to avoid division by zero below. + * This test ensures that callout handles NaN and either infinity. + */ + vcmpnle_uqpd dLargestFinite+__svml_dacosh_data_internal(%rip), %ymm9, %ymm10 + vcmpngt_uqpd %ymm8, %ymm9, %ymm11 -/* dU is needed later on */ - vsubpd %ymm8, %ymm9, %ymm6 + /* dU is needed later on */ + vsubpd %ymm8, %ymm9, %ymm6 -/* - * The following computation can go wrong for very large X, e.g. - * the X^2 - 1 = U * V can overflow. But for large X we have - * acosh(X) / log(2 X) - 1 =~= 1/(4 * X^2), so for X >= 2^30 - * we can just later stick X back into the log and tweak up the exponent. - * Actually we scale X by 2^-30 and tweak the exponent up by 31, - * to stay in the safe range for the later log computation. - * Compute a flag now telling us when to do this. - */ - vcmplt_oqpd dBigThreshold+__svml_dacosh_data_internal(%rip), %ymm9, %ymm7 + /* + * The following computation can go wrong for very large X, e.g. + * the X^2 - 1 = U * V can overflow. But for large X we have + * acosh(X) / log(2 X) - 1 =~= 1/(4 * X^2), so for X >= 2^30 + * we can just later stick X back into the log and tweak up the exponent. + * Actually we scale X by 2^-30 and tweak the exponent up by 31, + * to stay in the safe range for the later log computation. + * Compute a flag now telling us when to do this. + */ + vcmplt_oqpd dBigThreshold+__svml_dacosh_data_internal(%rip), %ymm9, %ymm7 -/* - * do the same thing but with NR iteration - * Finally, express Y + W = U * V accurately where Y has <= 29 bits - */ - vandpd dTopMask29+__svml_dacosh_data_internal(%rip), %ymm13, %ymm5 + /* + * do the same thing but with NR iteration + * Finally, express Y + W = U * V accurately where Y has <= 29 bits + */ + vandpd dTopMask29+__svml_dacosh_data_internal(%rip), %ymm13, %ymm5 -/* - * Compute R = 1/sqrt(Y + W) * (1 + d) - * Force R to <= 12 significant bits in case it isn't already - * This means that R * Y and R^2 * Y are exactly representable. - */ - vcvtpd2ps %ymm5, %xmm14 - vsubpd %ymm5, %ymm13, %ymm4 - vrsqrtps %xmm14, %xmm15 - vcvtps2pd %xmm15, %ymm0 - vandpd dTopMask12+__svml_dacosh_data_internal(%rip), %ymm0, %ymm2 - vorpd %ymm11, %ymm10, %ymm12 + /* + * Compute R = 1/sqrt(Y + W) * (1 + d) + * Force R to <= 12 significant bits in case it isn't already + * This means that R * Y and R^2 * Y are exactly representable. + */ + vcvtpd2ps %ymm5, %xmm14 + vsubpd %ymm5, %ymm13, %ymm4 + vrsqrtps %xmm14, %xmm15 + vcvtps2pd %xmm15, %ymm0 + vandpd dTopMask12+__svml_dacosh_data_internal(%rip), %ymm0, %ymm2 + vorpd %ymm11, %ymm10, %ymm12 -/* - * Compute S = (Y/sqrt(Y + W)) * (1 + d) - * and T = (W/sqrt(Y + W)) * (1 + d) - * so that S + T = sqrt(Y + W) * (1 + d) - * S is exact, and the rounding error in T is OK. - */ - vmulpd %ymm2, %ymm5, %ymm10 - vmulpd %ymm4, %ymm2, %ymm11 + /* + * Compute S = (Y/sqrt(Y + W)) * (1 + d) + * and T = (W/sqrt(Y + W)) * (1 + d) + * so that S + T = sqrt(Y + W) * (1 + d) + * S is exact, and the rounding error in T is OK. + */ + vmulpd %ymm2, %ymm5, %ymm10 + vmulpd %ymm4, %ymm2, %ymm11 -/* - * Compute e = -(2 * d + d^2) - * The first FMR is exact, and the rounding error in the other is acceptable - * since d and e are ~ 2^-12 - */ - vmovapd %ymm8, %ymm1 - vfnmadd231pd %ymm10, %ymm2, %ymm1 + /* + * Compute e = -(2 * d + d^2) + * The first FMR is exact, and the rounding error in the other is acceptable + * since d and e are ~ 2^-12 + */ + vmovapd %ymm8, %ymm1 + vfnmadd231pd %ymm10, %ymm2, %ymm1 -/* - * For low-accuracy versions, the computation can be done - * just as U + ((S + T) + (S + T) * Corr) - */ - vaddpd %ymm11, %ymm10, %ymm13 - vfnmadd231pd %ymm11, %ymm2, %ymm1 - vfmadd213pd dC4+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 - vfmadd213pd dC3+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 - vfmadd213pd dC2+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 - vfmadd213pd dC1+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 - vmovmskpd %ymm12, %eax - vmulpd %ymm3, %ymm1, %ymm12 + /* + * For low-accuracy versions, the computation can be done + * just as U + ((S + T) + (S + T) * Corr) + */ + vaddpd %ymm11, %ymm10, %ymm13 + vfnmadd231pd %ymm11, %ymm2, %ymm1 + vfmadd213pd dC4+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 + vfmadd213pd dC3+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 + vfmadd213pd dC2+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 + vfmadd213pd dC1+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 + vmovmskpd %ymm12, %eax + vmulpd %ymm3, %ymm1, %ymm12 -/* Now multiplex to the case X = 2^-30 * input, Xl = dL = 0 in the "big" case. */ - vmulpd XScale+__svml_dacosh_data_internal(%rip), %ymm9, %ymm3 - vfmadd213pd %ymm13, %ymm12, %ymm13 - vaddpd %ymm13, %ymm6, %ymm6 + /* Now multiplex to the case X = 2^-30 * input, Xl = dL = 0 in the "big" case. */ + vmulpd XScale+__svml_dacosh_data_internal(%rip), %ymm9, %ymm3 + vfmadd213pd %ymm13, %ymm12, %ymm13 + vaddpd %ymm13, %ymm6, %ymm6 -/* - * Now we feed into the log1p code, using H in place of _VARG1 and - * also adding L into Xl. - * compute 1+x as high, low parts - */ - vmaxpd %ymm6, %ymm8, %ymm4 - vminpd %ymm6, %ymm8, %ymm2 - vandpd SgnMask+__svml_dacosh_data_internal(%rip), %ymm6, %ymm14 - vcmplt_oqpd XThreshold+__svml_dacosh_data_internal(%rip), %ymm14, %ymm15 - vaddpd %ymm2, %ymm4, %ymm0 - vorpd XhMask+__svml_dacosh_data_internal(%rip), %ymm15, %ymm5 - vandpd %ymm5, %ymm0, %ymm6 - vblendvpd %ymm7, %ymm6, %ymm3, %ymm5 - vsubpd %ymm6, %ymm4, %ymm1 + /* + * Now we feed into the log1p code, using H in place of _VARG1 and + * also adding L into Xl. + * compute 1+x as high, low parts + */ + vmaxpd %ymm6, %ymm8, %ymm4 + vminpd %ymm6, %ymm8, %ymm2 + vandpd SgnMask+__svml_dacosh_data_internal(%rip), %ymm6, %ymm14 + vcmplt_oqpd XThreshold+__svml_dacosh_data_internal(%rip), %ymm14, %ymm15 + vaddpd %ymm2, %ymm4, %ymm0 + vorpd XhMask+__svml_dacosh_data_internal(%rip), %ymm15, %ymm5 + vandpd %ymm5, %ymm0, %ymm6 + vblendvpd %ymm7, %ymm6, %ymm3, %ymm5 + vsubpd %ymm6, %ymm4, %ymm1 -/* 2^ (-10-exp(X) ) */ - vmovupd ExpMask2+__svml_dacosh_data_internal(%rip), %ymm15 - vaddpd %ymm1, %ymm2, %ymm10 + /* 2^ (-10-exp(X) ) */ + vmovupd ExpMask2+__svml_dacosh_data_internal(%rip), %ymm15 + vaddpd %ymm1, %ymm2, %ymm10 -/* exponent bits */ - vpsrlq $20, %ymm5, %ymm2 + /* exponent bits */ + vpsrlq $20, %ymm5, %ymm2 -/* - * Now resume the main code. - * preserve mantissa, set input exponent to 2^(-10) - */ - vandpd ExpMask+__svml_dacosh_data_internal(%rip), %ymm5, %ymm11 - vorpd Two10+__svml_dacosh_data_internal(%rip), %ymm11, %ymm12 + /* + * Now resume the main code. + * preserve mantissa, set input exponent to 2^(-10) + */ + vandpd ExpMask+__svml_dacosh_data_internal(%rip), %ymm5, %ymm11 + vorpd Two10+__svml_dacosh_data_internal(%rip), %ymm11, %ymm12 -/* reciprocal approximation good to at least 11 bits */ - vcvtpd2ps %ymm12, %xmm13 - vrcpps %xmm13, %xmm14 + /* reciprocal approximation good to at least 11 bits */ + vcvtpd2ps %ymm12, %xmm13 + vrcpps %xmm13, %xmm14 -/* exponent*log(2.0) */ - vmovupd Threshold+__svml_dacosh_data_internal(%rip), %ymm13 - vcvtps2pd %xmm14, %ymm3 - vandpd %ymm7, %ymm10, %ymm4 + /* exponent*log(2.0) */ + vmovupd Threshold+__svml_dacosh_data_internal(%rip), %ymm13 + vcvtps2pd %xmm14, %ymm3 + vandpd %ymm7, %ymm10, %ymm4 -/* exponent of X needed to scale Xl */ - vandps ExpMask0+__svml_dacosh_data_internal(%rip), %ymm5, %ymm0 - vpsubq %ymm0, %ymm15, %ymm6 + /* exponent of X needed to scale Xl */ + vandps ExpMask0+__svml_dacosh_data_internal(%rip), %ymm5, %ymm0 + vpsubq %ymm0, %ymm15, %ymm6 -/* round reciprocal to nearest integer, will have 1+9 mantissa bits */ - vroundpd $0, %ymm3, %ymm3 - vextractf128 $1, %ymm2, %xmm1 - vshufps $221, %xmm1, %xmm2, %xmm10 + /* round reciprocal to nearest integer, will have 1+9 mantissa bits */ + vroundpd $0, %ymm3, %ymm3 + vextractf128 $1, %ymm2, %xmm1 + vshufps $221, %xmm1, %xmm2, %xmm10 -/* biased exponent in DP format */ - vcvtdq2pd %xmm10, %ymm12 + /* biased exponent in DP format */ + vcvtdq2pd %xmm10, %ymm12 -/* scale DblRcp */ - vmulpd %ymm6, %ymm3, %ymm2 + /* scale DblRcp */ + vmulpd %ymm6, %ymm3, %ymm2 -/* Add 31 to the exponent in the "large" case to get log(2 * input) */ - vaddpd dThirtyOne+__svml_dacosh_data_internal(%rip), %ymm12, %ymm11 + /* Add 31 to the exponent in the "large" case to get log(2 * input) */ + vaddpd dThirtyOne+__svml_dacosh_data_internal(%rip), %ymm12, %ymm11 -/* argument reduction */ - vfmsub213pd %ymm8, %ymm2, %ymm5 - vmulpd %ymm2, %ymm4, %ymm8 - vmovupd poly_coeff+64+__svml_dacosh_data_internal(%rip), %ymm2 - vblendvpd %ymm7, %ymm12, %ymm11, %ymm1 + /* argument reduction */ + vfmsub213pd %ymm8, %ymm2, %ymm5 + vmulpd %ymm2, %ymm4, %ymm8 + vmovupd poly_coeff+64+__svml_dacosh_data_internal(%rip), %ymm2 + vblendvpd %ymm7, %ymm12, %ymm11, %ymm1 -/* - * prepare table index - * table lookup - */ - vpsrlq $40, %ymm3, %ymm7 - vcmplt_oqpd %ymm3, %ymm13, %ymm3 - vandpd Bias+__svml_dacosh_data_internal(%rip), %ymm3, %ymm14 - vorpd Bias1+__svml_dacosh_data_internal(%rip), %ymm14, %ymm15 - vsubpd %ymm15, %ymm1, %ymm1 + /* + * prepare table index + * table lookup + */ + vpsrlq $40, %ymm3, %ymm7 + vcmplt_oqpd %ymm3, %ymm13, %ymm3 + vandpd Bias+__svml_dacosh_data_internal(%rip), %ymm3, %ymm14 + vorpd Bias1+__svml_dacosh_data_internal(%rip), %ymm14, %ymm15 + vsubpd %ymm15, %ymm1, %ymm1 -/* polynomial */ - vmovupd poly_coeff+__svml_dacosh_data_internal(%rip), %ymm3 - vmovd %xmm7, %edx - vextractf128 $1, %ymm7, %xmm10 - vpextrd $2, %xmm7, %ecx - vmulpd L2+__svml_dacosh_data_internal(%rip), %ymm1, %ymm7 - vaddpd %ymm8, %ymm5, %ymm1 - vmovd %xmm10, %esi - vsubpd %ymm5, %ymm1, %ymm5 - vfmadd213pd poly_coeff+32+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 - vfmadd213pd poly_coeff+96+__svml_dacosh_data_internal(%rip), %ymm1, %ymm2 - vsubpd %ymm5, %ymm8, %ymm4 - vmulpd %ymm1, %ymm1, %ymm8 - vfmadd213pd %ymm2, %ymm8, %ymm3 - movslq %edx, %rdx - movslq %esi, %rsi - vpextrd $2, %xmm10, %edi - movslq %ecx, %rcx - movslq %edi, %rdi + /* polynomial */ + vmovupd poly_coeff+__svml_dacosh_data_internal(%rip), %ymm3 + vmovd %xmm7, %edx + vextractf128 $1, %ymm7, %xmm10 + vpextrd $2, %xmm7, %ecx + vmulpd L2+__svml_dacosh_data_internal(%rip), %ymm1, %ymm7 + vaddpd %ymm8, %ymm5, %ymm1 + vmovd %xmm10, %esi + vsubpd %ymm5, %ymm1, %ymm5 + vfmadd213pd poly_coeff+32+__svml_dacosh_data_internal(%rip), %ymm1, %ymm3 + vfmadd213pd poly_coeff+96+__svml_dacosh_data_internal(%rip), %ymm1, %ymm2 + vsubpd %ymm5, %ymm8, %ymm4 + vmulpd %ymm1, %ymm1, %ymm8 + vfmadd213pd %ymm2, %ymm8, %ymm3 + movslq %edx, %rdx + movslq %esi, %rsi + vpextrd $2, %xmm10, %edi + movslq %ecx, %rcx + movslq %edi, %rdi -/* - * reconstruction - * VQFMA( D, R, P, R2, R ); - */ - vfmadd213pd %ymm4, %ymm8, %ymm3 - vmovsd (%r8,%rdx), %xmm0 - vmovsd (%r8,%rsi), %xmm11 - vmovhpd (%r8,%rcx), %xmm0, %xmm6 - vmovhpd (%r8,%rdi), %xmm11, %xmm12 - vinsertf128 $1, %xmm12, %ymm6, %ymm0 - vaddpd %ymm3, %ymm1, %ymm6 - vaddpd %ymm6, %ymm0, %ymm0 - vaddpd %ymm0, %ymm7, %ymm0 - testl %eax, %eax + /* + * reconstruction + * VQFMA( D, R, P, R2, R ); + */ + vfmadd213pd %ymm4, %ymm8, %ymm3 + vmovsd (%r8, %rdx), %xmm0 + vmovsd (%r8, %rsi), %xmm11 + vmovhpd (%r8, %rcx), %xmm0, %xmm6 + vmovhpd (%r8, %rdi), %xmm11, %xmm12 + vinsertf128 $1, %xmm12, %ymm6, %ymm0 + vaddpd %ymm3, %ymm1, %ymm6 + vaddpd %ymm6, %ymm0, %ymm0 + vaddpd %ymm0, %ymm7, %ymm0 + testl %eax, %eax -/* Go to special inputs processing branch */ - jne L(SPECIAL_VALUES_BRANCH) - # LOE rbx r12 r13 r14 r15 eax ymm0 ymm9 + /* Go to special inputs processing branch */ + jne L(SPECIAL_VALUES_BRANCH) + # LOE rbx r12 r13 r14 r15 eax ymm0 ymm9 -/* Restore registers - * and exit the function - */ + /* Restore registers + * and exit the function + */ L(EXIT): - movq %rbp, %rsp - popq %rbp - cfi_def_cfa(7, 8) - cfi_restore(6) - ret - cfi_def_cfa(6, 16) - cfi_offset(6, -16) + movq %rbp, %rsp + popq %rbp + cfi_def_cfa(7, 8) + cfi_restore(6) + ret + cfi_def_cfa(6, 16) + cfi_offset(6, -16) -/* Branch to process - * special inputs - */ + /* Branch to process + * special inputs + */ L(SPECIAL_VALUES_BRANCH): - vmovupd %ymm9, 32(%rsp) - vmovupd %ymm0, 64(%rsp) - # LOE rbx r12 r13 r14 r15 eax ymm0 + vmovupd %ymm9, 32(%rsp) + vmovupd %ymm0, 64(%rsp) + # LOE rbx r12 r13 r14 r15 eax ymm0 - xorl %edx, %edx - # LOE rbx r12 r13 r14 r15 eax edx + xorl %edx, %edx + # LOE rbx r12 r13 r14 r15 eax edx - vzeroupper - movq %r12, 16(%rsp) - /* DW_CFA_expression: r12 (r12) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -80; DW_OP_plus) */ - .cfi_escape 0x10, 0x0c, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xb0, 0xff, 0xff, 0xff, 0x22 - movl %edx, %r12d - movq %r13, 8(%rsp) - /* DW_CFA_expression: r13 (r13) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -88; DW_OP_plus) */ - .cfi_escape 0x10, 0x0d, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa8, 0xff, 0xff, 0xff, 0x22 - movl %eax, %r13d - movq %r14, (%rsp) - /* DW_CFA_expression: r14 (r14) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -96; DW_OP_plus) */ - .cfi_escape 0x10, 0x0e, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa0, 0xff, 0xff, 0xff, 0x22 - # LOE rbx r15 r12d r13d + vzeroupper + movq %r12, 16(%rsp) + /* DW_CFA_expression: r12 (r12) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -80; DW_OP_plus) */ + .cfi_escape 0x10, 0x0c, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xb0, 0xff, 0xff, 0xff, 0x22 + movl %edx, %r12d + movq %r13, 8(%rsp) + /* DW_CFA_expression: r13 (r13) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -88; DW_OP_plus) */ + .cfi_escape 0x10, 0x0d, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa8, 0xff, 0xff, 0xff, 0x22 + movl %eax, %r13d + movq %r14, (%rsp) + /* DW_CFA_expression: r14 (r14) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -96; DW_OP_plus) */ + .cfi_escape 0x10, 0x0e, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa0, 0xff, 0xff, 0xff, 0x22 + # LOE rbx r15 r12d r13d -/* Range mask - * bits check - */ + /* Range mask + * bits check + */ L(RANGEMASK_CHECK): - btl %r12d, %r13d + btl %r12d, %r13d -/* Call scalar math function */ - jc L(SCALAR_MATH_CALL) - # LOE rbx r15 r12d r13d + /* Call scalar math function */ + jc L(SCALAR_MATH_CALL) + # LOE rbx r15 r12d r13d -/* Special inputs - * processing loop - */ + /* Special inputs + * processing loop + */ L(SPECIAL_VALUES_LOOP): - incl %r12d - cmpl $4, %r12d + incl %r12d + cmpl $4, %r12d -/* Check bits in range mask */ - jl L(RANGEMASK_CHECK) - # LOE rbx r15 r12d r13d + /* Check bits in range mask */ + jl L(RANGEMASK_CHECK) + # LOE rbx r15 r12d r13d - movq 16(%rsp), %r12 - cfi_restore(12) - movq 8(%rsp), %r13 - cfi_restore(13) - movq (%rsp), %r14 - cfi_restore(14) - vmovupd 64(%rsp), %ymm0 + movq 16(%rsp), %r12 + cfi_restore(12) + movq 8(%rsp), %r13 + cfi_restore(13) + movq (%rsp), %r14 + cfi_restore(14) + vmovupd 64(%rsp), %ymm0 -/* Go to exit */ - jmp L(EXIT) - /* DW_CFA_expression: r12 (r12) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -80; DW_OP_plus) */ - .cfi_escape 0x10, 0x0c, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xb0, 0xff, 0xff, 0xff, 0x22 - /* DW_CFA_expression: r13 (r13) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -88; DW_OP_plus) */ - .cfi_escape 0x10, 0x0d, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa8, 0xff, 0xff, 0xff, 0x22 - /* DW_CFA_expression: r14 (r14) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -96; DW_OP_plus) */ - .cfi_escape 0x10, 0x0e, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa0, 0xff, 0xff, 0xff, 0x22 - # LOE rbx r12 r13 r14 r15 ymm0 + /* Go to exit */ + jmp L(EXIT) + /* DW_CFA_expression: r12 (r12) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -80; DW_OP_plus) */ + .cfi_escape 0x10, 0x0c, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xb0, 0xff, 0xff, 0xff, 0x22 + /* DW_CFA_expression: r13 (r13) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -88; DW_OP_plus) */ + .cfi_escape 0x10, 0x0d, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa8, 0xff, 0xff, 0xff, 0x22 + /* DW_CFA_expression: r14 (r14) (DW_OP_lit8; DW_OP_minus; DW_OP_const4s: -32; DW_OP_and; DW_OP_const4s: -96; DW_OP_plus) */ + .cfi_escape 0x10, 0x0e, 0x0e, 0x38, 0x1c, 0x0d, 0xe0, 0xff, 0xff, 0xff, 0x1a, 0x0d, 0xa0, 0xff, 0xff, 0xff, 0x22 + # LOE rbx r12 r13 r14 r15 ymm0 -/* Scalar math fucntion call - * to process special input - */ + /* Scalar math fucntion call + * to process special input + */ L(SCALAR_MATH_CALL): - movl %r12d, %r14d - movsd 32(%rsp,%r14,8), %xmm0 - call acosh@PLT - # LOE rbx r14 r15 r12d r13d xmm0 + movl %r12d, %r14d + movsd 32(%rsp, %r14, 8), %xmm0 + call acosh@PLT + # LOE rbx r14 r15 r12d r13d xmm0 - movsd %xmm0, 64(%rsp,%r14,8) + movsd %xmm0, 64(%rsp, %r14, 8) -/* Process special inputs in loop */ - jmp L(SPECIAL_VALUES_LOOP) - # LOE rbx r15 r12d r13d + /* Process special inputs in loop */ + jmp L(SPECIAL_VALUES_LOOP) + # LOE rbx r15 r12d r13d END(_ZGVdN4v_acosh_avx2) - .section .rodata, "a" - .align 32 + .section .rodata, "a" + .align 32 #ifdef __svml_dacosh_data_internal_typedef typedef unsigned int VUINT32; typedef struct { - __declspec(align(32)) VUINT32 Log_HA_table[(1<<10)+2][2]; - __declspec(align(32)) VUINT32 Log_LA_table[(1<<9)+1][2]; - __declspec(align(32)) VUINT32 po |
