codekingpro/portable-devtools
114k
1// Copyright 2017 The Go Authors. All rights reserved.2// Use of this source code is governed by a BSD-style3// license that can be found in the LICENSE file.4 5#define Ln2Hi 6.93147180369123816490e-016#define Ln2Lo 1.90821492927058770002e-107#define Log2e 1.44269504088896338700e+008#define Overflow 7.09782712893383973096e+029#define Underflow -7.45133219101941108420e+0210#define Overflow2 1.0239999999999999e+0311#define Underflow2 -1.0740e+0312#define NearZero 0x3e30000000000000 // 2**-2813#define PosInf 0x7ff000000000000014#define FracMask 0x000fffffffffffff15#define C1 0x3cb0000000000000 // 2**-5216#define P1 1.66666666666666657415e-01 // 0x3FC55555; 0x5555555517#define P2 -2.77777777770155933842e-03 // 0xBF66C16C; 0x16BEBD9318#define P3 6.61375632143793436117e-05 // 0x3F11566A; 0xAF25DE2C19#define P4 -1.65339022054652515390e-06 // 0xBEBBBD41; 0xC5D26BF120#define P5 4.13813679705723846039e-08 // 0x3E663769; 0x72BEA4D021 22// Exp returns e**x, the base-e exponential of x.23// This is an assembly implementation of the method used for function Exp in file exp.go.24//25// func Exp(x float64) float6426TEXT ·archExp(SB),$0-1627 FMOVD x+0(FP), F0 // F0 = x28 FCMPD F0, F029 BNE isNaN // x = NaN, return NaN30 FMOVD $Overflow, F131 FCMPD F1, F032 BGT overflow // x > Overflow, return PosInf33 FMOVD $Underflow, F134 FCMPD F1, F035 BLT underflow // x < Underflow, return 036 MOVD $NearZero, R037 FMOVD R0, F238 FABSD F0, F339 FMOVD $1.0, F1 // F1 = 1.040 FCMPD F2, F341 BLT nearzero // fabs(x) < NearZero, return 1 + x42 // argument reduction, x = k*ln2 + r, |r| <= 0.5*ln243 // computed as r = hi - lo for extra precision.44 FMOVD $Log2e, F245 FMOVD $0.5, F346 FNMSUBD F0, F3, F2, F4 // Log2e*x - 0.547 FMADDD F0, F3, F2, F3 // Log2e*x + 0.548 FCMPD $0.0, F049 FCSELD LT, F4, F3, F3 // F3 = k50 FCVTZSD F3, R1 // R1 = int(k)51 SCVTFD R1, F3 // F3 = float64(int(k))52 FMOVD $Ln2Hi, F4 // F4 = Ln2Hi53 FMOVD $Ln2Lo, F5 // F5 = Ln2Lo54 FMSUBD F3, F0, F4, F4 // F4 = hi = x - float64(int(k))*Ln2Hi55 FMULD F3, F5 // F5 = lo = float64(int(k)) * Ln2Lo56 FSUBD F5, F4, F6 // F6 = r = hi - lo57 FMULD F6, F6, F7 // F7 = t = r * r58 // compute y59 FMOVD $P5, F8 // F8 = P560 FMOVD $P4, F9 // F9 = P461 FMADDD F7, F9, F8, F13 // P4+t*P562 FMOVD $P3, F10 // F10 = P363 FMADDD F7, F10, F13, F13 // P3+t*(P4+t*P5)64 FMOVD $P2, F11 // F11 = P265 FMADDD F7, F11, F13, F13 // P2+t*(P3+t*(P4+t*P5))66 FMOVD $P1, F12 // F12 = P167 FMADDD F7, F12, F13, F13 // P1+t*(P2+t*(P3+t*(P4+t*P5)))68 FMSUBD F7, F6, F13, F13 // F13 = c = r - t*(P1+t*(P2+t*(P3+t*(P4+t*P5))))69 FMOVD $2.0, F1470 FSUBD F13, F1471 FMULD F6, F13, F1572 FDIVD F14, F15 // F15 = (r*c)/(2-c)73 FSUBD F15, F5, F15 // lo-(r*c)/(2-c)74 FSUBD F4, F15, F15 // (lo-(r*c)/(2-c))-hi75 FSUBD F15, F1, F16 // F16 = y = 1-((lo-(r*c)/(2-c))-hi)76 // inline Ldexp(y, k), benefit:77 // 1, no parameter pass overhead.78 // 2, skip unnecessary checks for Inf/NaN/Zero79 FMOVD F16, R080 AND $FracMask, R0, R2 // fraction81 LSR $52, R0, R5 // exponent82 ADD R1, R5 // R1 = int(k)83 CMP $1, R584 BGE normal85 ADD $52, R5 // denormal86 MOVD $C1, R887 FMOVD R8, F1 // m = 2**-5288normal:89 ORR R5<<52, R2, R090 FMOVD R0, F091 FMULD F1, F0 // return m * x92 FMOVD F0, ret+8(FP)93 RET94nearzero:95 FADDD F1, F096isNaN:97 FMOVD F0, ret+8(FP)98 RET99underflow:100 MOVD ZR, ret+8(FP)101 RET102overflow:103 MOVD $PosInf, R0104 MOVD R0, ret+8(FP)105 RET106 107 108// Exp2 returns 2**x, the base-2 exponential of x.109// This is an assembly implementation of the method used for function Exp2 in file exp.go.110//111// func Exp2(x float64) float64112TEXT ·archExp2(SB),$0-16113 FMOVD x+0(FP), F0 // F0 = x114 FCMPD F0, F0115 BNE isNaN // x = NaN, return NaN116 FMOVD $Overflow2, F1117 FCMPD F1, F0118 BGT overflow // x > Overflow, return PosInf119 FMOVD $Underflow2, F1120 FCMPD F1, F0121 BLT underflow // x < Underflow, return 0122 // argument reduction; x = r*lg(e) + k with |r| <= ln(2)/2123 // computed as r = hi - lo for extra precision.124 FMOVD $0.5, F2125 FSUBD F2, F0, F3 // x + 0.5126 FADDD F2, F0, F4 // x - 0.5127 FCMPD $0.0, F0128 FCSELD LT, F3, F4, F3 // F3 = k129 FCVTZSD F3, R1 // R1 = int(k)130 SCVTFD R1, F3 // F3 = float64(int(k))131 FSUBD F3, F0, F3 // t = x - float64(int(k))132 FMOVD $Ln2Hi, F4 // F4 = Ln2Hi133 FMOVD $Ln2Lo, F5 // F5 = Ln2Lo134 FMULD F3, F4 // F4 = hi = t * Ln2Hi135 FNMULD F3, F5 // F5 = lo = -t * Ln2Lo136 FSUBD F5, F4, F6 // F6 = r = hi - lo137 FMULD F6, F6, F7 // F7 = t = r * r138 // compute y139 FMOVD $P5, F8 // F8 = P5140 FMOVD $P4, F9 // F9 = P4141 FMADDD F7, F9, F8, F13 // P4+t*P5142 FMOVD $P3, F10 // F10 = P3143 FMADDD F7, F10, F13, F13 // P3+t*(P4+t*P5)144 FMOVD $P2, F11 // F11 = P2145 FMADDD F7, F11, F13, F13 // P2+t*(P3+t*(P4+t*P5))146 FMOVD $P1, F12 // F12 = P1147 FMADDD F7, F12, F13, F13 // P1+t*(P2+t*(P3+t*(P4+t*P5)))148 FMSUBD F7, F6, F13, F13 // F13 = c = r - t*(P1+t*(P2+t*(P3+t*(P4+t*P5))))149 FMOVD $2.0, F14150 FSUBD F13, F14151 FMULD F6, F13, F15152 FDIVD F14, F15 // F15 = (r*c)/(2-c)153 FMOVD $1.0, F1 // F1 = 1.0154 FSUBD F15, F5, F15 // lo-(r*c)/(2-c)155 FSUBD F4, F15, F15 // (lo-(r*c)/(2-c))-hi156 FSUBD F15, F1, F16 // F16 = y = 1-((lo-(r*c)/(2-c))-hi)157 // inline Ldexp(y, k), benefit:158 // 1, no parameter pass overhead.159 // 2, skip unnecessary checks for Inf/NaN/Zero160 FMOVD F16, R0161 AND $FracMask, R0, R2 // fraction162 LSR $52, R0, R5 // exponent163 ADD R1, R5 // R1 = int(k)164 CMP $1, R5165 BGE normal166 ADD $52, R5 // denormal167 MOVD $C1, R8168 FMOVD R8, F1 // m = 2**-52169normal:170 ORR R5<<52, R2, R0171 FMOVD R0, F0172 FMULD F1, F0 // return m * x173isNaN:174 FMOVD F0, ret+8(FP)175 RET176underflow:177 MOVD ZR, ret+8(FP)178 RET179overflow:180 MOVD $PosInf, R0181 MOVD R0, ret+8(FP)182 RET183 