Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
expm1.go245 linesDownload Raw Back to math
1// Copyright 2010 The Go Authors. All rights reserved.2// Use of this source code is governed by a BSD-style3// license that can be found in the LICENSE file.4 5package math6 7// The original C code, the long comment, and the constants8// below are from FreeBSD's /usr/src/lib/msun/src/s_expm1.c9// and came with this notice. The go code is a simplified10// version of the original C.11//12// ====================================================13// Copyright (C) 1993 by Sun Microsystems, Inc. All rights reserved.14//15// Developed at SunPro, a Sun Microsystems, Inc. business.16// Permission to use, copy, modify, and distribute this17// software is freely granted, provided that this notice18// is preserved.19// ====================================================20//21// expm1(x)22// Returns exp(x)-1, the exponential of x minus 1.23//24// Method25//   1. Argument reduction:26//      Given x, find r and integer k such that27//28//               x = k*ln2 + r,  |r| <= 0.5*ln2 ~ 0.3465829//30//      Here a correction term c will be computed to compensate31//      the error in r when rounded to a floating-point number.32//33//   2. Approximating expm1(r) by a special rational function on34//      the interval [0,0.34658]:35//      Since36//          r*(exp(r)+1)/(exp(r)-1) = 2+ r**2/6 - r**4/360 + ...37//      we define R1(r*r) by38//          r*(exp(r)+1)/(exp(r)-1) = 2+ r**2/6 * R1(r*r)39//      That is,40//          R1(r**2) = 6/r *((exp(r)+1)/(exp(r)-1) - 2/r)41//                   = 6/r * ( 1 + 2.0*(1/(exp(r)-1) - 1/r))42//                   = 1 - r**2/60 + r**4/2520 - r**6/100800 + ...43//      We use a special Reme algorithm on [0,0.347] to generate44//      a polynomial of degree 5 in r*r to approximate R1. The45//      maximum error of this polynomial approximation is bounded46//      by 2**-61. In other words,47//          R1(z) ~ 1.0 + Q1*z + Q2*z**2 + Q3*z**3 + Q4*z**4 + Q5*z**548//      where   Q1  =  -1.6666666666666567384E-2,49//              Q2  =   3.9682539681370365873E-4,50//              Q3  =  -9.9206344733435987357E-6,51//              Q4  =   2.5051361420808517002E-7,52//              Q5  =  -6.2843505682382617102E-9;53//      (where z=r*r, and the values of Q1 to Q5 are listed below)54//      with error bounded by55//          |                  5           |     -6156//          | 1.0+Q1*z+...+Q5*z   -  R1(z) | <= 257//          |                              |58//59//      expm1(r) = exp(r)-1 is then computed by the following60//      specific way which minimize the accumulation rounding error:61//                             2     362//                            r     r    [ 3 - (R1 + R1*r/2)  ]63//            expm1(r) = r + --- + --- * [--------------------]64//                            2     2    [ 6 - r*(3 - R1*r/2) ]65//66//      To compensate the error in the argument reduction, we use67//              expm1(r+c) = expm1(r) + c + expm1(r)*c68//                         ~ expm1(r) + c + r*c69//      Thus c+r*c will be added in as the correction terms for70//      expm1(r+c). Now rearrange the term to avoid optimization71//      screw up:72//                      (      2                                    2 )73//                      ({  ( r    [ R1 -  (3 - R1*r/2) ]  )  }    r  )74//       expm1(r+c)~r - ({r*(--- * [--------------------]-c)-c} - --- )75//                      ({  ( 2    [ 6 - r*(3 - R1*r/2) ]  )  }    2  )76//                      (                                             )77//78//                 = r - E79//   3. Scale back to obtain expm1(x):80//      From step 1, we have81//         expm1(x) = either 2**k*[expm1(r)+1] - 182//                  = or     2**k*[expm1(r) + (1-2**-k)]83//   4. Implementation notes:84//      (A). To save one multiplication, we scale the coefficient Qi85//           to Qi*2**i, and replace z by (x**2)/2.86//      (B). To achieve maximum accuracy, we compute expm1(x) by87//        (i)   if x < -56*ln2, return -1.0, (raise inexact if x!=inf)88//        (ii)  if k=0, return r-E89//        (iii) if k=-1, return 0.5*(r-E)-0.590//        (iv)  if k=1 if r < -0.25, return 2*((r+0.5)- E)91//                     else          return  1.0+2.0*(r-E);92//        (v)   if (k<-2||k>56) return 2**k(1-(E-r)) - 1 (or exp(x)-1)93//        (vi)  if k <= 20, return 2**k((1-2**-k)-(E-r)), else94//        (vii) return 2**k(1-((E+2**-k)-r))95//96// Special cases:97//      expm1(INF) is INF, expm1(NaN) is NaN;98//      expm1(-INF) is -1, and99//      for finite argument, only expm1(0)=0 is exact.100//101// Accuracy:102//      according to an error analysis, the error is always less than103//      1 ulp (unit in the last place).104//105// Misc. info.106//      For IEEE double107//          if x >  7.09782712893383973096e+02 then expm1(x) overflow108//109// Constants:110// The hexadecimal values are the intended ones for the following111// constants. The decimal values may be used, provided that the112// compiler will convert from decimal to binary accurately enough113// to produce the hexadecimal values shown.114//115 116// Expm1 returns e**x - 1, the base-e exponential of x minus 1.117// It is more accurate than [Exp](x) - 1 when x is near zero.118//119// Special cases are:120//121//	Expm1(+Inf) = +Inf122//	Expm1(-Inf) = -1123//	Expm1(NaN) = NaN124//125// Very large values overflow to -1 or +Inf.126func Expm1(x float64) float64 {127	if haveArchExpm1 {128		return archExpm1(x)129	}130	return expm1(x)131}132 133func expm1(x float64) float64 {134	const (135		Othreshold = 7.09782712893383973096e+02 // 0x40862E42FEFA39EF136		Ln2X56     = 3.88162421113569373274e+01 // 0x4043687a9f1af2b1137		Ln2HalfX3  = 1.03972077083991796413e+00 // 0x3ff0a2b23f3bab73138		Ln2Half    = 3.46573590279972654709e-01 // 0x3fd62e42fefa39ef139		Ln2Hi      = 6.93147180369123816490e-01 // 0x3fe62e42fee00000140		Ln2Lo      = 1.90821492927058770002e-10 // 0x3dea39ef35793c76141		InvLn2     = 1.44269504088896338700e+00 // 0x3ff71547652b82fe142		Tiny       = 1.0 / (1 << 54)            // 2**-54 = 0x3c90000000000000143		// scaled coefficients related to expm1144		Q1 = -3.33333333333331316428e-02 // 0xBFA11111111110F4145		Q2 = 1.58730158725481460165e-03  // 0x3F5A01A019FE5585146		Q3 = -7.93650757867487942473e-05 // 0xBF14CE199EAADBB7147		Q4 = 4.00821782732936239552e-06  // 0x3ED0CFCA86E65239148		Q5 = -2.01099218183624371326e-07 // 0xBE8AFDB76E09C32D149	)150 151	// special cases152	switch {153	case IsInf(x, 1) || IsNaN(x):154		return x155	case IsInf(x, -1):156		return -1157	}158 159	absx := x160	sign := false161	if x < 0 {162		absx = -absx163		sign = true164	}165 166	// filter out huge argument167	if absx >= Ln2X56 { // if |x| >= 56 * ln2168		if sign {169			return -1 // x < -56*ln2, return -1170		}171		if absx >= Othreshold { // if |x| >= 709.78...172			return Inf(1)173		}174	}175 176	// argument reduction177	var c float64178	var k int179	if absx > Ln2Half { // if  |x| > 0.5 * ln2180		var hi, lo float64181		if absx < Ln2HalfX3 { // and |x| < 1.5 * ln2182			if !sign {183				hi = x - Ln2Hi184				lo = Ln2Lo185				k = 1186			} else {187				hi = x + Ln2Hi188				lo = -Ln2Lo189				k = -1190			}191		} else {192			if !sign {193				k = int(InvLn2*x + 0.5)194			} else {195				k = int(InvLn2*x - 0.5)196			}197			t := float64(k)198			hi = x - t*Ln2Hi // t * Ln2Hi is exact here199			lo = t * Ln2Lo200		}201		x = hi - lo202		c = (hi - x) - lo203	} else if absx < Tiny { // when |x| < 2**-54, return x204		return x205	} else {206		k = 0207	}208 209	// x is now in primary range210	hfx := 0.5 * x211	hxs := x * hfx212	r1 := 1 + hxs*(Q1+hxs*(Q2+hxs*(Q3+hxs*(Q4+hxs*Q5))))213	t := 3 - r1*hfx214	e := hxs * ((r1 - t) / (6.0 - x*t))215	if k == 0 {216		return x - (x*e - hxs) // c is 0217	}218	e = (x*(e-c) - c)219	e -= hxs220	switch {221	case k == -1:222		return 0.5*(x-e) - 0.5223	case k == 1:224		if x < -0.25 {225			return -2 * (e - (x + 0.5))226		}227		return 1 + 2*(x-e)228	case k <= -2 || k > 56: // suffice to return exp(x)-1229		y := 1 - (e - x)230		y = Float64frombits(Float64bits(y) + uint64(k)<<52) // add k to y's exponent231		return y - 1232	}233	if k < 20 {234		t := Float64frombits(0x3ff0000000000000 - (0x20000000000000 >> uint(k))) // t=1-2**-k235		y := t - (e - x)236		y = Float64frombits(Float64bits(y) + uint64(k)<<52) // add k to y's exponent237		return y238	}239	t = Float64frombits(uint64(0x3ff-k) << 52) // 2**-k240	y := x - (e + t)241	y++242	y = Float64frombits(Float64bits(y) + uint64(k)<<52) // add k to y's exponent243	return y244}245 
codekingpro/portable-devtools · Team Ai