math: fix rare underflow issue in fma
the issue is described in commits1e5eb73545andffd8ac2dd5
This commit is contained in:
parent
4b539a826b
commit
8f438115f2
3 changed files with 55 additions and 13 deletions
|
|
@ -431,12 +431,24 @@ double fma(double x, double y, double z)
|
||||||
/*
|
/*
|
||||||
* There is no need to worry about double rounding in directed
|
* There is no need to worry about double rounding in directed
|
||||||
* rounding modes.
|
* rounding modes.
|
||||||
* TODO: underflow is not raised properly, example in downward rounding:
|
* But underflow may not be raised properly, example in downward rounding:
|
||||||
* fma(0x1.000000001p-1000, 0x1.000000001p-30, -0x1p-1066)
|
* fma(0x1.000000001p-1000, 0x1.000000001p-30, -0x1p-1066)
|
||||||
*/
|
*/
|
||||||
|
double ret;
|
||||||
|
#if defined(FE_INEXACT) && defined(FE_UNDERFLOW)
|
||||||
|
int e = fetestexcept(FE_INEXACT);
|
||||||
|
feclearexcept(FE_INEXACT);
|
||||||
|
#endif
|
||||||
fesetround(oround);
|
fesetround(oround);
|
||||||
adj = r.lo + xy.lo;
|
adj = r.lo + xy.lo;
|
||||||
return scalbn(r.hi + adj, spread);
|
ret = scalbn(r.hi + adj, spread);
|
||||||
|
#if defined(FE_INEXACT) && defined(FE_UNDERFLOW)
|
||||||
|
if (ilogb(ret) < -1022 && fetestexcept(FE_INEXACT))
|
||||||
|
feraiseexcept(FE_UNDERFLOW);
|
||||||
|
else if (e)
|
||||||
|
feraiseexcept(FE_INEXACT);
|
||||||
|
#endif
|
||||||
|
return ret;
|
||||||
}
|
}
|
||||||
|
|
||||||
adj = add_adjusted(r.lo, xy.lo);
|
adj = add_adjusted(r.lo, xy.lo);
|
||||||
|
|
|
||||||
|
|
@ -26,7 +26,8 @@
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <fenv.h>
|
#include <fenv.h>
|
||||||
#include "libm.h"
|
#include <math.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Fused multiply-add: Compute x * y + z with a single rounding error.
|
* Fused multiply-add: Compute x * y + z with a single rounding error.
|
||||||
|
|
@ -39,21 +40,35 @@ float fmaf(float x, float y, float z)
|
||||||
{
|
{
|
||||||
#pragma STDC FENV_ACCESS ON
|
#pragma STDC FENV_ACCESS ON
|
||||||
double xy, result;
|
double xy, result;
|
||||||
uint32_t hr, lr;
|
union {double f; uint64_t i;} u;
|
||||||
|
int e;
|
||||||
|
|
||||||
xy = (double)x * y;
|
xy = (double)x * y;
|
||||||
result = xy + z;
|
result = xy + z;
|
||||||
EXTRACT_WORDS(hr, lr, result);
|
u.f = result;
|
||||||
|
e = u.i>>52 & 0x7ff;
|
||||||
/* Common case: The double precision result is fine. */
|
/* Common case: The double precision result is fine. */
|
||||||
if ((lr & 0x1fffffff) != 0x10000000 || /* not a halfway case */
|
if ((u.i & 0x1fffffff) != 0x10000000 || /* not a halfway case */
|
||||||
(hr & 0x7ff00000) == 0x7ff00000 || /* NaN */
|
e == 0x7ff || /* NaN */
|
||||||
result - xy == z || /* exact */
|
result - xy == z || /* exact */
|
||||||
fegetround() != FE_TONEAREST) /* not round-to-nearest */
|
fegetround() != FE_TONEAREST) /* not round-to-nearest */
|
||||||
{
|
{
|
||||||
/*
|
/*
|
||||||
TODO: underflow is not raised correctly, example in
|
underflow may not be raised correctly, example:
|
||||||
downward rouding: fmaf(0x1p-120f, 0x1p-120f, 0x1p-149f)
|
fmaf(0x1p-120f, 0x1p-120f, 0x1p-149f)
|
||||||
*/
|
*/
|
||||||
|
#if defined(FE_INEXACT) && defined(FE_UNDERFLOW)
|
||||||
|
if (e < 0x3ff-126 && e >= 0x3ff-149 && fetestexcept(FE_INEXACT)) {
|
||||||
|
feclearexcept(FE_INEXACT);
|
||||||
|
/* TODO: gcc and clang bug workaround */
|
||||||
|
volatile float vz = z;
|
||||||
|
result = xy + vz;
|
||||||
|
if (fetestexcept(FE_INEXACT))
|
||||||
|
feraiseexcept(FE_UNDERFLOW);
|
||||||
|
else
|
||||||
|
feraiseexcept(FE_INEXACT);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
z = result;
|
z = result;
|
||||||
return z;
|
return z;
|
||||||
}
|
}
|
||||||
|
|
@ -68,8 +83,11 @@ float fmaf(float x, float y, float z)
|
||||||
volatile double vxy = xy; /* XXX work around gcc CSE bug */
|
volatile double vxy = xy; /* XXX work around gcc CSE bug */
|
||||||
double adjusted_result = vxy + z;
|
double adjusted_result = vxy + z;
|
||||||
fesetround(FE_TONEAREST);
|
fesetround(FE_TONEAREST);
|
||||||
if (result == adjusted_result)
|
if (result == adjusted_result) {
|
||||||
SET_LOW_WORD(adjusted_result, lr + 1);
|
u.f = adjusted_result;
|
||||||
|
u.i++;
|
||||||
|
adjusted_result = u.f;
|
||||||
|
}
|
||||||
z = adjusted_result;
|
z = adjusted_result;
|
||||||
return z;
|
return z;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -264,12 +264,24 @@ long double fmal(long double x, long double y, long double z)
|
||||||
/*
|
/*
|
||||||
* There is no need to worry about double rounding in directed
|
* There is no need to worry about double rounding in directed
|
||||||
* rounding modes.
|
* rounding modes.
|
||||||
* TODO: underflow is not raised correctly, example in downward rounding:
|
* But underflow may not be raised correctly, example in downward rounding:
|
||||||
* fmal(0x1.0000000001p-16000L, 0x1.0000000001p-400L, -0x1p-16440L)
|
* fmal(0x1.0000000001p-16000L, 0x1.0000000001p-400L, -0x1p-16440L)
|
||||||
*/
|
*/
|
||||||
|
long double ret;
|
||||||
|
#if defined(FE_INEXACT) && defined(FE_UNDERFLOW)
|
||||||
|
int e = fetestexcept(FE_INEXACT);
|
||||||
|
feclearexcept(FE_INEXACT);
|
||||||
|
#endif
|
||||||
fesetround(oround);
|
fesetround(oround);
|
||||||
adj = r.lo + xy.lo;
|
adj = r.lo + xy.lo;
|
||||||
return scalbnl(r.hi + adj, spread);
|
ret = scalbnl(r.hi + adj, spread);
|
||||||
|
#if defined(FE_INEXACT) && defined(FE_UNDERFLOW)
|
||||||
|
if (ilogbl(ret) < -16382 && fetestexcept(FE_INEXACT))
|
||||||
|
feraiseexcept(FE_UNDERFLOW);
|
||||||
|
else if (e)
|
||||||
|
feraiseexcept(FE_INEXACT);
|
||||||
|
#endif
|
||||||
|
return ret;
|
||||||
}
|
}
|
||||||
|
|
||||||
adj = add_adjusted(r.lo, xy.lo);
|
adj = add_adjusted(r.lo, xy.lo);
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue