diff options
Diffstat (limited to 'pl/math/atanf_common.h')
| -rw-r--r-- | pl/math/atanf_common.h | 33 |
1 files changed, 10 insertions, 23 deletions
diff --git a/pl/math/atanf_common.h b/pl/math/atanf_common.h index 37ca76dee2f7..8952e7e0078b 100644 --- a/pl/math/atanf_common.h +++ b/pl/math/atanf_common.h @@ -1,5 +1,5 @@ /* - * Single-precision polynomial evaluation function for scalar and vector + * Single-precision polynomial evaluation function for scalar * atan(x) and atan2(y,x). * * Copyright (c) 2021-2023, Arm Limited. @@ -10,26 +10,12 @@ #define PL_MATH_ATANF_COMMON_H #include "math_config.h" -#include "estrinf.h" - -#if V_SUPPORTED - -#include "v_math.h" - -#define FLT_T v_f32_t -#define P(i) v_f32 (__atanf_poly_data.poly[i]) - -#else - -#define FLT_T float -#define P(i) __atanf_poly_data.poly[i] - -#endif +#include "poly_scalar_f32.h" /* Polynomial used in fast atanf(x) and atan2f(y,x) implementations The order 7 polynomial P approximates (atan(sqrt(x))-sqrt(x))/x^(3/2). */ -static inline FLT_T -eval_poly (FLT_T z, FLT_T az, FLT_T shift) +static inline float +eval_poly (float z, float az, float shift) { /* Use 2-level Estrin scheme for P(z^2) with deg(P)=7. However, a standard implementation using z8 creates spurious underflow @@ -37,15 +23,16 @@ eval_poly (FLT_T z, FLT_T az, FLT_T shift) Therefore, we split the last fma into a mul and and an fma. Horner and single-level Estrin have higher errors that exceed threshold. */ - FLT_T z2 = z * z; - FLT_T z4 = z2 * z2; + float z2 = z * z; + float z4 = z2 * z2; /* Then assemble polynomial. */ - FLT_T y = FMA (z4, z4 * ESTRIN_3_ (z2, z4, P, 4), ESTRIN_3 (z2, z4, P)); - + float y = fmaf ( + z4, z4 * pairwise_poly_3_f32 (z2, z4, __atanf_poly_data.poly + 4), + pairwise_poly_3_f32 (z2, z4, __atanf_poly_data.poly)); /* Finalize: y = shift + z * P(z^2). */ - return FMA (y, z2 * az, az) + shift; + return fmaf (y, z2 * az, az) + shift; } #endif // PL_MATH_ATANF_COMMON_H |
