1*f3087befSAndrew Turner /*
2*f3087befSAndrew Turner * Double-precision SVE asinh(x) function.
3*f3087befSAndrew Turner *
4*f3087befSAndrew Turner * Copyright (c) 2022-2024, Arm Limited.
5*f3087befSAndrew Turner * SPDX-License-Identifier: MIT OR Apache-2.0 WITH LLVM-exception
6*f3087befSAndrew Turner */
7*f3087befSAndrew Turner
8*f3087befSAndrew Turner #include "sv_math.h"
9*f3087befSAndrew Turner #include "test_sig.h"
10*f3087befSAndrew Turner #include "test_defs.h"
11*f3087befSAndrew Turner
12*f3087befSAndrew Turner #define SignMask (0x8000000000000000)
13*f3087befSAndrew Turner #define One (0x3ff0000000000000)
14*f3087befSAndrew Turner #define Thres (0x5fe0000000000000) /* asuint64 (0x1p511). */
15*f3087befSAndrew Turner #define IndexMask (((1 << V_LOG_TABLE_BITS) - 1) << 1)
16*f3087befSAndrew Turner
17*f3087befSAndrew Turner static const struct data
18*f3087befSAndrew Turner {
19*f3087befSAndrew Turner double even_coeffs[9];
20*f3087befSAndrew Turner double ln2, p3, p1, p4, p0, p2, c1, c3, c5, c7, c9, c11, c13, c15, c17;
21*f3087befSAndrew Turner uint64_t off, mask;
22*f3087befSAndrew Turner
23*f3087befSAndrew Turner } data = {
24*f3087befSAndrew Turner /* Polynomial generated using Remez on [2^-26, 1]. */
25*f3087befSAndrew Turner .even_coeffs ={
26*f3087befSAndrew Turner -0x1.55555555554a7p-3,
27*f3087befSAndrew Turner -0x1.6db6db68332e6p-5,
28*f3087befSAndrew Turner -0x1.6e8b8b654a621p-6,
29*f3087befSAndrew Turner -0x1.c9871d10885afp-7,
30*f3087befSAndrew Turner -0x1.3ddca533e9f54p-7,
31*f3087befSAndrew Turner -0x1.b90c7099dd397p-8,
32*f3087befSAndrew Turner -0x1.d217026a669ecp-9,
33*f3087befSAndrew Turner -0x1.e0f37daef9127p-11,
34*f3087befSAndrew Turner -0x1.021a48685e287p-14, },
35*f3087befSAndrew Turner
36*f3087befSAndrew Turner .c1 = 0x1.3333333326c7p-4,
37*f3087befSAndrew Turner .c3 = 0x1.f1c71b26fb40dp-6,
38*f3087befSAndrew Turner .c5 = 0x1.1c4daa9e67871p-6,
39*f3087befSAndrew Turner .c7 = 0x1.7a16e8d9d2ecfp-7,
40*f3087befSAndrew Turner .c9 = 0x1.0becef748dafcp-7,
41*f3087befSAndrew Turner .c11 = 0x1.541f2bb1ffe51p-8,
42*f3087befSAndrew Turner .c13 = 0x1.0b5c7977aaf7p-9,
43*f3087befSAndrew Turner .c15 = 0x1.388b5fe542a6p-12,
44*f3087befSAndrew Turner .c17 = 0x1.93d4ba83d34dap-18,
45*f3087befSAndrew Turner
46*f3087befSAndrew Turner .ln2 = 0x1.62e42fefa39efp-1,
47*f3087befSAndrew Turner .p0 = -0x1.ffffffffffff7p-2,
48*f3087befSAndrew Turner .p1 = 0x1.55555555170d4p-2,
49*f3087befSAndrew Turner .p2 = -0x1.0000000399c27p-2,
50*f3087befSAndrew Turner .p3 = 0x1.999b2e90e94cap-3,
51*f3087befSAndrew Turner .p4 = -0x1.554e550bd501ep-3,
52*f3087befSAndrew Turner .off = 0x3fe6900900000000,
53*f3087befSAndrew Turner .mask = 0xfffULL << 52,
54*f3087befSAndrew Turner };
55*f3087befSAndrew Turner
56*f3087befSAndrew Turner static svfloat64_t NOINLINE
special_case(svfloat64_t x,svfloat64_t y,svbool_t special)57*f3087befSAndrew Turner special_case (svfloat64_t x, svfloat64_t y, svbool_t special)
58*f3087befSAndrew Turner {
59*f3087befSAndrew Turner return sv_call_f64 (asinh, x, y, special);
60*f3087befSAndrew Turner }
61*f3087befSAndrew Turner
62*f3087befSAndrew Turner static inline svfloat64_t
__sv_log_inline(svfloat64_t x,const struct data * d,const svbool_t pg)63*f3087befSAndrew Turner __sv_log_inline (svfloat64_t x, const struct data *d, const svbool_t pg)
64*f3087befSAndrew Turner {
65*f3087befSAndrew Turner /* Double-precision SVE log, copied from SVE log implementation with some
66*f3087befSAndrew Turner cosmetic modification and special-cases removed. See that file for details
67*f3087befSAndrew Turner of the algorithm used. */
68*f3087befSAndrew Turner
69*f3087befSAndrew Turner svuint64_t ix = svreinterpret_u64 (x);
70*f3087befSAndrew Turner svuint64_t i_off = svsub_x (pg, ix, d->off);
71*f3087befSAndrew Turner svuint64_t i
72*f3087befSAndrew Turner = svand_x (pg, svlsr_x (pg, i_off, (51 - V_LOG_TABLE_BITS)), IndexMask);
73*f3087befSAndrew Turner svuint64_t iz = svsub_x (pg, ix, svand_x (pg, i_off, d->mask));
74*f3087befSAndrew Turner svfloat64_t z = svreinterpret_f64 (iz);
75*f3087befSAndrew Turner
76*f3087befSAndrew Turner svfloat64_t invc = svld1_gather_index (pg, &__v_log_data.table[0].invc, i);
77*f3087befSAndrew Turner svfloat64_t logc = svld1_gather_index (pg, &__v_log_data.table[0].logc, i);
78*f3087befSAndrew Turner
79*f3087befSAndrew Turner svfloat64_t ln2_p3 = svld1rq (svptrue_b64 (), &d->ln2);
80*f3087befSAndrew Turner svfloat64_t p1_p4 = svld1rq (svptrue_b64 (), &d->p1);
81*f3087befSAndrew Turner
82*f3087befSAndrew Turner svfloat64_t r = svmla_x (pg, sv_f64 (-1.0), invc, z);
83*f3087befSAndrew Turner svfloat64_t kd
84*f3087befSAndrew Turner = svcvt_f64_x (pg, svasr_x (pg, svreinterpret_s64 (i_off), 52));
85*f3087befSAndrew Turner
86*f3087befSAndrew Turner svfloat64_t hi = svmla_lane (svadd_x (pg, logc, r), kd, ln2_p3, 0);
87*f3087befSAndrew Turner svfloat64_t r2 = svmul_x (svptrue_b64 (), r, r);
88*f3087befSAndrew Turner svfloat64_t y = svmla_lane (sv_f64 (d->p2), r, ln2_p3, 1);
89*f3087befSAndrew Turner svfloat64_t p = svmla_lane (sv_f64 (d->p0), r, p1_p4, 0);
90*f3087befSAndrew Turner
91*f3087befSAndrew Turner y = svmla_lane (y, r2, p1_p4, 1);
92*f3087befSAndrew Turner y = svmla_x (pg, p, r2, y);
93*f3087befSAndrew Turner y = svmla_x (pg, hi, r2, y);
94*f3087befSAndrew Turner return y;
95*f3087befSAndrew Turner }
96*f3087befSAndrew Turner
97*f3087befSAndrew Turner /* Double-precision implementation of SVE asinh(x).
98*f3087befSAndrew Turner asinh is very sensitive around 1, so it is impractical to devise a single
99*f3087befSAndrew Turner low-cost algorithm which is sufficiently accurate on a wide range of input.
100*f3087befSAndrew Turner Instead we use two different algorithms:
101*f3087befSAndrew Turner asinh(x) = sign(x) * log(|x| + sqrt(x^2 + 1) if |x| >= 1
102*f3087befSAndrew Turner = sign(x) * (|x| + |x|^3 * P(x^2)) otherwise
103*f3087befSAndrew Turner where log(x) is an optimized log approximation, and P(x) is a polynomial
104*f3087befSAndrew Turner shared with the scalar routine. The greatest observed error 2.51 ULP, in
105*f3087befSAndrew Turner |x| >= 1:
106*f3087befSAndrew Turner _ZGVsMxv_asinh(0x1.170469d024505p+0) got 0x1.e3181c43b0f36p-1
107*f3087befSAndrew Turner want 0x1.e3181c43b0f39p-1. */
SV_NAME_D1(asinh)108*f3087befSAndrew Turner svfloat64_t SV_NAME_D1 (asinh) (svfloat64_t x, const svbool_t pg)
109*f3087befSAndrew Turner {
110*f3087befSAndrew Turner const struct data *d = ptr_barrier (&data);
111*f3087befSAndrew Turner
112*f3087befSAndrew Turner svuint64_t ix = svreinterpret_u64 (x);
113*f3087befSAndrew Turner svuint64_t iax = svbic_x (pg, ix, SignMask);
114*f3087befSAndrew Turner svuint64_t sign = svand_x (pg, ix, SignMask);
115*f3087befSAndrew Turner svfloat64_t ax = svreinterpret_f64 (iax);
116*f3087befSAndrew Turner svbool_t ge1 = svcmpge (pg, iax, One);
117*f3087befSAndrew Turner svbool_t special = svcmpge (pg, iax, Thres);
118*f3087befSAndrew Turner
119*f3087befSAndrew Turner /* Option 1: |x| >= 1.
120*f3087befSAndrew Turner Compute asinh(x) according by asinh(x) = log(x + sqrt(x^2 + 1)). */
121*f3087befSAndrew Turner svfloat64_t option_1 = sv_f64 (0);
122*f3087befSAndrew Turner if (likely (svptest_any (pg, ge1)))
123*f3087befSAndrew Turner {
124*f3087befSAndrew Turner svfloat64_t x2 = svmul_x (svptrue_b64 (), ax, ax);
125*f3087befSAndrew Turner option_1 = __sv_log_inline (
126*f3087befSAndrew Turner svadd_x (pg, ax, svsqrt_x (pg, svadd_x (pg, x2, 1))), d, pg);
127*f3087befSAndrew Turner }
128*f3087befSAndrew Turner
129*f3087befSAndrew Turner /* Option 2: |x| < 1.
130*f3087befSAndrew Turner Compute asinh(x) using a polynomial.
131*f3087befSAndrew Turner The largest observed error in this region is 1.51 ULPs:
132*f3087befSAndrew Turner _ZGVsMxv_asinh(0x1.fe12bf8c616a2p-1) got 0x1.c1e649ee2681bp-1
133*f3087befSAndrew Turner want 0x1.c1e649ee2681dp-1. */
134*f3087befSAndrew Turner
135*f3087befSAndrew Turner svfloat64_t option_2 = sv_f64 (0);
136*f3087befSAndrew Turner if (likely (svptest_any (pg, svnot_z (pg, ge1))))
137*f3087befSAndrew Turner {
138*f3087befSAndrew Turner svfloat64_t x2 = svmul_x (svptrue_b64 (), ax, ax);
139*f3087befSAndrew Turner svfloat64_t x4 = svmul_x (svptrue_b64 (), x2, x2);
140*f3087befSAndrew Turner /* Order-17 Pairwise Horner scheme. */
141*f3087befSAndrew Turner svfloat64_t c13 = svld1rq (svptrue_b64 (), &d->c1);
142*f3087befSAndrew Turner svfloat64_t c57 = svld1rq (svptrue_b64 (), &d->c5);
143*f3087befSAndrew Turner svfloat64_t c911 = svld1rq (svptrue_b64 (), &d->c9);
144*f3087befSAndrew Turner svfloat64_t c1315 = svld1rq (svptrue_b64 (), &d->c13);
145*f3087befSAndrew Turner
146*f3087befSAndrew Turner svfloat64_t p01 = svmla_lane (sv_f64 (d->even_coeffs[0]), x2, c13, 0);
147*f3087befSAndrew Turner svfloat64_t p23 = svmla_lane (sv_f64 (d->even_coeffs[1]), x2, c13, 1);
148*f3087befSAndrew Turner svfloat64_t p45 = svmla_lane (sv_f64 (d->even_coeffs[2]), x2, c57, 0);
149*f3087befSAndrew Turner svfloat64_t p67 = svmla_lane (sv_f64 (d->even_coeffs[3]), x2, c57, 1);
150*f3087befSAndrew Turner svfloat64_t p89 = svmla_lane (sv_f64 (d->even_coeffs[4]), x2, c911, 0);
151*f3087befSAndrew Turner svfloat64_t p1011 = svmla_lane (sv_f64 (d->even_coeffs[5]), x2, c911, 1);
152*f3087befSAndrew Turner svfloat64_t p1213
153*f3087befSAndrew Turner = svmla_lane (sv_f64 (d->even_coeffs[6]), x2, c1315, 0);
154*f3087befSAndrew Turner svfloat64_t p1415
155*f3087befSAndrew Turner = svmla_lane (sv_f64 (d->even_coeffs[7]), x2, c1315, 1);
156*f3087befSAndrew Turner svfloat64_t p1617 = svmla_x (pg, sv_f64 (d->even_coeffs[8]), x2, d->c17);
157*f3087befSAndrew Turner
158*f3087befSAndrew Turner svfloat64_t p = svmla_x (pg, p1415, x4, p1617);
159*f3087befSAndrew Turner p = svmla_x (pg, p1213, x4, p);
160*f3087befSAndrew Turner p = svmla_x (pg, p1011, x4, p);
161*f3087befSAndrew Turner p = svmla_x (pg, p89, x4, p);
162*f3087befSAndrew Turner
163*f3087befSAndrew Turner p = svmla_x (pg, p67, x4, p);
164*f3087befSAndrew Turner p = svmla_x (pg, p45, x4, p);
165*f3087befSAndrew Turner
166*f3087befSAndrew Turner p = svmla_x (pg, p23, x4, p);
167*f3087befSAndrew Turner
168*f3087befSAndrew Turner p = svmla_x (pg, p01, x4, p);
169*f3087befSAndrew Turner
170*f3087befSAndrew Turner option_2 = svmla_x (pg, ax, p, svmul_x (svptrue_b64 (), x2, ax));
171*f3087befSAndrew Turner }
172*f3087befSAndrew Turner
173*f3087befSAndrew Turner if (unlikely (svptest_any (pg, special)))
174*f3087befSAndrew Turner return special_case (
175*f3087befSAndrew Turner x,
176*f3087befSAndrew Turner svreinterpret_f64 (sveor_x (
177*f3087befSAndrew Turner pg, svreinterpret_u64 (svsel (ge1, option_1, option_2)), sign)),
178*f3087befSAndrew Turner special);
179*f3087befSAndrew Turner
180*f3087befSAndrew Turner /* Choose the right option for each lane. */
181*f3087befSAndrew Turner svfloat64_t y = svsel (ge1, option_1, option_2);
182*f3087befSAndrew Turner return svreinterpret_f64 (sveor_x (pg, svreinterpret_u64 (y), sign));
183*f3087befSAndrew Turner }
184*f3087befSAndrew Turner
185*f3087befSAndrew Turner TEST_SIG (SV, D, 1, asinh, -10.0, 10.0)
186*f3087befSAndrew Turner TEST_ULP (SV_NAME_D1 (asinh), 2.52)
187*f3087befSAndrew Turner TEST_DISABLE_FENV (SV_NAME_D1 (asinh))
188*f3087befSAndrew Turner TEST_SYM_INTERVAL (SV_NAME_D1 (asinh), 0, 0x1p-26, 50000)
189*f3087befSAndrew Turner TEST_SYM_INTERVAL (SV_NAME_D1 (asinh), 0x1p-26, 1, 50000)
190*f3087befSAndrew Turner TEST_SYM_INTERVAL (SV_NAME_D1 (asinh), 1, 0x1p511, 50000)
191*f3087befSAndrew Turner TEST_SYM_INTERVAL (SV_NAME_D1 (asinh), 0x1p511, inf, 40000)
192*f3087befSAndrew Turner /* Test vector asinh 3 times, with control lane < 1, > 1 and special.
193*f3087befSAndrew Turner Ensures the v_sel is choosing the right option in all cases. */
194*f3087befSAndrew Turner TEST_CONTROL_VALUE (SV_NAME_D1 (asinh), 0.5)
195*f3087befSAndrew Turner TEST_CONTROL_VALUE (SV_NAME_D1 (asinh), 2)
196*f3087befSAndrew Turner TEST_CONTROL_VALUE (SV_NAME_D1 (asinh), 0x1p600)
197*f3087befSAndrew Turner CLOSE_SVE_ATTR
198