diff options
Diffstat (limited to 'libc')
29 files changed, 963 insertions, 643 deletions
diff --git a/libc/include/math.yaml b/libc/include/math.yaml index 3044ec3..007be23 100644 --- a/libc/include/math.yaml +++ b/libc/include/math.yaml @@ -33,14 +33,14 @@ functions: return_type: float arguments: - type: float - name: acoshf16 + - name: acoshf16 standards: - stdc return_type: _Float16 arguments: - type: _Float16 guard: LIBC_TYPES_HAS_FLOAT16 - name: acospif16 + - name: acospif16 standards: - stdc return_type: _Float16 diff --git a/libc/shared/math.h b/libc/shared/math.h index 617e466..e3c674c 100644 --- a/libc/shared/math.h +++ b/libc/shared/math.h @@ -13,6 +13,10 @@ #include "math/acos.h" #include "math/acosf.h" +#include "math/acosf16.h" +#include "math/acoshf.h" +#include "math/acoshf16.h" +#include "math/erff.h" #include "math/exp.h" #include "math/exp10.h" #include "math/exp10f.h" diff --git a/libc/shared/math/acosf16.h b/libc/shared/math/acosf16.h new file mode 100644 index 0000000..aaf6ed9 --- /dev/null +++ b/libc/shared/math/acosf16.h @@ -0,0 +1,29 @@ +//===-- Shared acosf16 function ---------------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SHARED_MATH_ACOSF16_H +#define LLVM_LIBC_SHARED_MATH_ACOSF16_H + +#include "include/llvm-libc-macros/float16-macros.h" + +#ifdef LIBC_TYPES_HAS_FLOAT16 + +#include "shared/libc_common.h" +#include "src/__support/math/acosf16.h" + +namespace LIBC_NAMESPACE_DECL { +namespace shared { + +using math::acosf16; + +} // namespace shared +} // namespace LIBC_NAMESPACE_DECL + +#endif // LIBC_TYPES_HAS_FLOAT16 + +#endif // LLVM_LIBC_SHARED_MATH_ACOSF16_H diff --git a/libc/shared/math/acoshf.h b/libc/shared/math/acoshf.h new file mode 100644 index 0000000..86bdbce --- /dev/null +++ b/libc/shared/math/acoshf.h @@ -0,0 +1,23 @@ +//===-- Shared acoshf function ----------------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SHARED_MATH_ACOSHF_H +#define LLVM_LIBC_SHARED_MATH_ACOSHF_H + +#include "shared/libc_common.h" +#include "src/__support/math/acoshf.h" + +namespace LIBC_NAMESPACE_DECL { +namespace shared { + +using math::acoshf; + +} // namespace shared +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SHARED_MATH_ACOSHF_H diff --git a/libc/shared/math/acoshf16.h b/libc/shared/math/acoshf16.h new file mode 100644 index 0000000..2f0bc6e --- /dev/null +++ b/libc/shared/math/acoshf16.h @@ -0,0 +1,29 @@ +//===-- Shared acoshf16 function --------------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SHARED_MATH_ACOSHF16_H +#define LLVM_LIBC_SHARED_MATH_ACOSHF16_H + +#include "include/llvm-libc-macros/float16-macros.h" +#include "shared/libc_common.h" + +#ifdef LIBC_TYPES_HAS_FLOAT16 + +#include "src/__support/math/acoshf16.h" + +namespace LIBC_NAMESPACE_DECL { +namespace shared { + +using math::acoshf16; + +} // namespace shared +} // namespace LIBC_NAMESPACE_DECL + +#endif // LIBC_TYPES_HAS_FLOAT16 + +#endif // LLVM_LIBC_SHARED_MATH_ACOSHF16_H diff --git a/libc/shared/math/erff.h b/libc/shared/math/erff.h new file mode 100644 index 0000000..d0cca15 --- /dev/null +++ b/libc/shared/math/erff.h @@ -0,0 +1,23 @@ +//===-- Shared erff function ------------------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SHARED_MATH_ERFF_H +#define LLVM_LIBC_SHARED_MATH_ERFF_H + +#include "shared/libc_common.h" +#include "src/__support/math/erff.h" + +namespace LIBC_NAMESPACE_DECL { +namespace shared { + +using math::erff; + +} // namespace shared +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SHARED_MATH_ERFF_H diff --git a/libc/src/__support/math/CMakeLists.txt b/libc/src/__support/math/CMakeLists.txt index fbe7e2c..9a8a4d1 100644 --- a/libc/src/__support/math/CMakeLists.txt +++ b/libc/src/__support/math/CMakeLists.txt @@ -32,6 +32,70 @@ add_header_library( ) add_header_library( + acosf16 + HDRS + acosf16.h + DEPENDS + libc.src.__support.FPUtil.cast + libc.src.__support.FPUtil.except_value_utils + libc.src.__support.FPUtil.fenv_impl + libc.src.__support.FPUtil.fp_bits + libc.src.__support.FPUtil.multiply_add + libc.src.__support.FPUtil.polyeval + libc.src.__support.FPUtil.sqrt + libc.src.__support.macros.optimization + libc.src.__support.macros.properties.types +) + +add_header_library( + acosh_float_constants + HDRS + acosh_float_constants.h + DEPENDS + libc.src.__support.macros.config +) + +add_header_library( + acoshf_utils + HDRS + acoshf_utils.h + DEPENDS + .acosh_float_constants + libc.src.__support.FPUtil.fp_bits + libc.src.__support.FPUtil.multiply_add + libc.src.__support.FPUtil.polyeval +) + +add_header_library( + acoshf + HDRS + acoshf.h + DEPENDS + .acoshf_utils + libc.src.__support.FPUtil.fenv_impl + libc.src.__support.FPUtil.fp_bits + libc.src.__support.FPUtil.multiply_add + libc.src.__support.FPUtil.sqrt + libc.src.__support.macros.optimization +) + +add_header_library( + acoshf16 + HDRS + acoshf16.h + DEPENDS + .acoshf_utils + libc.src.__support.FPUtil.cast + libc.src.__support.FPUtil.except_value_utils + libc.src.__support.FPUtil.fenv_impl + libc.src.__support.FPUtil.fp_bits + libc.src.__support.FPUtil.multiply_add + libc.src.__support.FPUtil.polyeval + libc.src.__support.FPUtil.sqrt + libc.src.__support.macros.optimization +) + +add_header_library( asin_utils HDRS asin_utils.h @@ -46,6 +110,18 @@ add_header_library( ) add_header_library( + erff + HDRS + erff.h + DEPENDS + libc.src.__support.FPUtil.fp_bits + libc.src.__support.FPUtil.except_value_utils + libc.src.__support.FPUtil.multiply_add + libc.src.__support.FPUtil.polyeval + libc.src.__support.macros.optimization +) + +add_header_library( exp_float_constants HDRS exp_float_constants.h diff --git a/libc/src/__support/math/acos.h b/libc/src/__support/math/acos.h index 7c9fc76..a52ead7 100644 --- a/libc/src/__support/math/acos.h +++ b/libc/src/__support/math/acos.h @@ -26,7 +26,6 @@ namespace math { static constexpr double acos(double x) { using DoubleDouble = fputil::DoubleDouble; - using Float128 = fputil::DyadicFloat<128>; using namespace asin_internal; using FPBits = fputil::FPBits<double>; diff --git a/libc/src/__support/math/acosf16.h b/libc/src/__support/math/acosf16.h new file mode 100644 index 0000000..58d3761 --- /dev/null +++ b/libc/src/__support/math/acosf16.h @@ -0,0 +1,164 @@ +//===-- Implementation header for acosf16 -----------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ACOSF16_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ACOSF16_H + +#include "include/llvm-libc-macros/float16-macros.h" + +#ifdef LIBC_TYPES_HAS_FLOAT16 + +#include "src/__support/FPUtil/FEnvImpl.h" +#include "src/__support/FPUtil/FPBits.h" +#include "src/__support/FPUtil/PolyEval.h" +#include "src/__support/FPUtil/cast.h" +#include "src/__support/FPUtil/except_value_utils.h" +#include "src/__support/FPUtil/multiply_add.h" +#include "src/__support/FPUtil/sqrt.h" +#include "src/__support/macros/optimization.h" + +namespace LIBC_NAMESPACE_DECL { + +namespace math { + +static constexpr float16 acosf16(float16 x) { + + // Generated by Sollya using the following command: + // > round(pi/2, SG, RN); + // > round(pi, SG, RN); + constexpr float PI_OVER_2 = 0x1.921fb6p0f; + constexpr float PI = 0x1.921fb6p1f; + +#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS + constexpr size_t N_EXCEPTS = 2; + + constexpr fputil::ExceptValues<float16, N_EXCEPTS> ACOSF16_EXCEPTS{{ + // (input, RZ output, RU offset, RD offset, RN offset) + {0xacaf, 0x3e93, 1, 0, 0}, + {0xb874, 0x4052, 1, 0, 1}, + }}; +#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS + + using FPBits = fputil::FPBits<float16>; + FPBits xbits(x); + + uint16_t x_u = xbits.uintval(); + uint16_t x_abs = x_u & 0x7fff; + uint16_t x_sign = x_u >> 15; + + // |x| > 0x1p0, |x| > 1, or x is NaN. + if (LIBC_UNLIKELY(x_abs > 0x3c00)) { + // acosf16(NaN) = NaN + if (xbits.is_nan()) { + if (xbits.is_signaling_nan()) { + fputil::raise_except_if_required(FE_INVALID); + return FPBits::quiet_nan().get_val(); + } + + return x; + } + + // 1 < |x| <= +/-inf + fputil::raise_except_if_required(FE_INVALID); + fputil::set_errno_if_required(EDOM); + + return FPBits::quiet_nan().get_val(); + } + + float xf = x; + +#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS + // Handle exceptional values + if (auto r = ACOSF16_EXCEPTS.lookup(x_u); LIBC_UNLIKELY(r.has_value())) + return r.value(); +#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS + + // |x| == 0x1p0, x is 1 or -1 + // if x is (-)1, return pi, else + // if x is (+)1, return 0 + if (LIBC_UNLIKELY(x_abs == 0x3c00)) + return fputil::cast<float16>(x_sign ? PI : 0.0f); + + float xsq = xf * xf; + + // |x| <= 0x1p-1, |x| <= 0.5 + if (x_abs <= 0x3800) { + // if x is 0, return pi/2 + if (LIBC_UNLIKELY(x_abs == 0)) + return fputil::cast<float16>(PI_OVER_2); + + // Note that: acos(x) = pi/2 + asin(-x) = pi/2 - asin(x) + // Degree-6 minimax polynomial of asin(x) generated by Sollya with: + // > P = fpminimax(asin(x)/x, [|0, 2, 4, 6, 8|], [|SG...|], [0, 0.5]); + float interm = + fputil::polyeval(xsq, 0x1.000002p0f, 0x1.554c2ap-3f, 0x1.3541ccp-4f, + 0x1.43b2d6p-5f, 0x1.a0d73ep-5f); + return fputil::cast<float16>(fputil::multiply_add(-xf, interm, PI_OVER_2)); + } + + // When |x| > 0.5, assume that 0.5 < |x| <= 1 + // + // Step-by-step range-reduction proof: + // 1: Let y = asin(x), such that, x = sin(y) + // 2: From complimentary angle identity: + // x = sin(y) = cos(pi/2 - y) + // 3: Let z = pi/2 - y, such that x = cos(z) + // 4: From double angle formula; cos(2A) = 1 - 2 * sin^2(A): + // z = 2A, z/2 = A + // cos(z) = 1 - 2 * sin^2(z/2) + // 5: Make sin(z/2) subject of the formula: + // sin(z/2) = sqrt((1 - cos(z))/2) + // 6: Recall [3]; x = cos(z). Therefore: + // sin(z/2) = sqrt((1 - x)/2) + // 7: Let u = (1 - x)/2 + // 8: Therefore: + // asin(sqrt(u)) = z/2 + // 2 * asin(sqrt(u)) = z + // 9: Recall [3]; z = pi/2 - y. Therefore: + // y = pi/2 - z + // y = pi/2 - 2 * asin(sqrt(u)) + // 10: Recall [1], y = asin(x). Therefore: + // asin(x) = pi/2 - 2 * asin(sqrt(u)) + // 11: Recall that: acos(x) = pi/2 + asin(-x) = pi/2 - asin(x) + // Therefore: + // acos(x) = pi/2 - (pi/2 - 2 * asin(sqrt(u))) + // acos(x) = 2 * asin(sqrt(u)) + // + // THE RANGE REDUCTION, HOW? + // 12: Recall [7], u = (1 - x)/2 + // 13: Since 0.5 < x <= 1, therefore: + // 0 <= u <= 0.25 and 0 <= sqrt(u) <= 0.5 + // + // Hence, we can reuse the same [0, 0.5] domain polynomial approximation for + // Step [11] as `sqrt(u)` is in range. + // When -1 < x <= -0.5, the identity: + // acos(x) = pi - acos(-x) + // allows us to compute for the negative x value (lhs) + // with a positive x value instead (rhs). + + float xf_abs = (xf < 0 ? -xf : xf); + float u = fputil::multiply_add(-0.5f, xf_abs, 0.5f); + float sqrt_u = fputil::sqrt<float>(u); + + // Degree-6 minimax polynomial of asin(x) generated by Sollya with: + // > P = fpminimax(asin(x)/x, [|0, 2, 4, 6, 8|], [|SG...|], [0, 0.5]); + float asin_sqrt_u = + sqrt_u * fputil::polyeval(u, 0x1.000002p0f, 0x1.554c2ap-3f, + 0x1.3541ccp-4f, 0x1.43b2d6p-5f, 0x1.a0d73ep-5f); + + return fputil::cast<float16>( + x_sign ? fputil::multiply_add(-2.0f, asin_sqrt_u, PI) : 2 * asin_sqrt_u); +} + +} // namespace math + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LIBC_TYPES_HAS_FLOAT16 + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ACOS_H diff --git a/libc/src/__support/math/acosh_float_constants.h b/libc/src/__support/math/acosh_float_constants.h new file mode 100644 index 0000000..2eb245d --- /dev/null +++ b/libc/src/__support/math/acosh_float_constants.h @@ -0,0 +1,114 @@ +//===-- Common constants for acoshf function --------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ACOSH_FLOAT_CONSTANTS_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ACOSH_FLOAT_CONSTANTS_H + +#include "src/__support/macros/config.h" + +namespace LIBC_NAMESPACE_DECL { + +namespace acoshf_internal { + +// Lookup table for (1/f) where f = 1 + n*2^(-7), n = 0..127. +static constexpr double ONE_OVER_F[128] = { + 0x1.0000000000000p+0, 0x1.fc07f01fc07f0p-1, 0x1.f81f81f81f820p-1, + 0x1.f44659e4a4271p-1, 0x1.f07c1f07c1f08p-1, 0x1.ecc07b301ecc0p-1, + 0x1.e9131abf0b767p-1, 0x1.e573ac901e574p-1, 0x1.e1e1e1e1e1e1ep-1, + 0x1.de5d6e3f8868ap-1, 0x1.dae6076b981dbp-1, 0x1.d77b654b82c34p-1, + 0x1.d41d41d41d41dp-1, 0x1.d0cb58f6ec074p-1, 0x1.cd85689039b0bp-1, + 0x1.ca4b3055ee191p-1, 0x1.c71c71c71c71cp-1, 0x1.c3f8f01c3f8f0p-1, + 0x1.c0e070381c0e0p-1, 0x1.bdd2b899406f7p-1, 0x1.bacf914c1bad0p-1, + 0x1.b7d6c3dda338bp-1, 0x1.b4e81b4e81b4fp-1, 0x1.b2036406c80d9p-1, + 0x1.af286bca1af28p-1, 0x1.ac5701ac5701bp-1, 0x1.a98ef606a63bep-1, + 0x1.a6d01a6d01a6dp-1, 0x1.a41a41a41a41ap-1, 0x1.a16d3f97a4b02p-1, + 0x1.9ec8e951033d9p-1, 0x1.9c2d14ee4a102p-1, 0x1.999999999999ap-1, + 0x1.970e4f80cb872p-1, 0x1.948b0fcd6e9e0p-1, 0x1.920fb49d0e229p-1, + 0x1.8f9c18f9c18fap-1, 0x1.8d3018d3018d3p-1, 0x1.8acb90f6bf3aap-1, + 0x1.886e5f0abb04ap-1, 0x1.8618618618618p-1, 0x1.83c977ab2beddp-1, + 0x1.8181818181818p-1, 0x1.7f405fd017f40p-1, 0x1.7d05f417d05f4p-1, + 0x1.7ad2208e0ecc3p-1, 0x1.78a4c8178a4c8p-1, 0x1.767dce434a9b1p-1, + 0x1.745d1745d1746p-1, 0x1.724287f46debcp-1, 0x1.702e05c0b8170p-1, + 0x1.6e1f76b4337c7p-1, 0x1.6c16c16c16c17p-1, 0x1.6a13cd1537290p-1, + 0x1.6816816816817p-1, 0x1.661ec6a5122f9p-1, 0x1.642c8590b2164p-1, + 0x1.623fa77016240p-1, 0x1.6058160581606p-1, 0x1.5e75bb8d015e7p-1, + 0x1.5c9882b931057p-1, 0x1.5ac056b015ac0p-1, 0x1.58ed2308158edp-1, + 0x1.571ed3c506b3ap-1, 0x1.5555555555555p-1, 0x1.5390948f40febp-1, + 0x1.51d07eae2f815p-1, 0x1.5015015015015p-1, 0x1.4e5e0a72f0539p-1, + 0x1.4cab88725af6ep-1, 0x1.4afd6a052bf5bp-1, 0x1.49539e3b2d067p-1, + 0x1.47ae147ae147bp-1, 0x1.460cbc7f5cf9ap-1, 0x1.446f86562d9fbp-1, + 0x1.42d6625d51f87p-1, 0x1.4141414141414p-1, 0x1.3fb013fb013fbp-1, + 0x1.3e22cbce4a902p-1, 0x1.3c995a47babe7p-1, 0x1.3b13b13b13b14p-1, + 0x1.3991c2c187f63p-1, 0x1.3813813813814p-1, 0x1.3698df3de0748p-1, + 0x1.3521cfb2b78c1p-1, 0x1.33ae45b57bcb2p-1, 0x1.323e34a2b10bfp-1, + 0x1.30d190130d190p-1, 0x1.2f684bda12f68p-1, 0x1.2e025c04b8097p-1, + 0x1.2c9fb4d812ca0p-1, 0x1.2b404ad012b40p-1, 0x1.29e4129e4129ep-1, + 0x1.288b01288b013p-1, 0x1.27350b8812735p-1, 0x1.25e22708092f1p-1, + 0x1.2492492492492p-1, 0x1.23456789abcdfp-1, 0x1.21fb78121fb78p-1, + 0x1.20b470c67c0d9p-1, 0x1.1f7047dc11f70p-1, 0x1.1e2ef3b3fb874p-1, + 0x1.1cf06ada2811dp-1, 0x1.1bb4a4046ed29p-1, 0x1.1a7b9611a7b96p-1, + 0x1.19453808ca29cp-1, 0x1.1811811811812p-1, 0x1.16e0689427379p-1, + 0x1.15b1e5f75270dp-1, 0x1.1485f0e0acd3bp-1, 0x1.135c81135c811p-1, + 0x1.12358e75d3033p-1, 0x1.1111111111111p-1, 0x1.0fef010fef011p-1, + 0x1.0ecf56be69c90p-1, 0x1.0db20a88f4696p-1, 0x1.0c9714fbcda3bp-1, + 0x1.0b7e6ec259dc8p-1, 0x1.0a6810a6810a7p-1, 0x1.0953f39010954p-1, + 0x1.0842108421084p-1, 0x1.073260a47f7c6p-1, 0x1.0624dd2f1a9fcp-1, + 0x1.05197f7d73404p-1, 0x1.0410410410410p-1, 0x1.03091b51f5e1ap-1, + 0x1.0204081020408p-1, 0x1.0101010101010p-1}; + +// Lookup table for log(f) = log(1 + n*2^(-7)) where n = 0..127. +static constexpr double LOG_F[128] = { + 0x0.0000000000000p+0, 0x1.fe02a6b106788p-8, 0x1.fc0a8b0fc03e3p-7, + 0x1.7b91b07d5b11ap-6, 0x1.f829b0e783300p-6, 0x1.39e87b9febd5fp-5, + 0x1.77458f632dcfcp-5, 0x1.b42dd711971bep-5, 0x1.f0a30c01162a6p-5, + 0x1.16536eea37ae0p-4, 0x1.341d7961bd1d0p-4, 0x1.51b073f06183fp-4, + 0x1.6f0d28ae56b4bp-4, 0x1.8c345d6319b20p-4, 0x1.a926d3a4ad563p-4, + 0x1.c5e548f5bc743p-4, 0x1.e27076e2af2e5p-4, 0x1.fec9131dbeabap-4, + 0x1.0d77e7cd08e59p-3, 0x1.1b72ad52f67a0p-3, 0x1.29552f81ff523p-3, + 0x1.371fc201e8f74p-3, 0x1.44d2b6ccb7d1ep-3, 0x1.526e5e3a1b437p-3, + 0x1.5ff3070a793d3p-3, 0x1.6d60fe719d21cp-3, 0x1.7ab890210d909p-3, + 0x1.87fa06520c910p-3, 0x1.9525a9cf456b4p-3, 0x1.a23bc1fe2b563p-3, + 0x1.af3c94e80bff2p-3, 0x1.bc286742d8cd6p-3, 0x1.c8ff7c79a9a21p-3, + 0x1.d5c216b4fbb91p-3, 0x1.e27076e2af2e5p-3, 0x1.ef0adcbdc5936p-3, + 0x1.fb9186d5e3e2ap-3, 0x1.0402594b4d040p-2, 0x1.0a324e27390e3p-2, + 0x1.1058bf9ae4ad5p-2, 0x1.1675cababa60ep-2, 0x1.1c898c16999fap-2, + 0x1.22941fbcf7965p-2, 0x1.2895a13de86a3p-2, 0x1.2e8e2bae11d30p-2, + 0x1.347dd9a987d54p-2, 0x1.3a64c556945e9p-2, 0x1.404308686a7e3p-2, + 0x1.4618bc21c5ec2p-2, 0x1.4be5f957778a0p-2, 0x1.51aad872df82dp-2, + 0x1.5767717455a6cp-2, 0x1.5d1bdbf5809cap-2, 0x1.62c82f2b9c795p-2, + 0x1.686c81e9b14aep-2, 0x1.6e08eaa2ba1e3p-2, 0x1.739d7f6bbd006p-2, + 0x1.792a55fdd47a2p-2, 0x1.7eaf83b82afc3p-2, 0x1.842d1da1e8b17p-2, + 0x1.89a3386c1425ap-2, 0x1.8f11e873662c7p-2, 0x1.947941c2116fap-2, + 0x1.99d958117e08ap-2, 0x1.9f323ecbf984bp-2, 0x1.a484090e5bb0ap-2, + 0x1.a9cec9a9a0849p-2, 0x1.af1293247786bp-2, 0x1.b44f77bcc8f62p-2, + 0x1.b9858969310fbp-2, 0x1.beb4d9da71b7bp-2, 0x1.c3dd7a7cdad4dp-2, + 0x1.c8ff7c79a9a21p-2, 0x1.ce1af0b85f3ebp-2, 0x1.d32fe7e00ebd5p-2, + 0x1.d83e7258a2f3ep-2, 0x1.dd46a04c1c4a0p-2, 0x1.e24881a7c6c26p-2, + 0x1.e744261d68787p-2, 0x1.ec399d2468cc0p-2, 0x1.f128f5faf06ecp-2, + 0x1.f6123fa7028acp-2, 0x1.faf588f78f31ep-2, 0x1.ffd2e0857f498p-2, + 0x1.02552a5a5d0fep-1, 0x1.04bdf9da926d2p-1, 0x1.0723e5c1cdf40p-1, + 0x1.0986f4f573520p-1, 0x1.0be72e4252a82p-1, 0x1.0e44985d1cc8bp-1, + 0x1.109f39e2d4c96p-1, 0x1.12f719593efbcp-1, 0x1.154c3d2f4d5e9p-1, + 0x1.179eabbd899a0p-1, 0x1.19ee6b467c96ep-1, 0x1.1c3b81f713c24p-1, + 0x1.1e85f5e7040d0p-1, 0x1.20cdcd192ab6dp-1, 0x1.23130d7bebf42p-1, + 0x1.2555bce98f7cbp-1, 0x1.2795e1289b11ap-1, 0x1.29d37fec2b08ap-1, + 0x1.2c0e9ed448e8bp-1, 0x1.2e47436e40268p-1, 0x1.307d7334f10bep-1, + 0x1.32b1339121d71p-1, 0x1.34e289d9ce1d3p-1, 0x1.37117b54747b5p-1, + 0x1.393e0d3562a19p-1, 0x1.3b68449fffc22p-1, 0x1.3d9026a7156fap-1, + 0x1.3fb5b84d16f42p-1, 0x1.41d8fe84672aep-1, 0x1.43f9fe2f9ce67p-1, + 0x1.4618bc21c5ec2p-1, 0x1.48353d1ea88dfp-1, 0x1.4a4f85db03ebbp-1, + 0x1.4c679afccee39p-1, 0x1.4e7d811b75bb0p-1, 0x1.50913cc01686bp-1, + 0x1.52a2d265bc5aap-1, 0x1.54b2467999497p-1, 0x1.56bf9d5b3f399p-1, + 0x1.58cadb5cd7989p-1, 0x1.5ad404c359f2cp-1, 0x1.5cdb1dc6c1764p-1, + 0x1.5ee02a9241675p-1, 0x1.60e32f44788d8p-1}; + +} // namespace acoshf_internal + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ACOSH_FLOAT_CONSTANTS_H diff --git a/libc/src/__support/math/acoshf.h b/libc/src/__support/math/acoshf.h new file mode 100644 index 0000000..f18f169 --- /dev/null +++ b/libc/src/__support/math/acoshf.h @@ -0,0 +1,86 @@ +//===-- Implementation header for acoshf ------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H + +#include "acoshf_utils.h" +#include "src/__support/FPUtil/FEnvImpl.h" +#include "src/__support/FPUtil/FPBits.h" +#include "src/__support/FPUtil/multiply_add.h" +#include "src/__support/FPUtil/sqrt.h" +#include "src/__support/macros/config.h" +#include "src/__support/macros/optimization.h" // LIBC_UNLIKELY + +namespace LIBC_NAMESPACE_DECL { + +namespace math { + +static constexpr float acoshf(float x) { + using namespace acoshf_internal; + using FPBits_t = typename fputil::FPBits<float>; + FPBits_t xbits(x); + + if (LIBC_UNLIKELY(x <= 1.0f)) { + if (x == 1.0f) + return 0.0f; + // x < 1. + fputil::set_errno_if_required(EDOM); + fputil::raise_except_if_required(FE_INVALID); + return FPBits_t::quiet_nan().get_val(); + } + +#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS + uint32_t x_u = xbits.uintval(); + if (LIBC_UNLIKELY(x_u >= 0x4f8ffb03)) { + if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) + return x; + + // Helper functions to set results for exceptional cases. + auto round_result_slightly_down = [](float r) -> float { + volatile float tmp = r; + tmp = tmp - 0x1.0p-25f; + return tmp; + }; + auto round_result_slightly_up = [](float r) -> float { + volatile float tmp = r; + tmp = tmp + 0x1.0p-25f; + return tmp; + }; + + switch (x_u) { + case 0x4f8ffb03: // x = 0x1.1ff606p32f + return round_result_slightly_up(0x1.6fdd34p4f); + case 0x5c569e88: // x = 0x1.ad3d1p57f + return round_result_slightly_up(0x1.45c146p5f); + case 0x5e68984e: // x = 0x1.d1309cp61f + return round_result_slightly_up(0x1.5c9442p5f); + case 0x655890d3: // x = 0x1.b121a6p75f + return round_result_slightly_down(0x1.a9a3f2p5f); + case 0x6eb1a8ec: // x = 0x1.6351d8p94f + return round_result_slightly_down(0x1.08b512p6f); + case 0x7997f30a: // x = 0x1.2fe614p116f + return round_result_slightly_up(0x1.451436p6f); + } + } +#else + if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) + return x; +#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS + + double x_d = static_cast<double>(x); + // acosh(x) = log(x + sqrt(x^2 - 1)) + return static_cast<float>(log_eval( + x_d + fputil::sqrt<double>(fputil::multiply_add(x_d, x_d, -1.0)))); +} + +} // namespace math + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H diff --git a/libc/src/__support/math/acoshf16.h b/libc/src/__support/math/acoshf16.h new file mode 100644 index 0000000..15e7f6a --- /dev/null +++ b/libc/src/__support/math/acoshf16.h @@ -0,0 +1,123 @@ +//===-- Implementation header for acoshf16 ----------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H + +#include "include/llvm-libc-macros/float16-macros.h" + +#ifdef LIBC_TYPES_HAS_FLOAT16 + +#include "acoshf_utils.h" +#include "src/__support/FPUtil/FEnvImpl.h" +#include "src/__support/FPUtil/FPBits.h" +#include "src/__support/FPUtil/PolyEval.h" +#include "src/__support/FPUtil/cast.h" +#include "src/__support/FPUtil/except_value_utils.h" +#include "src/__support/FPUtil/multiply_add.h" +#include "src/__support/FPUtil/sqrt.h" +#include "src/__support/macros/config.h" +#include "src/__support/macros/optimization.h" + +namespace LIBC_NAMESPACE_DECL { + +namespace math { + +static constexpr float16 acoshf16(float16 x) { + + using namespace acoshf_internal; + constexpr size_t N_EXCEPTS = 2; + constexpr fputil::ExceptValues<float16, N_EXCEPTS> ACOSHF16_EXCEPTS{{ + // (input, RZ output, RU offset, RD offset, RN offset) + // x = 0x1.6dcp+1, acoshf16(x) = 0x1.b6p+0 (RZ) + {0x41B7, 0x3ED8, 1, 0, 0}, + // x = 0x1.39p+0, acoshf16(x) = 0x1.4f8p-1 (RZ) + {0x3CE4, 0x393E, 1, 0, 1}, + }}; + + using FPBits = fputil::FPBits<float16>; + FPBits xbits(x); + uint16_t x_u = xbits.uintval(); + + // Check for NaN input first. + if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) { + if (xbits.is_signaling_nan()) { + fputil::raise_except_if_required(FE_INVALID); + return FPBits::quiet_nan().get_val(); + } + if (xbits.is_neg()) { + fputil::set_errno_if_required(EDOM); + fputil::raise_except_if_required(FE_INVALID); + return FPBits::quiet_nan().get_val(); + } + return x; + } + + // Domain error for inputs less than 1.0. + if (LIBC_UNLIKELY(x <= 1.0f)) { + if (x == 1.0f) + return FPBits::zero().get_val(); + fputil::set_errno_if_required(EDOM); + fputil::raise_except_if_required(FE_INVALID); + return FPBits::quiet_nan().get_val(); + } + + if (auto r = ACOSHF16_EXCEPTS.lookup(xbits.uintval()); + LIBC_UNLIKELY(r.has_value())) + return r.value(); + + float xf = x; + // High-precision polynomial approximation for inputs close to 1.0 + // ([1, 1.25)). + // + // Brief derivation: + // 1. Expand acosh(1 + delta) using Taylor series around delta=0: + // acosh(1 + delta) ≈ sqrt(2 * delta) * [1 - delta/12 + 3*delta^2/160 + // - 5*delta^3/896 + 35*delta^4/18432 + ...] + // 2. Truncate the series to fit accurately for delta in [0, 0.25]. + // 3. Polynomial coefficients (from sollya) used here are: + // P(delta) ≈ 1 - 0x1.555556p-4 * delta + 0x1.333334p-6 * delta^2 + // - 0x1.6db6dcp-8 * delta^3 + 0x1.f1c71cp-10 * delta^4 + // 4. The Sollya commands used to generate these coefficients were: + // > display = hexadecimal; + // > round(1/12, SG, RN); + // > round(3/160, SG, RN); + // > round(5/896, SG, RN); + // > round(35/18432, SG, RN); + // With hexadecimal display mode enabled, the outputs were: + // 0x1.555556p-4 + // 0x1.333334p-6 + // 0x1.6db6dcp-8 + // 0x1.f1c71cp-10 + // 5. The maximum absolute error, estimated using: + // dirtyinfnorm(acosh(1 + x) - sqrt(2*x) * P(x), [0, 0.25]) + // is: + // 0x1.d84281p-22 + if (LIBC_UNLIKELY(x_u < 0x3D00U)) { + float delta = xf - 1.0f; + float sqrt_2_delta = fputil::sqrt<float>(2.0 * delta); + float pe = fputil::polyeval(delta, 0x1p+0f, -0x1.555556p-4f, 0x1.333334p-6f, + -0x1.6db6dcp-8f, 0x1.f1c71cp-10f); + float approx = sqrt_2_delta * pe; + return fputil::cast<float16>(approx); + } + + // acosh(x) = log(x + sqrt(x^2 - 1)) + float sqrt_term = fputil::sqrt<float>(fputil::multiply_add(xf, xf, -1.0f)); + float result = static_cast<float>(log_eval(xf + sqrt_term)); + + return fputil::cast<float16>(result); +} + +} // namespace math + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LIBC_TYPES_HAS_FLOAT16 + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_H diff --git a/libc/src/__support/math/acoshf_utils.h b/libc/src/__support/math/acoshf_utils.h new file mode 100644 index 0000000..808c3dd --- /dev/null +++ b/libc/src/__support/math/acoshf_utils.h @@ -0,0 +1,60 @@ +//===-- Collection of utils for acoshf --------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_UTILS_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_UTILS_H + +#include "acosh_float_constants.h" +#include "src/__support/FPUtil/FPBits.h" +#include "src/__support/FPUtil/PolyEval.h" +#include "src/__support/FPUtil/multiply_add.h" + +namespace LIBC_NAMESPACE_DECL { + +namespace acoshf_internal { + +// x should be positive, normal finite value +LIBC_INLINE static double log_eval(double x) { + // For x = 2^ex * (1 + mx) + // log(x) = ex * log(2) + log(1 + mx) + using FPB = fputil::FPBits<double>; + FPB bs(x); + + double ex = static_cast<double>(bs.get_exponent()); + + // p1 is the leading 7 bits of mx, i.e. + // p1 * 2^(-7) <= m_x < (p1 + 1) * 2^(-7). + int p1 = static_cast<int>(bs.get_mantissa() >> (FPB::FRACTION_LEN - 7)); + + // Set bs to (1 + (mx - p1*2^(-7)) + bs.set_uintval(bs.uintval() & (FPB::FRACTION_MASK >> 7)); + bs.set_biased_exponent(FPB::EXP_BIAS); + // dx = (mx - p1*2^(-7)) / (1 + p1*2^(-7)). + double dx = (bs.get_val() - 1.0) * ONE_OVER_F[p1]; + + // Minimax polynomial of log(1 + dx) generated by Sollya with: + // > P = fpminimax(log(1 + x)/x, 6, [|D...|], [0, 2^-7]); + const double COEFFS[6] = {-0x1.ffffffffffffcp-2, 0x1.5555555552ddep-2, + -0x1.ffffffefe562dp-3, 0x1.9999817d3a50fp-3, + -0x1.554317b3f67a5p-3, 0x1.1dc5c45e09c18p-3}; + double dx2 = dx * dx; + double c1 = fputil::multiply_add(dx, COEFFS[1], COEFFS[0]); + double c2 = fputil::multiply_add(dx, COEFFS[3], COEFFS[2]); + double c3 = fputil::multiply_add(dx, COEFFS[5], COEFFS[4]); + + double p = fputil::polyeval(dx2, dx, c1, c2, c3); + double result = + fputil::multiply_add(ex, /*log(2)*/ 0x1.62e42fefa39efp-1, LOG_F[p1] + p); + return result; +} + +} // namespace acoshf_internal + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ACOSHF_UTILS_H diff --git a/libc/src/__support/math/erff.h b/libc/src/__support/math/erff.h new file mode 100644 index 0000000..e54ec77 --- /dev/null +++ b/libc/src/__support/math/erff.h @@ -0,0 +1,193 @@ +//===-- Implementation header for erff --------------------------*- C++ -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// + +#ifndef LLVM_LIBC_SRC___SUPPORT_MATH_ERFF_H +#define LLVM_LIBC_SRC___SUPPORT_MATH_ERFF_H + +#include "src/__support/FPUtil/FPBits.h" +#include "src/__support/FPUtil/except_value_utils.h" +#include "src/__support/FPUtil/multiply_add.h" +#include "src/__support/macros/config.h" +#include "src/__support/macros/optimization.h" // LIBC_UNLIKELY + +namespace LIBC_NAMESPACE_DECL { + +namespace math { + +static constexpr float erff(float x) { + + // Polynomials approximating erf(x)/x on ( k/8, (k + 1)/8 ) generated by + // Sollya with: > P = fpminimax(erf(x)/x, [|0, 2, 4, 6, 8, 10, 12, 14|], + // [|D...|], + // [k/8, (k + 1)/8]); + // for k = 0..31. + constexpr double COEFFS[32][8] = { + {0x1.20dd750429b6dp0, -0x1.812746b037753p-2, 0x1.ce2f219e8596ap-4, + -0x1.b82cdacb78fdap-6, 0x1.56479297dfda5p-8, -0x1.8b3ac5455ef02p-11, + -0x1.126fcac367e3bp-8, 0x1.2d0bdb3ba4984p-4}, + {0x1.20dd750429b6dp0, -0x1.812746b0379a8p-2, 0x1.ce2f21a03cf2ap-4, + -0x1.b82ce30de083ep-6, 0x1.565bcad3eb60fp-8, -0x1.c02c66f659256p-11, + 0x1.f92f673385229p-14, -0x1.def402648ae9p-17}, + {0x1.20dd750429b34p0, -0x1.812746b032dcep-2, 0x1.ce2f219d84aaep-4, + -0x1.b82ce22dcf139p-6, 0x1.565b9efcd4af1p-8, -0x1.c021f1af414bcp-11, + 0x1.f7c6d177eff82p-14, -0x1.c9e4410dcf865p-17}, + {0x1.20dd750426eabp0, -0x1.812746ae592c7p-2, 0x1.ce2f211525f14p-4, + -0x1.b82ccc125e63fp-6, 0x1.56596f261cfd3p-8, -0x1.bfde1ff8eeecfp-11, + 0x1.f31a9d15dc5d8p-14, -0x1.a5a4362844b3cp-17}, + {0x1.20dd75039c705p0, -0x1.812746777e74dp-2, 0x1.ce2f17af98a1bp-4, + -0x1.b82be4b817cbep-6, 0x1.564bec2e2962ep-8, -0x1.bee86f9da3558p-11, + 0x1.e9443689dc0ccp-14, -0x1.79c0f230805d8p-17}, + {0x1.20dd74f811211p0, -0x1.81274371a3e8fp-2, 0x1.ce2ec038262e5p-4, + -0x1.b8265b82c5e1fp-6, 0x1.5615a2e239267p-8, -0x1.bc63ae023dcebp-11, + 0x1.d87c2102f7e06p-14, -0x1.49584bea41d62p-17}, + {0x1.20dd746d063e3p0, -0x1.812729a8a950fp-2, 0x1.ce2cb0a2df232p-4, + -0x1.b80eca1f51278p-6, 0x1.5572e26c46815p-8, -0x1.b715e5638b65ep-11, + 0x1.bfbb195484968p-14, -0x1.177a565c15c52p-17}, + {0x1.20dd701b44486p0, -0x1.812691145f237p-2, 0x1.ce23a06b8cfd9p-4, + -0x1.b7c1dc7245288p-6, 0x1.53e92f7f397ddp-8, -0x1.ad97cc4acf0b2p-11, + 0x1.9f028b2b09b71p-14, -0x1.cdc4da08da8c1p-18}, + {0x1.20dd5715ac332p0, -0x1.8123e680bd0ebp-2, 0x1.ce0457aded691p-4, + -0x1.b6f52d52bed4p-6, 0x1.50c291b84414cp-8, -0x1.9ea246b1ad4a9p-11, + 0x1.77654674e0cap-14, -0x1.737c11a1bcebbp-18}, + {0x1.20dce6593e114p0, -0x1.811a59c02eadcp-2, 0x1.cdab53c7cd7d5p-4, + -0x1.b526d2e321eedp-6, 0x1.4b1d32cd8b994p-8, -0x1.8963143ec0a1ep-11, + 0x1.4ad5700e4db91p-14, -0x1.231e100e43ef2p-18}, + {0x1.20db48bfd5a62p0, -0x1.80fdd84f9e308p-2, 0x1.ccd340d462983p-4, + -0x1.b196a2928768p-6, 0x1.4210c2c13a0f7p-8, -0x1.6dbdfb4ff71aep-11, + 0x1.1bca2d17fbd71p-14, -0x1.bca36f90c7cf5p-19}, + {0x1.20d64b2f8f508p0, -0x1.80b4d4f19fa8bp-2, 0x1.cb088197262e3p-4, + -0x1.ab51fd02e5b99p-6, 0x1.34e1e5e81a632p-8, -0x1.4c66377b502cep-11, + 0x1.d9ad25066213cp-15, -0x1.4b0df7dd0cfa1p-19}, + {0x1.20c8fc1243576p0, -0x1.8010cb2009e27p-2, 0x1.c7a47e9299315p-4, + -0x1.a155be5683654p-6, 0x1.233502694997bp-8, -0x1.26c94b7d813p-11, + 0x1.8094f1de25fb9p-15, -0x1.e0e3d776c6eefp-20}, + {0x1.20a9bd1611bc1p0, -0x1.7ec7fbce83f9p-2, 0x1.c1d757d7317b7p-4, + -0x1.92c160cd589fp-6, 0x1.0d307269cc5c2p-8, -0x1.fda5b0d2d1879p-12, + 0x1.2fdd7b3b14a7fp-15, -0x1.54eed4a26af5ap-20}, + {0x1.20682834f943dp0, -0x1.7c73f747bf5a9p-2, 0x1.b8c2db4a9ffd1p-4, + -0x1.7f0e4ffe989ecp-6, 0x1.e7061eae4166ep-9, -0x1.ad36e873fff2dp-12, + 0x1.d39222396128ep-16, -0x1.d83dacec5ea6bp-21}, + {0x1.1feb8d12676d7p0, -0x1.7898347284afep-2, 0x1.aba3466b34451p-4, + -0x1.663adc573e2f9p-6, 0x1.ae99fb17c3e08p-9, -0x1.602f950ad5535p-12, + 0x1.5e9717490609dp-16, -0x1.3fca107bbc8d5p-21}, + {0x1.1f12fe3c536fap0, -0x1.72b1d1f22e6d3p-2, 0x1.99fc0eed4a896p-4, + -0x1.48db0a87bd8c6p-6, 0x1.73e368895aa61p-9, -0x1.19b35d5301fc8p-12, + 0x1.007987e4bb033p-16, -0x1.a7edcd4c2dc7p-22}, + {0x1.1db7b0df84d5dp0, -0x1.6a4e4a41cde02p-2, 0x1.83bbded16455dp-4, + -0x1.2809b3b36977ep-6, 0x1.39c08bab44679p-9, -0x1.b7b45a70ed119p-13, + 0x1.6e99b36410e7bp-17, -0x1.13619bb7ebc0cp-22}, + {0x1.1bb1c85c4a527p0, -0x1.5f23b99a249a3p-2, 0x1.694c91fa0d12cp-4, + -0x1.053e1ce11c72dp-6, 0x1.02bf72c50ea78p-9, -0x1.4f478fb56cb02p-13, + 0x1.005f80ecbe213p-17, -0x1.5f2446bde7f5bp-23}, + {0x1.18dec3bd51f9dp0, -0x1.5123f58346186p-2, 0x1.4b8a1ca536ab4p-4, + -0x1.c4243015cc723p-7, 0x1.a1a8a01d351efp-10, -0x1.f466b34f1d86bp-14, + 0x1.5f835eea0bf6ap-18, -0x1.b83165b939234p-24}, + {0x1.152804c3369f4p0, -0x1.4084cd4afd4bcp-2, 0x1.2ba2e836e47aap-4, + -0x1.800f2dfc6904bp-7, 0x1.4a6daf0669c59p-10, -0x1.6e326ab872317p-14, + 0x1.d9761a6a755a5p-19, -0x1.0fca33f9dd4b5p-24}, + {0x1.1087ad68356aap0, -0x1.2dbb044707459p-2, 0x1.0aea8ceaa0384p-4, + -0x1.40b516d52b3d2p-7, 0x1.00c9e05f01d22p-10, -0x1.076afb0dc0ff7p-14, + 0x1.39fadec400657p-19, -0x1.4b5761352e7e3p-25}, + {0x1.0b0a7a8ba4a22p0, -0x1.196990d22d4a1p-2, 0x1.d5551e6ac0c4dp-5, + -0x1.07cce1770bd1ap-7, 0x1.890347b8848bfp-11, -0x1.757ec96750b6ap-15, + 0x1.9b258a1e06bcep-20, -0x1.8fc6d22da7572p-26}, + {0x1.04ce2be70fb47p0, -0x1.0449e4b0b9cacp-2, 0x1.97f7424f4b0e7p-5, + -0x1.ac825439c42f4p-8, 0x1.28f5f65426dfbp-11, -0x1.05b699a90f90fp-15, + 0x1.0a888eecf4593p-20, -0x1.deace2b32bb31p-27}, + {0x1.fbf9fb0e11cc8p-1, -0x1.de2640856545ap-3, 0x1.5f5b1f47f851p-5, + -0x1.588bc71eb41b9p-8, 0x1.bc6a0a772f56dp-12, -0x1.6b9fad1f1657ap-16, + 0x1.573204ba66504p-21, -0x1.1d38065c94e44p-27}, + {0x1.ed8f18c99e031p-1, -0x1.b4cb6acd903b4p-3, 0x1.2c7f3dddd6fc1p-5, + -0x1.13052067df4ep-8, 0x1.4a5027444082fp-12, -0x1.f672bab0e2554p-17, + 0x1.b83c756348cc9p-22, -0x1.534f1a1079499p-28}, + {0x1.debd33044166dp-1, -0x1.8d7cd9053f7d8p-3, 0x1.ff9957fb3d6e7p-6, + -0x1.b50be55de0f36p-9, 0x1.e92c8ec53a628p-13, -0x1.5a4b88d508007p-17, + 0x1.1a27737559e26p-22, -0x1.942ae62cb2c14p-29}, + {0x1.cfdbf0386f3bdp-1, -0x1.68e33d93b0dc4p-3, 0x1.b2683d58f53dep-6, + -0x1.5a9174e70d26fp-9, 0x1.69ddd326d49cdp-13, -0x1.dd8f397a8219cp-18, + 0x1.6a755016ad4ddp-23, -0x1.e366e0139187dp-30}, + {0x1.c132adb8d7464p-1, -0x1.475a899f61b46p-3, 0x1.70a431397a77cp-6, + -0x1.12e3d35beeee2p-9, 0x1.0c16b05738333p-13, -0x1.4a47f873e144ep-18, + 0x1.d3d494c698c02p-24, -0x1.2302c59547fe5p-30}, + {0x1.b2f5fd05555e7p-1, -0x1.28feefbe03ec7p-3, 0x1.3923acbb3a676p-6, + -0x1.b4ff793cd6358p-10, 0x1.8ea0eb8c913bcp-14, -0x1.cb31ec2baceb1p-19, + 0x1.30011e7e80c04p-24, -0x1.617710635cb1dp-31}, + {0x1.a54853cd9593ep-1, -0x1.0dbdbaea4dc8ep-3, 0x1.0a93e2c20a0fdp-6, + -0x1.5c969ff401ea8p-10, 0x1.29e0cc64fe627p-14, -0x1.4160d8e9d3c2ap-19, + 0x1.8e7b67594624ap-25, -0x1.b1cf2c975b09bp-32}, + {0x1.983ceece09ff8p-1, -0x1.eacc78f7a2dp-4, 0x1.c74418410655fp-7, + -0x1.1756a050e441ep-10, 0x1.bff3650f7f548p-15, -0x1.c56c0217d3adap-20, + 0x1.07b4918d0b489p-25, -0x1.0d4be8c1c50f8p-32}, + }; + + using FPBits = typename fputil::FPBits<float>; + FPBits xbits(x); + + uint32_t x_u = xbits.uintval(); + uint32_t x_abs = x_u & 0x7fff'ffffU; + + if (LIBC_UNLIKELY(x_abs >= 0x4080'0000U)) { + constexpr float ONE[2] = {1.0f, -1.0f}; + constexpr float SMALL[2] = {-0x1.0p-25f, 0x1.0p-25f}; + + int sign = xbits.is_neg() ? 1 : 0; + + if (LIBC_UNLIKELY(x_abs >= 0x7f80'0000U)) { + if (xbits.is_signaling_nan()) { + fputil::raise_except_if_required(FE_INVALID); + return FPBits::quiet_nan().get_val(); + } + return (x_abs > 0x7f80'0000) ? x : ONE[sign]; + } + + return ONE[sign] + SMALL[sign]; + } + +#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS + // Exceptional mask = common 0 bits of 2 exceptional values. + constexpr uint32_t EXCEPT_MASK = 0x809a'6184U; + + if (LIBC_UNLIKELY((x_abs & EXCEPT_MASK) == 0)) { + // Exceptional values + if (LIBC_UNLIKELY(x_abs == 0x3f65'9229U)) // |x| = 0x1.cb2452p-1f + return x < 0.0f ? fputil::round_result_slightly_down(-0x1.972ea8p-1f) + : fputil::round_result_slightly_up(0x1.972ea8p-1f); + if (LIBC_UNLIKELY(x_abs == 0x4004'1e6aU)) // |x| = 0x1.083cd4p+1f + return x < 0.0f ? fputil::round_result_slightly_down(-0x1.fe3462p-1f) + : fputil::round_result_slightly_up(0x1.fe3462p-1f); + if (x_abs == 0U) + return x; + } +#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS + + // Polynomial approximation: + // erf(x) ~ x * (c0 + c1 * x^2 + c2 * x^4 + ... + c7 * x^14) + double xd = static_cast<double>(x); + double xsq = xd * xd; + + constexpr uint32_t EIGHT = 3 << FPBits::FRACTION_LEN; + int idx = static_cast<int>(FPBits(x_abs + EIGHT).get_val()); + + double x4 = xsq * xsq; + double c0 = fputil::multiply_add(xsq, COEFFS[idx][1], COEFFS[idx][0]); + double c1 = fputil::multiply_add(xsq, COEFFS[idx][3], COEFFS[idx][2]); + double c2 = fputil::multiply_add(xsq, COEFFS[idx][5], COEFFS[idx][4]); + double c3 = fputil::multiply_add(xsq, COEFFS[idx][7], COEFFS[idx][6]); + + double x8 = x4 * x4; + double p0 = fputil::multiply_add(x4, c1, c0); + double p1 = fputil::multiply_add(x4, c3, c2); + + return static_cast<float>(xd * fputil::multiply_add(x8, p1, p0)); +} + +} // namespace math + +} // namespace LIBC_NAMESPACE_DECL + +#endif // LLVM_LIBC_SRC___SUPPORT_MATH_ERFF_H diff --git a/libc/src/math/generic/CMakeLists.txt b/libc/src/math/generic/CMakeLists.txt index c90665d..408f99e 100644 --- a/libc/src/math/generic/CMakeLists.txt +++ b/libc/src/math/generic/CMakeLists.txt @@ -1295,12 +1295,8 @@ add_entrypoint_object( HDRS ../erff.h DEPENDS - .common_constants - libc.src.__support.FPUtil.fp_bits - libc.src.__support.FPUtil.except_value_utils - libc.src.__support.FPUtil.multiply_add - libc.src.__support.FPUtil.polyeval - libc.src.__support.macros.optimization + libc.src.__support.math.erff + libc.src.errno.errno ) add_entrypoint_object( @@ -1898,6 +1894,7 @@ add_object_library( common_constants.cpp DEPENDS libc.src.__support.math.exp_constants + libc.src.__support.math.acosh_float_constants libc.src.__support.number_pair ) @@ -3761,7 +3758,7 @@ add_header_library( DEPENDS .common_constants libc.src.__support.math.exp_utils - libc.src.__support.math.exp10f_utils + libc.src.__support.math.acoshf_utils libc.src.__support.macros.properties.cpu_features libc.src.errno.errno ) @@ -3871,12 +3868,7 @@ add_entrypoint_object( ../acoshf.h DEPENDS .explogxf - libc.src.__support.FPUtil.fenv_impl - libc.src.__support.FPUtil.fp_bits - libc.src.__support.FPUtil.multiply_add - libc.src.__support.FPUtil.polyeval - libc.src.__support.FPUtil.sqrt - libc.src.__support.macros.optimization + libc.src.__support.math.acoshf ) add_entrypoint_object( @@ -3886,18 +3878,8 @@ add_entrypoint_object( HDRS ../acoshf16.h DEPENDS - .explogxf - libc.hdr.errno_macros - libc.hdr.fenv_macros - libc.src.__support.FPUtil.cast - libc.src.__support.FPUtil.except_value_utils - libc.src.__support.FPUtil.fenv_impl - libc.src.__support.FPUtil.fp_bits - libc.src.__support.FPUtil.multiply_add - libc.src.__support.FPUtil.polyeval - libc.src.__support.FPUtil.sqrt - libc.src.__support.macros.optimization - libc.src.__support.macros.properties.types + libc.src.__support.math.acoshf16 + libc.src.errno.errno ) add_entrypoint_object( @@ -4040,17 +4022,8 @@ add_entrypoint_object( HDRS ../acosf16.h DEPENDS - libc.hdr.errno_macros - libc.hdr.fenv_macros - libc.src.__support.FPUtil.cast - libc.src.__support.FPUtil.except_value_utils - libc.src.__support.FPUtil.fenv_impl - libc.src.__support.FPUtil.fp_bits - libc.src.__support.FPUtil.multiply_add - libc.src.__support.FPUtil.polyeval - libc.src.__support.FPUtil.sqrt - libc.src.__support.macros.optimization - libc.src.__support.macros.properties.types + libc.src.__support.math.acosf16 + libc.src.errno.errno ) add_entrypoint_object( diff --git a/libc/src/math/generic/acosf16.cpp b/libc/src/math/generic/acosf16.cpp index 202a950..0bf85f8 100644 --- a/libc/src/math/generic/acosf16.cpp +++ b/libc/src/math/generic/acosf16.cpp @@ -8,144 +8,10 @@ //===----------------------------------------------------------------------===// #include "src/math/acosf16.h" -#include "hdr/errno_macros.h" -#include "hdr/fenv_macros.h" -#include "src/__support/FPUtil/FEnvImpl.h" -#include "src/__support/FPUtil/FPBits.h" -#include "src/__support/FPUtil/PolyEval.h" -#include "src/__support/FPUtil/cast.h" -#include "src/__support/FPUtil/except_value_utils.h" -#include "src/__support/FPUtil/multiply_add.h" -#include "src/__support/FPUtil/sqrt.h" -#include "src/__support/macros/optimization.h" +#include "src/__support/math/acosf16.h" namespace LIBC_NAMESPACE_DECL { -// Generated by Sollya using the following command: -// > round(pi/2, SG, RN); -// > round(pi, SG, RN); -static constexpr float PI_OVER_2 = 0x1.921fb6p0f; -static constexpr float PI = 0x1.921fb6p1f; +LLVM_LIBC_FUNCTION(float16, acosf16, (float16 x)) { return math::acosf16(x); } -#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS -static constexpr size_t N_EXCEPTS = 2; - -static constexpr fputil::ExceptValues<float16, N_EXCEPTS> ACOSF16_EXCEPTS{{ - // (input, RZ output, RU offset, RD offset, RN offset) - {0xacaf, 0x3e93, 1, 0, 0}, - {0xb874, 0x4052, 1, 0, 1}, -}}; -#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS - -LLVM_LIBC_FUNCTION(float16, acosf16, (float16 x)) { - using FPBits = fputil::FPBits<float16>; - FPBits xbits(x); - - uint16_t x_u = xbits.uintval(); - uint16_t x_abs = x_u & 0x7fff; - uint16_t x_sign = x_u >> 15; - - // |x| > 0x1p0, |x| > 1, or x is NaN. - if (LIBC_UNLIKELY(x_abs > 0x3c00)) { - // acosf16(NaN) = NaN - if (xbits.is_nan()) { - if (xbits.is_signaling_nan()) { - fputil::raise_except_if_required(FE_INVALID); - return FPBits::quiet_nan().get_val(); - } - - return x; - } - - // 1 < |x| <= +/-inf - fputil::raise_except_if_required(FE_INVALID); - fputil::set_errno_if_required(EDOM); - - return FPBits::quiet_nan().get_val(); - } - - float xf = x; - -#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS - // Handle exceptional values - if (auto r = ACOSF16_EXCEPTS.lookup(x_u); LIBC_UNLIKELY(r.has_value())) - return r.value(); -#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS - - // |x| == 0x1p0, x is 1 or -1 - // if x is (-)1, return pi, else - // if x is (+)1, return 0 - if (LIBC_UNLIKELY(x_abs == 0x3c00)) - return fputil::cast<float16>(x_sign ? PI : 0.0f); - - float xsq = xf * xf; - - // |x| <= 0x1p-1, |x| <= 0.5 - if (x_abs <= 0x3800) { - // if x is 0, return pi/2 - if (LIBC_UNLIKELY(x_abs == 0)) - return fputil::cast<float16>(PI_OVER_2); - - // Note that: acos(x) = pi/2 + asin(-x) = pi/2 - asin(x) - // Degree-6 minimax polynomial of asin(x) generated by Sollya with: - // > P = fpminimax(asin(x)/x, [|0, 2, 4, 6, 8|], [|SG...|], [0, 0.5]); - float interm = - fputil::polyeval(xsq, 0x1.000002p0f, 0x1.554c2ap-3f, 0x1.3541ccp-4f, - 0x1.43b2d6p-5f, 0x1.a0d73ep-5f); - return fputil::cast<float16>(fputil::multiply_add(-xf, interm, PI_OVER_2)); - } - - // When |x| > 0.5, assume that 0.5 < |x| <= 1 - // - // Step-by-step range-reduction proof: - // 1: Let y = asin(x), such that, x = sin(y) - // 2: From complimentary angle identity: - // x = sin(y) = cos(pi/2 - y) - // 3: Let z = pi/2 - y, such that x = cos(z) - // 4: From double angle formula; cos(2A) = 1 - 2 * sin^2(A): - // z = 2A, z/2 = A - // cos(z) = 1 - 2 * sin^2(z/2) - // 5: Make sin(z/2) subject of the formula: - // sin(z/2) = sqrt((1 - cos(z))/2) - // 6: Recall [3]; x = cos(z). Therefore: - // sin(z/2) = sqrt((1 - x)/2) - // 7: Let u = (1 - x)/2 - // 8: Therefore: - // asin(sqrt(u)) = z/2 - // 2 * asin(sqrt(u)) = z - // 9: Recall [3]; z = pi/2 - y. Therefore: - // y = pi/2 - z - // y = pi/2 - 2 * asin(sqrt(u)) - // 10: Recall [1], y = asin(x). Therefore: - // asin(x) = pi/2 - 2 * asin(sqrt(u)) - // 11: Recall that: acos(x) = pi/2 + asin(-x) = pi/2 - asin(x) - // Therefore: - // acos(x) = pi/2 - (pi/2 - 2 * asin(sqrt(u))) - // acos(x) = 2 * asin(sqrt(u)) - // - // THE RANGE REDUCTION, HOW? - // 12: Recall [7], u = (1 - x)/2 - // 13: Since 0.5 < x <= 1, therefore: - // 0 <= u <= 0.25 and 0 <= sqrt(u) <= 0.5 - // - // Hence, we can reuse the same [0, 0.5] domain polynomial approximation for - // Step [11] as `sqrt(u)` is in range. - // When -1 < x <= -0.5, the identity: - // acos(x) = pi - acos(-x) - // allows us to compute for the negative x value (lhs) - // with a positive x value instead (rhs). - - float xf_abs = (xf < 0 ? -xf : xf); - float u = fputil::multiply_add(-0.5f, xf_abs, 0.5f); - float sqrt_u = fputil::sqrt<float>(u); - - // Degree-6 minimax polynomial of asin(x) generated by Sollya with: - // > P = fpminimax(asin(x)/x, [|0, 2, 4, 6, 8|], [|SG...|], [0, 0.5]); - float asin_sqrt_u = - sqrt_u * fputil::polyeval(u, 0x1.000002p0f, 0x1.554c2ap-3f, - 0x1.3541ccp-4f, 0x1.43b2d6p-5f, 0x1.a0d73ep-5f); - - return fputil::cast<float16>( - x_sign ? fputil::multiply_add(-2.0f, asin_sqrt_u, PI) : 2 * asin_sqrt_u); -} } // namespace LIBC_NAMESPACE_DECL diff --git a/libc/src/math/generic/acoshf.cpp b/libc/src/math/generic/acoshf.cpp index c4927fa..5c04583 100644 --- a/libc/src/math/generic/acoshf.cpp +++ b/libc/src/math/generic/acoshf.cpp @@ -7,73 +7,11 @@ //===----------------------------------------------------------------------===// #include "src/math/acoshf.h" -#include "src/__support/FPUtil/FEnvImpl.h" -#include "src/__support/FPUtil/FPBits.h" -#include "src/__support/FPUtil/PolyEval.h" -#include "src/__support/FPUtil/multiply_add.h" -#include "src/__support/FPUtil/sqrt.h" -#include "src/__support/macros/config.h" -#include "src/__support/macros/optimization.h" // LIBC_UNLIKELY -#include "src/math/generic/common_constants.h" -#include "src/math/generic/explogxf.h" -namespace LIBC_NAMESPACE_DECL { - -LLVM_LIBC_FUNCTION(float, acoshf, (float x)) { - using FPBits_t = typename fputil::FPBits<float>; - FPBits_t xbits(x); - - if (LIBC_UNLIKELY(x <= 1.0f)) { - if (x == 1.0f) - return 0.0f; - // x < 1. - fputil::set_errno_if_required(EDOM); - fputil::raise_except_if_required(FE_INVALID); - return FPBits_t::quiet_nan().get_val(); - } +#include "src/__support/math/acoshf.h" -#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS - uint32_t x_u = xbits.uintval(); - if (LIBC_UNLIKELY(x_u >= 0x4f8ffb03)) { - if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) - return x; - - // Helper functions to set results for exceptional cases. - auto round_result_slightly_down = [](float r) -> float { - volatile float tmp = r; - tmp = tmp - 0x1.0p-25f; - return tmp; - }; - auto round_result_slightly_up = [](float r) -> float { - volatile float tmp = r; - tmp = tmp + 0x1.0p-25f; - return tmp; - }; - - switch (x_u) { - case 0x4f8ffb03: // x = 0x1.1ff606p32f - return round_result_slightly_up(0x1.6fdd34p4f); - case 0x5c569e88: // x = 0x1.ad3d1p57f - return round_result_slightly_up(0x1.45c146p5f); - case 0x5e68984e: // x = 0x1.d1309cp61f - return round_result_slightly_up(0x1.5c9442p5f); - case 0x655890d3: // x = 0x1.b121a6p75f - return round_result_slightly_down(0x1.a9a3f2p5f); - case 0x6eb1a8ec: // x = 0x1.6351d8p94f - return round_result_slightly_down(0x1.08b512p6f); - case 0x7997f30a: // x = 0x1.2fe614p116f - return round_result_slightly_up(0x1.451436p6f); - } - } -#else - if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) - return x; -#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS +namespace LIBC_NAMESPACE_DECL { - double x_d = static_cast<double>(x); - // acosh(x) = log(x + sqrt(x^2 - 1)) - return static_cast<float>(log_eval( - x_d + fputil::sqrt<double>(fputil::multiply_add(x_d, x_d, -1.0)))); -} +LLVM_LIBC_FUNCTION(float, acoshf, (float x)) { return math::acoshf(x); } } // namespace LIBC_NAMESPACE_DECL diff --git a/libc/src/math/generic/acoshf16.cpp b/libc/src/math/generic/acoshf16.cpp index 44783a8..bb3a91f 100644 --- a/libc/src/math/generic/acoshf16.cpp +++ b/libc/src/math/generic/acoshf16.cpp @@ -7,104 +7,10 @@ //===----------------------------------------------------------------------===// #include "src/math/acoshf16.h" -#include "explogxf.h" -#include "hdr/errno_macros.h" -#include "hdr/fenv_macros.h" -#include "src/__support/FPUtil/FEnvImpl.h" -#include "src/__support/FPUtil/FPBits.h" -#include "src/__support/FPUtil/PolyEval.h" -#include "src/__support/FPUtil/cast.h" -#include "src/__support/FPUtil/except_value_utils.h" -#include "src/__support/FPUtil/multiply_add.h" -#include "src/__support/FPUtil/sqrt.h" -#include "src/__support/common.h" -#include "src/__support/macros/config.h" -#include "src/__support/macros/optimization.h" +#include "src/__support/math/acoshf16.h" namespace LIBC_NAMESPACE_DECL { -static constexpr size_t N_EXCEPTS = 2; -static constexpr fputil::ExceptValues<float16, N_EXCEPTS> ACOSHF16_EXCEPTS{{ - // (input, RZ output, RU offset, RD offset, RN offset) - // x = 0x1.6dcp+1, acoshf16(x) = 0x1.b6p+0 (RZ) - {0x41B7, 0x3ED8, 1, 0, 0}, - // x = 0x1.39p+0, acoshf16(x) = 0x1.4f8p-1 (RZ) - {0x3CE4, 0x393E, 1, 0, 1}, -}}; - -LLVM_LIBC_FUNCTION(float16, acoshf16, (float16 x)) { - using FPBits = fputil::FPBits<float16>; - FPBits xbits(x); - uint16_t x_u = xbits.uintval(); - - // Check for NaN input first. - if (LIBC_UNLIKELY(xbits.is_inf_or_nan())) { - if (xbits.is_signaling_nan()) { - fputil::raise_except_if_required(FE_INVALID); - return FPBits::quiet_nan().get_val(); - } - if (xbits.is_neg()) { - fputil::set_errno_if_required(EDOM); - fputil::raise_except_if_required(FE_INVALID); - return FPBits::quiet_nan().get_val(); - } - return x; - } - - // Domain error for inputs less than 1.0. - if (LIBC_UNLIKELY(x <= 1.0f)) { - if (x == 1.0f) - return FPBits::zero().get_val(); - fputil::set_errno_if_required(EDOM); - fputil::raise_except_if_required(FE_INVALID); - return FPBits::quiet_nan().get_val(); - } - - if (auto r = ACOSHF16_EXCEPTS.lookup(xbits.uintval()); - LIBC_UNLIKELY(r.has_value())) - return r.value(); - - float xf = x; - // High-precision polynomial approximation for inputs close to 1.0 - // ([1, 1.25)). - // - // Brief derivation: - // 1. Expand acosh(1 + delta) using Taylor series around delta=0: - // acosh(1 + delta) ≈ sqrt(2 * delta) * [1 - delta/12 + 3*delta^2/160 - // - 5*delta^3/896 + 35*delta^4/18432 + ...] - // 2. Truncate the series to fit accurately for delta in [0, 0.25]. - // 3. Polynomial coefficients (from sollya) used here are: - // P(delta) ≈ 1 - 0x1.555556p-4 * delta + 0x1.333334p-6 * delta^2 - // - 0x1.6db6dcp-8 * delta^3 + 0x1.f1c71cp-10 * delta^4 - // 4. The Sollya commands used to generate these coefficients were: - // > display = hexadecimal; - // > round(1/12, SG, RN); - // > round(3/160, SG, RN); - // > round(5/896, SG, RN); - // > round(35/18432, SG, RN); - // With hexadecimal display mode enabled, the outputs were: - // 0x1.555556p-4 - // 0x1.333334p-6 - // 0x1.6db6dcp-8 - // 0x1.f1c71cp-10 - // 5. The maximum absolute error, estimated using: - // dirtyinfnorm(acosh(1 + x) - sqrt(2*x) * P(x), [0, 0.25]) - // is: - // 0x1.d84281p-22 - if (LIBC_UNLIKELY(x_u < 0x3D00U)) { - float delta = xf - 1.0f; - float sqrt_2_delta = fputil::sqrt<float>(2.0 * delta); - float pe = fputil::polyeval(delta, 0x1p+0f, -0x1.555556p-4f, 0x1.333334p-6f, - -0x1.6db6dcp-8f, 0x1.f1c71cp-10f); - float approx = sqrt_2_delta * pe; - return fputil::cast<float16>(approx); - } - - // acosh(x) = log(x + sqrt(x^2 - 1)) - float sqrt_term = fputil::sqrt<float>(fputil::multiply_add(xf, xf, -1.0f)); - float result = static_cast<float>(log_eval(xf + sqrt_term)); - - return fputil::cast<float16>(result); -} +LLVM_LIBC_FUNCTION(float16, acoshf16, (float16 x)) { return math::acoshf16(x); } } // namespace LIBC_NAMESPACE_DECL diff --git a/libc/src/math/generic/asinhf.cpp b/libc/src/math/generic/asinhf.cpp index 0bb7065..3aed3bc 100644 --- a/libc/src/math/generic/asinhf.cpp +++ b/libc/src/math/generic/asinhf.cpp @@ -19,6 +19,7 @@ namespace LIBC_NAMESPACE_DECL { LLVM_LIBC_FUNCTION(float, asinhf, (float x)) { + using namespace acoshf_internal; using FPBits_t = typename fputil::FPBits<float>; FPBits_t xbits(x); uint32_t x_u = xbits.uintval(); diff --git a/libc/src/math/generic/asinhf16.cpp b/libc/src/math/generic/asinhf16.cpp index 7878632..0a0b471 100644 --- a/libc/src/math/generic/asinhf16.cpp +++ b/libc/src/math/generic/asinhf16.cpp @@ -49,6 +49,7 @@ static constexpr fputil::ExceptValues<float16, N_EXCEPTS> ASINHF16_EXCEPTS{{ #endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS LLVM_LIBC_FUNCTION(float16, asinhf16, (float16 x)) { + using namespace acoshf_internal; using FPBits = fputil::FPBits<float16>; FPBits xbits(x); diff --git a/libc/src/math/generic/atanhf.cpp b/libc/src/math/generic/atanhf.cpp index f6fde76..602a8f0 100644 --- a/libc/src/math/generic/atanhf.cpp +++ b/libc/src/math/generic/atanhf.cpp @@ -16,6 +16,7 @@ namespace LIBC_NAMESPACE_DECL { LLVM_LIBC_FUNCTION(float, atanhf, (float x)) { + using namespace acoshf_internal; using FPBits = typename fputil::FPBits<float>; FPBits xbits(x); diff --git a/libc/src/math/generic/common_constants.cpp b/libc/src/math/generic/common_constants.cpp index 4dcf84d..42e3ff0 100644 --- a/libc/src/math/generic/common_constants.cpp +++ b/libc/src/math/generic/common_constants.cpp @@ -51,52 +51,6 @@ const float ONE_OVER_F_FLOAT[128] = { 0x1.08421p-1f, 0x1.07326p-1f, 0x1.0624dep-1f, 0x1.05198p-1f, 0x1.041042p-1f, 0x1.03091cp-1f, 0x1.020408p-1f, 0x1.010102p-1f}; -// Lookup table for (1/f) where f = 1 + n*2^(-7), n = 0..127. -const double ONE_OVER_F[128] = { - 0x1.0000000000000p+0, 0x1.fc07f01fc07f0p-1, 0x1.f81f81f81f820p-1, - 0x1.f44659e4a4271p-1, 0x1.f07c1f07c1f08p-1, 0x1.ecc07b301ecc0p-1, - 0x1.e9131abf0b767p-1, 0x1.e573ac901e574p-1, 0x1.e1e1e1e1e1e1ep-1, - 0x1.de5d6e3f8868ap-1, 0x1.dae6076b981dbp-1, 0x1.d77b654b82c34p-1, - 0x1.d41d41d41d41dp-1, 0x1.d0cb58f6ec074p-1, 0x1.cd85689039b0bp-1, - 0x1.ca4b3055ee191p-1, 0x1.c71c71c71c71cp-1, 0x1.c3f8f01c3f8f0p-1, - 0x1.c0e070381c0e0p-1, 0x1.bdd2b899406f7p-1, 0x1.bacf914c1bad0p-1, - 0x1.b7d6c3dda338bp-1, 0x1.b4e81b4e81b4fp-1, 0x1.b2036406c80d9p-1, - 0x1.af286bca1af28p-1, 0x1.ac5701ac5701bp-1, 0x1.a98ef606a63bep-1, - 0x1.a6d01a6d01a6dp-1, 0x1.a41a41a41a41ap-1, 0x1.a16d3f97a4b02p-1, - 0x1.9ec8e951033d9p-1, 0x1.9c2d14ee4a102p-1, 0x1.999999999999ap-1, - 0x1.970e4f80cb872p-1, 0x1.948b0fcd6e9e0p-1, 0x1.920fb49d0e229p-1, - 0x1.8f9c18f9c18fap-1, 0x1.8d3018d3018d3p-1, 0x1.8acb90f6bf3aap-1, - 0x1.886e5f0abb04ap-1, 0x1.8618618618618p-1, 0x1.83c977ab2beddp-1, - 0x1.8181818181818p-1, 0x1.7f405fd017f40p-1, 0x1.7d05f417d05f4p-1, - 0x1.7ad2208e0ecc3p-1, 0x1.78a4c8178a4c8p-1, 0x1.767dce434a9b1p-1, - 0x1.745d1745d1746p-1, 0x1.724287f46debcp-1, 0x1.702e05c0b8170p-1, - 0x1.6e1f76b4337c7p-1, 0x1.6c16c16c16c17p-1, 0x1.6a13cd1537290p-1, - 0x1.6816816816817p-1, 0x1.661ec6a5122f9p-1, 0x1.642c8590b2164p-1, - 0x1.623fa77016240p-1, 0x1.6058160581606p-1, 0x1.5e75bb8d015e7p-1, - 0x1.5c9882b931057p-1, 0x1.5ac056b015ac0p-1, 0x1.58ed2308158edp-1, - 0x1.571ed3c506b3ap-1, 0x1.5555555555555p-1, 0x1.5390948f40febp-1, - 0x1.51d07eae2f815p-1, 0x1.5015015015015p-1, 0x1.4e5e0a72f0539p-1, - 0x1.4cab88725af6ep-1, 0x1.4afd6a052bf5bp-1, 0x1.49539e3b2d067p-1, - 0x1.47ae147ae147bp-1, 0x1.460cbc7f5cf9ap-1, 0x1.446f86562d9fbp-1, - 0x1.42d6625d51f87p-1, 0x1.4141414141414p-1, 0x1.3fb013fb013fbp-1, - 0x1.3e22cbce4a902p-1, 0x1.3c995a47babe7p-1, 0x1.3b13b13b13b14p-1, - 0x1.3991c2c187f63p-1, 0x1.3813813813814p-1, 0x1.3698df3de0748p-1, - 0x1.3521cfb2b78c1p-1, 0x1.33ae45b57bcb2p-1, 0x1.323e34a2b10bfp-1, - 0x1.30d190130d190p-1, 0x1.2f684bda12f68p-1, 0x1.2e025c04b8097p-1, - 0x1.2c9fb4d812ca0p-1, 0x1.2b404ad012b40p-1, 0x1.29e4129e4129ep-1, - 0x1.288b01288b013p-1, 0x1.27350b8812735p-1, 0x1.25e22708092f1p-1, - 0x1.2492492492492p-1, 0x1.23456789abcdfp-1, 0x1.21fb78121fb78p-1, - 0x1.20b470c67c0d9p-1, 0x1.1f7047dc11f70p-1, 0x1.1e2ef3b3fb874p-1, - 0x1.1cf06ada2811dp-1, 0x1.1bb4a4046ed29p-1, 0x1.1a7b9611a7b96p-1, - 0x1.19453808ca29cp-1, 0x1.1811811811812p-1, 0x1.16e0689427379p-1, - 0x1.15b1e5f75270dp-1, 0x1.1485f0e0acd3bp-1, 0x1.135c81135c811p-1, - 0x1.12358e75d3033p-1, 0x1.1111111111111p-1, 0x1.0fef010fef011p-1, - 0x1.0ecf56be69c90p-1, 0x1.0db20a88f4696p-1, 0x1.0c9714fbcda3bp-1, - 0x1.0b7e6ec259dc8p-1, 0x1.0a6810a6810a7p-1, 0x1.0953f39010954p-1, - 0x1.0842108421084p-1, 0x1.073260a47f7c6p-1, 0x1.0624dd2f1a9fcp-1, - 0x1.05197f7d73404p-1, 0x1.0410410410410p-1, 0x1.03091b51f5e1ap-1, - 0x1.0204081020408p-1, 0x1.0101010101010p-1}; - // Lookup table for log(f) = log(1 + n*2^(-7)) where n = 0..127, // computed and stored as float precision constants. // Generated by Sollya with the following commands: @@ -136,52 +90,6 @@ const float LOG_F_FLOAT[128] = { 0x1.52a2d2p-1f, 0x1.54b246p-1f, 0x1.56bf9ep-1f, 0x1.58cadcp-1f, 0x1.5ad404p-1f, 0x1.5cdb1ep-1f, 0x1.5ee02ap-1f, 0x1.60e33p-1f}; -// Lookup table for log(f) = log(1 + n*2^(-7)) where n = 0..127. -const double LOG_F[128] = { - 0x0.0000000000000p+0, 0x1.fe02a6b106788p-8, 0x1.fc0a8b0fc03e3p-7, - 0x1.7b91b07d5b11ap-6, 0x1.f829b0e783300p-6, 0x1.39e87b9febd5fp-5, - 0x1.77458f632dcfcp-5, 0x1.b42dd711971bep-5, 0x1.f0a30c01162a6p-5, - 0x1.16536eea37ae0p-4, 0x1.341d7961bd1d0p-4, 0x1.51b073f06183fp-4, - 0x1.6f0d28ae56b4bp-4, 0x1.8c345d6319b20p-4, 0x1.a926d3a4ad563p-4, - 0x1.c5e548f5bc743p-4, 0x1.e27076e2af2e5p-4, 0x1.fec9131dbeabap-4, - 0x1.0d77e7cd08e59p-3, 0x1.1b72ad52f67a0p-3, 0x1.29552f81ff523p-3, - 0x1.371fc201e8f74p-3, 0x1.44d2b6ccb7d1ep-3, 0x1.526e5e3a1b437p-3, - 0x1.5ff3070a793d3p-3, 0x1.6d60fe719d21cp-3, 0x1.7ab890210d909p-3, - 0x1.87fa06520c910p-3, 0x1.9525a9cf456b4p-3, 0x1.a23bc1fe2b563p-3, - 0x1.af3c94e80bff2p-3, 0x1.bc286742d8cd6p-3, 0x1.c8ff7c79a9a21p-3, - 0x1.d5c216b4fbb91p-3, 0x1.e27076e2af2e5p-3, 0x1.ef0adcbdc5936p-3, - 0x1.fb9186d5e3e2ap-3, 0x1.0402594b4d040p-2, 0x1.0a324e27390e3p-2, - 0x1.1058bf9ae4ad5p-2, 0x1.1675cababa60ep-2, 0x1.1c898c16999fap-2, - 0x1.22941fbcf7965p-2, 0x1.2895a13de86a3p-2, 0x1.2e8e2bae11d30p-2, - 0x1.347dd9a987d54p-2, 0x1.3a64c556945e9p-2, 0x1.404308686a7e3p-2, - 0x1.4618bc21c5ec2p-2, 0x1.4be5f957778a0p-2, 0x1.51aad872df82dp-2, - 0x1.5767717455a6cp-2, 0x1.5d1bdbf5809cap-2, 0x1.62c82f2b9c795p-2, - 0x1.686c81e9b14aep-2, 0x1.6e08eaa2ba1e3p-2, 0x1.739d7f6bbd006p-2, - 0x1.792a55fdd47a2p-2, 0x1.7eaf83b82afc3p-2, 0x1.842d1da1e8b17p-2, - 0x1.89a3386c1425ap-2, 0x1.8f11e873662c7p-2, 0x1.947941c2116fap-2, - 0x1.99d958117e08ap-2, 0x1.9f323ecbf984bp-2, 0x1.a484090e5bb0ap-2, - 0x1.a9cec9a9a0849p-2, 0x1.af1293247786bp-2, 0x1.b44f77bcc8f62p-2, - 0x1.b9858969310fbp-2, 0x1.beb4d9da71b7bp-2, 0x1.c3dd7a7cdad4dp-2, - 0x1.c8ff7c79a9a21p-2, 0x1.ce1af0b85f3ebp-2, 0x1.d32fe7e00ebd5p-2, - 0x1.d83e7258a2f3ep-2, 0x1.dd46a04c1c4a0p-2, 0x1.e24881a7c6c26p-2, - 0x1.e744261d68787p-2, 0x1.ec399d2468cc0p-2, 0x1.f128f5faf06ecp-2, - 0x1.f6123fa7028acp-2, 0x1.faf588f78f31ep-2, 0x1.ffd2e0857f498p-2, - 0x1.02552a5a5d0fep-1, 0x1.04bdf9da926d2p-1, 0x1.0723e5c1cdf40p-1, - 0x1.0986f4f573520p-1, 0x1.0be72e4252a82p-1, 0x1.0e44985d1cc8bp-1, - 0x1.109f39e2d4c96p-1, 0x1.12f719593efbcp-1, 0x1.154c3d2f4d5e9p-1, - 0x1.179eabbd899a0p-1, 0x1.19ee6b467c96ep-1, 0x1.1c3b81f713c24p-1, - 0x1.1e85f5e7040d0p-1, 0x1.20cdcd192ab6dp-1, 0x1.23130d7bebf42p-1, - 0x1.2555bce98f7cbp-1, 0x1.2795e1289b11ap-1, 0x1.29d37fec2b08ap-1, - 0x1.2c0e9ed448e8bp-1, 0x1.2e47436e40268p-1, 0x1.307d7334f10bep-1, - 0x1.32b1339121d71p-1, 0x1.34e289d9ce1d3p-1, 0x1.37117b54747b5p-1, - 0x1.393e0d3562a19p-1, 0x1.3b68449fffc22p-1, 0x1.3d9026a7156fap-1, - 0x1.3fb5b84d16f42p-1, 0x1.41d8fe84672aep-1, 0x1.43f9fe2f9ce67p-1, - 0x1.4618bc21c5ec2p-1, 0x1.48353d1ea88dfp-1, 0x1.4a4f85db03ebbp-1, - 0x1.4c679afccee39p-1, 0x1.4e7d811b75bb0p-1, 0x1.50913cc01686bp-1, - 0x1.52a2d265bc5aap-1, 0x1.54b2467999497p-1, 0x1.56bf9d5b3f399p-1, - 0x1.58cadb5cd7989p-1, 0x1.5ad404c359f2cp-1, 0x1.5cdb1dc6c1764p-1, - 0x1.5ee02a9241675p-1, 0x1.60e32f44788d8p-1}; - // Range reduction constants for logarithms. // r(0) = 1, r(127) = 0.5 // r(k) = 2^-8 * ceil(2^8 * (1 - 2^-8) / (1 + k*2^-7)) diff --git a/libc/src/math/generic/common_constants.h b/libc/src/math/generic/common_constants.h index 291816a..72b1d564 100644 --- a/libc/src/math/generic/common_constants.h +++ b/libc/src/math/generic/common_constants.h @@ -11,6 +11,7 @@ #include "src/__support/FPUtil/triple_double.h" #include "src/__support/macros/config.h" +#include "src/__support/math/acosh_float_constants.h" #include "src/__support/math/exp_constants.h" #include "src/__support/number_pair.h" @@ -20,16 +21,10 @@ namespace LIBC_NAMESPACE_DECL { // computed and stored as float precision constants. extern const float ONE_OVER_F_FLOAT[128]; -// Lookup table for (1/f) where f = 1 + n*2^(-7), n = 0..127. -extern const double ONE_OVER_F[128]; - // Lookup table for log(f) = log(1 + n*2^(-7)) where n = 0..127, // computed and stored as float precision constants. extern const float LOG_F_FLOAT[128]; -// Lookup table for log(f) = log(1 + n*2^(-7)) where n = 0..127. -extern const double LOG_F[128]; - // Lookup table for range reduction constants r for logarithms. extern const float R[128]; diff --git a/libc/src/math/generic/erff.cpp b/libc/src/math/generic/erff.cpp index 44607a5..003b346 100644 --- a/libc/src/math/generic/erff.cpp +++ b/libc/src/math/generic/erff.cpp @@ -7,180 +7,10 @@ //===----------------------------------------------------------------------===// #include "src/math/erff.h" -#include "src/__support/FPUtil/FPBits.h" -#include "src/__support/FPUtil/PolyEval.h" -#include "src/__support/FPUtil/except_value_utils.h" -#include "src/__support/FPUtil/multiply_add.h" -#include "src/__support/common.h" -#include "src/__support/macros/config.h" -#include "src/__support/macros/optimization.h" // LIBC_UNLIKELY +#include "src/__support/math/erff.h" namespace LIBC_NAMESPACE_DECL { -// Polynomials approximating erf(x)/x on ( k/8, (k + 1)/8 ) generated by Sollya -// with: -// > P = fpminimax(erf(x)/x, [|0, 2, 4, 6, 8, 10, 12, 14|], [|D...|], -// [k/8, (k + 1)/8]); -// for k = 0..31. -constexpr double COEFFS[32][8] = { - {0x1.20dd750429b6dp0, -0x1.812746b037753p-2, 0x1.ce2f219e8596ap-4, - -0x1.b82cdacb78fdap-6, 0x1.56479297dfda5p-8, -0x1.8b3ac5455ef02p-11, - -0x1.126fcac367e3bp-8, 0x1.2d0bdb3ba4984p-4}, - {0x1.20dd750429b6dp0, -0x1.812746b0379a8p-2, 0x1.ce2f21a03cf2ap-4, - -0x1.b82ce30de083ep-6, 0x1.565bcad3eb60fp-8, -0x1.c02c66f659256p-11, - 0x1.f92f673385229p-14, -0x1.def402648ae9p-17}, - {0x1.20dd750429b34p0, -0x1.812746b032dcep-2, 0x1.ce2f219d84aaep-4, - -0x1.b82ce22dcf139p-6, 0x1.565b9efcd4af1p-8, -0x1.c021f1af414bcp-11, - 0x1.f7c6d177eff82p-14, -0x1.c9e4410dcf865p-17}, - {0x1.20dd750426eabp0, -0x1.812746ae592c7p-2, 0x1.ce2f211525f14p-4, - -0x1.b82ccc125e63fp-6, 0x1.56596f261cfd3p-8, -0x1.bfde1ff8eeecfp-11, - 0x1.f31a9d15dc5d8p-14, -0x1.a5a4362844b3cp-17}, - {0x1.20dd75039c705p0, -0x1.812746777e74dp-2, 0x1.ce2f17af98a1bp-4, - -0x1.b82be4b817cbep-6, 0x1.564bec2e2962ep-8, -0x1.bee86f9da3558p-11, - 0x1.e9443689dc0ccp-14, -0x1.79c0f230805d8p-17}, - {0x1.20dd74f811211p0, -0x1.81274371a3e8fp-2, 0x1.ce2ec038262e5p-4, - -0x1.b8265b82c5e1fp-6, 0x1.5615a2e239267p-8, -0x1.bc63ae023dcebp-11, - 0x1.d87c2102f7e06p-14, -0x1.49584bea41d62p-17}, - {0x1.20dd746d063e3p0, -0x1.812729a8a950fp-2, 0x1.ce2cb0a2df232p-4, - -0x1.b80eca1f51278p-6, 0x1.5572e26c46815p-8, -0x1.b715e5638b65ep-11, - 0x1.bfbb195484968p-14, -0x1.177a565c15c52p-17}, - {0x1.20dd701b44486p0, -0x1.812691145f237p-2, 0x1.ce23a06b8cfd9p-4, - -0x1.b7c1dc7245288p-6, 0x1.53e92f7f397ddp-8, -0x1.ad97cc4acf0b2p-11, - 0x1.9f028b2b09b71p-14, -0x1.cdc4da08da8c1p-18}, - {0x1.20dd5715ac332p0, -0x1.8123e680bd0ebp-2, 0x1.ce0457aded691p-4, - -0x1.b6f52d52bed4p-6, 0x1.50c291b84414cp-8, -0x1.9ea246b1ad4a9p-11, - 0x1.77654674e0cap-14, -0x1.737c11a1bcebbp-18}, - {0x1.20dce6593e114p0, -0x1.811a59c02eadcp-2, 0x1.cdab53c7cd7d5p-4, - -0x1.b526d2e321eedp-6, 0x1.4b1d32cd8b994p-8, -0x1.8963143ec0a1ep-11, - 0x1.4ad5700e4db91p-14, -0x1.231e100e43ef2p-18}, - {0x1.20db48bfd5a62p0, -0x1.80fdd84f9e308p-2, 0x1.ccd340d462983p-4, - -0x1.b196a2928768p-6, 0x1.4210c2c13a0f7p-8, -0x1.6dbdfb4ff71aep-11, - 0x1.1bca2d17fbd71p-14, -0x1.bca36f90c7cf5p-19}, - {0x1.20d64b2f8f508p0, -0x1.80b4d4f19fa8bp-2, 0x1.cb088197262e3p-4, - -0x1.ab51fd02e5b99p-6, 0x1.34e1e5e81a632p-8, -0x1.4c66377b502cep-11, - 0x1.d9ad25066213cp-15, -0x1.4b0df7dd0cfa1p-19}, - {0x1.20c8fc1243576p0, -0x1.8010cb2009e27p-2, 0x1.c7a47e9299315p-4, - -0x1.a155be5683654p-6, 0x1.233502694997bp-8, -0x1.26c94b7d813p-11, - 0x1.8094f1de25fb9p-15, -0x1.e0e3d776c6eefp-20}, - {0x1.20a9bd1611bc1p0, -0x1.7ec7fbce83f9p-2, 0x1.c1d757d7317b7p-4, - -0x1.92c160cd589fp-6, 0x1.0d307269cc5c2p-8, -0x1.fda5b0d2d1879p-12, - 0x1.2fdd7b3b14a7fp-15, -0x1.54eed4a26af5ap-20}, - {0x1.20682834f943dp0, -0x1.7c73f747bf5a9p-2, 0x1.b8c2db4a9ffd1p-4, - -0x1.7f0e4ffe989ecp-6, 0x1.e7061eae4166ep-9, -0x1.ad36e873fff2dp-12, - 0x1.d39222396128ep-16, -0x1.d83dacec5ea6bp-21}, - {0x1.1feb8d12676d7p0, -0x1.7898347284afep-2, 0x1.aba3466b34451p-4, - -0x1.663adc573e2f9p-6, 0x1.ae99fb17c3e08p-9, -0x1.602f950ad5535p-12, - 0x1.5e9717490609dp-16, -0x1.3fca107bbc8d5p-21}, - {0x1.1f12fe3c536fap0, -0x1.72b1d1f22e6d3p-2, 0x1.99fc0eed4a896p-4, - -0x1.48db0a87bd8c6p-6, 0x1.73e368895aa61p-9, -0x1.19b35d5301fc8p-12, - 0x1.007987e4bb033p-16, -0x1.a7edcd4c2dc7p-22}, - {0x1.1db7b0df84d5dp0, -0x1.6a4e4a41cde02p-2, 0x1.83bbded16455dp-4, - -0x1.2809b3b36977ep-6, 0x1.39c08bab44679p-9, -0x1.b7b45a70ed119p-13, - 0x1.6e99b36410e7bp-17, -0x1.13619bb7ebc0cp-22}, - {0x1.1bb1c85c4a527p0, -0x1.5f23b99a249a3p-2, 0x1.694c91fa0d12cp-4, - -0x1.053e1ce11c72dp-6, 0x1.02bf72c50ea78p-9, -0x1.4f478fb56cb02p-13, - 0x1.005f80ecbe213p-17, -0x1.5f2446bde7f5bp-23}, - {0x1.18dec3bd51f9dp0, -0x1.5123f58346186p-2, 0x1.4b8a1ca536ab4p-4, - -0x1.c4243015cc723p-7, 0x1.a1a8a01d351efp-10, -0x1.f466b34f1d86bp-14, - 0x1.5f835eea0bf6ap-18, -0x1.b83165b939234p-24}, - {0x1.152804c3369f4p0, -0x1.4084cd4afd4bcp-2, 0x1.2ba2e836e47aap-4, - -0x1.800f2dfc6904bp-7, 0x1.4a6daf0669c59p-10, -0x1.6e326ab872317p-14, - 0x1.d9761a6a755a5p-19, -0x1.0fca33f9dd4b5p-24}, - {0x1.1087ad68356aap0, -0x1.2dbb044707459p-2, 0x1.0aea8ceaa0384p-4, - -0x1.40b516d52b3d2p-7, 0x1.00c9e05f01d22p-10, -0x1.076afb0dc0ff7p-14, - 0x1.39fadec400657p-19, -0x1.4b5761352e7e3p-25}, - {0x1.0b0a7a8ba4a22p0, -0x1.196990d22d4a1p-2, 0x1.d5551e6ac0c4dp-5, - -0x1.07cce1770bd1ap-7, 0x1.890347b8848bfp-11, -0x1.757ec96750b6ap-15, - 0x1.9b258a1e06bcep-20, -0x1.8fc6d22da7572p-26}, - {0x1.04ce2be70fb47p0, -0x1.0449e4b0b9cacp-2, 0x1.97f7424f4b0e7p-5, - -0x1.ac825439c42f4p-8, 0x1.28f5f65426dfbp-11, -0x1.05b699a90f90fp-15, - 0x1.0a888eecf4593p-20, -0x1.deace2b32bb31p-27}, - {0x1.fbf9fb0e11cc8p-1, -0x1.de2640856545ap-3, 0x1.5f5b1f47f851p-5, - -0x1.588bc71eb41b9p-8, 0x1.bc6a0a772f56dp-12, -0x1.6b9fad1f1657ap-16, - 0x1.573204ba66504p-21, -0x1.1d38065c94e44p-27}, - {0x1.ed8f18c99e031p-1, -0x1.b4cb6acd903b4p-3, 0x1.2c7f3dddd6fc1p-5, - -0x1.13052067df4ep-8, 0x1.4a5027444082fp-12, -0x1.f672bab0e2554p-17, - 0x1.b83c756348cc9p-22, -0x1.534f1a1079499p-28}, - {0x1.debd33044166dp-1, -0x1.8d7cd9053f7d8p-3, 0x1.ff9957fb3d6e7p-6, - -0x1.b50be55de0f36p-9, 0x1.e92c8ec53a628p-13, -0x1.5a4b88d508007p-17, - 0x1.1a27737559e26p-22, -0x1.942ae62cb2c14p-29}, - {0x1.cfdbf0386f3bdp-1, -0x1.68e33d93b0dc4p-3, 0x1.b2683d58f53dep-6, - -0x1.5a9174e70d26fp-9, 0x1.69ddd326d49cdp-13, -0x1.dd8f397a8219cp-18, - 0x1.6a755016ad4ddp-23, -0x1.e366e0139187dp-30}, - {0x1.c132adb8d7464p-1, -0x1.475a899f61b46p-3, 0x1.70a431397a77cp-6, - -0x1.12e3d35beeee2p-9, 0x1.0c16b05738333p-13, -0x1.4a47f873e144ep-18, - 0x1.d3d494c698c02p-24, -0x1.2302c59547fe5p-30}, - {0x1.b2f5fd05555e7p-1, -0x1.28feefbe03ec7p-3, 0x1.3923acbb3a676p-6, - -0x1.b4ff793cd6358p-10, 0x1.8ea0eb8c913bcp-14, -0x1.cb31ec2baceb1p-19, - 0x1.30011e7e80c04p-24, -0x1.617710635cb1dp-31}, - {0x1.a54853cd9593ep-1, -0x1.0dbdbaea4dc8ep-3, 0x1.0a93e2c20a0fdp-6, - -0x1.5c969ff401ea8p-10, 0x1.29e0cc64fe627p-14, -0x1.4160d8e9d3c2ap-19, - 0x1.8e7b67594624ap-25, -0x1.b1cf2c975b09bp-32}, - {0x1.983ceece09ff8p-1, -0x1.eacc78f7a2dp-4, 0x1.c74418410655fp-7, - -0x1.1756a050e441ep-10, 0x1.bff3650f7f548p-15, -0x1.c56c0217d3adap-20, - 0x1.07b4918d0b489p-25, -0x1.0d4be8c1c50f8p-32}, -}; - -LLVM_LIBC_FUNCTION(float, erff, (float x)) { - using FPBits = typename fputil::FPBits<float>; - FPBits xbits(x); - - uint32_t x_u = xbits.uintval(); - uint32_t x_abs = x_u & 0x7fff'ffffU; - - if (LIBC_UNLIKELY(x_abs >= 0x4080'0000U)) { - const float ONE[2] = {1.0f, -1.0f}; - const float SMALL[2] = {-0x1.0p-25f, 0x1.0p-25f}; - - int sign = xbits.is_neg() ? 1 : 0; - - if (LIBC_UNLIKELY(x_abs >= 0x7f80'0000U)) { - if (xbits.is_signaling_nan()) { - fputil::raise_except_if_required(FE_INVALID); - return FPBits::quiet_nan().get_val(); - } - return (x_abs > 0x7f80'0000) ? x : ONE[sign]; - } - - return ONE[sign] + SMALL[sign]; - } - -#ifndef LIBC_MATH_HAS_SKIP_ACCURATE_PASS - // Exceptional mask = common 0 bits of 2 exceptional values. - constexpr uint32_t EXCEPT_MASK = 0x809a'6184U; - - if (LIBC_UNLIKELY((x_abs & EXCEPT_MASK) == 0)) { - // Exceptional values - if (LIBC_UNLIKELY(x_abs == 0x3f65'9229U)) // |x| = 0x1.cb2452p-1f - return x < 0.0f ? fputil::round_result_slightly_down(-0x1.972ea8p-1f) - : fputil::round_result_slightly_up(0x1.972ea8p-1f); - if (LIBC_UNLIKELY(x_abs == 0x4004'1e6aU)) // |x| = 0x1.083cd4p+1f - return x < 0.0f ? fputil::round_result_slightly_down(-0x1.fe3462p-1f) - : fputil::round_result_slightly_up(0x1.fe3462p-1f); - if (x_abs == 0U) - return x; - } -#endif // !LIBC_MATH_HAS_SKIP_ACCURATE_PASS - - // Polynomial approximation: - // erf(x) ~ x * (c0 + c1 * x^2 + c2 * x^4 + ... + c7 * x^14) - double xd = static_cast<double>(x); - double xsq = xd * xd; - - const uint32_t EIGHT = 3 << FPBits::FRACTION_LEN; - int idx = static_cast<int>(FPBits(x_abs + EIGHT).get_val()); - - double x4 = xsq * xsq; - double c0 = fputil::multiply_add(xsq, COEFFS[idx][1], COEFFS[idx][0]); - double c1 = fputil::multiply_add(xsq, COEFFS[idx][3], COEFFS[idx][2]); - double c2 = fputil::multiply_add(xsq, COEFFS[idx][5], COEFFS[idx][4]); - double c3 = fputil::multiply_add(xsq, COEFFS[idx][7], COEFFS[idx][6]); - - double x8 = x4 * x4; - double p0 = fputil::multiply_add(x4, c1, c0); - double p1 = fputil::multiply_add(x4, c3, c2); - - return static_cast<float>(xd * fputil::multiply_add(x8, p1, p0)); -} +LLVM_LIBC_FUNCTION(float, erff, (float x)) { return math::erff(x); } } // namespace LIBC_NAMESPACE_DECL diff --git a/libc/src/math/generic/explogxf.h b/libc/src/math/generic/explogxf.h index be4328a..a2a6d60 100644 --- a/libc/src/math/generic/explogxf.h +++ b/libc/src/math/generic/explogxf.h @@ -13,6 +13,7 @@ #include "src/__support/common.h" #include "src/__support/macros/properties/cpu_features.h" +#include "src/__support/math/acoshf_utils.h" #include "src/__support/math/exp10f_utils.h" #include "src/__support/math/exp_utils.h" @@ -163,41 +164,6 @@ LIBC_INLINE static float log_eval_f(float x) { return result; } -// x should be positive, normal finite value -LIBC_INLINE static double log_eval(double x) { - // For x = 2^ex * (1 + mx) - // log(x) = ex * log(2) + log(1 + mx) - using FPB = fputil::FPBits<double>; - FPB bs(x); - - double ex = static_cast<double>(bs.get_exponent()); - - // p1 is the leading 7 bits of mx, i.e. - // p1 * 2^(-7) <= m_x < (p1 + 1) * 2^(-7). - int p1 = static_cast<int>(bs.get_mantissa() >> (FPB::FRACTION_LEN - 7)); - - // Set bs to (1 + (mx - p1*2^(-7)) - bs.set_uintval(bs.uintval() & (FPB::FRACTION_MASK >> 7)); - bs.set_biased_exponent(FPB::EXP_BIAS); - // dx = (mx - p1*2^(-7)) / (1 + p1*2^(-7)). - double dx = (bs.get_val() - 1.0) * ONE_OVER_F[p1]; - - // Minimax polynomial of log(1 + dx) generated by Sollya with: - // > P = fpminimax(log(1 + x)/x, 6, [|D...|], [0, 2^-7]); - const double COEFFS[6] = {-0x1.ffffffffffffcp-2, 0x1.5555555552ddep-2, - -0x1.ffffffefe562dp-3, 0x1.9999817d3a50fp-3, - -0x1.554317b3f67a5p-3, 0x1.1dc5c45e09c18p-3}; - double dx2 = dx * dx; - double c1 = fputil::multiply_add(dx, COEFFS[1], COEFFS[0]); - double c2 = fputil::multiply_add(dx, COEFFS[3], COEFFS[2]); - double c3 = fputil::multiply_add(dx, COEFFS[5], COEFFS[4]); - - double p = fputil::polyeval(dx2, dx, c1, c2, c3); - double result = - fputil::multiply_add(ex, /*log(2)*/ 0x1.62e42fefa39efp-1, LOG_F[p1] + p); - return result; -} - } // namespace LIBC_NAMESPACE_DECL #endif // LLVM_LIBC_SRC_MATH_GENERIC_EXPLOGXF_H diff --git a/libc/src/math/generic/log1pf.cpp b/libc/src/math/generic/log1pf.cpp index 7f61429..16b1b34 100644 --- a/libc/src/math/generic/log1pf.cpp +++ b/libc/src/math/generic/log1pf.cpp @@ -37,6 +37,7 @@ namespace internal { // We don't need to treat denormal and 0 LIBC_INLINE float log(double x) { + using namespace acoshf_internal; constexpr double LOG_2 = 0x1.62e42fefa39efp-1; using FPBits = typename fputil::FPBits<double>; diff --git a/libc/test/shared/CMakeLists.txt b/libc/test/shared/CMakeLists.txt index 07a4e73..89b607d 100644 --- a/libc/test/shared/CMakeLists.txt +++ b/libc/test/shared/CMakeLists.txt @@ -9,6 +9,10 @@ add_fp_unittest( DEPENDS libc.src.__support.math.acos libc.src.__support.math.acosf + libc.src.__support.math.acosf16 + libc.src.__support.math.acoshf + libc.src.__support.math.acoshf16 + libc.src.__support.math.erff libc.src.__support.math.exp libc.src.__support.math.exp10 libc.src.__support.math.exp10f diff --git a/libc/test/shared/shared_math_test.cpp b/libc/test/shared/shared_math_test.cpp index 40fea3b..8d3cebd 100644 --- a/libc/test/shared/shared_math_test.cpp +++ b/libc/test/shared/shared_math_test.cpp @@ -14,6 +14,8 @@ TEST(LlvmLibcSharedMathTest, AllFloat16) { int exponent; + EXPECT_FP_EQ(0x0p+0f, LIBC_NAMESPACE::shared::acoshf16(1.0f)); + EXPECT_FP_EQ(0x1p+0f16, LIBC_NAMESPACE::shared::exp10f16(0.0f16)); EXPECT_FP_EQ(0x1p+0f16, LIBC_NAMESPACE::shared::expf16(0.0f16)); @@ -25,6 +27,8 @@ TEST(LlvmLibcSharedMathTest, AllFloat16) { EXPECT_FP_EQ_ALL_ROUNDING(0.75f16, LIBC_NAMESPACE::shared::frexpf16(24.0f, &exponent)); EXPECT_EQ(exponent, 5); + + EXPECT_FP_EQ(0x1.921fb6p+0f16, LIBC_NAMESPACE::shared::acosf16(0.0f16)); } #endif @@ -35,6 +39,8 @@ TEST(LlvmLibcSharedMathTest, AllFloat) { EXPECT_FP_EQ(0x1.921fb6p+0, LIBC_NAMESPACE::shared::acosf(0.0f)); EXPECT_FP_EQ(0x1p+0f, LIBC_NAMESPACE::shared::exp10f(0.0f)); EXPECT_FP_EQ(0x1p+0f, LIBC_NAMESPACE::shared::expf(0.0f)); + EXPECT_FP_EQ(0x0p+0f, LIBC_NAMESPACE::shared::erff(0.0f)); + EXPECT_FP_EQ(0x0p+0f, LIBC_NAMESPACE::shared::acoshf(1.0f)); EXPECT_FP_EQ_ALL_ROUNDING(0.75f, LIBC_NAMESPACE::shared::frexpf(24.0f, &exponent)); diff --git a/libc/test/src/math/explogxf_test.cpp b/libc/test/src/math/explogxf_test.cpp index ff1181e..49cc962 100644 --- a/libc/test/src/math/explogxf_test.cpp +++ b/libc/test/src/math/explogxf_test.cpp @@ -44,6 +44,7 @@ TEST_F(LlvmLibcExplogfTest, ExpInFloatRange) { } TEST_F(LlvmLibcExplogfTest, LogInFloatRange) { - CHECK_DATA(0.0f, inf, mpfr::Operation::Log, LIBC_NAMESPACE::log_eval, - f_normal, def_count, def_prec); + CHECK_DATA(0.0f, inf, mpfr::Operation::Log, + LIBC_NAMESPACE::acoshf_internal::log_eval, f_normal, def_count, + def_prec); } |