Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
simd_batch_detail.hpp
Go to the documentation of this file.
1#pragma once
2/**
3 * @file simd_batch_detail.hpp
4 * @brief Internal raw-double compute signatures for SIMD batch kernels.
5 *
6 * Module: src/simd/ (internal — do NOT include from public headers)
7 * Owner: AGT-08 — 2026-03-01
8 *
9 * Responsibility
10 * --------------
11 * Declare the low-level SIMD and scalar kernel signatures that operate on
12 * raw double pointers. These are implementation details; callers should
13 * use the public API in include/srfm/simd/simd_dispatch.hpp instead.
14 *
15 * Kernels
16 * -------
17 *
18 * Beta batch:
19 * Computes beta_i = |velocities[i]| / running_max for all i in [0, n).
20 * running_max is updated in-place: at each chunk boundary it is set to
21 * max(running_max, max(|velocities[chunk]|)). This guarantees:
22 * • beta_i ∈ [0.0, BETA_MAX_SAFE] always
23 * • running_max is monotonically non-decreasing across calls
24 *
25 * Gamma batch:
26 * Computes gamma_i = 1.0 / sqrt(1.0 − betas[i]²) for all i in [0, n).
27 * Input betas must be pre-validated (0 ≤ betas[i] ≤ BETA_MAX_SAFE).
28 * Clamping is applied before sqrt to prevent NaN.
29 *
30 * Guarantees
31 * ----------
32 * • All functions are noexcept.
33 * • Tail elements (n % LANE != 0) are handled with a scalar fallback.
34 * • Output pointers must be writable for n doubles; they need not be
35 * aligned (all loads/stores use unaligned intrinsics).
36 * • Input and output pointers must not alias (SRFM_RESTRICT).
37 *
38 * NOT Responsible For
39 * • Wrapping outputs into BetaVelocity / LorentzFactor (dispatch layer).
40 * • Selecting the correct kernel at runtime (simd_dispatch.cpp).
41 */
42
43#include <cstddef>
44
45// ── Portability: restrict keyword ─────────────────────────────────────────────
46// GCC/Clang use SRFM_RESTRICT; MSVC uses __restrict (no trailing underscores).
47#if defined(_MSC_VER)
48# define SRFM_RESTRICT __restrict
49#else
50# define SRFM_RESTRICT __restrict__
51#endif
52
53namespace srfm::simd::detail {
54
55// ── Beta kernels ───────────────────────────────────────────────────────────────
56
57/**
58 * @brief Scalar reference implementation of the beta batch kernel.
59 *
60 * Always available regardless of hardware. Used as the authoritative
61 * correctness reference and as the tail handler for wider kernels.
62 *
63 * @param velocities Input price velocities (any finite double).
64 * @param n Number of elements.
65 * @param running_max Current running maximum of |velocity|; updated in-place.
66 * @param out Output beta values, one per input velocity.
67 */
69 const double* SRFM_RESTRICT velocities,
70 std::size_t n,
71 double& running_max,
72 double* SRFM_RESTRICT out) noexcept;
73
74/**
75 * @brief AVX2 (256-bit, 4-wide) beta batch kernel.
76 *
77 * Requires AVX2 support (detected at runtime). Falls back to scalar for
78 * tail elements when n % 4 != 0.
79 */
81 const double* SRFM_RESTRICT velocities,
82 std::size_t n,
83 double& running_max,
84 double* SRFM_RESTRICT out) noexcept;
85
86/**
87 * @brief AVX-512F (512-bit, 8-wide) beta batch kernel.
88 *
89 * Requires AVX-512F support (detected at runtime). Falls back to scalar
90 * for tail elements when n % 8 != 0.
91 */
93 const double* SRFM_RESTRICT velocities,
94 std::size_t n,
95 double& running_max,
96 double* SRFM_RESTRICT out) noexcept;
97
98// ── Gamma kernels ──────────────────────────────────────────────────────────────
99
100/**
101 * @brief Scalar reference implementation of the gamma batch kernel.
102 *
103 * Computes gamma_i = 1.0 / sqrt(1.0 - betas[i]^2).
104 * Clamps betas[i] to BETA_CLAMP_SAFE before the sqrt to prevent NaN.
105 *
106 * @param betas Input beta values in [0, BETA_MAX_SAFE].
107 * @param n Number of elements.
108 * @param out Output gamma values, one per input beta.
109 */
111 const double* SRFM_RESTRICT betas,
112 std::size_t n,
113 double* SRFM_RESTRICT out) noexcept;
114
115/**
116 * @brief AVX2 (256-bit, 4-wide) gamma batch kernel.
117 */
119 const double* SRFM_RESTRICT betas,
120 std::size_t n,
121 double* SRFM_RESTRICT out) noexcept;
122
123/**
124 * @brief AVX-512F (512-bit, 8-wide) gamma batch kernel.
125 */
127 const double* SRFM_RESTRICT betas,
128 std::size_t n,
129 double* SRFM_RESTRICT out) noexcept;
130
131} // namespace srfm::simd::detail
void compute_gamma_avx512(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) gamma batch kernel.
void compute_beta_avx2(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) beta batch kernel.
Definition beta_avx2.cpp:44
void compute_gamma_scalar(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
Scalar reference implementation of the gamma batch kernel.
void compute_gamma_avx2(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) gamma batch kernel.
void compute_beta_scalar(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
Scalar reference implementation of the beta batch kernel.
void compute_beta_avx512(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) beta batch kernel.
#define SRFM_RESTRICT