Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
simd_dispatch.cpp
Go to the documentation of this file.
1/**
2 * @file simd_dispatch.cpp
3 * @brief Runtime dispatch: routes computeBetaBatch / computeGammaBatch to
4 * the widest available SIMD kernel at process start-up.
5 *
6 * Module: src/simd/
7 * Owner: AGT-08 — 2026-03-01
8 *
9 * Responsibility
10 * --------------
11 * Implements the two public batch functions and the BetaCalculator class
12 * declared in include/srfm/simd/simd_dispatch.hpp.
13 *
14 * Dispatch strategy
15 * -----------------
16 * detect_simd_level() is called once and cached by cpu_features.hpp.
17 * Based on the result, a compile-time-known function pointer is selected:
18 *
19 * AVX512F → detail::compute_beta_avx512 / compute_gamma_avx512
20 * AVX2 → detail::compute_beta_avx2 / compute_gamma_avx2
21 * * → detail::compute_beta_scalar / compute_gamma_scalar
22 *
23 * Wrapping raw doubles into BetaVelocity / LorentzFactor
24 * ------------------------------------------------------
25 * After the SIMD kernel fills a double[] buffer:
26 *
27 * • Beta: BetaVelocity::make(d).value() — the clamp in the kernel
28 * guarantees make() always returns a value, never nullopt.
29 *
30 * • Gamma: SimdGammaCompute::make(d) — uses the friend declaration added
31 * to LorentzFactor so that we avoid an extra sqrt() per element.
32 *
33 * Memory layout
34 * -------------
35 * Intermediate double buffers are stack-allocated for N ≤ STACK_THRESHOLD
36 * and heap-allocated (std::vector<double>) for larger batches, keeping
37 * the common hot-path (N ≈ 256) stack-resident and cache-hot.
38 */
39
40#include "srfm/simd/simd_dispatch.hpp" // public header (include/ on path)
41#include "simd_batch_detail.hpp" // internal detail header (src/simd/ on path)
42#include "srfm/simd/cpu_features.hpp" // runtime detection
43
44#include <vector>
45#include <cstddef>
46#include <memory>
47
48namespace srfm::simd {
49
50// ── Dispatch helpers ──────────────────────────────────────────────────────────
51
52namespace {
53
54/// Maximum number of doubles to store on the stack (avoid VLA; use alloca
55/// only on compilers that support it; fall back to heap beyond this limit).
56static constexpr std::size_t STACK_THRESHOLD = 1024;
57
58using BetaKernelFn = void(*)(const double*, std::size_t, double&, double*) noexcept;
59using GammaKernelFn = void(*)(const double*, std::size_t, double*) noexcept;
60
61/// Select the beta kernel function pointer at call time (single branch).
62[[nodiscard]] inline BetaKernelFn select_beta_kernel() noexcept {
63 switch (detect_simd_level()) {
66 default: return detail::compute_beta_scalar;
67 }
68}
69
70/// Select the gamma kernel function pointer at call time.
71[[nodiscard]] inline GammaKernelFn select_gamma_kernel() noexcept {
72 switch (detect_simd_level()) {
75 default: return detail::compute_gamma_scalar;
76 }
77}
78
79} // anonymous namespace
80
81// ── computeBetaBatch (free function) ─────────────────────────────────────────
82
83std::vector<srfm::momentum::BetaVelocity>
84computeBetaBatch(const std::vector<double>& velocities,
85 double& running_max) noexcept
86{
87 const std::size_t n = velocities.size();
88 std::vector<srfm::momentum::BetaVelocity> result;
89 if (n == 0) return result;
90 result.reserve(n);
91
92 // Allocate intermediate double buffer (heap for large N).
93 std::vector<double> buf(n);
94
95 // Run the dispatched kernel.
96 static const BetaKernelFn kernel = select_beta_kernel();
97 kernel(velocities.data(), n, running_max, buf.data());
98
99 // Wrap each computed beta double into a BetaVelocity.
100 // The kernel guarantees buf[i] ∈ [0, BETA_CLAMP_LIMIT], so make()
101 // always returns a value.
102 for (std::size_t i = 0; i < n; ++i) {
103 // make() validates; since kernel already clamped, this is always valid.
104 auto opt = srfm::momentum::BetaVelocity::make(buf[i]);
105 if (opt.has_value()) {
106 result.push_back(*opt);
107 } else {
108 // Defensive: push zero beta if (somehow) validation fails.
109 result.push_back(*srfm::momentum::BetaVelocity::make(0.0));
110 }
111 }
112
113 return result;
114}
115
116// ── computeGammaBatch (free function) ────────────────────────────────────────
117
118std::vector<srfm::momentum::LorentzFactor>
120 const std::vector<srfm::momentum::BetaVelocity>& betas) noexcept
121{
122 const std::size_t n = betas.size();
123 std::vector<srfm::momentum::LorentzFactor> result;
124 if (n == 0) return result;
125 result.reserve(n);
126
127 // Extract raw double values from BetaVelocity objects.
128 std::vector<double> beta_buf(n);
129 for (std::size_t i = 0; i < n; ++i) {
130 beta_buf[i] = betas[i].value();
131 }
132
133 // Gamma output buffer.
134 std::vector<double> gamma_buf(n);
135
136 // Run the dispatched kernel.
137 static const GammaKernelFn kernel = select_gamma_kernel();
138 kernel(beta_buf.data(), n, gamma_buf.data());
139
140 // Wrap each computed gamma double into a LorentzFactor using the
141 // friend accessor that avoids a redundant scalar sqrt per element.
142 for (std::size_t i = 0; i < n; ++i) {
143 result.push_back(SimdGammaCompute::make(gamma_buf[i]));
144 }
145
146 return result;
147}
148
149// ── BetaCalculator ────────────────────────────────────────────────────────────
150
152 : running_max_(0.0)
153 , simd_level_(detect_simd_level())
154{}
155
156std::vector<srfm::momentum::BetaVelocity>
158 const std::vector<double>& velocities) noexcept
159{
160 return srfm::simd::computeBetaBatch(velocities, running_max_);
161}
162
163std::vector<srfm::momentum::LorentzFactor>
165 const std::vector<srfm::momentum::BetaVelocity>& betas) noexcept
166{
167 return srfm::simd::computeGammaBatch(betas);
168}
169
170void BetaCalculator::reset() noexcept {
171 running_max_ = 0.0;
172}
173
174double BetaCalculator::running_max() const noexcept {
175 return running_max_;
176}
177
179 return simd_level_;
180}
181
182} // namespace srfm::simd
static std::optional< BetaVelocity > make(double value) noexcept
Validate and construct a BetaVelocity.
Definition momentum.cpp:18
std::vector< srfm::momentum::LorentzFactor > computeGammaBatch(const std::vector< srfm::momentum::BetaVelocity > &betas) noexcept
Compute γ_i = 1/√(1 − β_i²) for a batch of betas.
void reset() noexcept
Reset the running maximum to 0.0.
std::vector< srfm::momentum::BetaVelocity > computeBetaBatch(const std::vector< double > &velocities) noexcept
Compute β_i = |v_i| / running_max for a batch of velocities.
SimdLevel simd_level() const noexcept
Returns the SIMD level selected for this process.
double running_max() const noexcept
Returns the current running maximum (read-only).
BetaCalculator() noexcept
Construct a fresh calculator. running_max starts at 0.0.
Runtime SIMD capability detection for the SRFM acceleration layer.
void compute_gamma_avx512(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) gamma batch kernel.
void compute_beta_avx2(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) beta batch kernel.
Definition beta_avx2.cpp:44
void compute_gamma_scalar(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
Scalar reference implementation of the gamma batch kernel.
void compute_gamma_avx2(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) gamma batch kernel.
void compute_beta_scalar(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
Scalar reference implementation of the beta batch kernel.
void compute_beta_avx512(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) beta batch kernel.
std::vector< srfm::momentum::LorentzFactor > computeGammaBatch(const std::vector< srfm::momentum::BetaVelocity > &betas) noexcept
Compute γ_i = 1/√(1 − β_i²) for every element.
SimdLevel
Ordered enumeration of SIMD capability tiers.
@ AVX512F
AVX-512F — 512-bit; 8 doubles per ZMM register.
@ AVX2
AVX2 — 256-bit; 4 doubles per YMM register.
std::vector< srfm::momentum::BetaVelocity > computeBetaBatch(const std::vector< double > &velocities, double &running_max) noexcept
SimdLevel detect_simd_level() noexcept
Internal raw-double compute signatures for SIMD batch kernels.
Public API for SIMD-accelerated β and γ batch computation.
static srfm::momentum::LorentzFactor make(double gamma_val) noexcept