Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
beta_avx512.cpp
Go to the documentation of this file.
1/**
2 * @file beta_avx512.cpp
3 * @brief AVX-512F (512-bit, 8-wide) implementation of the beta batch kernel.
4 *
5 * Module: src/simd/
6 * Owner: AGT-08 — 2026-03-01
7 *
8 * Algorithm: Batch-max (8-wide SIMD)
9 * ------------------------------------
10 * 1. Pass 1 (vectorised): compute batch_max using 8-wide abs + _mm512_reduce_max_pd.
11 * 2. Update scalar running_max.
12 * 3. Pass 2 (vectorised): broadcast running_max, divide |v_i|/denom, clamp.
13 * 4. Tail (scalar): handle n % 8 remaining elements.
14 *
15 * Using the batch maximum as the single denominator for all elements in one
16 * call means the results are bit-identical to compute_beta_scalar() (which
17 * uses the same batch-max algorithm) regardless of SIMD width.
18 *
19 * Note: Compiled with -mavx512f / /arch:AVX512.
20 */
21
22#include "simd_batch_detail.hpp"
23#include "momentum/momentum.hpp"
24
25#include <immintrin.h>
26#include <cmath>
27
28namespace srfm::simd::detail {
29
30static constexpr double BETA_CLAMP_LIMIT =
32
34 const double* SRFM_RESTRICT velocities,
35 std::size_t n,
36 double& running_max,
37 double* SRFM_RESTRICT out) noexcept
38{
39 if (n == 0) return;
40
41 constexpr std::size_t LANE = 8;
42
43 // ── Pass 1: compute batch_max ────────────────────────────────────────────
44 __m512d vmax = _mm512_setzero_pd();
45 std::size_t i = 0;
46 for (; i + LANE <= n; i += LANE) {
47 __m512d v = _mm512_loadu_pd(velocities + i);
48 __m512d abs_v = _mm512_abs_pd(v);
49 vmax = _mm512_max_pd(vmax, abs_v);
50 }
51 // Reduce the 8-lane vector to a single scalar max.
52 double batch_max = _mm512_reduce_max_pd(vmax);
53 // Scalar tail for pass 1
54 for (; i < n; ++i) {
55 const double a = std::abs(velocities[i]);
56 if (a > batch_max) batch_max = a;
57 }
58
59 // Update running max
60 if (batch_max > running_max) running_max = batch_max;
61 const double denom = (running_max > 0.0) ? running_max : 1.0;
62
63 // ── Pass 2: compute betas ────────────────────────────────────────────────
64 const __m512d denom_v = _mm512_set1_pd(denom);
65 const __m512d clamp_v = _mm512_set1_pd(BETA_CLAMP_LIMIT);
66
67 i = 0;
68 for (; i + LANE <= n; i += LANE) {
69 __m512d v = _mm512_loadu_pd(velocities + i);
70 __m512d abs_v = _mm512_abs_pd(v);
71 __m512d beta = _mm512_div_pd(abs_v, denom_v);
72 beta = _mm512_min_pd(beta, clamp_v);
73 _mm512_storeu_pd(out + i, beta);
74 }
75 // Scalar tail for pass 2
76 for (; i < n; ++i) {
77 const double abs_v = std::abs(velocities[i]);
78 double beta = abs_v / denom;
79 if (beta > BETA_CLAMP_LIMIT) beta = BETA_CLAMP_LIMIT;
80 out[i] = beta;
81 }
82}
83
84} // namespace srfm::simd::detail
constexpr double BETA_MAX_SAFE
Definition momentum.hpp:62
static constexpr double BETA_CLAMP_LIMIT
Definition beta_avx2.cpp:30
void compute_beta_avx512(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) beta batch kernel.
Internal raw-double compute signatures for SIMD batch kernels.
#define SRFM_RESTRICT
Momentum-Velocity Signal Processor (AGT-03 / SRFM)