Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
beta_avx2.cpp
Go to the documentation of this file.
1/**
2 * @file beta_avx2.cpp
3 * @brief AVX2 (256-bit, 4-wide) implementation of the beta batch kernel.
4 *
5 * Module: src/simd/
6 * Owner: AGT-08 — 2026-03-01
7 *
8 * Algorithm: Batch-max (4-wide SIMD)
9 * ------------------------------------
10 * 1. Pass 1 (vectorised): compute batch_max using 4-wide SIMD abs + hmax.
11 * 2. Update scalar running_max.
12 * 3. Pass 2 (vectorised): broadcast running_max, divide |v_i| by it, clamp.
13 * 4. Tail (scalar): handle n % 4 remaining elements.
14 *
15 * Correctness guarantee: produces bit-identical results to compute_beta_scalar()
16 * for any input, because both use the same batch_max before dividing.
17 *
18 * Note: Compiled with -mavx2 / /arch:AVX2.
19 */
20
21#include "simd_batch_detail.hpp"
22#include "momentum/momentum.hpp"
23
24#include <immintrin.h>
25#include <cmath>
26#include <cstdint>
27
29
30static constexpr double BETA_CLAMP_LIMIT =
32
33static constexpr std::uint64_t ABS_MASK_U64 = 0x7FFF'FFFF'FFFF'FFFFu;
34
35// Horizontal max of 4-lane __m256d → scalar double.
36[[nodiscard]] static inline double hmax_pd_avx2(__m256d v) noexcept {
37 __m256d hi = _mm256_permute2f128_pd(v, v, 0x01);
38 __m256d mx1 = _mm256_max_pd(v, hi);
39 __m256d swp = _mm256_permute_pd(mx1, 0x05);
40 __m256d mx2 = _mm256_max_pd(mx1, swp);
41 return _mm256_cvtsd_f64(mx2);
42}
43
45 const double* SRFM_RESTRICT velocities,
46 std::size_t n,
47 double& running_max,
48 double* SRFM_RESTRICT out) noexcept
49{
50 if (n == 0) return;
51
52 constexpr std::size_t LANE = 4;
53
54 const __m256d abs_mask = _mm256_castsi256_pd(
55 _mm256_set1_epi64x(static_cast<long long>(ABS_MASK_U64)));
56
57 // ── Pass 1: compute batch_max ────────────────────────────────────────────
58 __m256d vmax = _mm256_setzero_pd();
59 std::size_t i = 0;
60 for (; i + LANE <= n; i += LANE) {
61 __m256d v = _mm256_loadu_pd(velocities + i);
62 __m256d abs_v = _mm256_and_pd(v, abs_mask);
63 vmax = _mm256_max_pd(vmax, abs_v);
64 }
65 double batch_max = hmax_pd_avx2(vmax);
66 // Scalar tail for pass 1
67 for (; i < n; ++i) {
68 const double a = std::abs(velocities[i]);
69 if (a > batch_max) batch_max = a;
70 }
71
72 // Update running max
73 if (batch_max > running_max) running_max = batch_max;
74 const double denom = (running_max > 0.0) ? running_max : 1.0;
75
76 // ── Pass 2: compute betas ────────────────────────────────────────────────
77 const __m256d denom_v = _mm256_set1_pd(denom);
78 const __m256d clamp_v = _mm256_set1_pd(BETA_CLAMP_LIMIT);
79
80 i = 0;
81 for (; i + LANE <= n; i += LANE) {
82 __m256d v = _mm256_loadu_pd(velocities + i);
83 __m256d abs_v = _mm256_and_pd(v, abs_mask);
84 __m256d beta = _mm256_div_pd(abs_v, denom_v);
85 beta = _mm256_min_pd(beta, clamp_v);
86 _mm256_storeu_pd(out + i, beta);
87 }
88 // Scalar tail for pass 2
89 for (; i < n; ++i) {
90 const double abs_v = std::abs(velocities[i]);
91 double beta = abs_v / denom;
92 if (beta > BETA_CLAMP_LIMIT) beta = BETA_CLAMP_LIMIT;
93 out[i] = beta;
94 }
95}
96
97} // namespace srfm::simd::detail
constexpr double BETA_MAX_SAFE
Definition momentum.hpp:62
static constexpr double BETA_CLAMP_LIMIT
Definition beta_avx2.cpp:30
static double hmax_pd_avx2(__m256d v) noexcept
Definition beta_avx2.cpp:36
void compute_beta_avx2(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) beta batch kernel.
Definition beta_avx2.cpp:44
static constexpr std::uint64_t ABS_MASK_U64
Definition beta_avx2.cpp:33
Internal raw-double compute signatures for SIMD batch kernels.
#define SRFM_RESTRICT
Momentum-Velocity Signal Processor (AGT-03 / SRFM)