Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
gamma_avx2.cpp
Go to the documentation of this file.
1/**
2 * @file gamma_avx2.cpp
3 * @brief AVX2 (256-bit, 4-wide) implementation of the gamma batch kernel.
4 *
5 * Module: src/simd/
6 * Owner: AGT-08 — 2026-03-01
7 *
8 * Responsibility
9 * --------------
10 * Vectorised gamma_i = 1.0 / sqrt(1.0 - betas[i]^2) for machines with
11 * AVX2 but without AVX-512F. Processes 4 doubles per SIMD cycle.
12 *
13 * Algorithm
14 * ---------
15 * For each 4-element chunk:
16 * 1. Load 4 betas: _mm256_loadu_pd
17 * 2. Clamp to BETA_CLAMP_LIMIT: _mm256_min_pd
18 * 3. Square: _mm256_mul_pd(b, b)
19 * 4. Subtract from 1.0: _mm256_sub_pd(ones, b2) → denom ∈ (0,1]
20 * 5. sqrt: _mm256_sqrt_pd(denom)
21 * 6. Divide 1.0 by sqrt: _mm256_div_pd(ones, sqrt_d) → gamma
22 * 7. Store: _mm256_storeu_pd
23 * Tail (n % 4 != 0): scalar fallback.
24 *
25 * Note: This file must be compiled with -mavx2 / /arch:AVX2.
26 */
27
28#include "simd_batch_detail.hpp"
29#include "momentum/momentum.hpp"
30
31#include <immintrin.h>
32#include <cmath>
33#include <cstddef>
34
35namespace srfm::simd::detail {
36
37static constexpr double BETA_CLAMP_LIMIT =
39
41 const double* SRFM_RESTRICT betas,
42 std::size_t n,
43 double* SRFM_RESTRICT out) noexcept
44{
45 constexpr std::size_t LANE = 4;
46
47 const __m256d ones = _mm256_set1_pd(1.0);
48 const __m256d clamp_v = _mm256_set1_pd(BETA_CLAMP_LIMIT);
49
50 std::size_t i = 0;
51
52 // ── Vectorised main loop ───────────────────────────────────────────────────
53 for (; i + LANE <= n; i += LANE) {
54 // 1. Load 4 betas.
55 __m256d b = _mm256_loadu_pd(betas + i);
56
57 // 2. Clamp each beta to BETA_CLAMP_LIMIT.
58 b = _mm256_min_pd(b, clamp_v);
59
60 // 3. Compute b² = b * b.
61 __m256d b2 = _mm256_mul_pd(b, b);
62
63 // 4. Compute 1 - b². Result is in (0, 1] for valid clamped betas.
64 __m256d denom = _mm256_sub_pd(ones, b2);
65
66 // 5. sqrt(1 - b²).
67 __m256d sqrt_d = _mm256_sqrt_pd(denom);
68
69 // 6. gamma = 1.0 / sqrt(1 - b²).
70 __m256d gamma = _mm256_div_pd(ones, sqrt_d);
71
72 // 7. Store.
73 _mm256_storeu_pd(out + i, gamma);
74 }
75
76 // ── Scalar tail ───────────────────────────────────────────────────────────
77 for (; i < n; ++i) {
78 const double b = (betas[i] > BETA_CLAMP_LIMIT) ? BETA_CLAMP_LIMIT : betas[i];
79 const double b2 = b * b;
80 out[i] = 1.0 / std::sqrt(1.0 - b2);
81 }
82}
83
84} // namespace srfm::simd::detail
constexpr double BETA_MAX_SAFE
Definition momentum.hpp:62
static constexpr double BETA_CLAMP_LIMIT
Definition beta_avx2.cpp:30
void compute_gamma_avx2(const double *__restrict__ betas, std::size_t n, double *__restrict__ out) noexcept
AVX2 (256-bit, 4-wide) gamma batch kernel.
Internal raw-double compute signatures for SIMD batch kernels.
#define SRFM_RESTRICT
Momentum-Velocity Signal Processor (AGT-03 / SRFM)