Special Relativity in Financial Modeling 1.0.0
Lorentz transforms, spacetime classification, and geodesic price paths for quantitative finance
Loading...
Searching...
No Matches
cpu_features.hpp
Go to the documentation of this file.
1#pragma once
2/**
3 * @file cpu_features.hpp
4 * @brief Runtime SIMD capability detection for the SRFM acceleration layer.
5 *
6 * Module: include/srfm/simd/
7 * Owner: AGT-08 (AVX-512 SIMD Acceleration) — 2026-03-01
8 *
9 * Responsibility
10 * --------------
11 * Probe the executing CPU at start-up for the widest SIMD instruction set
12 * available, and expose the result as a strongly-typed enum so that the
13 * dispatch layer can select the correct kernel at zero per-call overhead.
14 *
15 * Supported levels (highest wins):
16 * SCALAR — always available; pure C++ reference path
17 * SSE42 — SSE 4.2 (128-bit, 2 doubles per register)
18 * AVX2 — AVX2 (256-bit, 4 doubles per register)
19 * AVX512F — AVX-512F (512-bit, 8 doubles per register)
20 *
21 * Guarantees
22 * ----------
23 * • detect_simd_level() is thread-safe: result is computed once.
24 * • No dynamic allocation, no exceptions, noexcept throughout.
25 * • Portable: supports GCC/Clang (__builtin_cpu_supports) and
26 * MSVC (__cpuid / __cpuidex).
27 *
28 * NOT Responsible For
29 * -------------------
30 * • Verifying OS support (XSAVE/XGETBV) — assumed present if CPUID
31 * advertises the feature and the OS is modern (≥ Win10 / Linux ≥ 4.x).
32 * • AVX-512 sub-features beyond F (VL, DQ, BW) — only F is required.
33 */
34
35#include <cstdint>
36
37// ── Compiler-specific CPUID helpers ───────────────────────────────────────────
38
39#if defined(_MSC_VER)
40# include <intrin.h> // __cpuid, __cpuidex
41# include <immintrin.h> // _xgetbv
42#elif defined(__GNUC__) || defined(__clang__)
43# include <cpuid.h> // __get_cpuid, __get_cpuid_count
44# include <immintrin.h>
45#endif
46
47namespace srfm::simd {
48
49// ── SimdLevel ─────────────────────────────────────────────────────────────────
50
51/**
52 * @brief Ordered enumeration of SIMD capability tiers.
53 *
54 * Compare with < / <= to check "at least this level":
55 * @code
56 * if (detect_simd_level() >= SimdLevel::AVX2) { ... }
57 * @endcode
58 */
59enum class SimdLevel : std::uint8_t {
60 SCALAR = 0, ///< No SIMD; pure scalar C++ path.
61 SSE42 = 1, ///< SSE 4.2 — 128-bit; 2 doubles per YMM half.
62 AVX2 = 2, ///< AVX2 — 256-bit; 4 doubles per YMM register.
63 AVX512F = 3, ///< AVX-512F — 512-bit; 8 doubles per ZMM register.
64};
65
66// ── Internal CPUID utilities ───────────────────────────────────────────────────
67
68namespace detail {
69
70/// Thin portable wrapper around the CPUID instruction.
71/// @param leaf EAX input (function).
72/// @param subleaf ECX input (sub-function). 0 for most leaves.
73/// @param out Output array: [EAX, EBX, ECX, EDX].
74inline void cpuid(std::uint32_t leaf, std::uint32_t subleaf,
75 std::uint32_t out[4]) noexcept {
76#if defined(_MSC_VER)
77 // MSVC intrinsic signature: void __cpuidex(int[4], int, int)
78 int regs[4] = {0};
79 __cpuidex(regs, static_cast<int>(leaf), static_cast<int>(subleaf));
80 for (int i = 0; i < 4; ++i)
81 out[i] = static_cast<std::uint32_t>(regs[i]);
82#elif defined(__GNUC__) || defined(__clang__)
83 __cpuid_count(leaf, subleaf, out[0], out[1], out[2], out[3]);
84#else
85 // Fallback: assume no SIMD
86 out[0] = out[1] = out[2] = out[3] = 0u;
87#endif
88}
89
90/// Read XCR0 (used to verify OS XSAVE support for YMM/ZMM state).
91inline std::uint64_t xgetbv(std::uint32_t xcr) noexcept {
92#if defined(_MSC_VER)
93 return _xgetbv(xcr);
94#elif (defined(__GNUC__) || defined(__clang__)) && defined(__XSAVE__)
95 std::uint32_t lo, hi;
96 __asm__ volatile("xgetbv" : "=a"(lo), "=d"(hi) : "c"(xcr));
97 return (static_cast<std::uint64_t>(hi) << 32u) | lo;
98#else
99 (void)xcr;
100 return 0u; // Cannot read XCR0; assume OS doesn't save YMM/ZMM
101#endif
102}
103
104/// Probe OS support for YMM (AVX) register save/restore.
105inline bool os_saves_ymm() noexcept {
106 std::uint32_t regs[4] = {};
107 cpuid(0x1u, 0u, regs);
108 // ECX bit 27: OSXSAVE
109 const bool osxsave = (regs[2] >> 27u) & 1u;
110 if (!osxsave) return false;
111 // XCR0 bits 1 (SSE) and 2 (AVX/YMM) must both be set
112 const std::uint64_t xcr0 = xgetbv(0u);
113 return (xcr0 & 0x6u) == 0x6u;
114}
115
116/// Probe OS support for ZMM (AVX-512) register save/restore.
117inline bool os_saves_zmm() noexcept {
118 if (!os_saves_ymm()) return false;
119 // XCR0 bits 5 (opmask), 6 (ZMM hi256), 7 (ZMM hi16) must all be set
120 const std::uint64_t xcr0 = xgetbv(0u);
121 return (xcr0 & 0xE0u) == 0xE0u;
122}
123
124} // namespace detail
125
126// ── detect_simd_level ─────────────────────────────────────────────────────────
127
128/**
129 * @brief Detect the widest SIMD level available on the executing CPU.
130 *
131 * The result is computed once on first call and cached in a function-local
132 * static, so subsequent calls are effectively free (a single load).
133 *
134 * Probes (in order, highest wins):
135 * 1. AVX-512F — CPUID leaf 7 EBX bit 16, plus OS ZMM save.
136 * 2. AVX2 — CPUID leaf 7 EBX bit 5, plus OS YMM save.
137 * 3. SSE 4.2 — CPUID leaf 1 ECX bit 20.
138 * 4. SCALAR — unconditional fallback.
139 *
140 * @return The highest SimdLevel the current CPU and OS support.
141 *
142 * # Panics
143 * This function never panics.
144 *
145 * @example
146 * @code
147 * auto level = srfm::simd::detect_simd_level();
148 * if (level >= srfm::simd::SimdLevel::AVX512F) {
149 * // use AVX-512 path
150 * }
151 * @endcode
152 */
153[[nodiscard]] inline SimdLevel detect_simd_level() noexcept {
154 // Function-local static — initialized exactly once, thread-safe per C++11.
155 static const SimdLevel kLevel = []() noexcept -> SimdLevel {
156 // ── Leaf 0: get max supported leaf ───────────────────────────────────
157 std::uint32_t max_leaf[4] = {};
158 detail::cpuid(0u, 0u, max_leaf);
159 const std::uint32_t max_func = max_leaf[0];
160
161 // ── Leaf 1: SSE4.2 ───────────────────────────────────────────────────
162 if (max_func < 1u) return SimdLevel::SCALAR;
163 std::uint32_t leaf1[4] = {};
164 detail::cpuid(1u, 0u, leaf1);
165
166 const bool sse42 = (leaf1[2] >> 20u) & 1u; // ECX bit 20
167
168 // ── Leaf 7: AVX2, AVX-512F ───────────────────────────────────────────
169 if (max_func < 7u) return sse42 ? SimdLevel::SSE42 : SimdLevel::SCALAR;
170 std::uint32_t leaf7[4] = {};
171 detail::cpuid(7u, 0u, leaf7);
172
173 const bool avx2 = (leaf7[1] >> 5u) & 1u; // EBX bit 5
174 const bool avx512f = (leaf7[1] >> 16u) & 1u; // EBX bit 16
175
176 // ── Check OS register-save support ───────────────────────────────────
177 if (avx512f && detail::os_saves_zmm()) return SimdLevel::AVX512F;
178 if (avx2 && detail::os_saves_ymm()) return SimdLevel::AVX2;
179 if (sse42) return SimdLevel::SSE42;
180 return SimdLevel::SCALAR;
181 }();
182
183 return kLevel;
184}
185
186// ── Convenience predicates ────────────────────────────────────────────────────
187
188/// Returns true when AVX-512F is available on this CPU/OS.
189[[nodiscard]] inline bool has_avx512f() noexcept {
191}
192
193/// Returns true when AVX2 is available on this CPU/OS.
194[[nodiscard]] inline bool has_avx2() noexcept {
196}
197
198/// Returns true when SSE 4.2 is available on this CPU/OS.
199[[nodiscard]] inline bool has_sse42() noexcept {
201}
202
203/// Human-readable name for a SimdLevel value.
204[[nodiscard]] inline const char* simd_level_name(SimdLevel level) noexcept {
205 switch (level) {
206 case SimdLevel::SCALAR: return "SCALAR";
207 case SimdLevel::SSE42: return "SSE42";
208 case SimdLevel::AVX2: return "AVX2";
209 case SimdLevel::AVX512F: return "AVX512F";
210 default: return "UNKNOWN";
211 }
212}
213
214} // namespace srfm::simd
bool os_saves_zmm() noexcept
Probe OS support for ZMM (AVX-512) register save/restore.
void cpuid(std::uint32_t leaf, std::uint32_t subleaf, std::uint32_t out[4]) noexcept
std::uint64_t xgetbv(std::uint32_t xcr) noexcept
Read XCR0 (used to verify OS XSAVE support for YMM/ZMM state).
bool os_saves_ymm() noexcept
Probe OS support for YMM (AVX) register save/restore.
bool has_avx2() noexcept
Returns true when AVX2 is available on this CPU/OS.
const char * simd_level_name(SimdLevel level) noexcept
Human-readable name for a SimdLevel value.
bool has_avx512f() noexcept
Returns true when AVX-512F is available on this CPU/OS.
SimdLevel
Ordered enumeration of SIMD capability tiers.
@ SCALAR
No SIMD; pure scalar C++ path.
@ AVX512F
AVX-512F — 512-bit; 8 doubles per ZMM register.
@ SSE42
SSE 4.2 — 128-bit; 2 doubles per YMM half.
@ AVX2
AVX2 — 256-bit; 4 doubles per YMM register.
bool has_sse42() noexcept
Returns true when SSE 4.2 is available on this CPU/OS.
SimdLevel detect_simd_level() noexcept