41 constexpr std::size_t LANE = 8;
44 __m512d vmax = _mm512_setzero_pd();
46 for (; i + LANE <= n; i += LANE) {
47 __m512d v = _mm512_loadu_pd(velocities + i);
48 __m512d abs_v = _mm512_abs_pd(v);
49 vmax = _mm512_max_pd(vmax, abs_v);
52 double batch_max = _mm512_reduce_max_pd(vmax);
55 const double a = std::abs(velocities[i]);
56 if (a > batch_max) batch_max = a;
60 if (batch_max > running_max) running_max = batch_max;
61 const double denom = (running_max > 0.0) ? running_max : 1.0;
64 const __m512d denom_v = _mm512_set1_pd(denom);
68 for (; i + LANE <= n; i += LANE) {
69 __m512d v = _mm512_loadu_pd(velocities + i);
70 __m512d abs_v = _mm512_abs_pd(v);
71 __m512d beta = _mm512_div_pd(abs_v, denom_v);
72 beta = _mm512_min_pd(beta, clamp_v);
73 _mm512_storeu_pd(out + i, beta);
77 const double abs_v = std::abs(velocities[i]);
78 double beta = abs_v / denom;
void compute_beta_avx512(const double *__restrict__ velocities, std::size_t n, double &running_max, double *__restrict__ out) noexcept
AVX-512F (512-bit, 8-wide) beta batch kernel.