FastLED 3.10.6
Loading...
Searching...
No Matches
wave_perf_bench.h
Go to the documentation of this file.
1/*
2Wave-PDE-specific performance bench helpers. The wave simulator owns
3this code; AutoResearch (or any other RPC layer) is an *external
4binder* that calls into these helpers and translates results into its
5preferred wire format. The simulator side does not know about RPC
6topology, JSON, or any specific transport.
7
8Used by meta #3113 — see `agents/docs/research/wave_pde_alternatives.md`
9for the broader motivation. The RPC binder for these helpers lives in
10`examples/AutoResearch/AutoResearchRemote.cpp`.
11*/
12
13#pragma once
14
16#include "fl/stl/chrono.h"
17#include "fl/stl/int.h"
18#include "fl/stl/noexcept.h"
19#include "fl/stl/stdint.h"
20#include "fl/stl/unique_ptr.h"
21
22namespace fl {
23
24namespace wave_perf {
25
26// Result of a single-shot wave-PDE perf measurement.
28 bool success = false;
29 u32 total_us = 0;
30 double us_per_update = 0.0;
33};
34
35// Result of a multi-repeat stability check.
37 bool success = false;
38 u32 repeats = 0;
39 double mean_us_per_update = 0.0;
41 double std_dev_pct = 0.0;
42};
43
44// Bench config range gates. Tight enough to refuse pathological RPC
45// payloads (e.g. accidentally requesting a 256-MB grid) without
46// constraining real experiments.
47inline bool isValidGrid(u32 W, u32 H) FL_NO_EXCEPT {
48 return W >= 4 && H >= 4 && W <= 1024 && H <= 1024;
49}
50inline bool isValidIterations(u32 iterations) FL_NO_EXCEPT {
51 return iterations >= 1 && iterations <= 100000;
52}
53inline bool isValidRepeats(u32 repeats) FL_NO_EXCEPT {
54 return repeats >= 2 && repeats <= 64;
55}
56
57// Build a wave simulator seeded with a deterministic +/-0.5 checkerboard.
58// Bench helpers below use this so every run starts from the same state.
59// The +/-0.5 amplitude is deliberate — avoids the +/-1.0 float_to_fixed
60// asymmetry noted in #3083's audit.
62 u32 W, u32 H, LaplacianStencil stencil) FL_NO_EXCEPT {
63 auto sim = fl::make_unique<WaveSimulation2D_Real>(W, H, 0.16f, 6.0f);
64 sim->setHalfDuplex(false);
65 sim->setStencil(stencil);
66 for (u32 y = 0; y < H; ++y) {
67 for (u32 x = 0; x < W; ++x) {
68 sim->setf(static_cast<fl::size>(x),
69 static_cast<fl::size>(y),
70 ((x + y) & 1u) ? -0.5f : 0.5f);
71 }
72 }
73 sim->update(); // warm-up: prime caches, get past init quirks
74 return sim;
75}
76
77// Single-shot: time `iterations` calls to update() and report
78// per-cell-per-update cost. The corresponding `loads_only=true` mode
79// (memory-bound baseline) lives in `runWavePerfLoadsOnly` so the
80// signatures stay simple — the gap between the two reveals memory
81// vs compute cost (critical for the ESP32-S3 PSRAM analysis from
82// #3114).
83inline WavePerfResult runWavePerf(u32 W, u32 H, u32 iterations,
86 if (!isValidGrid(W, H) || !isValidIterations(iterations)) {
87 return r;
88 }
89 auto sim = makeBenchSim(W, H, stencil);
90 const u32 t0 = fl::micros();
91 for (u32 i = 0; i < iterations; ++i) {
92 sim->update();
93 }
94 const u32 t1 = fl::micros();
95 r.success = true;
96 r.total_us = t1 - t0;
97 const double cells_per_update = static_cast<double>(W) * H;
98 r.us_per_update = static_cast<double>(r.total_us)
99 / static_cast<double>(iterations);
100 r.us_per_cell_per_update = r.us_per_update / cells_per_update;
102 ? (1.0e6 / r.us_per_update) : 0.0;
103 return r;
104}
105
106// Memory-bound baseline: read every inner cell `iterations` times into
107// a volatile sink. Touches the same memory pattern as update() with
108// none of the kernel arithmetic. The gap between this and runWavePerf
109// is the memory-vs-compute breakdown.
110inline WavePerfResult runWavePerfLoadsOnly(u32 W, u32 H, u32 iterations,
113 if (!isValidGrid(W, H) || !isValidIterations(iterations)) {
114 return r;
115 }
116 auto sim = makeBenchSim(W, H, stencil);
117 volatile fl::i32 sink = 0;
118 const u32 t0 = fl::micros();
119 for (u32 i = 0; i < iterations; ++i) {
120 fl::i32 acc = 0;
121 for (u32 y = 0; y < H; ++y) {
122 for (u32 x = 0; x < W; ++x) {
123 acc += sim->geti16(static_cast<fl::size>(x),
124 static_cast<fl::size>(y));
125 }
126 }
127 sink += acc;
128 }
129 const u32 t1 = fl::micros();
130 (void)sink;
131 r.success = true;
132 r.total_us = t1 - t0;
133 const double cells_per_update = static_cast<double>(W) * H;
134 r.us_per_update = static_cast<double>(r.total_us)
135 / static_cast<double>(iterations);
136 r.us_per_cell_per_update = r.us_per_update / cells_per_update;
138 ? (1.0e6 / r.us_per_update) : 0.0;
139 return r;
140}
141
142// Repeat-stability: run `repeats` measurements of `runWavePerf` and
143// report mean + std_dev_pct of us_per_update. Caller checks
144// std_dev_pct < 5%; higher → IRQ noise or RPC framing jitter is
145// contaminating the data, mark UNTRUSTED.
147 u32 W, u32 H, u32 iterations, u32 repeats,
150 if (!isValidGrid(W, H) || !isValidIterations(iterations)
151 || !isValidRepeats(repeats)) {
152 return r;
153 }
154 // Single sim reused across repeats — the warm-up cost is paid once
155 // and each repeat measures steady-state cost.
156 auto sim = makeBenchSim(W, H, stencil);
157 double sum = 0.0;
158 double sum_sq = 0.0;
159 for (u32 j = 0; j < repeats; ++j) {
160 const u32 t0 = fl::micros();
161 for (u32 i = 0; i < iterations; ++i) {
162 sim->update();
163 }
164 const u32 t1 = fl::micros();
165 const double us_per_update =
166 static_cast<double>(t1 - t0) / static_cast<double>(iterations);
167 sum += us_per_update;
168 sum_sq += us_per_update * us_per_update;
169 }
170 const double mean = sum / static_cast<double>(repeats);
171 const double variance = (sum_sq / static_cast<double>(repeats))
172 - (mean * mean);
173 double std_dev = 0.0;
174 if (variance > 0.0) {
175 // Newton-Raphson sqrt — avoids pulling in <cmath> on platforms
176 // where it's costly. 6 iterations gives ~14 significant digits.
177 double s = variance;
178 for (int k = 0; k < 6; ++k) {
179 s = 0.5 * (s + variance / s);
180 }
181 std_dev = s;
182 }
183 r.success = true;
184 r.repeats = repeats;
185 r.mean_us_per_update = mean;
186 r.std_dev_us_per_update = std_dev;
187 r.std_dev_pct = (mean > 0.0) ? (100.0 * std_dev / mean) : 0.0;
188 return r;
189}
190
191} // namespace wave_perf
192} // namespace fl
static const int H
Definition PerfDisc.ino:21
static const int W
Definition PerfDisc.ino:20
FastLED chrono implementation - duration types for time measurements.
bool isValidRepeats(u32 repeats) FL_NO_EXCEPT
WavePerfRepeatResult runWavePerfRepeat(u32 W, u32 H, u32 iterations, u32 repeats, LaplacianStencil stencil) FL_NO_EXCEPT
fl::unique_ptr< WaveSimulation2D_Real > makeBenchSim(u32 W, u32 H, LaplacianStencil stencil) FL_NO_EXCEPT
WavePerfResult runWavePerf(u32 W, u32 H, u32 iterations, LaplacianStencil stencil) FL_NO_EXCEPT
bool isValidIterations(u32 iterations) FL_NO_EXCEPT
bool isValidGrid(u32 W, u32 H) FL_NO_EXCEPT
WavePerfResult runWavePerfLoadsOnly(u32 W, u32 H, u32 iterations, LaplacianStencil stencil) FL_NO_EXCEPT
fl::enable_if<!fl::is_array< T >::value, unique_ptr< T > >::type make_unique(Args &&... args) FL_NO_EXCEPT
Definition unique_ptr.h:261
fl::u32 micros()
Universal microsecond timer - returns microseconds since system startup.
InputGamut g FL_NO_EXCEPT
Definition rgbw.h:121
Base definition for an LED controller.
Definition crgb.hpp:179