FastLED 3.10.6
Loading...
Searching...
No Matches

◆ update()

void fl::WaveSimulation2D_Real::update ( )

Definition at line 275 of file wave_simulation_real.cpp.hpp.

275 {
276 i16 *curr = (whichGrid == 0 ? grid1.data() : grid2.data());
277 i16 *next = (whichGrid == 0 ? grid2.data() : grid1.data());
278
279 // Update horizontal boundaries.
280 for (fl::size j = 0; j < height + 2; ++j) {
281 if (mXCylindrical) {
282 curr[j * stride + 0] = curr[j * stride + width];
283 curr[j * stride + (width + 1)] = curr[j * stride + 1];
284 } else {
285 curr[j * stride + 0] = curr[j * stride + 1];
286 curr[j * stride + (width + 1)] = curr[j * stride + width];
287 }
288 }
289
290 // Update vertical boundaries.
291 for (fl::size i = 0; i < width + 2; ++i) {
292 curr[0 * stride + i] = curr[1 * stride + i];
293 curr[(height + 1) * stride + i] = curr[height * stride + i];
294 }
295
296 i32 mCourantSq32 = static_cast<i32>(mCourantSq);
297 // Hoist the lower-saturation bound — see WaveSimulation1D_Real::update()
298 // for the rationale. Fold half-duplex into the per-cell clamp instead
299 // of running a second pass over the grid (especially expensive on
300 // PSRAM-backed grids).
301 const i32 q15_min = mHalfDuplex ? 0 : -32768;
302
303 // Update each inner cell. Per-row hoist: lift j*stride out of the inner
304 // loop, set up row pointers to curr/next/above/below, and mark them
305 // FL_RESTRICT_PARAM so the optimizer knows curr and next don't alias.
306 // This gives the compiler a clean dependency picture across iterations
307 // (it can vectorize / interleave loads without re-deriving the address).
308 //
309 // Outer-level branch on the stencil keeps the inner loop branchless;
310 // the two kernels are otherwise identical (Q15 multiply, damping,
311 // clamp). FivePoint is the backward-compatible default; the wrapper
312 // class WaveSimulation2D auto-selects NinePointIsotropic at high
313 // super-sample factors where the anisotropy of the 5-point stencil
314 // becomes visually obvious.
315 const bool nine_point = (mStencil == LaplacianStencil::NinePointIsotropic);
316 for (fl::size j = 1; j <= height; ++j) {
317 const fl::size row = j * stride;
318 const i16* FL_RESTRICT_PARAM row_curr = curr + row;
319 const i16* FL_RESTRICT_PARAM row_above = curr + (row - stride);
320 const i16* FL_RESTRICT_PARAM row_below = curr + (row + stride);
321 i16* FL_RESTRICT_PARAM row_next = next + row;
322 for (fl::size i = 1; i <= width; ++i) {
323 const i32 c = row_curr[i];
324 i32 laplacian;
325 if (nine_point) {
326 // 9-point isotropic Laplacian, scaled up by 6 to keep
327 // everything in integer arithmetic:
328 // 6 * lap = (NW+NE+SW+SE) + 4*(N+S+E+W) - 20*C
329 // The /6 is folded into the term computation below so we
330 // pay it once per cell rather than four times.
331 const i32 diag = (i32)row_above[i - 1] + row_above[i + 1] +
332 row_below[i - 1] + row_below[i + 1];
333 const i32 nbr = (i32)row_above[i] + row_below[i] +
334 row_curr[i - 1] + row_curr[i + 1];
335 laplacian = diag + (nbr << 2) - 20 * c;
336 } else {
337 // Standard 5-point Laplacian: N + S + E + W - 4*C.
338 laplacian = (i32)row_curr[i + 1] + row_curr[i - 1] +
339 row_above[i] + row_below[i] - (c << 2);
340 }
341 // Promote to i64 before the multiply. With the 2D CFL clamp at
342 // 0.5, the 5-point worst case product is ~4.3e9 and the
343 // 9-point (scaled by 6, so |lap| ~6x larger) is ~2.6e10 — both
344 // past i32 max (2.15e9). The i64 promote also acts as a safety
345 // net if the clamp is ever regressed.
346 i64 product = static_cast<i64>(mCourantSq32) * laplacian;
347 i32 term;
348 if (nine_point) {
349 // Undo the x6 scaling on the 9-point Laplacian here.
350 // Integer division by a compile-time-constant 6 is lowered
351 // to a reciprocal-multiply on Cortex-M4+ and to a small
352 // shift+add on AVR; either way it's well under the cost
353 // of computing the Laplacian itself.
354 term = static_cast<i32>((product >> 15) / 6);
355 } else {
356 term = static_cast<i32>(product >> 15);
357 }
358 // f = -next[index] + 2 * curr[index] + mCourantSq * laplacian.
359 i32 f = -(i32)row_next[i] + (c << 1) + term;
360
361 // Apply damping with the precomputed Q15 decay multiplier —
362 // see the 1D update for the rationale.
363 f = static_cast<i32>(
364 (static_cast<i64>(f) * mDampDecayQ15) >> 15);
365
366 // Clamp f into [q15_min, 32767] in a single step — subsumes
367 // both the Q15 saturation clamp and the half-duplex zero pass.
368 row_next[i] = static_cast<i16>(fl::clamp(f, q15_min, static_cast<i32>(32767)));
369 }
370 }
371
372 // Swap the roles of the grids.
373 whichGrid ^= 1;
374}
#define FL_RESTRICT_PARAM
constexpr enable_if< is_fixed_point< T >::value, T >::type clamp(T x, T lo, T hi) FL_NO_EXCEPT
fl::i64 i64
Definition stdint.h:221

References fl::clamp(), FL_RESTRICT_PARAM, grid1, grid2, height, mCourantSq, mDampDecayQ15, mHalfDuplex, mStencil, mXCylindrical, fl::NinePointIsotropic, stride, whichGrid, and width.

+ Here is the call graph for this function: