FastLED 3.10.6
Loading...
Searching...
No Matches
minimp3.h
Go to the documentation of this file.
1#ifndef MINIMP3_H
2#define MINIMP3_H
3// SPDX-License-Identifier: CC0-1.0
4/*
5 https://github.com/lieff/minimp3
6 To the extent possible under law, the author(s) have dedicated all copyright and related and neighboring rights to this software to the public domain worldwide.
7 This software is distributed without any warranty.
8 See <http://creativecommons.org/publicdomain/zero/1.0/>.
9*/
11#include "fl/math/int_asm.h"
12#include "fl/stl/cstring.h"
13#include "fl/stl/noexcept.h"
14#include "fl/stl/stdint.h"
15
16#define MINIMP3_MAX_SAMPLES_PER_FRAME (1152*2)
17
18
19/* FastLED: intrinsics headers are included here, at global scope, rather than
20 next to the code that uses them.
21
22 The implementation section runs inside `namespace fl { namespace
23 MINIMP3_NAMESPACE {`, so an #include there declares int32x4_t and friends
24 *inside* that namespace. With one instantiation that is merely untidy; with
25 several -- which the fixed-vs-float and SIMD-vs-scalar gates require -- the
26 header guard means only the first variant gets the types and every later one
27 fails to compile with "did you mean minimp3_float_probe::int32x4_t". Found by
28 the macOS ARM64 CI job, which is the only place NEON is actually built. */
29#if (defined(_MSC_VER) && (defined(_M_IX86) || defined(_M_X64))) || \
30 ((defined(__i386__) || defined(__x86_64__)) && defined(__SSE2__))
31#include <immintrin.h>
32#if defined(_MSC_VER)
33#include <intrin.h> /* __cpuid, for the SSE4.1 run-time check */
34#elif defined(__GNUC__) || defined(__clang__)
35#include <cpuid.h>
36#endif
37#elif defined(__ARM_NEON) || defined(__aarch64__) || defined(_M_ARM64)
38#include <arm_neon.h>
39#endif
40
41/* FastLED: fixed-point mode. MINIMP3_HAVE_FIXED_POINT reports whether this
42 instantiation actually built the integer DSP path. The macro is recomputed on
43 every (re-)inclusion so that a float variant and a fixed variant can coexist
44 in one translation unit.
45
46 HOW TO TURN IT ON, AND HOW NOT TO. The selection changes MINIMP3_SCRATCH_SIZE
47 and the layout of the scratch arena, so every translation unit that sees this
48 header must agree on it. Build-wide, that means defining
49 FASTLED_MP3_FIXED_POINT as a compiler flag, which is why it is honoured here:
50 a flag reaches every TU, and a #define in one .cpp does not. Since fixed
51 point became the header's default (FastLED#4056) the flag is a no-op for a
52 normal build; it is kept because it is the documented, safe way to say
53 "fixed point everywhere" and because removing it would silently change the
54 meaning of any build that already passes it.
55
56 Defining MINIMP3_FIXED_POINT directly in a single translation unit -- the
57 unity build, say -- while `fl/codec/mp3.cpp.hpp` includes this header without
58 it gives the two a different sizeof(mp3dec_scratch_t). The caller then
59 allocates the smaller arena and the decoder writes the larger one, and the
60 symptom is a glibc "malloc.c: assertion failed" from somewhere unrelated
61 rather than anything pointing at MP3. Confirmed by doing exactly that.
62
63 The test harness in tests/fl/codec/minimp3_variants.hpp is the one safe
64 exception: it sets the macro per-inclusion but also renames MINIMP3_NAMESPACE,
65 so its types are distinct from production's and never meet them. */
66#if defined(MINIMP3_FIXED_POINT) && defined(MINIMP3_FLOAT_POINT)
67#error "MINIMP3_FIXED_POINT and MINIMP3_FLOAT_POINT are mutually exclusive"
68#endif
69
70/* A translation unit that names MINIMP3_FLOAT_POINT means it, so the build-wide
71 flag does not override it. That is what lets the audit and reference TUs pin
72 the float variant while a build-wide -DFASTLED_MP3_FIXED_POINT is in force,
73 and it is why the pin is not silently ignorable. */
74#if defined(FASTLED_MP3_FIXED_POINT) && !defined(MINIMP3_FIXED_POINT) && \
75 !defined(MINIMP3_FLOAT_POINT)
76#define MINIMP3_FIXED_POINT 1
77#endif
78
79/* FastLED: fixed-point is the default DSP for this component.
80
81 Decided here, in the header, rather than by a build flag. The selection
82 changes MINIMP3_SCRATCH_SIZE, so every translation unit that sees this header
83 must agree on it; a flag can be passed to one TU and forgotten for another,
84 and the failure mode is a glibc malloc assertion from somewhere unrelated
85 rather than anything pointing at MP3. Deciding it in the header makes
86 disagreement impossible for a normal build.
87
88 MINIMP3_FLOAT_POINT opts back into upstream's float pipeline. It exists for
89 the golden harness, which builds both and holds the fixed path to a bounded
90 distance from the float one; it is not a supported production configuration
91 and must never be set for only part of a build. */
92#if !defined(MINIMP3_FIXED_POINT) && !defined(MINIMP3_FLOAT_POINT)
93#define MINIMP3_FIXED_POINT 1
94/* Marks the definition above as the header's, not the caller's, so the release
95 block at the end of this file can take it back. Without that, a first,
96 default inclusion would leave MINIMP3_FIXED_POINT defined and a later
97 MINIMP3_FLOAT_POINT-pinned inclusion in the same translation unit would trip
98 the mutual-exclusion #error above -- making the build order-dependent on
99 which header happened to pull minimp3 in first. */
100#define MINIMP3_FIXED_POINT_IS_HEADER_DEFAULT 1
101#endif
102
103#undef MINIMP3_HAVE_FIXED_POINT
104#if defined(MINIMP3_FIXED_POINT)
105#define MINIMP3_HAVE_FIXED_POINT 1
106#else
107#define MINIMP3_HAVE_FIXED_POINT 0
108#endif
109
110/* FastLED: 1 once the integer kernels have actually replaced the float
111 pipeline for this instantiation. Deliberately separate from
112 MINIMP3_HAVE_FIXED_POINT: the golden gate has to be able to tell "this build
113 accepted -DMINIMP3_FIXED_POINT" apart from "this build decodes with integer
114 arithmetic", and during the staged conversion those are not the same thing.
115 The fixed-point kernels flip this to MINIMP3_HAVE_FIXED_POINT. */
116#undef MINIMP3_DSP_INTEGER
117#define MINIMP3_DSP_INTEGER MINIMP3_HAVE_FIXED_POINT
118
119/* Scratch arena size. The fixed-point build needs a little more than the float
120 build because scalefactor gains are carried as mantissa+exponent rather than
121 as a single float, so each variant gets its own figure and the ledger records
122 both rather than charging one build for the other's arena.
123 A static_assert on mp3dec_scratch_internal_t below enforces both. */
124#undef MINIMP3_SCRATCH_SIZE
125#if MINIMP3_HAVE_FIXED_POINT
126#define MINIMP3_SCRATCH_SIZE 7936
127#else
128#define MINIMP3_SCRATCH_SIZE 7808
129#endif
130
131/* FastLED: number of fractional bits in a fixed-point DSP sample.
132
133 Measured over the *full* upstream conformance suite -- all 83 vectors that
134 ship a reference PCM, not the handful vendored under tests/data -- the float
135 pipeline peaks at 3.89, in `l3-si_huff`. Every other vector with a reference
136 peaks at 0.63 or below. minimp3 normalises internally and the polyphase
137 window carries the scaling back up to int16 only at the very end, so one Q
138 format serves the whole pipeline; block floating point is needed only where
139 the dynamic range genuinely is unbounded (the scalefactor gains and
140 x**(4/3), carried as mantissa+exponent instead).
141
142 Q26 in int32 leaves range +/-32 -- about 8x that measured peak -- and a
143 resolution of 2**-26, roughly 1/1000 of an int16 LSB at output scale. Every
144 multiply widens to int64 first, so this is the only headroom that has to be
145 reasoned about.
146
147 That 3.89 is why dequantised samples are clamped to the Q26 range rather
148 than to +/-1. They were clamped to +/-1 until FastLED#4127, on the strength
149 of a "peaks near 0.65" measurement taken over the two vectors that were
150 vendored at the time. `l3-si_huff` reaches 2.44 at the huffman stage and the
151 clamp truncated it, costing 81 dB on that vector -- 26.58 dB against float's
152 107.95, far below the 60 dB ISO floor. The measurement was sound; the sample
153 it was taken over was 2 of 83.
154
155 One vector, `l3-nonstandard-big-iscf`, drives the pipeline to 230767. It is
156 a deliberately malformed stream with no reference PCM, and saturating there
157 is the intended behaviour rather than a range failure. */
158#undef MINIMP3_FRAC_BITS
159#define MINIMP3_FRAC_BITS 26
160
161/* FastLED: per-stage observation hook. Define MINIMP3_STAGE_DUMP to a callable
162 accepting (int stage, int channel, const T *buf, int count) where T is the
163 variant's DSP sample type. The golden harness uses it to compare the fixed
164 and float pipelines one stage at a time, which is what localizes a numeric
165 regression to a single kernel instead of to "the decoder". */
166#define MINIMP3_STAGE_HUFFMAN 0
167#define MINIMP3_STAGE_STEREO 1
168#define MINIMP3_STAGE_ANTIALIAS 2
169#define MINIMP3_STAGE_IMDCT 3
170#define MINIMP3_STAGE_DCT2 4
171#define MINIMP3_STAGE_COUNT 5
172
173/* FastLED: the decoder is instantiated more than once in the same program so
174 that the fixed-point build can be compared against the float build inside a
175 single test binary. Each instantiation lands in its own C++ namespace, so
176 the header carries no C linkage: `extern "C"` ignores namespaces and the
177 variants would collide at link time. Re-include with MINIMP3_H and
178 _MINIMP3_IMPLEMENTATION_GUARD undefined and MINIMP3_NAMESPACE set to a
179 fresh name to build another variant. */
180#ifndef MINIMP3_NAMESPACE
181#define MINIMP3_NAMESPACE third_party
182#endif
183
184#ifdef __cplusplus
185namespace fl {
186namespace MINIMP3_NAMESPACE {
187#endif
188
193
194/* Sample type flowing through the DSP pipeline: Q(MINIMP3_FRAC_BITS) integers
195 in the fixed-point build, plain floats otherwise. Both are 4 bytes wide, so
196 the persistent decoder state below keeps its size either way. */
197#if MINIMP3_HAVE_FIXED_POINT
198typedef int32_t mp3d_dsp_t;
199#else
200typedef float mp3d_dsp_t;
201#endif
202
203typedef struct mp3dec_t
204{
205 mp3d_dsp_t mdct_overlap[2][9*32], qmf_state[(15 + 18)*2*32];
207 unsigned char header[4], reserv_buf[511];
209
215
217/* 1 when this instantiation was asked for the integer DSP path, 0 for float. */
219/* 1 when this instantiation actually decodes with integer arithmetic. */
221/* 1 when this instantiation compiled integer SIMD kernels. Reported
222 separately from the two above for the same reason they are separate from
223 each other: "the build accepted the flag" and "the build actually vectorised"
224 are different claims, and only the second is worth gating on. */
226#ifndef MINIMP3_FLOAT_OUTPUT
228#else /* MINIMP3_FLOAT_OUTPUT */
229typedef float mp3d_sample_t;
230void mp3dec_f32_to_s16(const float *in, int16_t *out, int num_samples) FL_NO_EXCEPT;
231#endif /* MINIMP3_FLOAT_OUTPUT */
233 const uint8_t *mp3, int mp3_bytes,
234 mp3d_sample_t *pcm,
236
237#ifdef __cplusplus
238} /* namespace MINIMP3_NAMESPACE */
239} /* namespace fl */
240#endif /* __cplusplus */
241
242#endif /* MINIMP3_H */
243#if defined(MINIMP3_IMPLEMENTATION) && !defined(_MINIMP3_IMPLEMENTATION_GUARD)
244#define _MINIMP3_IMPLEMENTATION_GUARD
245
246#ifdef __cplusplus
247namespace fl {
248namespace MINIMP3_NAMESPACE {
249#endif
250
251#define MAX_FREE_FORMAT_FRAME_SIZE 2304 /* more than ISO spec's */
252#ifndef MAX_FRAME_SYNC_MATCHES
253#define MAX_FRAME_SYNC_MATCHES 10
254#endif /* MAX_FRAME_SYNC_MATCHES */
255
256/* ---------------------------------------------------------------------------
257 Inline policy.
258
259 Two things are wanted from the same source and they conflict. While hunting
260 hotspots the hot functions must stay separable, or the profiler attributes
261 their cost to whatever inlined them -- that is how `mp3d_synth_granule` came
262 to read as 53% of the decode when a third of it was the DCT-32. While
263 shipping, the opposite: inline everything and let the compiler see through
264 it, because the wins in this decoder have repeatedly come from removing call
265 overhead and spill traffic around small leaves.
266
267 Three modes, selected at build time:
268
269 (default) the shipping policy. Leaves force-inlined; hot
270 blocks left to the compiler except where a
271 measurement said otherwise.
272 MINIMP3_PROFILE_ATTRIBUTION keep the hot blocks out of line so
273 callgrind_annotate can attribute them
274 separately. Slower; for measurement only.
275 MINIMP3_MAX_INLINE force-inline everything and ask for O3 on the
276 kernels. Largest .text; use when flash is free
277 and the last few percent matter.
278
279 MP3D_LEAF marks small helpers that should essentially always be inlined.
280 MP3D_HOT marks the big kernels whose attribution matters when profiling.
281
282 Do not assume MAX_INLINE is fastest without measuring. -O3 on the polyphase
283 emits 395 memory operations against a hand-written 188, and the biggest win
284 in this file came from *out*-of-lining mp3d_synth so the register allocator
285 stopped spilling across it. Measure on the device: the 64-bit host has
286 ranked these changes backwards twice.
287 --------------------------------------------------------------------------- */
288#if defined(MINIMP3_PROFILE_ATTRIBUTION) && defined(MINIMP3_MAX_INLINE)
289#error "MINIMP3_PROFILE_ATTRIBUTION and MINIMP3_MAX_INLINE are mutually exclusive"
290#endif
291
292#if defined(MINIMP3_MAX_INLINE)
293#define MP3D_LEAF FASTLED_FORCE_INLINE
294#define MP3D_HOT FASTLED_FORCE_INLINE
295#define MP3D_KERNEL FL_OPTIMIZE_FUNCTION
296#elif defined(MINIMP3_PROFILE_ATTRIBUTION)
297#define MP3D_LEAF static
298#define MP3D_HOT FL_NO_INLINE static
299#define MP3D_KERNEL FL_NO_INLINE
300#else
301#define MP3D_LEAF FL_ALWAYS_INLINE
302#define MP3D_HOT FL_NO_INLINE static
303#define MP3D_KERNEL FL_OPTIMIZE_FUNCTION
304#endif
305
306
307#define MAX_L3_FRAME_PAYLOAD_BYTES MAX_FREE_FORMAT_FRAME_SIZE /* MUST be >= 320000/8/32000*1152 = 1440 */
308
309#define MAX_BITRESERVOIR_BYTES 511
310#define SHORT_BLOCK_TYPE 2
311#define STOP_BLOCK_TYPE 3
312#define MODE_MONO 3
313#define MODE_JOINT_STEREO 1
314#define HDR_SIZE 4
315#define HDR_IS_MONO(h) (((h[3]) & 0xC0) == 0xC0)
316#define HDR_IS_MS_STEREO(h) (((h[3]) & 0xE0) == 0x60)
317#define HDR_IS_FREE_FORMAT(h) (((h[2]) & 0xF0) == 0)
318#define HDR_IS_CRC(h) (!((h[1]) & 1))
319#define HDR_TEST_PADDING(h) ((h[2]) & 0x2)
320#define HDR_TEST_MPEG1(h) ((h[1]) & 0x8)
321#define HDR_TEST_NOT_MPEG25(h) ((h[1]) & 0x10)
322#define HDR_TEST_I_STEREO(h) ((h[3]) & 0x10)
323#define HDR_TEST_MS_STEREO(h) ((h[3]) & 0x20)
324#define HDR_GET_STEREO_MODE(h) (((h[3]) >> 6) & 3)
325#define HDR_GET_STEREO_MODE_EXT(h) (((h[3]) >> 4) & 3)
326#define HDR_GET_LAYER(h) (((h[1]) >> 1) & 3)
327#define HDR_GET_BITRATE(h) ((h[2]) >> 4)
328#define HDR_GET_SAMPLE_RATE(h) (((h[2]) >> 2) & 3)
329#define HDR_GET_MY_SAMPLE_RATE(h) (HDR_GET_SAMPLE_RATE(h) + (((h[1] >> 3) & 1) + ((h[1] >> 4) & 1))*3)
330#define HDR_IS_FRAME_576(h) ((h[1] & 14) == 2)
331#define HDR_IS_LAYER_1(h) ((h[1] & 6) == 6)
332
333#define BITS_DEQUANTIZER_OUT -1
334#define MAX_SCF (255 + BITS_DEQUANTIZER_OUT*4 - 210)
335#define MAX_SCFI ((MAX_SCF + 3) & ~3)
336
337#define MINIMP3_MIN(a, b) ((a) > (b) ? (b) : (a))
338#define MINIMP3_MAX(a, b) ((a) < (b) ? (b) : (a))
339
340#if MINIMP3_HAVE_FIXED_POINT
341#if defined(MINIMP3_FLOAT_OUTPUT)
342#error "MINIMP3_FIXED_POINT and MINIMP3_FLOAT_OUTPUT are mutually exclusive"
343#endif
344#include "third_party/minimp3/minimp3_fixed_tables.h" // ok cpp include
345
346/* FastLED: the fixed-point arithmetic contract.
347
348 ONE rounding rule for the whole decoder: add half, then arithmetic-shift
349 right. Because the shift floors, that is round-half-toward-+infinity, NOT
350 round-half-away-from-zero -- -1.5 rounds to -1. Stated precisely because
351 Phase 4 (#4055) has to prove SSE2/NEON kernels bit-identical to this scalar
352 path, and it must implement this rule rather than the one the phrase
353 "rounding" usually implies. It is also the cheaper rule to vectorise, which
354 is why it is worth keeping rather than "correcting".
355
356 (Two deliberate exceptions, both documented at their definitions: the Python
357 table generator rounds half away from zero, because it runs once at build
358 time and symmetry matters for a constant; and mp3d_scale_pcm reproduces the
359 float build's own output rounding quirk so the two builds' PCM agrees.)
360
361 Every product widens to int64 before shifting, so a 32x32 multiply can never
362 overflow. Shifts are written to keep UBSan quiet: no left shift of a
363 negative value anywhere (those go through MP3D_SHL, which multiplies), and
364 every variable shift distance is clamped to [0, 63] by its caller. */
365/* Symmetric on purpose: L3_change_sign negates samples in place, and negating
366 INT32_MIN is signed-overflow UB. Giving up one representable value buys a
367 pipeline where every sample can be negated unconditionally. */
368#define MP3D_SAT_MAX ((int32_t)0x7fffffff)
369#define MP3D_SAT_MIN (-MP3D_SAT_MAX)
370
371/* int32 add and subtract with defined wraparound. Signed overflow is undefined
372 in C; unsigned is specified to wrap, and the round trip through uint32_t is
373 the standard way to get the wrapping the hardware does anyway without telling
374 the optimiser it may assume otherwise. Compiles to one instruction. */
375#define MP3D_WRAP_ADD(x, y) \
376 ((int32_t)((uint32_t)(x) + (uint32_t)(y)))
377#define MP3D_WRAP_SUB(x, y) \
378 ((int32_t)((uint32_t)(x) - (uint32_t)(y)))
379
380/* Saturating narrow of a 64-bit accumulator to a Q(MINIMP3_FRAC_BITS) sample. */
381static int32_t mp3d_sat64(int64_t value) FL_NO_EXCEPT
382{
383 if (value > MP3D_SAT_MAX) return MP3D_SAT_MAX;
384 if (value < MP3D_SAT_MIN) return MP3D_SAT_MIN;
385 return (int32_t)value;
386}
387
388/* FastLED: the small arithmetic helpers are force-inlined, and on a 32-bit
389 target that is worth more than any amount of tuning inside them.
390
391 gcc at -Os declines to inline mp3d_mulshift, mp3d_narrow_q30 and
392 mp3d_scale_pcm -- they are called from 36, 12 and 10 sites respectively, and
393 the size heuristic sees three functions worth keeping once. They are also
394 the innermost helpers in the decoder, which is the part the heuristic cannot
395 see: mp3d_scale_pcm alone runs 1.46M times over the audit corpus.
396
397 mp3d_mulshift is the worst of the three, and for an extra reason: `bits` is
398 a parameter, so out of line the compiler must emit a *variable* 64-bit shift
399 and a runtime rounding term -- 42 instructions on riscv32, plus call
400 overhead. Every one of its call sites passes a literal (27, 29, 30 or 31),
401 which folds that down to a handful. Inlining it alone took the C6 from
402 102,461 us to 56,105 us; adding the other two brought it to 50,933 us, a
403 ratio against Helix of 1.46x where it started at 2.84x.
404
405 The cost is 1750 bytes on the codec translation unit's .text, +14% measured
406 against the unmodified header (the mp3d_scale_pcm rewrite below gives 46 of
407 those bytes back, so the inlining alone accounts for 1796). Nothing about
408 the arithmetic changes: the host CPU-audit checksum and the device's FNV-1a
409 over decoded PCM are both unchanged, so this is pure codegen.
410
411 x86-64 sees none of this -- clang at -O2 already inlines all three, which is
412 exactly why the host callgrind number cannot be the authority here. */
413
414/* value * coefficient, where the coefficient is Q`bits`.
415
416 Saturates on the way down to 32 bits. The coefficients are not all <= 1 --
417 the DCT-32 secants reach 10.19 -- so the product can leave int32 range for an
418 operand far below MP3D_SAT_MAX, which means the saturating adds around it
419 never get the chance to fire. Concretely, at Q27 a 10.19 secant wraps once
420 the operand passes about 3.14 in real units, an order of magnitude under the
421 format's +/-32; a mutated stream that drives dequantisation to the +/-1 clamp
422 reaches that comfortably by the time mid/side, antialias and the IMDCT have
423 each added gain. Wrapping there produces a sign-flipped full-scale sample --
424 an audible bang instead of a bounded clip.
425
426 Worth stating why this cannot be left to the fuzz gate: narrowing an int64 to
427 int32 is implementation-defined, not undefined, so UBSan does not see it. */
428/* `(value * coef + 2^(Bits-1)) >> Bits`, via fl::math::mul_shift_round32.
429 Templated on the shift rather than taking it as a parameter: every one of the
430 36 call sites passes a literal (27, 29, 30 or 31), and a runtime `bits` forces
431 a variable 64-bit shift plus a runtime rounding term -- 42 riscv32
432 instructions against 10.
433
434 The result is no longer saturated. On riscv32 -Os the saturating form was 25
435 instructions, 15 of them the 64-bit clamp, and that clamp is very nearly
436 never taken: across the 83 ISO conformance vectors it fired 268 times in
437 40,613,225 calls, all 268 from one malformed stream
438 (`l3-nonstandard-big-iscf`), and zero times in 21,067,559 calls on real
439 music. Same reasoning, same evidence and same regression vector as the
440 butterfly clamp removed earlier.
441
442 Not to be confused with FastLED#4127's dequant clamp, which is a different
443 clamp on a different quantity and is untouched. */
444template <int Bits>
445MP3D_LEAF int32_t mp3d_mulshift(int32_t value, int32_t coef) FL_NO_EXCEPT
446{
447 return fl::math::mul_shift_round32<Bits>(value, coef);
448}
449
450/* value * Coef, where Coef is a compile-time Q`Bits` coefficient -- the same
451 arithmetic as mp3d_mulshift<Bits>(value, Coef), computed from the high half
452 of the product instead of the low one.
453
454 This is the one thing that made the DCT-32 the only stage where this decoder
455 still lost to Helix on riscv32. Both do the same 80 multiplies per 32-point
456 DCT; Helix spends two instructions on each (`mulh`, then a left shift to
457 undo the coefficient's Q format) and this decoder spent nine:
458
459 mul mulh lui add sltu add srli slli or
460
461 The seven after `mul` are all there to round at bit `Bits` rather than at
462 bit 32, which needs the low half of the product, the carry out of adding
463 2^(Bits-1) to it, and a 64-bit funnel shift. Helix does not round at all --
464 it truncates, which is where its 20 dB of accuracy went.
465
466 The identity below rounds at bit 32 instead, which `mulh` does for free, and
467 still lands on exactly the same bits. Write Coef = W*2^Bits + F with
468 0 <= F < 2^Bits; then
469
470 floor((v*Coef + 2^(Bits-1)) / 2^Bits)
471 = v*W + floor((v*F + 2^(Bits-1)) / 2^Bits)
472 = v*W + floor((v*(F << (32-Bits)) + 2^31) / 2^32)
473
474 and the second term is `mulh(v, F32) + (lo(v*F32) >> 31)`, because adding
475 2^31 to an unsigned low half carries exactly when its top bit is set. Both
476 W and F32 are compile-time constants.
477
478 F32 is kept *signed*. F << (32-Bits) is a Q32 fraction and exceeds int32
479 whenever the coefficient's fractional part is at least a half, which is most
480 of them; taking that bit pattern as a negative int32 subtracts 2^32 from it,
481 so the missing 2^32/2^32 = 1 goes back on the integer side as W+1. That
482 keeps the multiply a plain `mulh` -- gcc lowers a genuinely unsigned
483 constant operand into srai/mul/mul/mulhu/add, five instructions, which
484 throws the win away.
485
486 Cost on riscv32 -Os, excluding the lui/addi that materialises the
487 coefficient (both forms pay that, and both hoist it out of the k loop):
488
489 | coefficient | before | after |
490 |------------------------------|--------|-------|
491 | tan(pi/16) Q31, W = 0 | 8 | 4 |
492 | cos(pi/4) Q31, W = 1 | 8 | 5 |
493 | 1/(2cos(7pi/16)) Q29, W = 3 | 8 | 7 |
494 | secants Q27, W = 0 or 1 | 8 | 4..5 |
495
496 Bit-exactness is not an argument here, it is a proof: the identity was
497 checked against mp3d_mulshift for all 2^32 int32 inputs against each of the
498 33 coefficients the DCT-32 instantiates, and again for the seven Q31
499 cosines of the 9-point DCT-III / 3-point IDCT and the Q30 pow43 term,
500 with zero mismatches. The device checksum and the PSNR are unchanged, and
501 they are the tripwire if a transcribed constant is ever wrong.
502
503 Defined here rather than in minimp3_synth_fixed.h because the IMDCT's
504 L3_dct3_9, L3_idct3 and mp3d_pow43 above that include point use it too.
505 On riscv32 -Os those ten call sites were 7 instructions each through
506 mp3d_mulshift and are 4 here -- W is 0 for all of them, the nine Q31
507 cosine sites and the Q30 pow43 term alike. An ESP32-C6 decodes the
508 fixture in 44,993 us against 45,186 us before (-0.43%, spread 0.00%
509 either side), for -46 bytes of .text.
510
511 Only where the product does *not* already fit in one register. The whole
512 trick is to avoid materialising the low half, so it is a pessimisation
513 wherever a single instruction hands you all 64 bits: on x86-64 the portable
514 form is an `imul` and a `shrd`, and routing the DCT-32 and the IMDCT through
515 this identity instead cost +7.5% in L3_dct3_9 and +0.95% of the whole
516 fixed-point decode, measured under Callgrind. `fl::math::mul_shift_round32`
517 guards the same identity the same way and for the same reason -- see
518 src/platforms/int_asm.h, which also records why a 64-bit host cannot be the
519 authority on this file.
520
521 The test is register width, not pointer width in principle; pointer width is
522 what the preprocessor can actually see, and it separates the two cases for
523 every target this decoder is built for. wasm32 lands on the narrow path
524 despite having a native i64 -- which is what it already did before this
525 guard existed, so it is status quo rather than a decision; there is no
526 wasm codegen bound recorded to check it against.
527
528 Not applied to the SIMD kernels: MP3D_V_MULSHIFT saturates on the narrow,
529 which this form cannot do without giving back what it saved, and the vector
530 targets have no scarcity of multiply throughput to fix. */
531#if (defined(__SIZEOF_POINTER__) && __SIZEOF_POINTER__ >= 8) || \
532 defined(_WIN64) || defined(__LP64__) || defined(_LP64)
533#define MP3D_MULSHIFT_K_WIDE_PRODUCT 1
534#else
535#define MP3D_MULSHIFT_K_WIDE_PRODUCT 0
536#endif
537
538template <int Bits, int32_t Coef>
539MP3D_LEAF int32_t mp3d_mulshift_k(int32_t value) FL_NO_EXCEPT
540{
541 /* Every coefficient this is used with is a positive secant or cosine.
542 The split relies on that: `Coef >> Bits` is implementation-defined for a negative
543 value, and W would have to round the other way. Outside the #if because it
544 is a property of the coefficient, not of one lowering. */
545 static_assert(Coef > 0, "mp3d_mulshift_k: coefficient must be positive");
546#if MP3D_MULSHIFT_K_WIDE_PRODUCT
547 return mp3d_mulshift<Bits>(value, Coef);
548#else
549 const int32_t frac = (int32_t)((uint32_t)Coef << (32 - Bits));
550 const int32_t whole = (int32_t)(Coef >> Bits) + (frac < 0 ? 1 : 0);
551 const int64_t product = (int64_t)value * (int64_t)frac;
552 const uint32_t rounded = (uint32_t)(uint64_t)(product >> 32) +
553 ((uint32_t)(uint64_t)product >> 31);
554 return (int32_t)((uint32_t)value * (uint32_t)whole + rounded);
555#endif
556}
557
558/* Left shift that is defined for negative inputs, with saturation. `shift` is
559 assumed to be in [0, 62]; callers clamp before calling. */
560static int32_t mp3d_shl_sat(int32_t value, int shift) FL_NO_EXCEPT
561{
562 int64_t wide;
563 if (shift >= 32)
564 {
565 return value == 0 ? 0 : (value > 0 ? MP3D_SAT_MAX : MP3D_SAT_MIN);
566 }
567 wide = (int64_t)value * ((int64_t)1 << shift);
568 if (wide > MP3D_SAT_MAX) return MP3D_SAT_MAX;
569 if (wide < MP3D_SAT_MIN) return MP3D_SAT_MIN;
570 return (int32_t)wide;
571}
572
573/* Arithmetic right shift with the same round-half-toward-+infinity rule,
574 tolerating shifts that discard the value entirely. */
575static int32_t mp3d_shr_round(int32_t value, int shift) FL_NO_EXCEPT
576{
577 if (shift <= 0)
578 {
579 return value;
580 }
581 if (shift > 31)
582 {
583 return 0;
584 }
585 return (int32_t)(((int64_t)value + ((int64_t)1 << (shift - 1))) >> shift);
586}
587
588/* Narrow a Q30 accumulator to a sample, rounding once (half toward +infinity,
589 as above) for the whole accumulation rather than once per product.
590
591 The narrow wraps rather than saturating (FastLED#4139, third instance).
592
593 This is the IMDCT's output stage: L3_imdct36's twiddle-and-window loop and
594 L3_imdct12 between them call it four times per output, and it is the only
595 caller of mp3d_sat64 left on the IMDCT path. Out of line the saturating form
596 is two 64-bit comparisons against 64-bit constants, which a 32-bit target
597 pays for as a compare/branch pair per half.
598
599 Instrumented the same way the butterfly and mulshift clamps were before they
600 went, counting actual clamps rather than calls:
601
602 83 ISO conformance vectors 14,685,696 calls 273 clamps 0.0019%
603 repo corpus (real music) 2,635,776 calls 0 clamps 0%
604
605 and all 273 come from l3-nonstandard-big-iscf.bit, the same malformed
606 intensity-stereo vector that accounts for every clamp the other two ever
607 took. The other 82 vectors clamp zero times. So on conformant input the two
608 forms produce identical bits, because the branch is never taken.
609
610 The narrowing is written through uint32_t rather than as a plain int64 ->
611 int32 conversion. That conversion is implementation-defined, not undefined,
612 so it would not be a correctness bug -- but the wrapping form is specified,
613 compiles to the same instruction, and matches MP3D_WRAP_ADD's idiom.
614
615 The vector path's mp3d_v_narrow still saturates, because there vqmovn_s64
616 and its SSE equivalent do it for free. That divergence is pre-existing --
617 MP3D_V_MULSHIFT already saturates where the scalar mp3d_mulshift no longer
618 does -- and it is unobservable on everything except the one malformed
619 vector; the SIMD-equals-scalar gate runs the real-music corpus, where
620 neither form ever clamps. */
621MP3D_LEAF int32_t mp3d_narrow_q30(int64_t acc) FL_NO_EXCEPT
622{
623 return (int32_t)(uint32_t)((acc + ((int64_t)1 << 29)) >> 30);
624}
625
626/* Butterfly add/subtract with defined wraparound (FastLED#4139).
627
628 These were saturating, via a 64-bit intermediate, and that was 24.5% of the
629 whole decode on x86-64 and 32.4% of mp3d_DCT_II's instructions on riscv32.
630 Removing it is worth 28% of Layer III decode time on an ESP32-C6: the gap to
631 the retired Helix backend goes from 3.99x to 2.84x, and real-time margin from
632 1.51x to 2.09x.
633
634 It costs nothing, on four separate measurements:
635
636 - Output is bit-identical on conformant content. Instrumenting the clamp
637 showed it firing 4,609 times in 154,115,265 calls across the 83 ISO
638 vectors -- and every one of those from a single stream,
639 l3-nonstandard-big-iscf. On the repo's real-music corpus: 82,537,927
640 calls, zero clamps.
641 - All 83 ISO vectors still pass the 60 dB floor, fixed and float.
642 - ASan/UBSan clean across all 83 bitstreams. Unsigned arithmetic wraps by
643 definition, so this is specified behaviour, not the undefined overflow
644 FastLED#4133 removed.
645 - On the one vector that did clamp, the audible result is unchanged:
646 peak 32768 either way, mean |x| 16649 -> 17011, near-full-scale samples
647 46.4% -> 47.3%. The old comment argued the clamp bought "a bounded clip
648 instead of an audible bang"; measured, that stream is a wall of
649 near-full-scale noise with or without it.
650
651 Note this is NOT FastLED#4127's clamp. That one bounds *dequantised samples*
652 and was worth 81 dB; it is untouched, as is mp3d_sat64 itself, which is still
653 used by mp3d_mulshift and mp3d_narrow_q30 where the 64-bit intermediate is
654 real and the narrowing has to be bounded.
655
656 A branchless 32-bit saturating form was tried earlier and reverted at +24%
657 text for ~3% of decode time. This is a different change: not a cheaper clamp,
658 but no clamp where none was doing anything. */
659static int32_t mp3d_add_sat(int32_t a, int32_t b) FL_NO_EXCEPT
660{
661 return MP3D_WRAP_ADD(a, b);
662}
663
664static int32_t mp3d_sub_sat(int32_t a, int32_t b) FL_NO_EXCEPT
665{
666 return MP3D_WRAP_SUB(a, b);
667}
668
669/* One Q26 unit of 1.0, and the clamp applied to dequantised samples.
670
671 Clamping to +/-1.0 rather than to the Q format's +/-32 is deliberate. The
672 measured legitimate peak is 0.65, so this costs nothing on real streams, and
673 it is what bounds every downstream butterfly: a fuzzed bitstream can ask for
674 an astronomically large dequantised value, and without this the DCT-32
675 secants (up to 10.19) stacked on three levels of adds would overflow int32
676 and hand UBSan a signed-overflow report. */
677
678static int32_t mp3d_clamp_sample(int64_t value) FL_NO_EXCEPT
679{
680 if (value > MP3D_SAT_MAX) return MP3D_SAT_MAX;
681 if (value < MP3D_SAT_MIN) return MP3D_SAT_MIN;
682 return (int32_t)value;
683}
684
685/* Combine a mantissa/exponent pair (value = mant * 2**(exp - 30)) into a
686 Q(MINIMP3_FRAC_BITS) sample, clamped. */
687static int32_t mp3d_scale_to_q(int32_t mant, int exp) FL_NO_EXCEPT
688{
689 const int shift = 30 - MINIMP3_FRAC_BITS - exp;
690 if (mant == 0)
691 {
692 return 0;
693 }
694 if (shift > 0)
695 {
696 return mp3d_clamp_sample(mp3d_shr_round(mant, shift));
697 }
698 return mp3d_clamp_sample((int64_t)mp3d_shl_sat(mant, -shift));
699}
700#endif /* MINIMP3_HAVE_FIXED_POINT */
701
702/* FastLED: see MINIMP3_STAGE_* in the public section. Compiles away entirely
703 unless the including translation unit asked for stage observation. */
704#ifdef MINIMP3_STAGE_DUMP
705#define MP3D_STAGE(stage, ch, buf, n) MINIMP3_STAGE_DUMP((stage), (ch), (buf), (n))
706#else
707#define MP3D_STAGE(stage, ch, buf, n) ((void)0)
708#endif
709
710/* FastLED: the vector kernels come in two families that must not be confused.
711 The `HAVE_SIMD` block below is the float one, and it stays exactly as
712 upstream wrote it -- the fixed-point build must never compile float vector
713 code. The integer kernels live in their own MP3D_HAVE_INT_SIMD block and are
714 selected only for the fixed build. */
715#if MINIMP3_HAVE_FIXED_POINT
716/* Keep upstream's float SIMD out of the fixed build without disturbing the
717 caller's own MINIMP3_NO_SIMD, which still has to mean "scalar everywhere"
718 and is what the Phase 4 opt-out proof relies on. */
719#define MP3D_FLOAT_SIMD_OFF
720#endif
721
722#if !defined(MINIMP3_NO_SIMD) && !defined(MP3D_FLOAT_SIMD_OFF)
723
724#if !defined(MINIMP3_ONLY_SIMD) && (defined(_M_X64) || defined(__x86_64__) || defined(__aarch64__) || defined(_M_ARM64))
725/* x64 always have SSE2, arm64 always have neon, no need for generic code */
726#define MINIMP3_ONLY_SIMD
727#endif /* SIMD checks... */
728
729#if (defined(_MSC_VER) && (defined(_M_IX86) || defined(_M_X64))) || ((defined(__i386__) || defined(__x86_64__)) && defined(__SSE2__))
730#if defined(_MSC_VER)
731#include <intrin.h>
732#endif /* defined(_MSC_VER) */
733#include <immintrin.h>
734#define HAVE_SSE 1
735#define HAVE_SIMD 1
736#define VSTORE _mm_storeu_ps
737#define VLD _mm_loadu_ps
738#define VSET _mm_set1_ps
739#define VADD _mm_add_ps
740#define VSUB _mm_sub_ps
741#define VMUL _mm_mul_ps
742#define VMAC(a, x, y) _mm_add_ps(a, _mm_mul_ps(x, y))
743#define VMSB(a, x, y) _mm_sub_ps(a, _mm_mul_ps(x, y))
744#define VMUL_S(x, s) _mm_mul_ps(x, _mm_set1_ps(s))
745#define VREV(x) _mm_shuffle_ps(x, x, _MM_SHUFFLE(0, 1, 2, 3))
746typedef __m128 f4;
747#if defined(_MSC_VER) || defined(MINIMP3_ONLY_SIMD)
748#define minimp3_cpuid __cpuid
749#else /* defined(_MSC_VER) || defined(MINIMP3_ONLY_SIMD) */
750static __inline__ __attribute__((always_inline)) void minimp3_cpuid(int CPUInfo[], const int InfoType)
751{
752#if defined(__PIC__)
753 __asm__ __volatile__(
754#if defined(__x86_64__)
755 "push %%rbx\n"
756 "cpuid\n"
757 "xchgl %%ebx, %1\n"
758 "pop %%rbx\n"
759#else /* defined(__x86_64__) */
760 "xchgl %%ebx, %1\n"
761 "cpuid\n"
762 "xchgl %%ebx, %1\n"
763#endif /* defined(__x86_64__) */
764 : "=a" (CPUInfo[0]), "=r" (CPUInfo[1]), "=c" (CPUInfo[2]), "=d" (CPUInfo[3])
765 : "a" (InfoType));
766#else /* defined(__PIC__) */
767 __asm__ __volatile__(
768 "cpuid"
769 : "=a" (CPUInfo[0]), "=b" (CPUInfo[1]), "=c" (CPUInfo[2]), "=d" (CPUInfo[3])
770 : "a" (InfoType));
771#endif /* defined(__PIC__)*/
772}
773#endif /* defined(_MSC_VER) || defined(MINIMP3_ONLY_SIMD) */
774static int have_simd(void) FL_NO_EXCEPT
775{
776#ifdef MINIMP3_ONLY_SIMD
777 return 1;
778#else /* MINIMP3_ONLY_SIMD */
779 static int g_have_simd;
780 int CPUInfo[4];
781#ifdef MINIMP3_TEST
782 static int g_counter;
783 if (g_counter++ > 100)
784 return 0;
785#endif /* MINIMP3_TEST */
786 if (g_have_simd)
787 goto end;
788 minimp3_cpuid(CPUInfo, 0);
789 g_have_simd = 1;
790 if (CPUInfo[0] > 0)
791 {
792 minimp3_cpuid(CPUInfo, 1);
793 g_have_simd = (CPUInfo[3] & (1 << 26)) + 1; /* SSE2 */
794 }
795end:
796 return g_have_simd - 1;
797#endif /* MINIMP3_ONLY_SIMD */
798}
799#elif defined(__ARM_NEON) || defined(__aarch64__) || defined(_M_ARM64)
800#include <arm_neon.h>
801#define HAVE_SSE 0
802#define HAVE_SIMD 1
803#define VSTORE vst1q_f32
804#define VLD vld1q_f32
805#define VSET vmovq_n_f32
806#define VADD vaddq_f32
807#define VSUB vsubq_f32
808#define VMUL vmulq_f32
809#define VMAC(a, x, y) vmlaq_f32(a, x, y)
810#define VMSB(a, x, y) vmlsq_f32(a, x, y)
811#define VMUL_S(x, s) vmulq_f32(x, vmovq_n_f32(s))
812#define VREV(x) vcombine_f32(vget_high_f32(vrev64q_f32(x)), vget_low_f32(vrev64q_f32(x)))
813typedef float32x4_t f4;
814static int have_simd()
815{ /* TODO: detect neon for !MINIMP3_ONLY_SIMD */
816 return 1;
817}
818#else /* SIMD checks... */
819#define HAVE_SSE 0
820#define HAVE_SIMD 0
821#ifdef MINIMP3_ONLY_SIMD
822#error MINIMP3_ONLY_SIMD used, but SSE/NEON not enabled
823#endif /* MINIMP3_ONLY_SIMD */
824#endif /* SIMD checks... */
825#else /* !defined(MINIMP3_NO_SIMD) && !defined(MP3D_FLOAT_SIMD_OFF) */
826#define HAVE_SIMD 0
827#endif /* !defined(MINIMP3_NO_SIMD) && !defined(MP3D_FLOAT_SIMD_OFF) */
828
829/* FastLED: integer SIMD for the fixed-point path (FastLED/FastLED#4055).
830 Deliberately a separate detection block from upstream's float one above --
831 the two select different instruction sets and must never both be live.
832 MINIMP3_NO_SIMD suppresses this as well, which is what makes the scalar
833 opt-out proof meaningful. */
834#if MINIMP3_HAVE_FIXED_POINT && !defined(MINIMP3_NO_SIMD)
835#if (defined(_MSC_VER) && (defined(_M_IX86) || defined(_M_X64))) || \
836 ((defined(__i386__) || defined(__x86_64__)) && defined(__SSE2__))
837#define MP3D_HAVE_INT_SIMD 1
838#define MP3D_INT_SIMD_SSE 1
839typedef __m128i mp3d_i32x4;
840#elif defined(__ARM_NEON) || defined(__aarch64__) || defined(_M_ARM64)
841#define MP3D_HAVE_INT_SIMD 1
842#define MP3D_INT_SIMD_NEON 1
843typedef int32x4_t mp3d_i32x4;
844#endif
845#endif /* MINIMP3_HAVE_FIXED_POINT && !MINIMP3_NO_SIMD */
846
847#ifndef MP3D_HAVE_INT_SIMD
848#define MP3D_HAVE_INT_SIMD 0
849#endif
850
851/* 1 once integer kernels actually replace scalar ones. Separate from
852 MP3D_HAVE_INT_SIMD -- having the intrinsics available is not the same as
853 using them, and the gate is on the second. */
854#undef MP3D_SIMD_KERNELS_LIVE
855#define MP3D_SIMD_KERNELS_LIVE MP3D_HAVE_INT_SIMD
856
857#if defined(__ARM_ARCH) && (__ARM_ARCH >= 6) && !defined(__aarch64__) && !defined(_M_ARM64) && !defined(__ARM_ARCH_6M__)
858#define HAVE_ARMV6 1
859static __inline__ __attribute__((always_inline)) int32_t minimp3_clip_int16_arm(int32_t a)
860{
861 int32_t x = 0;
862 __asm__ ("ssat %0, #16, %1" : "=r"(x) : "r"(a));
863 return x;
864}
865#else
866#define HAVE_ARMV6 0
867#endif
868
869typedef struct
870{
871 const uint8_t *buf;
872 int pos, limit;
873} bs_t;
874
875typedef struct
876{
877#if MINIMP3_HAVE_FIXED_POINT
878 /* Same mantissa/exponent split the Layer III gains use: the Layer I/II
879 dequantiser steps are around 1e-7 before a runtime scale that can move
880 them 21 binary places either way, so they do not fit one Q format.
881
882 int8_t, unlike the Layer III gains' int16_t, because this array lives on
883 the stack inside mp3dec_decode_frame_r rather than in the heap scratch
884 arena, and 192 entries of it is the difference between meeting the 2 KiB
885 decode-stack budget and missing it. The exponents here span [-37, -1]:
886 the table's own range plus the runtime 2**(21 - b/3) scale. */
887 int32_t scf_mant[3*64];
888 int8_t scf_exp[3*64];
889#else
890 float scf[3*64];
891#endif
892 uint8_t total_bands, stereo_bands, bitalloc[64], scfcod[64];
893} L12_scale_info;
894
895typedef struct
896{
897 uint8_t tab_offset, code_tab_width, band_count;
898} L12_subband_alloc_t;
899
900typedef struct
901{
902 const uint8_t *sfbtab;
903 uint16_t part_23_length, big_values, scalefac_compress;
904 uint8_t global_gain, block_type, mixed_block_flag, n_long_sfb, n_short_sfb;
905 uint8_t table_select[3], region_count[3], subblock_gain[3];
906 uint8_t preflag, scalefac_scale, count1_table, scfsi;
907} L3_gr_info_t;
908
909typedef struct
910{
911 bs_t bs;
912 /* Layer III's main-data window and Layer I/II's scale info share storage:
913 a frame is one or the other, never both (FastLED#4116).
914
915 L12_scale_info is ~1090 bytes in the fixed-point build -- mantissa plus
916 exponent for 192 scalefactors -- and it used to sit on the stack inside
917 mp3dec_decode_frame_r, where it was almost the entire 1480-byte frame and
918 the difference between meeting the 2 KiB decode-stack budget and missing
919 it. Moving it to the heap arena would normally cost working RAM, of which
920 there were only 692 bytes spare against the 24 KiB budget. Overlapping it
921 with maindata costs nothing instead: maindata is at least 1951 bytes, so
922 the union sizes to maindata and MINIMP3_SCRATCH_SIZE does not move.
923
924 The exclusivity is structural, not incidental. maindata is written only
925 by L3_restore_reservoir and read only by the Layer III bitstream reader;
926 the Layer I/II path never touches it, and never runs in the same call. */
927 union
928 {
929 uint8_t maindata[MAX_BITRESERVOIR_BYTES + MAX_L3_FRAME_PAYLOAD_BYTES];
930 L12_scale_info l12;
931 } u;
932 L3_gr_info_t gr_info[4];
933 mp3d_dsp_t grbuf[2][576];
934#if MINIMP3_HAVE_FIXED_POINT
935 /* Scalefactor gains span roughly 2**-181 to 2**10, so they are the one
936 quantity in the pipeline that genuinely needs an exponent of its own.
937 int16 rather than int8 because that range does not fit a signed byte. */
938 int32_t scf_mant[40];
939 int16_t scf_exp[40];
940#else
941 float scf[40];
942#endif
943 uint8_t ist_pos[2][39];
944} mp3dec_scratch_internal_t;
945
946#ifdef __cplusplus
947static_assert(sizeof(mp3dec_scratch_internal_t) <= MINIMP3_SCRATCH_SIZE,
948 "MINIMP3_SCRATCH_SIZE is too small");
949/* The whole point of overlapping L12_scale_info with maindata is that it is
950 free. If L12_scale_info ever outgrows maindata the union starts costing
951 working RAM silently, which is exactly what this change exists to avoid. */
952static_assert(sizeof(L12_scale_info) <=
953 MAX_BITRESERVOIR_BYTES + MAX_L3_FRAME_PAYLOAD_BYTES,
954 "L12_scale_info no longer fits inside maindata");
955static_assert(alignof(mp3dec_scratch_internal_t) <= alignof(mp3dec_scratch_t),
956 "mp3dec_scratch_t alignment is too small");
957#endif
958
959static void bs_init(bs_t *bs, const uint8_t *data, int bytes) FL_NO_EXCEPT
960{
961 bs->buf = data;
962 bs->pos = 0;
963 bs->limit = bytes*8;
964}
965
966/* Force-inlined for riscv32 -Os, which leaves all four of these out of line.
967
968 FastLED#4117 force-inlined the DSP leaves and left the bitstream ones
969 alone, on the reading that minimp3 was already ahead of the retired Helix
970 backend on Huffman and dequantisation. It was not: that comparison counted
971 only the rows Callgrind attributed to this header and dropped the ones it
972 attributed to fl/math's int_asm.h for the same functions. Summed properly,
973 mp3dec_decode_frame_r is 43.6M instructions against Helix's 37.7M for
974 DecodeHuffman plus DequantBlock -- 5.9M behind, not 1.6M ahead.
975
976 Inlining the four leaves gcc leaves out of line at -Os is worth
977 41,634 -> 40,868 us on an ESP32-C6, 1.8% of the whole Layer III decode
978 (0.97% after normalising against the Helix reference measured in the same
979 two flashes, which moved 0.9%). Output is bit-identical: combined FNV-1a
980 0xc6b632ab, unchanged. It costs 1,716 bytes of .text.
981
982 mp3d_l12_scale is in the set on the same reasoning but honestly cannot be
983 credited with any of that number: it is Layer I/II code and the benchmark
984 above is Layer III only. It is kept because it is the same situation --
985 216 bytes out of line at -Os on a per-sample leaf -- and because removing
986 it would invalidate the measurement the other three come from. */
987MP3D_LEAF uint32_t get_bits(bs_t *bs, int n) FL_NO_EXCEPT
988{
989 uint32_t next, cache = 0, s = bs->pos & 7;
990 int shl = n + s;
991 const uint8_t *p = bs->buf + (bs->pos >> 3);
992 if ((bs->pos += n) > bs->limit)
993 return 0;
994 next = *p++ & (255 >> s);
995 while ((shl -= 8) > 0)
996 {
997 cache |= next << shl;
998 next = *p++;
999 }
1000 return cache | (next >> -shl);
1001}
1002
1003static int hdr_valid(const uint8_t *h) FL_NO_EXCEPT
1004{
1005 return h[0] == 0xff &&
1006 ((h[1] & 0xF0) == 0xf0 || (h[1] & 0xFE) == 0xe2) &&
1007 (HDR_GET_LAYER(h) != 0) &&
1008 (HDR_GET_BITRATE(h) != 15) &&
1009 (HDR_GET_SAMPLE_RATE(h) != 3);
1010}
1011
1012static int hdr_compare(const uint8_t *h1, const uint8_t *h2) FL_NO_EXCEPT
1013{
1014 return hdr_valid(h2) &&
1015 ((h1[1] ^ h2[1]) & 0xFE) == 0 &&
1016 ((h1[2] ^ h2[2]) & 0x0C) == 0 &&
1017 !(HDR_IS_FREE_FORMAT(h1) ^ HDR_IS_FREE_FORMAT(h2));
1018}
1019
1020static unsigned hdr_bitrate_kbps(const uint8_t *h) FL_NO_EXCEPT
1021{
1022 static const uint8_t halfrate[2][3][15] = {
1023 { { 0,4,8,12,16,20,24,28,32,40,48,56,64,72,80 }, { 0,4,8,12,16,20,24,28,32,40,48,56,64,72,80 }, { 0,16,24,28,32,40,48,56,64,72,80,88,96,112,128 } },
1024 { { 0,16,20,24,28,32,40,48,56,64,80,96,112,128,160 }, { 0,16,24,28,32,40,48,56,64,80,96,112,128,160,192 }, { 0,16,32,48,64,80,96,112,128,144,160,176,192,208,224 } },
1025 };
1026 return 2*halfrate[!!HDR_TEST_MPEG1(h)][HDR_GET_LAYER(h) - 1][HDR_GET_BITRATE(h)];
1027}
1028
1029static unsigned hdr_sample_rate_hz(const uint8_t *h) FL_NO_EXCEPT
1030{
1031 static const unsigned g_hz[3] = { 44100, 48000, 32000 };
1032 return g_hz[HDR_GET_SAMPLE_RATE(h)] >> (int)!HDR_TEST_MPEG1(h) >> (int)!HDR_TEST_NOT_MPEG25(h);
1033}
1034
1035static unsigned hdr_frame_samples(const uint8_t *h) FL_NO_EXCEPT
1036{
1037 return HDR_IS_LAYER_1(h) ? 384 : (1152 >> (int)HDR_IS_FRAME_576(h));
1038}
1039
1040static int hdr_frame_bytes(const uint8_t *h, int free_format_size) FL_NO_EXCEPT
1041{
1042 int frame_bytes = hdr_frame_samples(h)*hdr_bitrate_kbps(h)*125/hdr_sample_rate_hz(h);
1043 if (HDR_IS_LAYER_1(h))
1044 {
1045 frame_bytes &= ~3; /* slot align */
1046 }
1047 return frame_bytes ? frame_bytes : free_format_size;
1048}
1049
1050static int hdr_padding(const uint8_t *h) FL_NO_EXCEPT
1051{
1052 return HDR_TEST_PADDING(h) ? (HDR_IS_LAYER_1(h) ? 4 : 1) : 0;
1053}
1054
1055#ifndef MINIMP3_ONLY_MP3
1056static const L12_subband_alloc_t *L12_subband_alloc_table(const uint8_t *hdr, L12_scale_info *sci) FL_NO_EXCEPT
1057{
1058 const L12_subband_alloc_t *alloc;
1059 int mode = HDR_GET_STEREO_MODE(hdr);
1060 int nbands, stereo_bands = (mode == MODE_MONO) ? 0 : (mode == MODE_JOINT_STEREO) ? (HDR_GET_STEREO_MODE_EXT(hdr) << 2) + 4 : 32;
1061
1062 if (HDR_IS_LAYER_1(hdr))
1063 {
1064 static const L12_subband_alloc_t g_alloc_L1[] = { { 76, 4, 32 } };
1065 alloc = g_alloc_L1;
1066 nbands = 32;
1067 } else if (!HDR_TEST_MPEG1(hdr))
1068 {
1069 static const L12_subband_alloc_t g_alloc_L2M2[] = { { 60, 4, 4 }, { 44, 3, 7 }, { 44, 2, 19 } };
1070 alloc = g_alloc_L2M2;
1071 nbands = 30;
1072 } else
1073 {
1074 static const L12_subband_alloc_t g_alloc_L2M1[] = { { 0, 4, 3 }, { 16, 4, 8 }, { 32, 3, 12 }, { 40, 2, 7 } };
1075 int sample_rate_idx = HDR_GET_SAMPLE_RATE(hdr);
1076 unsigned kbps = hdr_bitrate_kbps(hdr) >> (int)(mode != MODE_MONO);
1077 if (!kbps) /* free-format */
1078 {
1079 kbps = 192;
1080 }
1081
1082 alloc = g_alloc_L2M1;
1083 nbands = 27;
1084 if (kbps < 56)
1085 {
1086 static const L12_subband_alloc_t g_alloc_L2M1_lowrate[] = { { 44, 4, 2 }, { 44, 3, 10 } };
1087 alloc = g_alloc_L2M1_lowrate;
1088 nbands = sample_rate_idx == 2 ? 12 : 8;
1089 } else if (kbps >= 96 && sample_rate_idx != 1)
1090 {
1091 nbands = 30;
1092 }
1093 }
1094
1095 sci->total_bands = (uint8_t)nbands;
1096 sci->stereo_bands = (uint8_t)MINIMP3_MIN(stereo_bands, nbands);
1097
1098 return alloc;
1099}
1100
1101#if MINIMP3_HAVE_FIXED_POINT
1102/* raw quantised value * Layer I/II step, landed in Q(MINIMP3_FRAC_BITS). */
1103/* Force-inlined with get_bits above; see that comment. */
1104MP3D_LEAF int32_t mp3d_l12_scale(int32_t raw, int32_t mant, int exp) FL_NO_EXCEPT
1105{
1106 int64_t product = (int64_t)raw * (int64_t)mant;
1107 int shift;
1108 if (product == 0)
1109 {
1110 return 0;
1111 }
1112 shift = 30 - MINIMP3_FRAC_BITS - exp;
1113 if (shift >= 63)
1114 {
1115 return 0;
1116 }
1117 if (shift <= 0)
1118 {
1119 return mp3d_clamp_sample(
1120 (int64_t)mp3d_shl_sat(mp3d_sat64(product), -shift));
1121 }
1122 product = (product + ((int64_t)1 << (shift - 1))) >> shift;
1123 return mp3d_clamp_sample(product);
1124}
1125
1126static void L12_read_scalefactors(bs_t *bs, uint8_t *pba, uint8_t *scfcod, int bands, int32_t *scf_mant, int8_t *scf_exp) FL_NO_EXCEPT
1127{
1128 int i, m;
1129 for (i = 0; i < bands; i++)
1130 {
1131 int32_t mant = 0;
1132 int exp = 0;
1133 int ba = *pba++;
1134 int mask = ba ? 4 + ((19 >> scfcod[i]) & 3) : 0;
1135 for (m = 4; m; m >>= 1)
1136 {
1137 if (mask & m)
1138 {
1139 const int b = get_bits(bs, 6);
1140 const int idx = ba*3 - 6 + b % 3;
1141 mant = g_deq_L12_mant[idx];
1142 /* The float build multiplies by (1 << 21 >> b/3); in
1143 mantissa/exponent form that is purely an exponent shift. */
1144 exp = g_deq_L12_exp[idx] + 21 - b/3;
1145 }
1146 *scf_mant++ = mant;
1147 *scf_exp++ = (int8_t)exp;
1148 }
1149 }
1150}
1151#else
1152static void L12_read_scalefactors(bs_t *bs, uint8_t *pba, uint8_t *scfcod, int bands, float *scf) FL_NO_EXCEPT
1153{
1154 static const float g_deq_L12[18*3] = {
1155#define DQ(x) 9.53674316e-07f/x, 7.56931807e-07f/x, 6.00777173e-07f/x
1156 DQ(3),DQ(7),DQ(15),DQ(31),DQ(63),DQ(127),DQ(255),DQ(511),DQ(1023),DQ(2047),DQ(4095),DQ(8191),DQ(16383),DQ(32767),DQ(65535),DQ(3),DQ(5),DQ(9)
1157 };
1158 int i, m;
1159 for (i = 0; i < bands; i++)
1160 {
1161 float s = 0;
1162 int ba = *pba++;
1163 int mask = ba ? 4 + ((19 >> scfcod[i]) & 3) : 0;
1164 for (m = 4; m; m >>= 1)
1165 {
1166 if (mask & m)
1167 {
1168 int b = get_bits(bs, 6);
1169 s = g_deq_L12[ba*3 - 6 + b % 3]*(1 << 21 >> b/3);
1170 }
1171 *scf++ = s;
1172 }
1173 }
1174}
1175#endif /* MINIMP3_HAVE_FIXED_POINT */
1176
1177static void L12_read_scale_info(const uint8_t *hdr, bs_t *bs, L12_scale_info *sci) FL_NO_EXCEPT
1178{
1179 static const uint8_t g_bitalloc_code_tab[] = {
1180 0,17, 3, 4, 5,6,7, 8,9,10,11,12,13,14,15,16,
1181 0,17,18, 3,19,4,5, 6,7, 8, 9,10,11,12,13,16,
1182 0,17,18, 3,19,4,5,16,
1183 0,17,18,16,
1184 0,17,18,19, 4,5,6, 7,8, 9,10,11,12,13,14,15,
1185 0,17,18, 3,19,4,5, 6,7, 8, 9,10,11,12,13,14,
1186 0, 2, 3, 4, 5,6,7, 8,9,10,11,12,13,14,15,16
1187 };
1188 const L12_subband_alloc_t *subband_alloc = L12_subband_alloc_table(hdr, sci);
1189
1190 int i, k = 0, ba_bits = 0;
1191 const uint8_t *ba_code_tab = g_bitalloc_code_tab;
1192
1193 for (i = 0; i < sci->total_bands; i++)
1194 {
1195 uint8_t ba;
1196 if (i == k)
1197 {
1198 k += subband_alloc->band_count;
1199 ba_bits = subband_alloc->code_tab_width;
1200 ba_code_tab = g_bitalloc_code_tab + subband_alloc->tab_offset;
1201 subband_alloc++;
1202 }
1203 ba = ba_code_tab[get_bits(bs, ba_bits)];
1204 sci->bitalloc[2*i] = ba;
1205 if (i < sci->stereo_bands)
1206 {
1207 ba = ba_code_tab[get_bits(bs, ba_bits)];
1208 }
1209 sci->bitalloc[2*i + 1] = sci->stereo_bands ? ba : 0;
1210 }
1211
1212 for (i = 0; i < 2*sci->total_bands; i++)
1213 {
1214 sci->scfcod[i] = sci->bitalloc[i] ? HDR_IS_LAYER_1(hdr) ? 2 : get_bits(bs, 2) : 6;
1215 }
1216
1217#if MINIMP3_HAVE_FIXED_POINT
1218 L12_read_scalefactors(bs, sci->bitalloc, sci->scfcod, sci->total_bands*2, sci->scf_mant, sci->scf_exp);
1219#else
1220 L12_read_scalefactors(bs, sci->bitalloc, sci->scfcod, sci->total_bands*2, sci->scf);
1221#endif
1222
1223 for (i = sci->stereo_bands; i < sci->total_bands; i++)
1224 {
1225 sci->bitalloc[2*i + 1] = 0;
1226 }
1227}
1228
1229/* Writes the raw quantised integers, exactly as the float build writes raw
1230 integer-valued floats. L12_apply_scf_384 is what turns them into
1231 Q(MINIMP3_FRAC_BITS) samples, because the step size is not known until the
1232 scalefactors are applied. */
1233static int L12_dequantize_granule(mp3d_dsp_t *grbuf, bs_t *bs, L12_scale_info *sci, int group_size) FL_NO_EXCEPT
1234{
1235 int i, j, k, choff = 576;
1236 for (j = 0; j < 4; j++)
1237 {
1238 mp3d_dsp_t *dst = grbuf + group_size*j;
1239 for (i = 0; i < 2*sci->total_bands; i++)
1240 {
1241 int ba = sci->bitalloc[i];
1242 if (ba != 0)
1243 {
1244 if (ba < 17)
1245 {
1246 int half = (1 << (ba - 1)) - 1;
1247 for (k = 0; k < group_size; k++)
1248 {
1249 dst[k] = (mp3d_dsp_t)((int)get_bits(bs, ba) - half);
1250 }
1251 } else
1252 {
1253 unsigned mod = (2 << (ba - 17)) + 1; /* 3, 5, 9 */
1254 unsigned code = get_bits(bs, mod + 2 - (mod >> 3)); /* 5, 7, 10 */
1255 for (k = 0; k < group_size; k++, code /= mod)
1256 {
1257 dst[k] = (mp3d_dsp_t)((int)(code % mod - mod/2));
1258 }
1259 }
1260 }
1261 dst += choff;
1262 choff = 18 - choff;
1263 }
1264 }
1265 return group_size*4;
1266}
1267
1268#if MINIMP3_HAVE_FIXED_POINT
1269static void L12_apply_scf_384(L12_scale_info *sci, const int32_t *scf_mant, const int8_t *scf_exp, mp3d_dsp_t *dst) FL_NO_EXCEPT
1270{
1271 int i, k;
1272 memcpy(dst + 576 + sci->stereo_bands*18, dst + sci->stereo_bands*18, (sci->total_bands - sci->stereo_bands)*18*sizeof(mp3d_dsp_t));
1273 for (i = 0; i < sci->total_bands; i++, dst += 18, scf_mant += 6, scf_exp += 6)
1274 {
1275 for (k = 0; k < 12; k++)
1276 {
1277 dst[k + 0] = mp3d_l12_scale(dst[k + 0], scf_mant[0], scf_exp[0]);
1278 dst[k + 576] = mp3d_l12_scale(dst[k + 576], scf_mant[3], scf_exp[3]);
1279 }
1280 }
1281}
1282#else
1283static void L12_apply_scf_384(L12_scale_info *sci, const float *scf, float *dst) FL_NO_EXCEPT
1284{
1285 int i, k;
1286 memcpy(dst + 576 + sci->stereo_bands*18, dst + sci->stereo_bands*18, (sci->total_bands - sci->stereo_bands)*18*sizeof(float));
1287 for (i = 0; i < sci->total_bands; i++, dst += 18, scf += 6)
1288 {
1289 for (k = 0; k < 12; k++)
1290 {
1291 dst[k + 0] *= scf[0];
1292 dst[k + 576] *= scf[3];
1293 }
1294 }
1295}
1296#endif /* MINIMP3_HAVE_FIXED_POINT */
1297#endif /* MINIMP3_ONLY_MP3 */
1298
1299static int L3_read_side_info(bs_t *bs, L3_gr_info_t *gr, const uint8_t *hdr) FL_NO_EXCEPT
1300{
1301 static const uint8_t g_scf_long[8][23] = {
1302 { 6,6,6,6,6,6,8,10,12,14,16,20,24,28,32,38,46,52,60,68,58,54,0 },
1303 { 12,12,12,12,12,12,16,20,24,28,32,40,48,56,64,76,90,2,2,2,2,2,0 },
1304 { 6,6,6,6,6,6,8,10,12,14,16,20,24,28,32,38,46,52,60,68,58,54,0 },
1305 { 6,6,6,6,6,6,8,10,12,14,16,18,22,26,32,38,46,54,62,70,76,36,0 },
1306 { 6,6,6,6,6,6,8,10,12,14,16,20,24,28,32,38,46,52,60,68,58,54,0 },
1307 { 4,4,4,4,4,4,6,6,8,8,10,12,16,20,24,28,34,42,50,54,76,158,0 },
1308 { 4,4,4,4,4,4,6,6,6,8,10,12,16,18,22,28,34,40,46,54,54,192,0 },
1309 { 4,4,4,4,4,4,6,6,8,10,12,16,20,24,30,38,46,56,68,84,102,26,0 }
1310 };
1311 static const uint8_t g_scf_short[8][40] = {
1312 { 4,4,4,4,4,4,4,4,4,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,30,30,30,40,40,40,18,18,18,0 },
1313 { 8,8,8,8,8,8,8,8,8,12,12,12,16,16,16,20,20,20,24,24,24,28,28,28,36,36,36,2,2,2,2,2,2,2,2,2,26,26,26,0 },
1314 { 4,4,4,4,4,4,4,4,4,6,6,6,6,6,6,8,8,8,10,10,10,14,14,14,18,18,18,26,26,26,32,32,32,42,42,42,18,18,18,0 },
1315 { 4,4,4,4,4,4,4,4,4,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,32,32,32,44,44,44,12,12,12,0 },
1316 { 4,4,4,4,4,4,4,4,4,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,30,30,30,40,40,40,18,18,18,0 },
1317 { 4,4,4,4,4,4,4,4,4,4,4,4,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,22,22,22,30,30,30,56,56,56,0 },
1318 { 4,4,4,4,4,4,4,4,4,4,4,4,6,6,6,6,6,6,10,10,10,12,12,12,14,14,14,16,16,16,20,20,20,26,26,26,66,66,66,0 },
1319 { 4,4,4,4,4,4,4,4,4,4,4,4,6,6,6,8,8,8,12,12,12,16,16,16,20,20,20,26,26,26,34,34,34,42,42,42,12,12,12,0 }
1320 };
1321 static const uint8_t g_scf_mixed[8][40] = {
1322 { 6,6,6,6,6,6,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,30,30,30,40,40,40,18,18,18,0 },
1323 { 12,12,12,4,4,4,8,8,8,12,12,12,16,16,16,20,20,20,24,24,24,28,28,28,36,36,36,2,2,2,2,2,2,2,2,2,26,26,26,0 },
1324 { 6,6,6,6,6,6,6,6,6,6,6,6,8,8,8,10,10,10,14,14,14,18,18,18,26,26,26,32,32,32,42,42,42,18,18,18,0 },
1325 { 6,6,6,6,6,6,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,32,32,32,44,44,44,12,12,12,0 },
1326 { 6,6,6,6,6,6,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,24,24,24,30,30,30,40,40,40,18,18,18,0 },
1327 { 4,4,4,4,4,4,6,6,4,4,4,6,6,6,8,8,8,10,10,10,12,12,12,14,14,14,18,18,18,22,22,22,30,30,30,56,56,56,0 },
1328 { 4,4,4,4,4,4,6,6,4,4,4,6,6,6,6,6,6,10,10,10,12,12,12,14,14,14,16,16,16,20,20,20,26,26,26,66,66,66,0 },
1329 { 4,4,4,4,4,4,6,6,4,4,4,6,6,6,8,8,8,12,12,12,16,16,16,20,20,20,26,26,26,34,34,34,42,42,42,12,12,12,0 }
1330 };
1331
1332 unsigned tables, scfsi = 0;
1333 int main_data_begin, part_23_sum = 0;
1334 int sr_idx = HDR_GET_MY_SAMPLE_RATE(hdr); sr_idx -= (sr_idx != 0);
1335 int gr_count = HDR_IS_MONO(hdr) ? 1 : 2;
1336
1337 if (HDR_TEST_MPEG1(hdr))
1338 {
1339 gr_count *= 2;
1340 main_data_begin = get_bits(bs, 9);
1341 scfsi = get_bits(bs, 7 + gr_count);
1342 } else
1343 {
1344 main_data_begin = get_bits(bs, 8 + gr_count) >> gr_count;
1345 }
1346
1347 do
1348 {
1349 if (HDR_IS_MONO(hdr))
1350 {
1351 scfsi <<= 4;
1352 }
1353 gr->part_23_length = (uint16_t)get_bits(bs, 12);
1354 part_23_sum += gr->part_23_length;
1355 gr->big_values = (uint16_t)get_bits(bs, 9);
1356 if (gr->big_values > 288)
1357 {
1358 return -1;
1359 }
1360 gr->global_gain = (uint8_t)get_bits(bs, 8);
1361 gr->scalefac_compress = (uint16_t)get_bits(bs, HDR_TEST_MPEG1(hdr) ? 4 : 9);
1362 gr->sfbtab = g_scf_long[sr_idx];
1363 gr->n_long_sfb = 22;
1364 gr->n_short_sfb = 0;
1365 if (get_bits(bs, 1))
1366 {
1367 gr->block_type = (uint8_t)get_bits(bs, 2);
1368 if (!gr->block_type)
1369 {
1370 return -1;
1371 }
1372 gr->mixed_block_flag = (uint8_t)get_bits(bs, 1);
1373 gr->region_count[0] = 7;
1374 gr->region_count[1] = 255;
1375 if (gr->block_type == SHORT_BLOCK_TYPE)
1376 {
1377 scfsi &= 0x0F0F;
1378 if (!gr->mixed_block_flag)
1379 {
1380 gr->region_count[0] = 8;
1381 gr->sfbtab = g_scf_short[sr_idx];
1382 gr->n_long_sfb = 0;
1383 gr->n_short_sfb = 39;
1384 } else
1385 {
1386 gr->sfbtab = g_scf_mixed[sr_idx];
1387 gr->n_long_sfb = HDR_TEST_MPEG1(hdr) ? 8 : 6;
1388 gr->n_short_sfb = 30;
1389 }
1390 }
1391 tables = get_bits(bs, 10);
1392 tables <<= 5;
1393 gr->subblock_gain[0] = (uint8_t)get_bits(bs, 3);
1394 gr->subblock_gain[1] = (uint8_t)get_bits(bs, 3);
1395 gr->subblock_gain[2] = (uint8_t)get_bits(bs, 3);
1396 } else
1397 {
1398 gr->block_type = 0;
1399 gr->mixed_block_flag = 0;
1400 tables = get_bits(bs, 15);
1401 gr->region_count[0] = (uint8_t)get_bits(bs, 4);
1402 gr->region_count[1] = (uint8_t)get_bits(bs, 3);
1403 gr->region_count[2] = 255;
1404 }
1405 gr->table_select[0] = (uint8_t)(tables >> 10);
1406 gr->table_select[1] = (uint8_t)((tables >> 5) & 31);
1407 gr->table_select[2] = (uint8_t)((tables) & 31);
1408 gr->preflag = HDR_TEST_MPEG1(hdr) ? get_bits(bs, 1) : (gr->scalefac_compress >= 500);
1409 gr->scalefac_scale = (uint8_t)get_bits(bs, 1);
1410 gr->count1_table = (uint8_t)get_bits(bs, 1);
1411 gr->scfsi = (uint8_t)((scfsi >> 12) & 15);
1412 scfsi <<= 4;
1413 gr++;
1414 } while(--gr_count);
1415
1416 if (part_23_sum + bs->pos > bs->limit + main_data_begin*8)
1417 {
1418 return -1;
1419 }
1420
1421 return main_data_begin;
1422}
1423
1424static void L3_read_scalefactors(uint8_t *scf, uint8_t *ist_pos, const uint8_t *scf_size, const uint8_t *scf_count, bs_t *bitbuf, int scfsi) FL_NO_EXCEPT
1425{
1426 int i, k;
1427 for (i = 0; i < 4 && scf_count[i]; i++, scfsi *= 2)
1428 {
1429 int cnt = scf_count[i];
1430 if (scfsi & 8)
1431 {
1432 memcpy(scf, ist_pos, cnt);
1433 } else
1434 {
1435 int bits = scf_size[i];
1436 if (!bits)
1437 {
1438 memset(scf, 0, cnt);
1439 memset(ist_pos, 0, cnt);
1440 } else
1441 {
1442 int max_scf = (scfsi < 0) ? (1 << bits) - 1 : -1;
1443 for (k = 0; k < cnt; k++)
1444 {
1445 int s = get_bits(bitbuf, bits);
1446 ist_pos[k] = (s == max_scf ? -1 : s);
1447 scf[k] = s;
1448 }
1449 }
1450 }
1451 ist_pos += cnt;
1452 scf += cnt;
1453 }
1454 scf[0] = scf[1] = scf[2] = 0;
1455}
1456
1457#if MINIMP3_HAVE_FIXED_POINT
1458/* Split a gain exponent expressed in quarter-powers of two into the
1459 mantissa/exponent pair the dequantiser multiplies with:
1460 2**(quarters/4) == g_expfrac_q30[r] * 2**(a - 30) for quarters == 4a + r. */
1461static void mp3d_gain_from_quarters(int quarters, int32_t *mant,
1462 int *exp) FL_NO_EXCEPT
1463{
1464 /* Floor division that stays correct for negative `quarters` without
1465 shifting a negative value, which is what UBSan objects to. */
1466 int r = quarters % 4;
1467 int a;
1468 if (r < 0)
1469 {
1470 r += 4;
1471 }
1472 a = (quarters - r) / 4;
1473 *mant = g_expfrac_q30[r];
1474 *exp = a;
1475}
1476
1477/* x**(4/3) as mantissa * 2**(exp - 30).
1478
1479 Below 129 this is a table lookup, laid out exactly like upstream's float
1480 table so the Huffman loop keeps folding the sign into the index. Above it,
1481 upstream interpolates from the same table with a quadratic in the fractional
1482 part; that is reproduced here in Q30, which is the most delicate arithmetic
1483 in the conversion and is covered exhaustively by a unit test over every
1484 reachable x. */
1485static int32_t mp3d_pow43(int x, int *exp) FL_NO_EXCEPT
1486{
1487 int mult_log2 = 8;
1488 int sign, num, den, idx;
1489 int32_t frac, poly, term;
1490
1491 if (x < 129)
1492 {
1493 *exp = g_pow43_exp[16 + x];
1494 return g_pow43_mant[16 + x];
1495 }
1496 if (x < 1024)
1497 {
1498 mult_log2 = 4;
1499 x <<= 3;
1500 }
1501 sign = 2*x & 64;
1502 num = (x & 63) - sign;
1503 den = (x & ~63) + sign;
1504 /* Multiply rather than shift: `num` is negative for half of all inputs
1505 and shifting a negative value left is UB, which the differential fuzzer
1506 caught under UBSan. */
1507 frac = (int32_t)(((int64_t)num * ((int64_t)1 << 30)) / den);
1508 term = MP3D_Q30_POW43_C1 + mp3d_mulshift_k<30, MP3D_Q30_POW43_C2>(frac);
1509 /* Shift by 31 rather than 30: the polynomial reaches 1.72, which would
1510 carry a normalised mantissa past INT32_MAX. The lost bit is paid back
1511 in the exponent. */
1512 poly = ((int32_t)1 << 30) + mp3d_mulshift<30>(frac, term);
1513 idx = 16 + ((x + sign) >> 6);
1514 *exp = g_pow43_exp[idx] + mult_log2 + 1;
1515 return mp3d_mulshift<31>(g_pow43_mant[idx], poly);
1516}
1517
1518/* scalefactor gain * x**(4/3), landed in Q(MINIMP3_FRAC_BITS) and clamped.
1519 Both inputs carry their own exponent, so this is where the pipeline's one
1520 unbounded quantity collapses into the fixed Q format. */
1521/* Force-inlined with get_bits above; see that comment. */
1522MP3D_LEAF int32_t mp3d_dequant(int32_t gain_mant, int gain_exp,
1523 int32_t pow_mant, int pow_exp) FL_NO_EXCEPT
1524{
1525 int64_t product;
1526 int shift;
1527 if (pow_mant == 0)
1528 {
1529 return 0;
1530 }
1531 product = (int64_t)gain_mant * (int64_t)pow_mant;
1532 if (product == 0)
1533 {
1534 return 0;
1535 }
1536 /* value = product * 2**(gain_exp + pow_exp - 60); want Q(FRAC_BITS). */
1537 shift = 60 - MINIMP3_FRAC_BITS - gain_exp - pow_exp;
1538 if (shift >= 63)
1539 {
1540 return 0;
1541 }
1542 if (shift <= 0)
1543 {
1544 return product > 0 ? MP3D_SAT_MAX : MP3D_SAT_MIN;
1545 }
1546 product = (product + ((int64_t)1 << (shift - 1))) >> shift;
1547 return mp3d_clamp_sample(product);
1548}
1549#else
1550static float L3_ldexp_q2(float y, int exp_q2) FL_NO_EXCEPT
1551{
1552 static const float g_expfrac[4] = { 9.31322575e-10f,7.83145814e-10f,6.58544508e-10f,5.53767716e-10f };
1553 int e;
1554 do
1555 {
1556 e = MINIMP3_MIN(30*4, exp_q2);
1557 y *= g_expfrac[e & 3]*(1 << 30 >> (e >> 2));
1558 } while ((exp_q2 -= e) > 0);
1559 return y;
1560}
1561#endif /* MINIMP3_HAVE_FIXED_POINT */
1562
1563/* The bitstream half of scalefactor decoding is shared: only how the decoded
1564 integers are turned into gains differs between the two builds. Keeping
1565 `iscf`/`ist_pos` handling common is deliberate -- `ist_pos` drives intensity
1566 stereo positioning, and any drift there would be a parsing divergence rather
1567 than a rounding one. */
1568#if MINIMP3_HAVE_FIXED_POINT
1569static void L3_decode_scalefactors(const uint8_t *hdr, uint8_t *ist_pos, bs_t *bs, const L3_gr_info_t *gr, int32_t *scf_mant, int16_t *scf_exp, int ch) FL_NO_EXCEPT
1570#else
1571static void L3_decode_scalefactors(const uint8_t *hdr, uint8_t *ist_pos, bs_t *bs, const L3_gr_info_t *gr, float *scf, int ch) FL_NO_EXCEPT
1572#endif
1573{
1574 static const uint8_t g_scf_partitions[3][28] = {
1575 { 6,5,5, 5,6,5,5,5,6,5, 7,3,11,10,0,0, 7, 7, 7,0, 6, 6,6,3, 8, 8,5,0 },
1576 { 8,9,6,12,6,9,9,9,6,9,12,6,15,18,0,0, 6,15,12,0, 6,12,9,6, 6,18,9,0 },
1577 { 9,9,6,12,9,9,9,9,9,9,12,6,18,18,0,0,12,12,12,0,12, 9,9,6,15,12,9,0 }
1578 };
1579 const uint8_t *scf_partition = g_scf_partitions[!!gr->n_short_sfb + !gr->n_long_sfb];
1580 uint8_t scf_size[4], iscf[40];
1581 int i, scf_shift = gr->scalefac_scale + 1, gain_exp, scfsi = gr->scfsi;
1582#if !MINIMP3_HAVE_FIXED_POINT
1583 float gain;
1584#endif
1585
1586 if (HDR_TEST_MPEG1(hdr))
1587 {
1588 static const uint8_t g_scfc_decode[16] = { 0,1,2,3, 12,5,6,7, 9,10,11,13, 14,15,18,19 };
1589 int part = g_scfc_decode[gr->scalefac_compress];
1590 scf_size[1] = scf_size[0] = (uint8_t)(part >> 2);
1591 scf_size[3] = scf_size[2] = (uint8_t)(part & 3);
1592 } else
1593 {
1594 static const uint8_t g_mod[6*4] = { 5,5,4,4,5,5,4,1,4,3,1,1,5,6,6,1,4,4,4,1,4,3,1,1 };
1595 int k, modprod, sfc, ist = HDR_TEST_I_STEREO(hdr) && ch;
1596 sfc = gr->scalefac_compress >> ist;
1597 for (k = ist*3*4; sfc >= 0; sfc -= modprod, k += 4)
1598 {
1599 for (modprod = 1, i = 3; i >= 0; i--)
1600 {
1601 scf_size[i] = (uint8_t)(sfc / modprod % g_mod[k + i]);
1602 modprod *= g_mod[k + i];
1603 }
1604 }
1605 scf_partition += k;
1606 scfsi = -16;
1607 }
1608 L3_read_scalefactors(iscf, ist_pos, scf_size, scf_partition, bs, scfsi);
1609
1610 if (gr->n_short_sfb)
1611 {
1612 int sh = 3 - scf_shift;
1613 for (i = 0; i < gr->n_short_sfb; i += 3)
1614 {
1615 iscf[gr->n_long_sfb + i + 0] += gr->subblock_gain[0] << sh;
1616 iscf[gr->n_long_sfb + i + 1] += gr->subblock_gain[1] << sh;
1617 iscf[gr->n_long_sfb + i + 2] += gr->subblock_gain[2] << sh;
1618 }
1619 } else if (gr->preflag)
1620 {
1621 static const uint8_t g_preamp[10] = { 1,1,1,1,2,2,3,3,3,2 };
1622 for (i = 0; i < 10; i++)
1623 {
1624 iscf[11 + i] += g_preamp[i];
1625 }
1626 }
1627
1628 gain_exp = gr->global_gain + BITS_DEQUANTIZER_OUT*4 - 210 - (HDR_IS_MS_STEREO(hdr) ? 2 : 0);
1629#if MINIMP3_HAVE_FIXED_POINT
1630 /* The float build computes 2**11 * 2**((gain_exp - 44 - iscf*2**shift)/4)
1631 by repeated multiplication. In quarter-power terms that whole expression
1632 is just an integer exponent, so the fixed build carries the integer and
1633 looks the fractional quarter up -- exact, and with none of
1634 L3_ldexp_q2's rounding. */
1635 for (i = 0; i < (int)(gr->n_long_sfb + gr->n_short_sfb); i++)
1636 {
1637 int32_t mant;
1638 int exp;
1639 mp3d_gain_from_quarters(gain_exp - MAX_SCFI - (iscf[i] << scf_shift),
1640 &mant, &exp);
1641 scf_mant[i] = mant;
1642 scf_exp[i] = (int16_t)(exp + (MAX_SCFI/4));
1643 }
1644#else
1645 gain = L3_ldexp_q2(1 << (MAX_SCFI/4), MAX_SCFI - gain_exp);
1646 for (i = 0; i < (int)(gr->n_long_sfb + gr->n_short_sfb); i++)
1647 {
1648 scf[i] = L3_ldexp_q2(gain, iscf[i] << scf_shift);
1649 }
1650#endif
1651}
1652
1653#if MINIMP3_HAVE_FIXED_POINT
1654static int32_t mp3d_huff_escape(int32_t gain_mant, int gain_exp, int lsb,
1655 int negative) FL_NO_EXCEPT
1656{
1657 int pow_exp;
1658 const int32_t pow_mant = mp3d_pow43(lsb, &pow_exp);
1659 const int32_t value = mp3d_dequant(gain_mant, gain_exp, pow_mant, pow_exp);
1660 return negative ? -value : value;
1661}
1662
1663/* Force-inlined with get_bits above; see that comment. */
1664MP3D_LEAF int32_t mp3d_huff_one(int32_t gain_mant, int gain_exp,
1665 int negative) FL_NO_EXCEPT
1666{
1667 const int32_t value = mp3d_scale_to_q(gain_mant, gain_exp);
1668 return negative ? -value : value;
1669}
1670
1671/* The Huffman loop itself is bit-exact between the two builds -- only the four
1672 places that turn a decoded magnitude into a sample differ, so they go
1673 through these and the loop stays single-sourced. Frame acceptance and bit
1674 consumption are parsing properties, and the golden gate treats any
1675 divergence in them as a hard failure rather than a rounding one. */
1676#define MP3D_HUFF_SCF_ARGS const int32_t *scf_mant, const int16_t *scf_exp
1677#define MP3D_HUFF_ONE_VARS int32_t one_m = 0; int one_e = 0
1678#define MP3D_HUFF_NEXT_SCF() (one_m = *scf_mant++, one_e = *scf_exp++)
1679#define MP3D_HUFF_ESC(lsb, neg) mp3d_huff_escape(one_m, one_e, (lsb), (neg))
1680#define MP3D_HUFF_TAB(idx) mp3d_dequant(one_m, one_e, g_pow43_mant[idx], \
1681 g_pow43_exp[idx])
1682#define MP3D_HUFF_ONE(neg) mp3d_huff_one(one_m, one_e, (neg))
1683#else
1684#define MP3D_HUFF_SCF_ARGS const float *scf
1685#define MP3D_HUFF_ONE_VARS float one = 0.0f
1686#define MP3D_HUFF_NEXT_SCF() (one = *scf++)
1687#define MP3D_HUFF_ESC(lsb, neg) (one*L3_pow_43(lsb)*((neg) ? -1 : 1))
1688#define MP3D_HUFF_TAB(idx) (g_pow43[idx]*one)
1689#define MP3D_HUFF_ONE(neg) ((neg) ? -one : one)
1690
1691static const float g_pow43[129 + 16] = {
1692 0,-1,-2.519842f,-4.326749f,-6.349604f,-8.549880f,-10.902724f,-13.390518f,-16.000000f,-18.720754f,-21.544347f,-24.463781f,-27.473142f,-30.567351f,-33.741992f,-36.993181f,
1693 0,1,2.519842f,4.326749f,6.349604f,8.549880f,10.902724f,13.390518f,16.000000f,18.720754f,21.544347f,24.463781f,27.473142f,30.567351f,33.741992f,36.993181f,40.317474f,43.711787f,47.173345f,50.699631f,54.288352f,57.937408f,61.644865f,65.408941f,69.227979f,73.100443f,77.024898f,81.000000f,85.024491f,89.097188f,93.216975f,97.382800f,101.593667f,105.848633f,110.146801f,114.487321f,118.869381f,123.292209f,127.755065f,132.257246f,136.798076f,141.376907f,145.993119f,150.646117f,155.335327f,160.060199f,164.820202f,169.614826f,174.443577f,179.305980f,184.201575f,189.129918f,194.090580f,199.083145f,204.107210f,209.162385f,214.248292f,219.364564f,224.510845f,229.686789f,234.892058f,240.126328f,245.389280f,250.680604f,256.000000f,261.347174f,266.721841f,272.123723f,277.552547f,283.008049f,288.489971f,293.998060f,299.532071f,305.091761f,310.676898f,316.287249f,321.922592f,327.582707f,333.267377f,338.976394f,344.709550f,350.466646f,356.247482f,362.051866f,367.879608f,373.730522f,379.604427f,385.501143f,391.420496f,397.362314f,403.326427f,409.312672f,415.320884f,421.350905f,427.402579f,433.475750f,439.570269f,445.685987f,451.822757f,457.980436f,464.158883f,470.357960f,476.577530f,482.817459f,489.077615f,495.357868f,501.658090f,507.978156f,514.317941f,520.677324f,527.056184f,533.454404f,539.871867f,546.308458f,552.764065f,559.238575f,565.731879f,572.243870f,578.774440f,585.323483f,591.890898f,598.476581f,605.080431f,611.702349f,618.342238f,625.000000f,631.675540f,638.368763f,645.079578f
1694};
1695
1696static float L3_pow_43(int x) FL_NO_EXCEPT
1697{
1698 float frac;
1699 int sign, mult = 256;
1700
1701 if (x < 129)
1702 {
1703 return g_pow43[16 + x];
1704 }
1705
1706 if (x < 1024)
1707 {
1708 mult = 16;
1709 x <<= 3;
1710 }
1711
1712 sign = 2*x & 64;
1713 frac = (float)((x & 63) - sign) / ((x & ~63) + sign);
1714 return g_pow43[16 + ((x + sign) >> 6)]*(1.f + frac*((4.f/3) + frac*(2.f/9)))*mult;
1715}
1716#endif /* MINIMP3_HAVE_FIXED_POINT */
1717
1718static void L3_huffman(mp3d_dsp_t *dst, bs_t *bs, const L3_gr_info_t *gr_info, MP3D_HUFF_SCF_ARGS, int layer3gr_limit) FL_NO_EXCEPT
1719{
1720 static const int16_t tabs[] = { 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1721 785,785,785,785,784,784,784,784,513,513,513,513,513,513,513,513,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,
1722 -255,1313,1298,1282,785,785,785,785,784,784,784,784,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,290,288,
1723 -255,1313,1298,1282,769,769,769,769,529,529,529,529,529,529,529,529,528,528,528,528,528,528,528,528,512,512,512,512,512,512,512,512,290,288,
1724 -253,-318,-351,-367,785,785,785,785,784,784,784,784,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,819,818,547,547,275,275,275,275,561,560,515,546,289,274,288,258,
1725 -254,-287,1329,1299,1314,1312,1057,1057,1042,1042,1026,1026,784,784,784,784,529,529,529,529,529,529,529,529,769,769,769,769,768,768,768,768,563,560,306,306,291,259,
1726 -252,-413,-477,-542,1298,-575,1041,1041,784,784,784,784,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,-383,-399,1107,1092,1106,1061,849,849,789,789,1104,1091,773,773,1076,1075,341,340,325,309,834,804,577,577,532,532,516,516,832,818,803,816,561,561,531,531,515,546,289,289,288,258,
1727 -252,-429,-493,-559,1057,1057,1042,1042,529,529,529,529,529,529,529,529,784,784,784,784,769,769,769,769,512,512,512,512,512,512,512,512,-382,1077,-415,1106,1061,1104,849,849,789,789,1091,1076,1029,1075,834,834,597,581,340,340,339,324,804,833,532,532,832,772,818,803,817,787,816,771,290,290,290,290,288,258,
1728 -253,-349,-414,-447,-463,1329,1299,-479,1314,1312,1057,1057,1042,1042,1026,1026,785,785,785,785,784,784,784,784,769,769,769,769,768,768,768,768,-319,851,821,-335,836,850,805,849,341,340,325,336,533,533,579,579,564,564,773,832,578,548,563,516,321,276,306,291,304,259,
1729 -251,-572,-733,-830,-863,-879,1041,1041,784,784,784,784,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,-511,-527,-543,1396,1351,1381,1366,1395,1335,1380,-559,1334,1138,1138,1063,1063,1350,1392,1031,1031,1062,1062,1364,1363,1120,1120,1333,1348,881,881,881,881,375,374,359,373,343,358,341,325,791,791,1123,1122,-703,1105,1045,-719,865,865,790,790,774,774,1104,1029,338,293,323,308,-799,-815,833,788,772,818,803,816,322,292,307,320,561,531,515,546,289,274,288,258,
1730 -251,-525,-605,-685,-765,-831,-846,1298,1057,1057,1312,1282,785,785,785,785,784,784,784,784,769,769,769,769,512,512,512,512,512,512,512,512,1399,1398,1383,1367,1382,1396,1351,-511,1381,1366,1139,1139,1079,1079,1124,1124,1364,1349,1363,1333,882,882,882,882,807,807,807,807,1094,1094,1136,1136,373,341,535,535,881,775,867,822,774,-591,324,338,-671,849,550,550,866,864,609,609,293,336,534,534,789,835,773,-751,834,804,308,307,833,788,832,772,562,562,547,547,305,275,560,515,290,290,
1731 -252,-397,-477,-557,-622,-653,-719,-735,-750,1329,1299,1314,1057,1057,1042,1042,1312,1282,1024,1024,785,785,785,785,784,784,784,784,769,769,769,769,-383,1127,1141,1111,1126,1140,1095,1110,869,869,883,883,1079,1109,882,882,375,374,807,868,838,881,791,-463,867,822,368,263,852,837,836,-543,610,610,550,550,352,336,534,534,865,774,851,821,850,805,593,533,579,564,773,832,578,578,548,548,577,577,307,276,306,291,516,560,259,259,
1732 -250,-2107,-2507,-2764,-2909,-2974,-3007,-3023,1041,1041,1040,1040,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,-767,-1052,-1213,-1277,-1358,-1405,-1469,-1535,-1550,-1582,-1614,-1647,-1662,-1694,-1726,-1759,-1774,-1807,-1822,-1854,-1886,1565,-1919,-1935,-1951,-1967,1731,1730,1580,1717,-1983,1729,1564,-1999,1548,-2015,-2031,1715,1595,-2047,1714,-2063,1610,-2079,1609,-2095,1323,1323,1457,1457,1307,1307,1712,1547,1641,1700,1699,1594,1685,1625,1442,1442,1322,1322,-780,-973,-910,1279,1278,1277,1262,1276,1261,1275,1215,1260,1229,-959,974,974,989,989,-943,735,478,478,495,463,506,414,-1039,1003,958,1017,927,942,987,957,431,476,1272,1167,1228,-1183,1256,-1199,895,895,941,941,1242,1227,1212,1135,1014,1014,490,489,503,487,910,1013,985,925,863,894,970,955,1012,847,-1343,831,755,755,984,909,428,366,754,559,-1391,752,486,457,924,997,698,698,983,893,740,740,908,877,739,739,667,667,953,938,497,287,271,271,683,606,590,712,726,574,302,302,738,736,481,286,526,725,605,711,636,724,696,651,589,681,666,710,364,467,573,695,466,466,301,465,379,379,709,604,665,679,316,316,634,633,436,436,464,269,424,394,452,332,438,363,347,408,393,448,331,422,362,407,392,421,346,406,391,376,375,359,1441,1306,-2367,1290,-2383,1337,-2399,-2415,1426,1321,-2431,1411,1336,-2447,-2463,-2479,1169,1169,1049,1049,1424,1289,1412,1352,1319,-2495,1154,1154,1064,1064,1153,1153,416,390,360,404,403,389,344,374,373,343,358,372,327,357,342,311,356,326,1395,1394,1137,1137,1047,1047,1365,1392,1287,1379,1334,1364,1349,1378,1318,1363,792,792,792,792,1152,1152,1032,1032,1121,1121,1046,1046,1120,1120,1030,1030,-2895,1106,1061,1104,849,849,789,789,1091,1076,1029,1090,1060,1075,833,833,309,324,532,532,832,772,818,803,561,561,531,560,515,546,289,274,288,258,
1733 -250,-1179,-1579,-1836,-1996,-2124,-2253,-2333,-2413,-2477,-2542,-2574,-2607,-2622,-2655,1314,1313,1298,1312,1282,785,785,785,785,1040,1040,1025,1025,768,768,768,768,-766,-798,-830,-862,-895,-911,-927,-943,-959,-975,-991,-1007,-1023,-1039,-1055,-1070,1724,1647,-1103,-1119,1631,1767,1662,1738,1708,1723,-1135,1780,1615,1779,1599,1677,1646,1778,1583,-1151,1777,1567,1737,1692,1765,1722,1707,1630,1751,1661,1764,1614,1736,1676,1763,1750,1645,1598,1721,1691,1762,1706,1582,1761,1566,-1167,1749,1629,767,766,751,765,494,494,735,764,719,749,734,763,447,447,748,718,477,506,431,491,446,476,461,505,415,430,475,445,504,399,460,489,414,503,383,474,429,459,502,502,746,752,488,398,501,473,413,472,486,271,480,270,-1439,-1455,1357,-1471,-1487,-1503,1341,1325,-1519,1489,1463,1403,1309,-1535,1372,1448,1418,1476,1356,1462,1387,-1551,1475,1340,1447,1402,1386,-1567,1068,1068,1474,1461,455,380,468,440,395,425,410,454,364,467,466,464,453,269,409,448,268,432,1371,1473,1432,1417,1308,1460,1355,1446,1459,1431,1083,1083,1401,1416,1458,1445,1067,1067,1370,1457,1051,1051,1291,1430,1385,1444,1354,1415,1400,1443,1082,1082,1173,1113,1186,1066,1185,1050,-1967,1158,1128,1172,1097,1171,1081,-1983,1157,1112,416,266,375,400,1170,1142,1127,1065,793,793,1169,1033,1156,1096,1141,1111,1155,1080,1126,1140,898,898,808,808,897,897,792,792,1095,1152,1032,1125,1110,1139,1079,1124,882,807,838,881,853,791,-2319,867,368,263,822,852,837,866,806,865,-2399,851,352,262,534,534,821,836,594,594,549,549,593,593,533,533,848,773,579,579,564,578,548,563,276,276,577,576,306,291,516,560,305,305,275,259,
1734 -251,-892,-2058,-2620,-2828,-2957,-3023,-3039,1041,1041,1040,1040,769,769,769,769,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,256,-511,-527,-543,-559,1530,-575,-591,1528,1527,1407,1526,1391,1023,1023,1023,1023,1525,1375,1268,1268,1103,1103,1087,1087,1039,1039,1523,-604,815,815,815,815,510,495,509,479,508,463,507,447,431,505,415,399,-734,-782,1262,-815,1259,1244,-831,1258,1228,-847,-863,1196,-879,1253,987,987,748,-767,493,493,462,477,414,414,686,669,478,446,461,445,474,429,487,458,412,471,1266,1264,1009,1009,799,799,-1019,-1276,-1452,-1581,-1677,-1757,-1821,-1886,-1933,-1997,1257,1257,1483,1468,1512,1422,1497,1406,1467,1496,1421,1510,1134,1134,1225,1225,1466,1451,1374,1405,1252,1252,1358,1480,1164,1164,1251,1251,1238,1238,1389,1465,-1407,1054,1101,-1423,1207,-1439,830,830,1248,1038,1237,1117,1223,1148,1236,1208,411,426,395,410,379,269,1193,1222,1132,1235,1221,1116,976,976,1192,1162,1177,1220,1131,1191,963,963,-1647,961,780,-1663,558,558,994,993,437,408,393,407,829,978,813,797,947,-1743,721,721,377,392,844,950,828,890,706,706,812,859,796,960,948,843,934,874,571,571,-1919,690,555,689,421,346,539,539,944,779,918,873,932,842,903,888,570,570,931,917,674,674,-2575,1562,-2591,1609,-2607,1654,1322,1322,1441,1441,1696,1546,1683,1593,1669,1624,1426,1426,1321,1321,1639,1680,1425,1425,1305,1305,1545,1668,1608,1623,1667,1592,1638,1666,1320,1320,1652,1607,1409,1409,1304,1304,1288,1288,1664,1637,1395,1395,1335,1335,1622,1636,1394,1394,1319,1319,1606,1621,1392,1392,1137,1137,1137,1137,345,390,360,375,404,373,1047,-2751,-2767,-2783,1062,1121,1046,-2799,1077,-2815,1106,1061,789,789,1105,1104,263,355,310,340,325,354,352,262,339,324,1091,1076,1029,1090,1060,1075,833,833,788,788,1088,1028,818,818,803,803,561,561,531,531,816,771,546,546,289,274,288,258,
1735 -253,-317,-381,-446,-478,-509,1279,1279,-811,-1179,-1451,-1756,-1900,-2028,-2189,-2253,-2333,-2414,-2445,-2511,-2526,1313,1298,-2559,1041,1041,1040,1040,1025,1025,1024,1024,1022,1007,1021,991,1020,975,1019,959,687,687,1018,1017,671,671,655,655,1016,1015,639,639,758,758,623,623,757,607,756,591,755,575,754,559,543,543,1009,783,-575,-621,-685,-749,496,-590,750,749,734,748,974,989,1003,958,988,973,1002,942,987,957,972,1001,926,986,941,971,956,1000,910,985,925,999,894,970,-1071,-1087,-1102,1390,-1135,1436,1509,1451,1374,-1151,1405,1358,1480,1420,-1167,1507,1494,1389,1342,1465,1435,1450,1326,1505,1310,1493,1373,1479,1404,1492,1464,1419,428,443,472,397,736,526,464,464,486,457,442,471,484,482,1357,1449,1434,1478,1388,1491,1341,1490,1325,1489,1463,1403,1309,1477,1372,1448,1418,1433,1476,1356,1462,1387,-1439,1475,1340,1447,1402,1474,1324,1461,1371,1473,269,448,1432,1417,1308,1460,-1711,1459,-1727,1441,1099,1099,1446,1386,1431,1401,-1743,1289,1083,1083,1160,1160,1458,1445,1067,1067,1370,1457,1307,1430,1129,1129,1098,1098,268,432,267,416,266,400,-1887,1144,1187,1082,1173,1113,1186,1066,1050,1158,1128,1143,1172,1097,1171,1081,420,391,1157,1112,1170,1142,1127,1065,1169,1049,1156,1096,1141,1111,1155,1080,1126,1154,1064,1153,1140,1095,1048,-2159,1125,1110,1137,-2175,823,823,1139,1138,807,807,384,264,368,263,868,838,853,791,867,822,852,837,866,806,865,790,-2319,851,821,836,352,262,850,805,849,-2399,533,533,835,820,336,261,578,548,563,577,532,532,832,772,562,562,547,547,305,275,560,515,290,290,288,258 };
1736 static const uint8_t tab32[] = { 130,162,193,209,44,28,76,140,9,9,9,9,9,9,9,9,190,254,222,238,126,94,157,157,109,61,173,205 };
1737 static const uint8_t tab33[] = { 252,236,220,204,188,172,156,140,124,108,92,76,60,44,28,12 };
1738 static const int16_t tabindex[2*16] = { 0,32,64,98,0,132,180,218,292,364,426,538,648,746,0,1126,1460,1460,1460,1460,1460,1460,1460,1460,1842,1842,1842,1842,1842,1842,1842,1842 };
1739 static const uint8_t g_linbits[] = { 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,3,4,6,8,10,13,4,5,6,7,8,9,11,13 };
1740
1741#define PEEK_BITS(n) (bs_cache >> (32 - n))
1742#define FLUSH_BITS(n) { bs_cache <<= (n); bs_sh += (n); }
1743#define CHECK_BITS while (bs_sh >= 0) { bs_cache |= (uint32_t)*bs_next_ptr++ << bs_sh; bs_sh -= 8; }
1744#define BSPOS ((bs_next_ptr - bs->buf)*8 - 24 + bs_sh)
1745
1746 MP3D_HUFF_ONE_VARS;
1747 int ireg = 0, big_val_cnt = gr_info->big_values;
1748 const uint8_t *sfb = gr_info->sfbtab;
1749 const uint8_t *bs_next_ptr = bs->buf + bs->pos/8;
1750 uint32_t bs_cache = (((bs_next_ptr[0]*256u + bs_next_ptr[1])*256u + bs_next_ptr[2])*256u + bs_next_ptr[3]) << (bs->pos & 7);
1751 int pairs_to_decode, np, bs_sh = (bs->pos & 7) - 8;
1752 bs_next_ptr += 4;
1753
1754 while (big_val_cnt > 0)
1755 {
1756 int tab_num = gr_info->table_select[ireg];
1757 int sfb_cnt = gr_info->region_count[ireg++];
1758 const int16_t *codebook = tabs + tabindex[tab_num];
1759 int linbits = g_linbits[tab_num];
1760 if (linbits)
1761 {
1762 do
1763 {
1764 np = *sfb++ / 2;
1765 pairs_to_decode = MINIMP3_MIN(big_val_cnt, np);
1766 MP3D_HUFF_NEXT_SCF();
1767 do
1768 {
1769 int j, w = 5;
1770 int leaf = codebook[PEEK_BITS(w)];
1771 while (leaf < 0)
1772 {
1773 FLUSH_BITS(w);
1774 w = leaf & 7;
1775 leaf = codebook[PEEK_BITS(w) - (leaf >> 3)];
1776 }
1777 FLUSH_BITS(leaf >> 8);
1778
1779 for (j = 0; j < 2; j++, dst++, leaf >>= 4)
1780 {
1781 int lsb = leaf & 0x0F;
1782 if (lsb == 15)
1783 {
1784 lsb += PEEK_BITS(linbits);
1785 FLUSH_BITS(linbits);
1786 CHECK_BITS;
1787 *dst = MP3D_HUFF_ESC(lsb, (int32_t)bs_cache < 0);
1788 } else
1789 {
1790 *dst = MP3D_HUFF_TAB(16 + lsb - 16*(bs_cache >> 31));
1791 }
1792 FLUSH_BITS(lsb ? 1 : 0);
1793 }
1794 CHECK_BITS;
1795 } while (--pairs_to_decode);
1796 } while ((big_val_cnt -= np) > 0 && --sfb_cnt >= 0);
1797 } else
1798 {
1799 do
1800 {
1801 np = *sfb++ / 2;
1802 pairs_to_decode = MINIMP3_MIN(big_val_cnt, np);
1803 MP3D_HUFF_NEXT_SCF();
1804 do
1805 {
1806 int j, w = 5;
1807 int leaf = codebook[PEEK_BITS(w)];
1808 while (leaf < 0)
1809 {
1810 FLUSH_BITS(w);
1811 w = leaf & 7;
1812 leaf = codebook[PEEK_BITS(w) - (leaf >> 3)];
1813 }
1814 FLUSH_BITS(leaf >> 8);
1815
1816 for (j = 0; j < 2; j++, dst++, leaf >>= 4)
1817 {
1818 int lsb = leaf & 0x0F;
1819 *dst = MP3D_HUFF_TAB(16 + lsb - 16*(bs_cache >> 31));
1820 FLUSH_BITS(lsb ? 1 : 0);
1821 }
1822 CHECK_BITS;
1823 } while (--pairs_to_decode);
1824 } while ((big_val_cnt -= np) > 0 && --sfb_cnt >= 0);
1825 }
1826 }
1827
1828 for (np = 1 - big_val_cnt;; dst += 4)
1829 {
1830 const uint8_t *codebook_count1 = (gr_info->count1_table) ? tab33 : tab32;
1831 int leaf = codebook_count1[PEEK_BITS(4)];
1832 if (!(leaf & 8))
1833 {
1834 leaf = codebook_count1[(leaf >> 3) + (bs_cache << 4 >> (32 - (leaf & 3)))];
1835 }
1836 FLUSH_BITS(leaf & 7);
1837 if (BSPOS > layer3gr_limit)
1838 {
1839 break;
1840 }
1841#define RELOAD_SCALEFACTOR if (!--np) { np = *sfb++/2; if (!np) break; MP3D_HUFF_NEXT_SCF(); }
1842#define DEQ_COUNT1(s) if (leaf & (128 >> s)) { dst[s] = MP3D_HUFF_ONE((int32_t)bs_cache < 0); FLUSH_BITS(1) }
1843 RELOAD_SCALEFACTOR;
1844 DEQ_COUNT1(0);
1845 DEQ_COUNT1(1);
1846 RELOAD_SCALEFACTOR;
1847 DEQ_COUNT1(2);
1848 DEQ_COUNT1(3);
1849 CHECK_BITS;
1850 }
1851
1852 bs->pos = layer3gr_limit;
1853}
1854
1855static void L3_midside_stereo(mp3d_dsp_t *left, int n) FL_NO_EXCEPT
1856{
1857 int i = 0;
1858 mp3d_dsp_t *right = left + 576;
1859#if HAVE_SIMD
1860 if (have_simd())
1861 {
1862 for (; i < n - 3; i += 4)
1863 {
1864 f4 vl = VLD(left + i);
1865 f4 vr = VLD(right + i);
1866 VSTORE(left + i, VADD(vl, vr));
1867 VSTORE(right + i, VSUB(vl, vr));
1868 }
1869#ifdef __GNUC__
1870 /* Workaround for spurious -Waggressive-loop-optimizations warning from gcc.
1871 * For more info see: https://github.com/lieff/minimp3/issues/88
1872 */
1873 if (__builtin_constant_p(n % 4 == 0) && n % 4 == 0)
1874 return;
1875#endif
1876 }
1877#endif /* HAVE_SIMD */
1878 for (; i < n; i++)
1879 {
1880#if MINIMP3_HAVE_FIXED_POINT
1881 /* Saturating rather than wrapping: mid/side sums can exceed the
1882 clamped input range on a hostile stream, and a wrap would be both
1883 signed-overflow UB and an audible discontinuity. */
1884 const int32_t a = left[i];
1885 const int32_t b = right[i];
1886 left[i] = mp3d_add_sat(a, b);
1887 right[i] = mp3d_sub_sat(a, b);
1888#else
1889 float a = left[i];
1890 float b = right[i];
1891 left[i] = a + b;
1892 right[i] = a - b;
1893#endif
1894 }
1895}
1896
1897/* Intensity-stereo gains are Q30 in the fixed build: they reach sqrt(2) after
1898 the mid/side rescale, so Q30 is the widest format that still holds them. */
1899#if MINIMP3_HAVE_FIXED_POINT
1900typedef int32_t mp3d_coef_t;
1901#else
1902typedef float mp3d_coef_t;
1903#endif
1904
1905static void L3_intensity_stereo_band(mp3d_dsp_t *left, int n, mp3d_coef_t kl, mp3d_coef_t kr) FL_NO_EXCEPT
1906{
1907 int i;
1908 for (i = 0; i < n; i++)
1909 {
1910#if MINIMP3_HAVE_FIXED_POINT
1911 left[i + 576] = mp3d_mulshift<30>(left[i], kr);
1912 left[i] = mp3d_mulshift<30>(left[i], kl);
1913#else
1914 left[i + 576] = left[i]*kr;
1915 left[i] = left[i]*kl;
1916#endif
1917 }
1918}
1919
1920static void L3_stereo_top_band(const mp3d_dsp_t *right, const uint8_t *sfb, int nbands, int max_band[3]) FL_NO_EXCEPT // ok array parameter
1921{
1922 int i, k;
1923
1924 max_band[0] = max_band[1] = max_band[2] = -1;
1925
1926 for (i = 0; i < nbands; i++)
1927 {
1928 for (k = 0; k < sfb[i]; k += 2)
1929 {
1930 if (right[k] != 0 || right[k + 1] != 0)
1931 {
1932 max_band[i % 3] = i;
1933 break;
1934 }
1935 }
1936 right += sfb[i];
1937 }
1938}
1939
1940#if MINIMP3_HAVE_FIXED_POINT
1941/* MPEG-2 intensity gain 2**(-quarters/4), as a Q30 coefficient. The exponent
1942 is always <= 0 here, so this is the same quarter-power split the scalefactor
1943 gains use, collapsed by a right shift instead of carried. */
1944static int32_t mp3d_ist_gain_q30(int quarters) FL_NO_EXCEPT
1945{
1946 int32_t mant;
1947 int exp;
1948 mp3d_gain_from_quarters(-quarters, &mant, &exp);
1949 return mp3d_shr_round(mant, -exp);
1950}
1951#endif
1952
1953static void L3_stereo_process(mp3d_dsp_t *left, const uint8_t *ist_pos, const uint8_t *sfb, const uint8_t *hdr, int max_band[3], int mpeg2_sh) FL_NO_EXCEPT // ok array parameter
1954{
1955#if !MINIMP3_HAVE_FIXED_POINT
1956 static const float g_pan[7*2] = { 0,1,0.21132487f,0.78867513f,0.36602540f,0.63397460f,0.5f,0.5f,0.63397460f,0.36602540f,0.78867513f,0.21132487f,1,0 };
1957#endif
1958 unsigned i, max_pos = HDR_TEST_MPEG1(hdr) ? 7 : 64;
1959
1960 for (i = 0; sfb[i]; i++)
1961 {
1962 unsigned ipos = ist_pos[i];
1963 if ((int)i > max_band[i % 3] && ipos < max_pos)
1964 {
1965#if MINIMP3_HAVE_FIXED_POINT
1966 int32_t kl, kr;
1967 const int32_t s = HDR_TEST_MS_STEREO(hdr) ? MP3D_Q30_SQRT2
1968 : ((int32_t)1 << 30);
1969 if (HDR_TEST_MPEG1(hdr))
1970 {
1971 kl = g_pan_q30[2*ipos];
1972 kr = g_pan_q30[2*ipos + 1];
1973 } else
1974 {
1975 kl = (int32_t)1 << 30;
1976 kr = mp3d_ist_gain_q30((int)((ipos + 1) >> 1) << mpeg2_sh);
1977 if (ipos & 1)
1978 {
1979 kl = kr;
1980 kr = (int32_t)1 << 30;
1981 }
1982 }
1983 L3_intensity_stereo_band(left, sfb[i], mp3d_mulshift<30>(kl, s),
1984 mp3d_mulshift<30>(kr, s));
1985#else
1986 float kl, kr, s = HDR_TEST_MS_STEREO(hdr) ? 1.41421356f : 1;
1987 if (HDR_TEST_MPEG1(hdr))
1988 {
1989 kl = g_pan[2*ipos];
1990 kr = g_pan[2*ipos + 1];
1991 } else
1992 {
1993 kl = 1;
1994 kr = L3_ldexp_q2(1, (ipos + 1) >> 1 << mpeg2_sh);
1995 if (ipos & 1)
1996 {
1997 kl = kr;
1998 kr = 1;
1999 }
2000 }
2001 L3_intensity_stereo_band(left, sfb[i], kl*s, kr*s);
2002#endif
2003 } else if (HDR_TEST_MS_STEREO(hdr))
2004 {
2005 L3_midside_stereo(left, sfb[i]);
2006 }
2007 left += sfb[i];
2008 }
2009}
2010
2011static void L3_intensity_stereo(mp3d_dsp_t *left, uint8_t *ist_pos, const L3_gr_info_t *gr, const uint8_t *hdr) FL_NO_EXCEPT
2012{
2013 int max_band[3], n_sfb = gr->n_long_sfb + gr->n_short_sfb;
2014 int i, max_blocks = gr->n_short_sfb ? 3 : 1;
2015
2016 L3_stereo_top_band(left + 576, gr->sfbtab, n_sfb, max_band);
2017 if (gr->n_long_sfb)
2018 {
2019 max_band[0] = max_band[1] = max_band[2] = MINIMP3_MAX(MINIMP3_MAX(max_band[0], max_band[1]), max_band[2]);
2020 }
2021 for (i = 0; i < max_blocks; i++)
2022 {
2023 int default_pos = HDR_TEST_MPEG1(hdr) ? 3 : 0;
2024 int itop = n_sfb - max_blocks + i;
2025 int prev = itop - max_blocks;
2026 ist_pos[itop] = max_band[i] >= prev ? default_pos : ist_pos[prev];
2027 }
2028 L3_stereo_process(left, ist_pos, gr->sfbtab, hdr, max_band, gr[1].scalefac_compress & 1);
2029}
2030
2031static void L3_reorder(mp3d_dsp_t *grbuf, mp3d_dsp_t *scratch, const uint8_t *sfb) FL_NO_EXCEPT
2032{
2033 int i, len;
2034 mp3d_dsp_t *src = grbuf, *dst = scratch;
2035
2036 for (;0 != (len = *sfb); sfb += 3, src += 2*len)
2037 {
2038 for (i = 0; i < len; i++, src++)
2039 {
2040 *dst++ = src[0*len];
2041 *dst++ = src[1*len];
2042 *dst++ = src[2*len];
2043 }
2044 }
2045 memcpy(grbuf, scratch, (dst - scratch)*sizeof(mp3d_dsp_t));
2046}
2047
2048static void L3_antialias(mp3d_dsp_t *grbuf, int nbands) FL_NO_EXCEPT
2049{
2050#if !MINIMP3_HAVE_FIXED_POINT
2051 static const float g_aa[2][8] = {
2052 {0.85749293f,0.88174200f,0.94962865f,0.98331459f,0.99551782f,0.99916056f,0.99989920f,0.99999316f},
2053 {0.51449576f,0.47173197f,0.31337745f,0.18191320f,0.09457419f,0.04096558f,0.01419856f,0.00369997f}
2054 };
2055#endif
2056
2057 for (; nbands > 0; nbands--, grbuf += 18)
2058 {
2059 int i = 0;
2060#if HAVE_SIMD
2061 if (have_simd()) for (; i < 8; i += 4)
2062 {
2063 f4 vu = VLD(grbuf + 18 + i);
2064 f4 vd = VLD(grbuf + 14 - i);
2065 f4 vc0 = VLD(g_aa[0] + i);
2066 f4 vc1 = VLD(g_aa[1] + i);
2067 vd = VREV(vd);
2068 VSTORE(grbuf + 18 + i, VSUB(VMUL(vu, vc0), VMUL(vd, vc1)));
2069 vd = VADD(VMUL(vu, vc1), VMUL(vd, vc0));
2070 VSTORE(grbuf + 14 - i, VREV(vd));
2071 }
2072#endif /* HAVE_SIMD */
2073#ifndef MINIMP3_ONLY_SIMD
2074 for(; i < 8; i++)
2075 {
2076#if MINIMP3_HAVE_FIXED_POINT
2077 const int32_t u = grbuf[18 + i];
2078 const int32_t d = grbuf[17 - i];
2079 grbuf[18 + i] = mp3d_sub_sat(mp3d_mulshift<31>(u, g_aa_cs_q31[i]),
2080 mp3d_mulshift<31>(d, g_aa_ca_q31[i]));
2081 grbuf[17 - i] = mp3d_add_sat(mp3d_mulshift<31>(u, g_aa_ca_q31[i]),
2082 mp3d_mulshift<31>(d, g_aa_cs_q31[i]));
2083#else
2084 float u = grbuf[18 + i];
2085 float d = grbuf[17 - i];
2086 grbuf[18 + i] = u*g_aa[0][i] - d*g_aa[1][i];
2087 grbuf[17 - i] = u*g_aa[1][i] + d*g_aa[0][i];
2088#endif
2089 }
2090#endif /* MINIMP3_ONLY_SIMD */
2091 }
2092}
2093
2094#if MINIMP3_HAVE_FIXED_POINT
2095/* 9-point DCT-III. Same butterfly graph as the float build, with every
2096 constant in Q31 and every add saturating. The two halvings are arithmetic
2097 right shifts with the pipeline's rounding rule rather than a multiply by a
2098 Q31 0.5, which would cost a 64-bit multiply for an exact power of two. */
2099static void L3_dct3_9(int32_t *y) FL_NO_EXCEPT
2100{
2101 int32_t s0, s1, s2, s3, s4, s5, s6, s7, s8, t0, t2, t4;
2102
2103 s0 = y[0]; s2 = y[2]; s4 = y[4]; s6 = y[6]; s8 = y[8];
2104 t0 = mp3d_add_sat(s0, mp3d_shr_round(s6, 1));
2105 s0 = mp3d_sub_sat(s0, s6);
2106 t4 = mp3d_mulshift_k<31, MP3D_Q31_COS_PI_9>(mp3d_add_sat(s4, s2));
2107 t2 = mp3d_mulshift_k<31, MP3D_Q31_COS_2PI_9>(mp3d_add_sat(s8, s2));
2108 s6 = mp3d_mulshift_k<31, MP3D_Q31_COS_4PI_9>(mp3d_sub_sat(s4, s8));
2109 s4 = mp3d_add_sat(s4, mp3d_sub_sat(s8, s2));
2110
2111 s2 = mp3d_sub_sat(s0, mp3d_shr_round(s4, 1));
2112 y[4] = mp3d_add_sat(s4, s0);
2113 s8 = mp3d_add_sat(mp3d_sub_sat(t0, t2), s6);
2114 s0 = mp3d_add_sat(mp3d_sub_sat(t0, t4), t2);
2115 s4 = mp3d_sub_sat(mp3d_add_sat(t0, t4), s6);
2116
2117 s1 = y[1]; s3 = y[3]; s5 = y[5]; s7 = y[7];
2118
2119 s3 = mp3d_mulshift_k<31, MP3D_Q31_COS_PI_6>(s3);
2120 t0 = mp3d_mulshift_k<31, MP3D_Q31_COS_PI_18>(mp3d_add_sat(s5, s1));
2121 t4 = mp3d_mulshift_k<31, MP3D_Q31_COS_7PI_18>(mp3d_sub_sat(s5, s7));
2122 t2 = mp3d_mulshift_k<31, MP3D_Q31_COS_5PI_18>(mp3d_add_sat(s1, s7));
2123 s1 = mp3d_mulshift_k<31, MP3D_Q31_COS_PI_6>(mp3d_sub_sat(mp3d_sub_sat(s1, s5), s7));
2124
2125 s5 = mp3d_sub_sat(mp3d_sub_sat(t0, s3), t2);
2126 s7 = mp3d_sub_sat(mp3d_sub_sat(t4, s3), t0);
2127 s3 = mp3d_sub_sat(mp3d_add_sat(t4, s3), t2);
2128
2129 y[0] = mp3d_sub_sat(s4, s7);
2130 y[1] = mp3d_add_sat(s2, s1);
2131 y[2] = mp3d_sub_sat(s0, s3);
2132 y[3] = mp3d_add_sat(s8, s5);
2133 y[5] = mp3d_sub_sat(s8, s5);
2134 y[6] = mp3d_add_sat(s0, s3);
2135 y[7] = mp3d_sub_sat(s2, s1);
2136 y[8] = mp3d_add_sat(s4, s7);
2137}
2138
2139/* The closing twiddle-and-window loop of L3_imdct36, factored out so it can be
2140 defined below the SIMD primitives it dispatches to while its caller stays
2141 here (#4109). Plain and unattributed on purpose: the vector path and the
2142 runtime capability check live inside the definition, so this declaration
2143 needs none of the SIMD macros, which this point in the file precedes. */
2144static void mp3d_imdct36_twiddle(int32_t *grbuf, int32_t *overlap,
2145 const int32_t *window, const int32_t *co,
2146 const int32_t *si) FL_NO_EXCEPT;
2147
2148/* -O3 for the same reason as mp3d_DCT_II above; the two were measured
2149 together. */
2150MP3D_KERNEL static void L3_imdct36(int32_t *grbuf, int32_t *overlap, const int32_t *window, int nbands) FL_NO_EXCEPT
2151{
2152 int i, j;
2153
2154 for (j = 0; j < nbands; j++, grbuf += 18, overlap += 9)
2155 {
2156 int32_t co[9], si[9];
2157 co[0] = -grbuf[0];
2158 si[0] = grbuf[17];
2159 for (i = 0; i < 4; i++)
2160 {
2161 si[8 - 2*i] = mp3d_sub_sat(grbuf[4*i + 1], grbuf[4*i + 2]);
2162 co[1 + 2*i] = mp3d_add_sat(grbuf[4*i + 1], grbuf[4*i + 2]);
2163 si[7 - 2*i] = mp3d_sub_sat(grbuf[4*i + 4], grbuf[4*i + 3]);
2164 co[2 + 2*i] = -mp3d_add_sat(grbuf[4*i + 3], grbuf[4*i + 4]);
2165 }
2166 L3_dct3_9(co);
2167 L3_dct3_9(si);
2168
2169 si[1] = -si[1];
2170 si[3] = -si[3];
2171 si[5] = -si[5];
2172 si[7] = -si[7];
2173
2174 /* The twiddle and window products accumulate in int64 before a single
2175 rounded narrow, so each output carries one rounding step rather than
2176 two. That matters: this pair of multiply-accumulates runs 18 times
2177 per band per granule, and per-term rounding would bias the overlap
2178 state, which then feeds the next frame. */
2179 mp3d_imdct36_twiddle(grbuf, overlap, window, co, si);
2180 }
2181}
2182
2183static void L3_idct3(int32_t x0, int32_t x1, int32_t x2, int32_t *dst) FL_NO_EXCEPT
2184{
2185 const int32_t m1 = mp3d_mulshift_k<31, MP3D_Q31_COS_PI_6>(x1);
2186 const int32_t a1 = mp3d_sub_sat(x0, mp3d_shr_round(x2, 1));
2187 dst[1] = mp3d_add_sat(x0, x2);
2188 dst[0] = mp3d_add_sat(a1, m1);
2189 dst[2] = mp3d_sub_sat(a1, m1);
2190}
2191
2192static void L3_imdct12(int32_t *x, int32_t *dst, int32_t *overlap) FL_NO_EXCEPT
2193{
2194 int32_t co[3], si[3];
2195 int i;
2196
2197 L3_idct3(-x[0], mp3d_add_sat(x[6], x[3]), mp3d_add_sat(x[12], x[9]), co);
2198 L3_idct3(x[15], mp3d_sub_sat(x[12], x[9]), mp3d_sub_sat(x[6], x[3]), si);
2199 si[1] = -si[1];
2200
2201 for (i = 0; i < 3; i++)
2202 {
2203 const int32_t ovl = overlap[i];
2204 const int32_t sum = mp3d_narrow_q30(
2205 (int64_t)co[i]*g_twid3_q30[3 + i] +
2206 (int64_t)si[i]*g_twid3_q30[0 + i]);
2207 overlap[i] = mp3d_narrow_q30(
2208 (int64_t)co[i]*g_twid3_q30[0 + i] -
2209 (int64_t)si[i]*g_twid3_q30[3 + i]);
2210 dst[i] = mp3d_narrow_q30(
2211 (int64_t)ovl*g_twid3_q30[2 - i] - (int64_t)sum*g_twid3_q30[5 - i]);
2212 dst[5 - i] = mp3d_narrow_q30(
2213 (int64_t)ovl*g_twid3_q30[5 - i] + (int64_t)sum*g_twid3_q30[2 - i]);
2214 }
2215}
2216
2217static void L3_imdct_short(int32_t *grbuf, int32_t *overlap, int nbands) FL_NO_EXCEPT
2218{
2219 for (;nbands > 0; nbands--, overlap += 9, grbuf += 18)
2220 {
2221 int32_t tmp[18];
2222 memcpy(tmp, grbuf, sizeof(tmp));
2223 memcpy(grbuf, overlap, 6*sizeof(int32_t));
2224 L3_imdct12(tmp, grbuf + 6, overlap + 6);
2225 L3_imdct12(tmp + 1, grbuf + 12, overlap + 6);
2226 L3_imdct12(tmp + 2, overlap, overlap + 6);
2227 }
2228}
2229
2230static void L3_change_sign(int32_t *grbuf) FL_NO_EXCEPT
2231{
2232 int b, i;
2233 for (b = 0, grbuf += 18; b < 32; b += 2, grbuf += 36)
2234 for (i = 1; i < 18; i += 2)
2235 grbuf[i] = -grbuf[i];
2236}
2237
2238static void L3_imdct_gr(int32_t *grbuf, int32_t *overlap, unsigned block_type, unsigned n_long_bands) FL_NO_EXCEPT
2239{
2240 /* Selected with a conditional rather than through a two-entry pointer
2241 table like the float build uses. The table would be a relocated array,
2242 which costs a .rel entry and load-time relocation work on exactly the
2243 embedded targets this path exists for. */
2244 if (n_long_bands)
2245 {
2246 L3_imdct36(grbuf, overlap, g_mdct_window_normal_q30, n_long_bands);
2247 grbuf += 18*n_long_bands;
2248 overlap += 9*n_long_bands;
2249 }
2250 if (block_type == SHORT_BLOCK_TYPE)
2251 L3_imdct_short(grbuf, overlap, 32 - n_long_bands);
2252 else
2253 L3_imdct36(grbuf, overlap,
2254 block_type == STOP_BLOCK_TYPE ? g_mdct_window_stop_q30
2256 32 - n_long_bands);
2257}
2258#else
2259static void L3_dct3_9(float *y) FL_NO_EXCEPT
2260{
2261 float s0, s1, s2, s3, s4, s5, s6, s7, s8, t0, t2, t4;
2262
2263 s0 = y[0]; s2 = y[2]; s4 = y[4]; s6 = y[6]; s8 = y[8];
2264 t0 = s0 + s6*0.5f;
2265 s0 -= s6;
2266 t4 = (s4 + s2)*0.93969262f;
2267 t2 = (s8 + s2)*0.76604444f;
2268 s6 = (s4 - s8)*0.17364818f;
2269 s4 += s8 - s2;
2270
2271 s2 = s0 - s4*0.5f;
2272 y[4] = s4 + s0;
2273 s8 = t0 - t2 + s6;
2274 s0 = t0 - t4 + t2;
2275 s4 = t0 + t4 - s6;
2276
2277 s1 = y[1]; s3 = y[3]; s5 = y[5]; s7 = y[7];
2278
2279 s3 *= 0.86602540f;
2280 t0 = (s5 + s1)*0.98480775f;
2281 t4 = (s5 - s7)*0.34202014f;
2282 t2 = (s1 + s7)*0.64278761f;
2283 s1 = (s1 - s5 - s7)*0.86602540f;
2284
2285 s5 = t0 - s3 - t2;
2286 s7 = t4 - s3 - t0;
2287 s3 = t4 + s3 - t2;
2288
2289 y[0] = s4 - s7;
2290 y[1] = s2 + s1;
2291 y[2] = s0 - s3;
2292 y[3] = s8 + s5;
2293 y[5] = s8 - s5;
2294 y[6] = s0 + s3;
2295 y[7] = s2 - s1;
2296 y[8] = s4 + s7;
2297}
2298
2299static void L3_imdct36(float *grbuf, float *overlap, const float *window, int nbands) FL_NO_EXCEPT
2300{
2301 int i, j;
2302 static const float g_twid9[18] = {
2303 0.73727734f,0.79335334f,0.84339145f,0.88701083f,0.92387953f,0.95371695f,0.97629601f,0.99144486f,0.99904822f,0.67559021f,0.60876143f,0.53729961f,0.46174861f,0.38268343f,0.30070580f,0.21643961f,0.13052619f,0.04361938f
2304 };
2305
2306 for (j = 0; j < nbands; j++, grbuf += 18, overlap += 9)
2307 {
2308 float co[9], si[9];
2309 co[0] = -grbuf[0];
2310 si[0] = grbuf[17];
2311 for (i = 0; i < 4; i++)
2312 {
2313 si[8 - 2*i] = grbuf[4*i + 1] - grbuf[4*i + 2];
2314 co[1 + 2*i] = grbuf[4*i + 1] + grbuf[4*i + 2];
2315 si[7 - 2*i] = grbuf[4*i + 4] - grbuf[4*i + 3];
2316 co[2 + 2*i] = -(grbuf[4*i + 3] + grbuf[4*i + 4]);
2317 }
2318 L3_dct3_9(co);
2319 L3_dct3_9(si);
2320
2321 si[1] = -si[1];
2322 si[3] = -si[3];
2323 si[5] = -si[5];
2324 si[7] = -si[7];
2325
2326 i = 0;
2327
2328#if HAVE_SIMD
2329 if (have_simd()) for (; i < 8; i += 4)
2330 {
2331 f4 vovl = VLD(overlap + i);
2332 f4 vc = VLD(co + i);
2333 f4 vs = VLD(si + i);
2334 f4 vr0 = VLD(g_twid9 + i);
2335 f4 vr1 = VLD(g_twid9 + 9 + i);
2336 f4 vw0 = VLD(window + i);
2337 f4 vw1 = VLD(window + 9 + i);
2338 f4 vsum = VADD(VMUL(vc, vr1), VMUL(vs, vr0));
2339 VSTORE(overlap + i, VSUB(VMUL(vc, vr0), VMUL(vs, vr1)));
2340 VSTORE(grbuf + i, VSUB(VMUL(vovl, vw0), VMUL(vsum, vw1)));
2341 vsum = VADD(VMUL(vovl, vw1), VMUL(vsum, vw0));
2342 VSTORE(grbuf + 14 - i, VREV(vsum));
2343 }
2344#endif /* HAVE_SIMD */
2345 for (; i < 9; i++)
2346 {
2347 float ovl = overlap[i];
2348 float sum = co[i]*g_twid9[9 + i] + si[i]*g_twid9[0 + i];
2349 overlap[i] = co[i]*g_twid9[0 + i] - si[i]*g_twid9[9 + i];
2350 grbuf[i] = ovl*window[0 + i] - sum*window[9 + i];
2351 grbuf[17 - i] = ovl*window[9 + i] + sum*window[0 + i];
2352 }
2353 }
2354}
2355
2356static void L3_idct3(float x0, float x1, float x2, float *dst) FL_NO_EXCEPT
2357{
2358 float m1 = x1*0.86602540f;
2359 float a1 = x0 - x2*0.5f;
2360 dst[1] = x0 + x2;
2361 dst[0] = a1 + m1;
2362 dst[2] = a1 - m1;
2363}
2364
2365static void L3_imdct12(float *x, float *dst, float *overlap) FL_NO_EXCEPT
2366{
2367 static const float g_twid3[6] = { 0.79335334f,0.92387953f,0.99144486f, 0.60876143f,0.38268343f,0.13052619f };
2368 float co[3], si[3];
2369 int i;
2370
2371 L3_idct3(-x[0], x[6] + x[3], x[12] + x[9], co);
2372 L3_idct3(x[15], x[12] - x[9], x[6] - x[3], si);
2373 si[1] = -si[1];
2374
2375 for (i = 0; i < 3; i++)
2376 {
2377 float ovl = overlap[i];
2378 float sum = co[i]*g_twid3[3 + i] + si[i]*g_twid3[0 + i];
2379 overlap[i] = co[i]*g_twid3[0 + i] - si[i]*g_twid3[3 + i];
2380 dst[i] = ovl*g_twid3[2 - i] - sum*g_twid3[5 - i];
2381 dst[5 - i] = ovl*g_twid3[5 - i] + sum*g_twid3[2 - i];
2382 }
2383}
2384
2385static void L3_imdct_short(float *grbuf, float *overlap, int nbands) FL_NO_EXCEPT
2386{
2387 for (;nbands > 0; nbands--, overlap += 9, grbuf += 18)
2388 {
2389 float tmp[18];
2390 memcpy(tmp, grbuf, sizeof(tmp));
2391 memcpy(grbuf, overlap, 6*sizeof(float));
2392 L3_imdct12(tmp, grbuf + 6, overlap + 6);
2393 L3_imdct12(tmp + 1, grbuf + 12, overlap + 6);
2394 L3_imdct12(tmp + 2, overlap, overlap + 6);
2395 }
2396}
2397
2398static void L3_change_sign(float *grbuf) FL_NO_EXCEPT
2399{
2400 int b, i;
2401 for (b = 0, grbuf += 18; b < 32; b += 2, grbuf += 36)
2402 for (i = 1; i < 18; i += 2)
2403 grbuf[i] = -grbuf[i];
2404}
2405
2406static void L3_imdct_gr(float *grbuf, float *overlap, unsigned block_type, unsigned n_long_bands) FL_NO_EXCEPT
2407{
2408 static const float g_mdct_window[2][18] = {
2409 { 0.99904822f,0.99144486f,0.97629601f,0.95371695f,0.92387953f,0.88701083f,0.84339145f,0.79335334f,0.73727734f,0.04361938f,0.13052619f,0.21643961f,0.30070580f,0.38268343f,0.46174861f,0.53729961f,0.60876143f,0.67559021f },
2410 { 1,1,1,1,1,1,0.99144486f,0.92387953f,0.79335334f,0,0,0,0,0,0,0.13052619f,0.38268343f,0.60876143f }
2411 };
2412 if (n_long_bands)
2413 {
2414 L3_imdct36(grbuf, overlap, g_mdct_window[0], n_long_bands);
2415 grbuf += 18*n_long_bands;
2416 overlap += 9*n_long_bands;
2417 }
2418 if (block_type == SHORT_BLOCK_TYPE)
2419 L3_imdct_short(grbuf, overlap, 32 - n_long_bands);
2420 else
2421 L3_imdct36(grbuf, overlap, g_mdct_window[block_type == STOP_BLOCK_TYPE], 32 - n_long_bands);
2422}
2423#endif /* MINIMP3_HAVE_FIXED_POINT */
2424
2425
2426static void L3_save_reservoir(mp3dec_t *h, mp3dec_scratch_internal_t *s) FL_NO_EXCEPT
2427{
2428 int pos = (s->bs.pos + 7)/8u;
2429 int remains = s->bs.limit/8u - pos;
2430 if (remains > MAX_BITRESERVOIR_BYTES)
2431 {
2432 pos += remains - MAX_BITRESERVOIR_BYTES;
2433 remains = MAX_BITRESERVOIR_BYTES;
2434 }
2435 if (remains > 0)
2436 {
2437 memmove(h->reserv_buf, s->u.maindata + pos, remains);
2438 }
2439 h->reserv = remains;
2440}
2441
2442static int L3_restore_reservoir(mp3dec_t *h, bs_t *bs, mp3dec_scratch_internal_t *s, int main_data_begin) FL_NO_EXCEPT
2443{
2444 int frame_bytes = (bs->limit - bs->pos)/8;
2445 int bytes_have = MINIMP3_MIN(h->reserv, main_data_begin);
2446 memcpy(s->u.maindata, h->reserv_buf + MINIMP3_MAX(0, h->reserv - main_data_begin), MINIMP3_MIN(h->reserv, main_data_begin));
2447 memcpy(s->u.maindata + bytes_have, bs->buf + bs->pos/8, frame_bytes);
2448 bs_init(&s->bs, s->u.maindata, bytes_have + frame_bytes);
2449 return h->reserv >= main_data_begin;
2450}
2451
2452/* The scalefactor gains reach the dequantiser as one array in the float build
2453 and as a mantissa/exponent pair in the fixed build; this keeps L3_decode
2454 itself identical between the two. */
2455#if MINIMP3_HAVE_FIXED_POINT
2456#define MP3D_SCF_ARGS(s) (s)->scf_mant, (s)->scf_exp
2457#else
2458#define MP3D_SCF_ARGS(s) (s)->scf
2459#endif
2460
2461static void L3_decode(mp3dec_t *h, mp3dec_scratch_internal_t *s, L3_gr_info_t *gr_info, int nch) FL_NO_EXCEPT
2462{
2463 int ch;
2464
2465 for (ch = 0; ch < nch; ch++)
2466 {
2467 int layer3gr_limit = s->bs.pos + gr_info[ch].part_23_length;
2468 /* Clear only what L3_huffman will not itself define, and only for the
2469 channels this granule decodes. The granule buffer has three regions:
2470
2471 [0, 2*big_values) the big_values loop stores every sample. Each
2472 band writes 2*min(big_val_cnt, np) and takes np
2473 off the count, so `dst` lands exactly on
2474 grbuf + 2*big_values whether the last band is
2475 whole or partial. Pre-clearing it is dead.
2476 [2*big_values, ..) the count1 loop stores *sparsely* -- DEQ_COUNT1
2477 writes a sample only when its quadruple bit is
2478 set -- so the gaps have to already read zero.
2479 (.., 576) rzero. Never written; must read zero.
2480
2481 So the clear starts at 2*big_values and runs to the end of the
2482 granule. `big_values <= 288` is enforced in L3_read_side_info, so the
2483 length is never negative, and every g_scf_* band width is even and
2484 each table sums to exactly 576, so `*sfb++/2` never truncates and the
2485 big_values accounting is exact.
2486
2487 A mono granule needs no clear of grbuf[1] at all: `mp3d_synth` takes
2488 `xr = xl + 576*(nch - 1)`, so `xr == xl` and every read of it sits
2489 behind `if (nch == 2)`. This loop running `ch < nch` covers that. */
2490 memset(s->grbuf[ch] + 2*gr_info[ch].big_values, 0,
2491 (576 - 2*(int)gr_info[ch].big_values)*sizeof(mp3d_dsp_t));
2492 L3_decode_scalefactors(h->header, s->ist_pos[ch], &s->bs, gr_info + ch, MP3D_SCF_ARGS(s), ch);
2493 L3_huffman(s->grbuf[ch], &s->bs, gr_info + ch, MP3D_SCF_ARGS(s), layer3gr_limit);
2494 MP3D_STAGE(MINIMP3_STAGE_HUFFMAN, ch, s->grbuf[ch], 576);
2495 }
2496
2497 if (HDR_TEST_I_STEREO(h->header))
2498 {
2499 L3_intensity_stereo(s->grbuf[0], s->ist_pos[1], gr_info, h->header);
2500 } else if (HDR_IS_MS_STEREO(h->header))
2501 {
2502 L3_midside_stereo(s->grbuf[0], 576);
2503 }
2504 MP3D_STAGE(MINIMP3_STAGE_STEREO, 0, s->grbuf[0], 576*nch);
2505
2506 for (ch = 0; ch < nch; ch++, gr_info++)
2507 {
2508 int aa_bands = 31;
2509 int n_long_bands = (gr_info->mixed_block_flag ? 2 : 0) << (int)(HDR_GET_MY_SAMPLE_RATE(h->header) == 2);
2510
2511 if (gr_info->n_short_sfb)
2512 {
2513 aa_bands = n_long_bands - 1;
2514 L3_reorder(s->grbuf[ch] + n_long_bands*18,
2515 h->qmf_state + 15*2*32,
2516 gr_info->sfbtab + gr_info->n_long_sfb);
2517 }
2518
2519 L3_antialias(s->grbuf[ch], aa_bands);
2520 MP3D_STAGE(MINIMP3_STAGE_ANTIALIAS, ch, s->grbuf[ch], 576);
2521 L3_imdct_gr(s->grbuf[ch], h->mdct_overlap[ch], gr_info->block_type, n_long_bands);
2522 L3_change_sign(s->grbuf[ch]);
2523 MP3D_STAGE(MINIMP3_STAGE_IMDCT, ch, s->grbuf[ch], 576);
2524 }
2525}
2526
2527#if MINIMP3_HAVE_FIXED_POINT
2528/* The fixed-point synthesis back-end -- DCT-32, polyphase and the integer
2529 SIMD kernels -- lives in its own file. It is 57% of a decode and is where
2530 optimisation work happens; see the header comment there. */
2531#include "minimp3_synth_fixed.h"
2532#else
2533static void mp3d_DCT_II(float *grbuf, int n) FL_NO_EXCEPT
2534{
2535 static const float g_sec[24] = {
2536 10.19000816f,0.50060302f,0.50241929f,3.40760851f,0.50547093f,0.52249861f,2.05778098f,0.51544732f,0.56694406f,1.48416460f,0.53104258f,0.64682180f,1.16943991f,0.55310392f,0.78815460f,0.97256821f,0.58293498f,1.06067765f,0.83934963f,0.62250412f,1.72244716f,0.74453628f,0.67480832f,5.10114861f
2537 };
2538 int i, k = 0;
2539#if HAVE_SIMD
2540 if (have_simd()) for (; k < n; k += 4)
2541 {
2542 f4 t[4][8], *x;
2543 float *y = grbuf + k;
2544
2545 for (x = t[0], i = 0; i < 8; i++, x++)
2546 {
2547 f4 x0 = VLD(&y[i*18]);
2548 f4 x1 = VLD(&y[(15 - i)*18]);
2549 f4 x2 = VLD(&y[(16 + i)*18]);
2550 f4 x3 = VLD(&y[(31 - i)*18]);
2551 f4 t0 = VADD(x0, x3);
2552 f4 t1 = VADD(x1, x2);
2553 f4 t2 = VMUL_S(VSUB(x1, x2), g_sec[3*i + 0]);
2554 f4 t3 = VMUL_S(VSUB(x0, x3), g_sec[3*i + 1]);
2555 x[0] = VADD(t0, t1);
2556 x[8] = VMUL_S(VSUB(t0, t1), g_sec[3*i + 2]);
2557 x[16] = VADD(t3, t2);
2558 x[24] = VMUL_S(VSUB(t3, t2), g_sec[3*i + 2]);
2559 }
2560 for (x = t[0], i = 0; i < 4; i++, x += 8)
2561 {
2562 f4 x0 = x[0], x1 = x[1], x2 = x[2], x3 = x[3], x4 = x[4], x5 = x[5], x6 = x[6], x7 = x[7], xt;
2563 xt = VSUB(x0, x7); x0 = VADD(x0, x7);
2564 x7 = VSUB(x1, x6); x1 = VADD(x1, x6);
2565 x6 = VSUB(x2, x5); x2 = VADD(x2, x5);
2566 x5 = VSUB(x3, x4); x3 = VADD(x3, x4);
2567 x4 = VSUB(x0, x3); x0 = VADD(x0, x3);
2568 x3 = VSUB(x1, x2); x1 = VADD(x1, x2);
2569 x[0] = VADD(x0, x1);
2570 x[4] = VMUL_S(VSUB(x0, x1), 0.70710677f);
2571 x5 = VADD(x5, x6);
2572 x6 = VMUL_S(VADD(x6, x7), 0.70710677f);
2573 x7 = VADD(x7, xt);
2574 x3 = VMUL_S(VADD(x3, x4), 0.70710677f);
2575 x5 = VSUB(x5, VMUL_S(x7, 0.198912367f)); /* rotate by PI/8 */
2576 x7 = VADD(x7, VMUL_S(x5, 0.382683432f));
2577 x5 = VSUB(x5, VMUL_S(x7, 0.198912367f));
2578 x0 = VSUB(xt, x6); xt = VADD(xt, x6);
2579 x[1] = VMUL_S(VADD(xt, x7), 0.50979561f);
2580 x[2] = VMUL_S(VADD(x4, x3), 0.54119611f);
2581 x[3] = VMUL_S(VSUB(x0, x5), 0.60134488f);
2582 x[5] = VMUL_S(VADD(x0, x5), 0.89997619f);
2583 x[6] = VMUL_S(VSUB(x4, x3), 1.30656302f);
2584 x[7] = VMUL_S(VSUB(xt, x7), 2.56291556f);
2585 }
2586
2587 if (k > n - 3)
2588 {
2589#if HAVE_SSE
2590#define VSAVE2(i, v) _mm_storel_pi((__m64 *)(void*)&y[i*18], v)
2591#else /* HAVE_SSE */
2592#define VSAVE2(i, v) vst1_f32((float32_t *)&y[i*18], vget_low_f32(v))
2593#endif /* HAVE_SSE */
2594 for (i = 0; i < 7; i++, y += 4*18)
2595 {
2596 f4 s = VADD(t[3][i], t[3][i + 1]);
2597 VSAVE2(0, t[0][i]);
2598 VSAVE2(1, VADD(t[2][i], s));
2599 VSAVE2(2, VADD(t[1][i], t[1][i + 1]));
2600 VSAVE2(3, VADD(t[2][1 + i], s));
2601 }
2602 VSAVE2(0, t[0][7]);
2603 VSAVE2(1, VADD(t[2][7], t[3][7]));
2604 VSAVE2(2, t[1][7]);
2605 VSAVE2(3, t[3][7]);
2606 } else
2607 {
2608#define VSAVE4(i, v) VSTORE(&y[i*18], v)
2609 for (i = 0; i < 7; i++, y += 4*18)
2610 {
2611 f4 s = VADD(t[3][i], t[3][i + 1]);
2612 VSAVE4(0, t[0][i]);
2613 VSAVE4(1, VADD(t[2][i], s));
2614 VSAVE4(2, VADD(t[1][i], t[1][i + 1]));
2615 VSAVE4(3, VADD(t[2][1 + i], s));
2616 }
2617 VSAVE4(0, t[0][7]);
2618 VSAVE4(1, VADD(t[2][7], t[3][7]));
2619 VSAVE4(2, t[1][7]);
2620 VSAVE4(3, t[3][7]);
2621 }
2622 } else
2623#endif /* HAVE_SIMD */
2624#ifdef MINIMP3_ONLY_SIMD
2625 {} /* for HAVE_SIMD=1, MINIMP3_ONLY_SIMD=1 case we do not need non-intrinsic "else" branch */
2626#else /* MINIMP3_ONLY_SIMD */
2627 for (; k < n; k++)
2628 {
2629 float t[4][8], *x, *y = grbuf + k;
2630
2631 for (x = t[0], i = 0; i < 8; i++, x++)
2632 {
2633 float x0 = y[i*18];
2634 float x1 = y[(15 - i)*18];
2635 float x2 = y[(16 + i)*18];
2636 float x3 = y[(31 - i)*18];
2637 float t0 = x0 + x3;
2638 float t1 = x1 + x2;
2639 float t2 = (x1 - x2)*g_sec[3*i + 0];
2640 float t3 = (x0 - x3)*g_sec[3*i + 1];
2641 x[0] = t0 + t1;
2642 x[8] = (t0 - t1)*g_sec[3*i + 2];
2643 x[16] = t3 + t2;
2644 x[24] = (t3 - t2)*g_sec[3*i + 2];
2645 }
2646 for (x = t[0], i = 0; i < 4; i++, x += 8)
2647 {
2648 float x0 = x[0], x1 = x[1], x2 = x[2], x3 = x[3], x4 = x[4], x5 = x[5], x6 = x[6], x7 = x[7], xt;
2649 xt = x0 - x7; x0 += x7;
2650 x7 = x1 - x6; x1 += x6;
2651 x6 = x2 - x5; x2 += x5;
2652 x5 = x3 - x4; x3 += x4;
2653 x4 = x0 - x3; x0 += x3;
2654 x3 = x1 - x2; x1 += x2;
2655 x[0] = x0 + x1;
2656 x[4] = (x0 - x1)*0.70710677f;
2657 x5 = x5 + x6;
2658 x6 = (x6 + x7)*0.70710677f;
2659 x7 = x7 + xt;
2660 x3 = (x3 + x4)*0.70710677f;
2661 x5 -= x7*0.198912367f; /* rotate by PI/8 */
2662 x7 += x5*0.382683432f;
2663 x5 -= x7*0.198912367f;
2664 x0 = xt - x6; xt += x6;
2665 x[1] = (xt + x7)*0.50979561f;
2666 x[2] = (x4 + x3)*0.54119611f;
2667 x[3] = (x0 - x5)*0.60134488f;
2668 x[5] = (x0 + x5)*0.89997619f;
2669 x[6] = (x4 - x3)*1.30656302f;
2670 x[7] = (xt - x7)*2.56291556f;
2671
2672 }
2673 for (i = 0; i < 7; i++, y += 4*18)
2674 {
2675 y[0*18] = t[0][i];
2676 y[1*18] = t[2][i] + t[3][i] + t[3][i + 1];
2677 y[2*18] = t[1][i] + t[1][i + 1];
2678 y[3*18] = t[2][i + 1] + t[3][i] + t[3][i + 1];
2679 }
2680 y[0*18] = t[0][7];
2681 y[1*18] = t[2][7] + t[3][7];
2682 y[2*18] = t[1][7];
2683 y[3*18] = t[3][7];
2684 }
2685#endif /* MINIMP3_ONLY_SIMD */
2686}
2687
2688#ifndef MINIMP3_FLOAT_OUTPUT
2689static int16_t mp3d_scale_pcm(float sample) FL_NO_EXCEPT
2690{
2691#if HAVE_ARMV6
2692 int32_t s32 = (int32_t)(sample + .5f);
2693 s32 -= (s32 < 0);
2694 int16_t s = (int16_t)minimp3_clip_int16_arm(s32);
2695#else
2696 if (sample >= 32766.5) return (int16_t) 32767;
2697 if (sample <= -32767.5) return (int16_t)-32768;
2698 int16_t s = (int16_t)(sample + .5f);
2699 s -= (s < 0); /* away from zero, to be compliant */
2700#endif
2701 return s;
2702}
2703#else /* MINIMP3_FLOAT_OUTPUT */
2704static float mp3d_scale_pcm(float sample)
2705{
2706 return sample*(1.f/32768.f);
2707}
2708#endif /* MINIMP3_FLOAT_OUTPUT */
2709
2710static void mp3d_synth_pair(mp3d_sample_t *pcm, int nch, const float *z) FL_NO_EXCEPT
2711{
2712 float a;
2713 a = (z[14*64] - z[ 0]) * 29;
2714 a += (z[ 1*64] + z[13*64]) * 213;
2715 a += (z[12*64] - z[ 2*64]) * 459;
2716 a += (z[ 3*64] + z[11*64]) * 2037;
2717 a += (z[10*64] - z[ 4*64]) * 5153;
2718 a += (z[ 5*64] + z[ 9*64]) * 6574;
2719 a += (z[ 8*64] - z[ 6*64]) * 37489;
2720 a += z[ 7*64] * 75038;
2721 pcm[0] = mp3d_scale_pcm(a);
2722
2723 z += 2;
2724 a = z[14*64] * 104;
2725 a += z[12*64] * 1567;
2726 a += z[10*64] * 9727;
2727 a += z[ 8*64] * 64019;
2728 a += z[ 6*64] * -9975;
2729 a += z[ 4*64] * -45;
2730 a += z[ 2*64] * 146;
2731 a += z[ 0*64] * -5;
2732 pcm[16*nch] = mp3d_scale_pcm(a);
2733}
2734
2735static void mp3d_synth(float *xl, mp3d_sample_t *dstl, int nch, float *lins) FL_NO_EXCEPT
2736{
2737 int i;
2738 float *xr = xl + 576*(nch - 1);
2739 mp3d_sample_t *dstr = dstl + (nch - 1);
2740
2741 static const float g_win[] = {
2742 -1,26,-31,208,218,401,-519,2063,2000,4788,-5517,7134,5959,35640,-39336,74992,
2743 -1,24,-35,202,222,347,-581,2080,1952,4425,-5879,7640,5288,33791,-41176,74856,
2744 -1,21,-38,196,225,294,-645,2087,1893,4063,-6237,8092,4561,31947,-43006,74630,
2745 -1,19,-41,190,227,244,-711,2085,1822,3705,-6589,8492,3776,30112,-44821,74313,
2746 -1,17,-45,183,228,197,-779,2075,1739,3351,-6935,8840,2935,28289,-46617,73908,
2747 -1,16,-49,176,228,153,-848,2057,1644,3004,-7271,9139,2037,26482,-48390,73415,
2748 -2,14,-53,169,227,111,-919,2032,1535,2663,-7597,9389,1082,24694,-50137,72835,
2749 -2,13,-58,161,224,72,-991,2001,1414,2330,-7910,9592,70,22929,-51853,72169,
2750 -2,11,-63,154,221,36,-1064,1962,1280,2006,-8209,9750,-998,21189,-53534,71420,
2751 -2,10,-68,147,215,2,-1137,1919,1131,1692,-8491,9863,-2122,19478,-55178,70590,
2752 -3,9,-73,139,208,-29,-1210,1870,970,1388,-8755,9935,-3300,17799,-56778,69679,
2753 -3,8,-79,132,200,-57,-1283,1817,794,1095,-8998,9966,-4533,16155,-58333,68692,
2754 -4,7,-85,125,189,-83,-1356,1759,605,814,-9219,9959,-5818,14548,-59838,67629,
2755 -4,7,-91,117,177,-106,-1428,1698,402,545,-9416,9916,-7154,12980,-61289,66494,
2756 -5,6,-97,111,163,-127,-1498,1634,185,288,-9585,9838,-8540,11455,-62684,65290
2757 };
2758 float *zlin = lins + 15*64;
2759 const float *w = g_win;
2760
2761 zlin[4*15] = xl[18*16];
2762 zlin[4*15 + 1] = xr[18*16];
2763 zlin[4*15 + 2] = xl[0];
2764 zlin[4*15 + 3] = xr[0];
2765
2766 zlin[4*31] = xl[1 + 18*16];
2767 zlin[4*31 + 1] = xr[1 + 18*16];
2768 zlin[4*31 + 2] = xl[1];
2769 zlin[4*31 + 3] = xr[1];
2770
2771 mp3d_synth_pair(dstr, nch, lins + 4*15 + 1);
2772 mp3d_synth_pair(dstr + 32*nch, nch, lins + 4*15 + 64 + 1);
2773 mp3d_synth_pair(dstl, nch, lins + 4*15);
2774 mp3d_synth_pair(dstl + 32*nch, nch, lins + 4*15 + 64);
2775
2776#if HAVE_SIMD
2777 if (have_simd()) for (i = 14; i >= 0; i--)
2778 {
2779#define VLOAD(k) f4 w0 = VSET(*w++); f4 w1 = VSET(*w++); f4 vz = VLD(&zlin[4*i - 64*k]); f4 vy = VLD(&zlin[4*i - 64*(15 - k)]);
2780#define V0(k) { VLOAD(k) b = VADD(VMUL(vz, w1), VMUL(vy, w0)) ; a = VSUB(VMUL(vz, w0), VMUL(vy, w1)); }
2781#define V1(k) { VLOAD(k) b = VADD(b, VADD(VMUL(vz, w1), VMUL(vy, w0))); a = VADD(a, VSUB(VMUL(vz, w0), VMUL(vy, w1))); }
2782#define V2(k) { VLOAD(k) b = VADD(b, VADD(VMUL(vz, w1), VMUL(vy, w0))); a = VADD(a, VSUB(VMUL(vy, w1), VMUL(vz, w0))); }
2783 f4 a, b;
2784 zlin[4*i] = xl[18*(31 - i)];
2785 zlin[4*i + 1] = xr[18*(31 - i)];
2786 zlin[4*i + 2] = xl[1 + 18*(31 - i)];
2787 zlin[4*i + 3] = xr[1 + 18*(31 - i)];
2788 zlin[4*i + 64] = xl[1 + 18*(1 + i)];
2789 zlin[4*i + 64 + 1] = xr[1 + 18*(1 + i)];
2790 zlin[4*i - 64 + 2] = xl[18*(1 + i)];
2791 zlin[4*i - 64 + 3] = xr[18*(1 + i)];
2792
2793 V0(0) V2(1) V1(2) V2(3) V1(4) V2(5) V1(6) V2(7)
2794
2795 {
2796#ifndef MINIMP3_FLOAT_OUTPUT
2797#if HAVE_SSE
2798 static const f4 g_max = { 32767.0f, 32767.0f, 32767.0f, 32767.0f };
2799 static const f4 g_min = { -32768.0f, -32768.0f, -32768.0f, -32768.0f };
2800 __m128i pcm8 = _mm_packs_epi32(_mm_cvtps_epi32(_mm_max_ps(_mm_min_ps(a, g_max), g_min)),
2801 _mm_cvtps_epi32(_mm_max_ps(_mm_min_ps(b, g_max), g_min)));
2802 dstr[(15 - i)*nch] = _mm_extract_epi16(pcm8, 1);
2803 dstr[(17 + i)*nch] = _mm_extract_epi16(pcm8, 5);
2804 dstl[(15 - i)*nch] = _mm_extract_epi16(pcm8, 0);
2805 dstl[(17 + i)*nch] = _mm_extract_epi16(pcm8, 4);
2806 dstr[(47 - i)*nch] = _mm_extract_epi16(pcm8, 3);
2807 dstr[(49 + i)*nch] = _mm_extract_epi16(pcm8, 7);
2808 dstl[(47 - i)*nch] = _mm_extract_epi16(pcm8, 2);
2809 dstl[(49 + i)*nch] = _mm_extract_epi16(pcm8, 6);
2810#else /* HAVE_SSE */
2811 int16x4_t pcma, pcmb;
2812 a = VADD(a, VSET(0.5f));
2813 b = VADD(b, VSET(0.5f));
2814 pcma = vqmovn_s32(vqaddq_s32(vcvtq_s32_f32(a), vreinterpretq_s32_u32(vcltq_f32(a, VSET(0)))));
2815 pcmb = vqmovn_s32(vqaddq_s32(vcvtq_s32_f32(b), vreinterpretq_s32_u32(vcltq_f32(b, VSET(0)))));
2816 vst1_lane_s16(dstr + (15 - i)*nch, pcma, 1);
2817 vst1_lane_s16(dstr + (17 + i)*nch, pcmb, 1);
2818 vst1_lane_s16(dstl + (15 - i)*nch, pcma, 0);
2819 vst1_lane_s16(dstl + (17 + i)*nch, pcmb, 0);
2820 vst1_lane_s16(dstr + (47 - i)*nch, pcma, 3);
2821 vst1_lane_s16(dstr + (49 + i)*nch, pcmb, 3);
2822 vst1_lane_s16(dstl + (47 - i)*nch, pcma, 2);
2823 vst1_lane_s16(dstl + (49 + i)*nch, pcmb, 2);
2824#endif /* HAVE_SSE */
2825
2826#else /* MINIMP3_FLOAT_OUTPUT */
2827
2828 static const f4 g_scale = { 1.0f/32768.0f, 1.0f/32768.0f, 1.0f/32768.0f, 1.0f/32768.0f };
2829 a = VMUL(a, g_scale);
2830 b = VMUL(b, g_scale);
2831#if HAVE_SSE
2832 _mm_store_ss(dstr + (15 - i)*nch, _mm_shuffle_ps(a, a, _MM_SHUFFLE(1, 1, 1, 1)));
2833 _mm_store_ss(dstr + (17 + i)*nch, _mm_shuffle_ps(b, b, _MM_SHUFFLE(1, 1, 1, 1)));
2834 _mm_store_ss(dstl + (15 - i)*nch, _mm_shuffle_ps(a, a, _MM_SHUFFLE(0, 0, 0, 0)));
2835 _mm_store_ss(dstl + (17 + i)*nch, _mm_shuffle_ps(b, b, _MM_SHUFFLE(0, 0, 0, 0)));
2836 _mm_store_ss(dstr + (47 - i)*nch, _mm_shuffle_ps(a, a, _MM_SHUFFLE(3, 3, 3, 3)));
2837 _mm_store_ss(dstr + (49 + i)*nch, _mm_shuffle_ps(b, b, _MM_SHUFFLE(3, 3, 3, 3)));
2838 _mm_store_ss(dstl + (47 - i)*nch, _mm_shuffle_ps(a, a, _MM_SHUFFLE(2, 2, 2, 2)));
2839 _mm_store_ss(dstl + (49 + i)*nch, _mm_shuffle_ps(b, b, _MM_SHUFFLE(2, 2, 2, 2)));
2840#else /* HAVE_SSE */
2841 vst1q_lane_f32(dstr + (15 - i)*nch, a, 1);
2842 vst1q_lane_f32(dstr + (17 + i)*nch, b, 1);
2843 vst1q_lane_f32(dstl + (15 - i)*nch, a, 0);
2844 vst1q_lane_f32(dstl + (17 + i)*nch, b, 0);
2845 vst1q_lane_f32(dstr + (47 - i)*nch, a, 3);
2846 vst1q_lane_f32(dstr + (49 + i)*nch, b, 3);
2847 vst1q_lane_f32(dstl + (47 - i)*nch, a, 2);
2848 vst1q_lane_f32(dstl + (49 + i)*nch, b, 2);
2849#endif /* HAVE_SSE */
2850#endif /* MINIMP3_FLOAT_OUTPUT */
2851 }
2852 } else
2853#endif /* HAVE_SIMD */
2854#ifdef MINIMP3_ONLY_SIMD
2855 {} /* for HAVE_SIMD=1, MINIMP3_ONLY_SIMD=1 case we do not need non-intrinsic "else" branch */
2856#else /* MINIMP3_ONLY_SIMD */
2857 for (i = 14; i >= 0; i--)
2858 {
2859#define LOAD(k) float w0 = *w++; float w1 = *w++; float *vz = &zlin[4*i - k*64]; float *vy = &zlin[4*i - (15 - k)*64];
2860#define S0(k) { int j; LOAD(k); for (j = 0; j < 4; j++) b[j] = vz[j]*w1 + vy[j]*w0, a[j] = vz[j]*w0 - vy[j]*w1; }
2861#define S1(k) { int j; LOAD(k); for (j = 0; j < 4; j++) b[j] += vz[j]*w1 + vy[j]*w0, a[j] += vz[j]*w0 - vy[j]*w1; }
2862#define S2(k) { int j; LOAD(k); for (j = 0; j < 4; j++) b[j] += vz[j]*w1 + vy[j]*w0, a[j] += vy[j]*w1 - vz[j]*w0; }
2863 float a[4], b[4];
2864
2865 zlin[4*i] = xl[18*(31 - i)];
2866 zlin[4*i + 1] = xr[18*(31 - i)];
2867 zlin[4*i + 2] = xl[1 + 18*(31 - i)];
2868 zlin[4*i + 3] = xr[1 + 18*(31 - i)];
2869 zlin[4*(i + 16)] = xl[1 + 18*(1 + i)];
2870 zlin[4*(i + 16) + 1] = xr[1 + 18*(1 + i)];
2871 zlin[4*(i - 16) + 2] = xl[18*(1 + i)];
2872 zlin[4*(i - 16) + 3] = xr[18*(1 + i)];
2873
2874 S0(0) S2(1) S1(2) S2(3) S1(4) S2(5) S1(6) S2(7)
2875
2876 dstr[(15 - i)*nch] = mp3d_scale_pcm(a[1]);
2877 dstr[(17 + i)*nch] = mp3d_scale_pcm(b[1]);
2878 dstl[(15 - i)*nch] = mp3d_scale_pcm(a[0]);
2879 dstl[(17 + i)*nch] = mp3d_scale_pcm(b[0]);
2880 dstr[(47 - i)*nch] = mp3d_scale_pcm(a[3]);
2881 dstr[(49 + i)*nch] = mp3d_scale_pcm(b[3]);
2882 dstl[(47 - i)*nch] = mp3d_scale_pcm(a[2]);
2883 dstl[(49 + i)*nch] = mp3d_scale_pcm(b[2]);
2884 }
2885#endif /* MINIMP3_ONLY_SIMD */
2886}
2887
2888static void mp3d_synth_granule(float *qmf_state, float *grbuf, int nbands,
2889 int nch, mp3d_sample_t *pcm) FL_NO_EXCEPT
2890{
2891 int i;
2892 float *lins = qmf_state;
2893 for (i = 0; i < nch; i++)
2894 {
2895 mp3d_DCT_II(grbuf + 576*i, nbands);
2896 MP3D_STAGE(MINIMP3_STAGE_DCT2, i, grbuf + 576*i, 18*nbands);
2897 }
2898
2899 for (i = 0; i < nbands; i += 2)
2900 {
2901 mp3d_synth(grbuf + i, pcm + 32*nch*i, nch, lins + i*64);
2902 }
2903#ifndef MINIMP3_NONSTANDARD_BUT_LOGICAL
2904 if (nch == 1)
2905 {
2906 for (i = 0; i < 15*64; i += 2)
2907 {
2908 qmf_state[i] = lins[nbands*64 + i];
2909 }
2910 } else
2911#endif /* MINIMP3_NONSTANDARD_BUT_LOGICAL */
2912 {
2913 memmove(qmf_state, lins + nbands*64, sizeof(float)*15*64);
2914 }
2915}
2916
2917#endif /* MINIMP3_HAVE_FIXED_POINT */
2918
2919static int mp3d_match_frame(const uint8_t *hdr, int mp3_bytes, int frame_bytes) FL_NO_EXCEPT
2920{
2921 int i, nmatch;
2922 for (i = 0, nmatch = 0; nmatch < MAX_FRAME_SYNC_MATCHES; nmatch++)
2923 {
2924 i += hdr_frame_bytes(hdr + i, frame_bytes) + hdr_padding(hdr + i);
2925 if (i + HDR_SIZE > mp3_bytes)
2926 return nmatch > 0;
2927 if (!hdr_compare(hdr, hdr + i))
2928 return 0;
2929 }
2930 return 1;
2931}
2932
2933static int mp3d_find_frame(const uint8_t *mp3, int mp3_bytes, int *free_format_bytes, int *ptr_frame_bytes) FL_NO_EXCEPT
2934{
2935 int i, k;
2936 for (i = 0; i < mp3_bytes - HDR_SIZE; i++, mp3++)
2937 {
2938 if (hdr_valid(mp3))
2939 {
2940 int frame_bytes = hdr_frame_bytes(mp3, *free_format_bytes);
2941 int frame_and_padding = frame_bytes + hdr_padding(mp3);
2942
2943 for (k = HDR_SIZE; !frame_bytes && k < MAX_FREE_FORMAT_FRAME_SIZE && i + 2*k < mp3_bytes - HDR_SIZE; k++)
2944 {
2945 if (hdr_compare(mp3, mp3 + k))
2946 {
2947 int fb = k - hdr_padding(mp3);
2948 int nextfb = fb + hdr_padding(mp3 + k);
2949 if (i + k + nextfb + HDR_SIZE > mp3_bytes || !hdr_compare(mp3, mp3 + k + nextfb))
2950 continue;
2951 frame_and_padding = k;
2952 frame_bytes = fb;
2953 *free_format_bytes = fb;
2954 }
2955 }
2956 if ((frame_bytes && i + frame_and_padding <= mp3_bytes &&
2957 mp3d_match_frame(mp3, mp3_bytes - i, frame_bytes)) ||
2958 (!i && frame_and_padding == mp3_bytes))
2959 {
2960 *ptr_frame_bytes = frame_and_padding;
2961 return i;
2962 }
2963 *free_format_bytes = 0;
2964 }
2965 }
2966 *ptr_frame_bytes = 0;
2967 return mp3_bytes;
2968}
2969
2971{
2973}
2974
2976{
2977 return MINIMP3_DSP_INTEGER;
2978}
2979
2981{
2982#if MP3D_SIMD_KERNELS_LIVE
2983 /* Compile-time availability is not the question -- on x86 the kernels are
2984 compiled for any SSE2 build but only reached when the run-time SSE4.1
2985 check passes. Reporting the compile-time answer would make the
2986 bit-exactness gate compare a scalar decode against another scalar decode
2987 and call it a pass, and would make the perf gate assert a ratio on two
2988 identical runs, i.e. on timer noise. */
2989 return MP3D_SIMD_AVAILABLE();
2990#else
2991 return 0;
2992#endif
2993}
2994
2996{
2997 dec->header[0] = 0;
2998}
2999
3000int mp3dec_decode_frame_r(mp3dec_t *dec, mp3dec_scratch_t *scratch_storage,
3001 const uint8_t *mp3, int mp3_bytes,
3002 mp3d_sample_t *pcm,
3003 mp3dec_frame_info_t *info) FL_NO_EXCEPT
3004{
3005 int i = 0, igr, frame_size = 0, success = 1;
3006 const uint8_t *hdr;
3007 bs_t bs_frame[1];
3008 mp3dec_scratch_internal_t *scratch =
3009 (mp3dec_scratch_internal_t *)(void *)scratch_storage->buffer;
3010
3011 if (mp3_bytes > 4 && dec->header[0] == 0xff && hdr_compare(dec->header, mp3))
3012 {
3013 frame_size = hdr_frame_bytes(mp3, dec->free_format_bytes) + hdr_padding(mp3);
3014 if (frame_size != mp3_bytes && (frame_size + HDR_SIZE > mp3_bytes || !hdr_compare(mp3, mp3 + frame_size)))
3015 {
3016 frame_size = 0;
3017 }
3018 }
3019 if (!frame_size)
3020 {
3021 memset(dec, 0, sizeof(mp3dec_t));
3022 i = mp3d_find_frame(mp3, mp3_bytes, &dec->free_format_bytes, &frame_size);
3023 if (!frame_size || i + frame_size > mp3_bytes)
3024 {
3025 info->frame_bytes = i;
3026 return 0;
3027 }
3028 }
3029
3030 hdr = mp3 + i;
3031 memcpy(dec->header, hdr, HDR_SIZE);
3032 info->frame_bytes = i + frame_size;
3033 info->frame_offset = i;
3034 info->channels = HDR_IS_MONO(hdr) ? 1 : 2;
3035 info->hz = hdr_sample_rate_hz(hdr);
3036 info->layer = 4 - HDR_GET_LAYER(hdr);
3037 info->bitrate_kbps = hdr_bitrate_kbps(hdr);
3038
3039 if (!pcm)
3040 {
3041 return hdr_frame_samples(hdr);
3042 }
3043
3044 bs_init(bs_frame, hdr + HDR_SIZE, frame_size - HDR_SIZE);
3045 if (HDR_IS_CRC(hdr))
3046 {
3047 get_bits(bs_frame, 16);
3048 }
3049
3050 if (info->layer == 3)
3051 {
3052 int main_data_begin = L3_read_side_info(bs_frame, scratch->gr_info, hdr);
3053 if (main_data_begin < 0 || bs_frame->pos > bs_frame->limit)
3054 {
3055 mp3dec_init(dec);
3056 return 0;
3057 }
3058 success = L3_restore_reservoir(dec, bs_frame, scratch, main_data_begin);
3059 if (success)
3060 {
3061 for (igr = 0; igr < (HDR_TEST_MPEG1(hdr) ? 2 : 1); igr++, pcm += 576*info->channels)
3062 {
3063 L3_decode(dec, scratch, scratch->gr_info + igr*info->channels, info->channels);
3064 mp3d_synth_granule(dec->qmf_state, scratch->grbuf[0], 18, info->channels, pcm);
3065 }
3066 }
3067 L3_save_reservoir(dec, scratch);
3068 } else
3069 {
3070#ifdef MINIMP3_ONLY_MP3
3071 return 0;
3072#else /* MINIMP3_ONLY_MP3 */
3073 /* Aliases the arena rather than the stack; see the union above. */
3074 L12_scale_info *sci = &scratch->u.l12;
3075 L12_read_scale_info(hdr, bs_frame, sci);
3076
3077 memset(scratch->grbuf[0], 0, 576*2*sizeof(mp3d_dsp_t));
3078 for (i = 0, igr = 0; igr < 3; igr++)
3079 {
3080 if (12 == (i += L12_dequantize_granule(scratch->grbuf[0] + i, bs_frame, sci, info->layer | 1)))
3081 {
3082 i = 0;
3083#if MINIMP3_HAVE_FIXED_POINT
3084 L12_apply_scf_384(sci, sci->scf_mant + igr, sci->scf_exp + igr, scratch->grbuf[0]);
3085#else
3086 L12_apply_scf_384(sci, sci->scf + igr, scratch->grbuf[0]);
3087#endif
3088 mp3d_synth_granule(dec->qmf_state, scratch->grbuf[0], 12, info->channels, pcm);
3089 memset(scratch->grbuf[0], 0, 576*2*sizeof(mp3d_dsp_t));
3090 pcm += 384*info->channels;
3091 }
3092 if (bs_frame->pos > bs_frame->limit)
3093 {
3094 mp3dec_init(dec);
3095 return 0;
3096 }
3097 }
3098#endif /* MINIMP3_ONLY_MP3 */
3099 }
3100 return success*hdr_frame_samples(dec->header);
3101}
3102
3103#ifdef MINIMP3_FLOAT_OUTPUT
3104void mp3dec_f32_to_s16(const float *in, int16_t *out, int num_samples) FL_NO_EXCEPT
3105{
3106 int i = 0;
3107#if HAVE_SIMD
3108 int aligned_count = num_samples & ~7;
3109 for(; i < aligned_count; i += 8)
3110 {
3111 static const f4 g_scale = { 32768.0f, 32768.0f, 32768.0f, 32768.0f };
3112 f4 a = VMUL(VLD(&in[i ]), g_scale);
3113 f4 b = VMUL(VLD(&in[i+4]), g_scale);
3114#if HAVE_SSE
3115 static const f4 g_max = { 32767.0f, 32767.0f, 32767.0f, 32767.0f };
3116 static const f4 g_min = { -32768.0f, -32768.0f, -32768.0f, -32768.0f };
3117 __m128i pcm8 = _mm_packs_epi32(_mm_cvtps_epi32(_mm_max_ps(_mm_min_ps(a, g_max), g_min)),
3118 _mm_cvtps_epi32(_mm_max_ps(_mm_min_ps(b, g_max), g_min)));
3119 out[i ] = _mm_extract_epi16(pcm8, 0);
3120 out[i+1] = _mm_extract_epi16(pcm8, 1);
3121 out[i+2] = _mm_extract_epi16(pcm8, 2);
3122 out[i+3] = _mm_extract_epi16(pcm8, 3);
3123 out[i+4] = _mm_extract_epi16(pcm8, 4);
3124 out[i+5] = _mm_extract_epi16(pcm8, 5);
3125 out[i+6] = _mm_extract_epi16(pcm8, 6);
3126 out[i+7] = _mm_extract_epi16(pcm8, 7);
3127#else /* HAVE_SSE */
3128 int16x4_t pcma, pcmb;
3129 a = VADD(a, VSET(0.5f));
3130 b = VADD(b, VSET(0.5f));
3131 pcma = vqmovn_s32(vqaddq_s32(vcvtq_s32_f32(a), vreinterpretq_s32_u32(vcltq_f32(a, VSET(0)))));
3132 pcmb = vqmovn_s32(vqaddq_s32(vcvtq_s32_f32(b), vreinterpretq_s32_u32(vcltq_f32(b, VSET(0)))));
3133 vst1_lane_s16(out+i , pcma, 0);
3134 vst1_lane_s16(out+i+1, pcma, 1);
3135 vst1_lane_s16(out+i+2, pcma, 2);
3136 vst1_lane_s16(out+i+3, pcma, 3);
3137 vst1_lane_s16(out+i+4, pcmb, 0);
3138 vst1_lane_s16(out+i+5, pcmb, 1);
3139 vst1_lane_s16(out+i+6, pcmb, 2);
3140 vst1_lane_s16(out+i+7, pcmb, 3);
3141#endif /* HAVE_SSE */
3142 }
3143#endif /* HAVE_SIMD */
3144 for(; i < num_samples; i++)
3145 {
3146 float sample = in[i] * 32768.0f;
3147 if (sample >= 32766.5)
3148 out[i] = (int16_t) 32767;
3149 else if (sample <= -32767.5)
3150 out[i] = (int16_t)-32768;
3151 else
3152 {
3153 int16_t s = (int16_t)(sample + .5f);
3154 s -= (s < 0); /* away from zero, to be compliant */
3155 out[i] = s;
3156 }
3157 }
3158}
3159#endif /* MINIMP3_FLOAT_OUTPUT */
3160#ifdef __cplusplus
3161} /* namespace MINIMP3_NAMESPACE */
3162} /* namespace fl */
3163#endif
3164
3165/* FastLED: release every macro the implementation owns so the header can be
3166 included a second time in the same translation unit under a different
3167 MINIMP3_NAMESPACE. Without this the fixed variant would inherit the float
3168 variant's SIMD configuration (HAVE_SIMD / MINIMP3_ONLY_SIMD in particular,
3169 which would compile the scalar integer kernels out entirely) and every
3170 redefinition that differs between the two would be a hard error. */
3171#undef MP3D_STAGE
3172#undef MP3D_SCF_ARGS
3173#undef MP3D_HUFF_SCF_ARGS
3174#undef MP3D_HUFF_ONE_VARS
3175#undef MP3D_HUFF_NEXT_SCF
3176#undef MP3D_HUFF_ESC
3177#undef MP3D_HUFF_TAB
3178#undef MP3D_HUFF_ONE
3179#undef MP3D_WRAP_ADD
3180#undef MP3D_WRAP_SUB
3181#undef MP3D_LEAF
3182#undef MP3D_HOT
3183#undef MP3D_KERNEL
3184#undef MP3D_SAT_MAX
3185#undef MP3D_SAT_MIN
3186#undef MP3D_PCM_HALF
3187/* Only release MINIMP3_NO_SIMD if this header is what set it; a caller that
3188 asked for the scalar build must keep getting it on a re-include. */
3189#undef MP3D_FLOAT_SIMD_OFF
3190#undef MP3D_V_ZERO64
3191#undef MP3D_V_LOAD4
3192#undef MP3D_V_SPLAT
3193#undef MP3D_V_MULV_HI
3194#undef MP3D_V_MULV_LO
3195#undef MP3D_V_REV4
3196#undef MP3D_V_MUL_LO
3197#undef MP3D_V_MUL_HI
3198#undef MP3D_V_ADD64
3199#undef MP3D_V_SUB64
3200#undef MP3D_V_GET64
3201#undef MP3D_V_PREP
3202#undef MP3D_V_ADDSAT
3203#undef MP3D_V_SUBSAT
3204#undef MP3D_V_MULSHIFT
3205#undef MP3D_V_STORE4
3206#undef MP3D_HAVE_INT_SIMD
3207#undef MP3D_INT_SIMD_SSE
3208#undef MP3D_INT_SIMD_NEON
3209#undef MP3D_SIMD_KERNELS_LIVE
3210#undef MP3D_SIMD_AVAILABLE
3211#undef MP3D_SIMD_TARGET
3212#undef MP3D_MULSHIFT_K_WIDE_PRODUCT
3213#undef BITS_DEQUANTIZER_OUT
3214#undef BSPOS
3215#undef CHECK_BITS
3216#undef DEQ_COUNT1
3217#undef DQ
3218#undef FLUSH_BITS
3219#undef HAVE_ARMV6
3220#undef HAVE_SIMD
3221#undef HAVE_SSE
3222#undef HDR_GET_BITRATE
3223#undef HDR_GET_LAYER
3224#undef HDR_GET_MY_SAMPLE_RATE
3225#undef HDR_GET_SAMPLE_RATE
3226#undef HDR_GET_STEREO_MODE
3227#undef HDR_GET_STEREO_MODE_EXT
3228#undef HDR_IS_CRC
3229#undef HDR_IS_FRAME_576
3230#undef HDR_IS_FREE_FORMAT
3231#undef HDR_IS_LAYER_1
3232#undef HDR_IS_MONO
3233#undef HDR_IS_MS_STEREO
3234#undef HDR_SIZE
3235#undef HDR_TEST_I_STEREO
3236#undef HDR_TEST_MPEG1
3237#undef HDR_TEST_MS_STEREO
3238#undef HDR_TEST_NOT_MPEG25
3239#undef HDR_TEST_PADDING
3240#undef LOAD
3241#undef MAX_BITRESERVOIR_BYTES
3242#undef MAX_FRAME_SYNC_MATCHES
3243#undef MAX_FREE_FORMAT_FRAME_SIZE
3244#undef MAX_L3_FRAME_PAYLOAD_BYTES
3245#undef MAX_SCF
3246#undef MAX_SCFI
3247#undef minimp3_cpuid
3248#undef MINIMP3_MAX
3249#undef MINIMP3_MIN
3250#undef MINIMP3_ONLY_SIMD
3251#undef MODE_JOINT_STEREO
3252#undef MODE_MONO
3253#undef MP3D_SYNTH_CHAIN
3254#undef PEEK_BITS
3255#undef RELOAD_SCALEFACTOR
3256#undef S0
3257#undef S1
3258#undef S2
3259#undef SHORT_BLOCK_TYPE
3260#undef STOP_BLOCK_TYPE
3261#undef V0
3262#undef V1
3263#undef V2
3264#undef VADD
3265#undef VLD
3266#undef VLOAD
3267#undef VMAC
3268#undef VMSB
3269#undef VMUL
3270#undef VMUL_S
3271#undef VREV
3272#undef VSAVE2
3273#undef VSAVE4
3274#undef VSET
3275#undef VSTORE
3276#undef VSUB
3277
3278#endif /* MINIMP3_IMPLEMENTATION && !_MINIMP3_IMPLEMENTATION_GUARD */
3279
3280/* Outside every guard above, so it runs on header-only inclusions too: give
3281 back the variant default this header set for itself, leaving a caller's own
3282 MINIMP3_FIXED_POINT alone. See MINIMP3_FIXED_POINT_IS_HEADER_DEFAULT. */
3283#if defined(MINIMP3_FIXED_POINT_IS_HEADER_DEFAULT)
3284#undef MINIMP3_FIXED_POINT
3285#undef MINIMP3_FIXED_POINT_IS_HEADER_DEFAULT
3286#endif
int y
Definition simple.h:93
int x
Definition simple.h:92
uint8_t pos
Definition Blur.ino:11
uint32_t z[NUM_LAYERS]
Definition Fire2023.h:93
static uint32_t t
Definition Luminova.h:55
Integer arithmetic with a guaranteed instruction shape.
void mp3dec_init(mp3dec_t *dec) FL_NO_EXCEPT
#define MINIMP3_SCRATCH_SIZE
Definition minimp3.h:128
int mp3dec_is_fixed_point(void) FL_NO_EXCEPT
#define MINIMP3_NAMESPACE
Definition minimp3.h:181
#define MINIMP3_STAGE_STEREO
Definition minimp3.h:167
#define MINIMP3_STAGE_HUFFMAN
Definition minimp3.h:166
#define MINIMP3_STAGE_ANTIALIAS
Definition minimp3.h:168
#define MINIMP3_DSP_INTEGER
Definition minimp3.h:117
int mp3dec_dsp_uses_simd(void) FL_NO_EXCEPT
int mp3dec_decode_frame_r(mp3dec_t *dec, mp3dec_scratch_t *scratch, const uint8_t *mp3, int mp3_bytes, mp3d_sample_t *pcm, mp3dec_frame_info_t *info) FL_NO_EXCEPT
int16_t mp3d_sample_t
Definition minimp3.h:227
union mp3dec_scratch_t mp3dec_scratch_t
#define MINIMP3_FRAC_BITS
Definition minimp3.h:159
#define MINIMP3_STAGE_DCT2
Definition minimp3.h:170
struct mp3dec_t mp3dec_t
#define MINIMP3_STAGE_IMDCT
Definition minimp3.h:169
int mp3dec_dsp_is_integer(void) FL_NO_EXCEPT
#define MINIMP3_HAVE_FIXED_POINT
Definition minimp3.h:105
float mp3d_dsp_t
Definition minimp3.h:200
mp3d_dsp_t mdct_overlap[2][9 *32]
Definition minimp3.h:205
int reserv
Definition minimp3.h:206
unsigned char reserv_buf[511]
Definition minimp3.h:207
unsigned char header[4]
Definition minimp3.h:207
mp3d_dsp_t qmf_state[(15+18) *2 *32]
Definition minimp3.h:205
uint8_t buffer[MINIMP3_SCRATCH_SIZE]
Definition minimp3.h:213
uint64_t alignment
Definition minimp3.h:212
int free_format_bytes
Definition minimp3.h:206
#define MP3D_Q30_POW43_C1
static const int32_t g_mdct_window_normal_q30[18]
static const int32_t g_expfrac_q30[4]
static const int32_t g_twid3_q30[6]
static const int8_t g_deq_L12_exp[54]
static const int32_t g_pow43_mant[145]
#define MP3D_Q30_SQRT2
static const int32_t g_mdct_window_stop_q30[18]
static const int32_t g_aa_ca_q31[8]
static const int32_t g_aa_cs_q31[8]
static const int32_t g_deq_L12_mant[54]
static const int32_t g_pan_q30[14]
static const int8_t g_pow43_exp[145]
static MP3D_KERNEL void mp3d_DCT_II(int32_t *grbuf, int n) FL_NO_EXCEPT
#define S1(k, st)
static void mp3d_synth_granule(int32_t *qmf_state, int32_t *grbuf, int nbands, int nch, mp3d_sample_t *pcm) FL_NO_EXCEPT
MP3D_HOT void mp3d_synth(int32_t *xl, mp3d_sample_t *dstl, int nch, int32_t *lins) FL_NO_EXCEPT
static MP3D_KERNEL void mp3d_imdct36_twiddle(int32_t *grbuf, int32_t *overlap, const int32_t *window, const int32_t *co, const int32_t *si) FL_NO_EXCEPT
#define S0(k, st)
static void mp3d_synth_pair(mp3d_sample_t *pcm, int nch, const int32_t *z) FL_NO_EXCEPT
#define S2(k, st)
MP3D_LEAF mp3d_sample_t mp3d_scale_pcm(int64_t sample) FL_NO_EXCEPT
FASTLED_FORCE_INLINE i32 mul_shift_round32(i32 x, i32 y) FL_NO_EXCEPT
Definition int_asm.h:41
const dec_t dec
Definition ios.cpp.hpp:13
void * memcpy(void *dest, const void *src, size_t n) FL_NO_EXCEPT
void * memmove(void *dest, const void *src, size_t n) FL_NO_EXCEPT
CRGB sample(const CRGB *grid, const XYMap &xyMap, float x, float y, SampleMode mode)
Sample a pixel from a 2D CRGB grid at floating-point coordinates.
Definition sample.cpp.hpp:9
void * memset(void *s, int c, size_t n) FL_NO_EXCEPT
constexpr T * end(T(&array)[N]) FL_NO_EXCEPT
enable_if< is_fixed_point< T >::value, T >::type exp(T x) FL_NO_EXCEPT
constexpr enable_if< is_fixed_point< T >::value, int >::type sign(T x) FL_NO_EXCEPT
constexpr enable_if< is_fixed_point< T >::value, T >::type mod(T a, T b) FL_NO_EXCEPT
Base definition for an LED controller.
Definition crgb.hpp:179
#define FL_NO_EXCEPT
unsigned char uint8_t
Definition stdint.h:208
fl::i16 int16_t
Definition stdint.h:214
fl::u32 uint32_t
Definition stdint.h:218
signed char int8_t
Definition stdint.h:209
fl::i32 int32_t
Definition stdint.h:219
fl::u16 uint16_t
Definition stdint.h:213