FastLED 3.10.6
Loading...
Searching...
No Matches

◆ mp3d_synth()

MP3D_HOT void mp3d_synth ( int32_t * xl,
mp3d_sample_t * dstl,
int nch,
int32_t * lins )

Definition at line 747 of file minimp3_synth_fixed.h.

748{
749 int i;
750 int32_t *xr = xl + 576*(nch - 1);
751 mp3d_sample_t *dstr = dstl + (nch - 1);
752
753 static const int32_t g_win[] = {
754 -1,26,-31,208,218,401,-519,2063,2000,4788,-5517,7134,5959,35640,-39336,74992,
755 -1,24,-35,202,222,347,-581,2080,1952,4425,-5879,7640,5288,33791,-41176,74856,
756 -1,21,-38,196,225,294,-645,2087,1893,4063,-6237,8092,4561,31947,-43006,74630,
757 -1,19,-41,190,227,244,-711,2085,1822,3705,-6589,8492,3776,30112,-44821,74313,
758 -1,17,-45,183,228,197,-779,2075,1739,3351,-6935,8840,2935,28289,-46617,73908,
759 -1,16,-49,176,228,153,-848,2057,1644,3004,-7271,9139,2037,26482,-48390,73415,
760 -2,14,-53,169,227,111,-919,2032,1535,2663,-7597,9389,1082,24694,-50137,72835,
761 -2,13,-58,161,224,72,-991,2001,1414,2330,-7910,9592,70,22929,-51853,72169,
762 -2,11,-63,154,221,36,-1064,1962,1280,2006,-8209,9750,-998,21189,-53534,71420,
763 -2,10,-68,147,215,2,-1137,1919,1131,1692,-8491,9863,-2122,19478,-55178,70590,
764 -3,9,-73,139,208,-29,-1210,1870,970,1388,-8755,9935,-3300,17799,-56778,69679,
765 -3,8,-79,132,200,-57,-1283,1817,794,1095,-8998,9966,-4533,16155,-58333,68692,
766 -4,7,-85,125,189,-83,-1356,1759,605,814,-9219,9959,-5818,14548,-59838,67629,
767 -4,7,-91,117,177,-106,-1428,1698,402,545,-9416,9916,-7154,12980,-61289,66494,
768 -5,6,-97,111,163,-127,-1498,1634,185,288,-9585,9838,-8540,11455,-62684,65290
769 };
770 int32_t *zlin = lins + 15*64;
771 const int32_t *w = g_win;
772 /* Lane stride through the four-lane accumulator.
773
774 The four lanes are (left, right) x (subband i, subband i+1). In mono
775 `xr == xl` and `dstr == dstl`, so lanes 1 and 3 recompute lanes 0 and 2
776 and their stores are immediately overwritten by the lane 0/2 stores that
777 follow them -- half the polyphase filterbank, thrown away. Helix avoids
778 it by shipping a separate PolyphaseMono; here it is the same function
779 with a two-lane tap chain, which is worth 34% of a mono decode on an
780 ESP32-C6 (136,546 -> 89,699 us over 60 frames of 16 kHz MPEG-2 Layer III,
781 bit-identical output).
782
783 Nothing else reads the odd lanes in mono. The granule's mono carry loop
784 (`for (i = 0; i < 15*64; i += 2)`) already saves only even indices, so
785 the odd lanes of the history are stale in mono either way -- today they
786 are read and discarded, which is exactly the work this removes.
787
788 The lane step is a literal in each of two copies of the tap chain rather
789 than a variable in one copy. One copy with `j += jstep` was tried first,
790 on the reasoning that riscv32 -Os does not unroll the four-lane loop
791 anyway (mp3d_synth_granule has exactly 32 `mulh` -- 8 taps x 4 products,
792 not 8 x 4 x 4), so the step ought to have been free. Measured on the C6
793 it was not: the stereo Layer III path went 46,294 -> 47,665 us, +3.0%,
794 against a helix reference that moved 0.01% between the same two runs.
795 Two chains with literal steps cost .text and nothing else. */
796#if MP3D_HAVE_INT_SIMD
797 const int use_simd = MP3D_SIMD_AVAILABLE();
798#endif
799
800 zlin[4*15] = xl[18*16];
801 zlin[4*15 + 2] = xl[0];
802
803 zlin[4*31] = xl[1 + 18*16];
804 zlin[4*31 + 2] = xl[1];
805
806 if (nch == 2)
807 {
808 zlin[4*15 + 1] = xr[18*16];
809 zlin[4*15 + 3] = xr[0];
810 zlin[4*31 + 1] = xr[1 + 18*16];
811 zlin[4*31 + 3] = xr[1];
812
813 mp3d_synth_pair(dstr, nch, lins + 4*15 + 1);
814 mp3d_synth_pair(dstr + 32*nch, nch, lins + 4*15 + 64 + 1);
815 }
816 mp3d_synth_pair(dstl, nch, lins + 4*15);
817 mp3d_synth_pair(dstl + 32*nch, nch, lins + 4*15 + 64);
818
819 for (i = 14; i >= 0; i--)
820 {
821/* One tap of one lane pair. S0/S1/S2 are the original chain unchanged: S0
822 opens the accumulators, S1 adds and S2 adds with the `a` operands swapped.
823 The arithmetic, the operand order and the accumulation order are identical
824 to the four-lane form these replaced, which is why the PCM checksum does
825 not move.
826
827 Two things changed, and only the second one is arithmetic-free by accident.
828
829 The chain used to carry all four lanes -- eight int64 accumulators, held in
830 `int64_t a[4], b[4]` -- across the whole eight-tap run. On riscv32 -Os that
831 array is a stack slot, and the tap body paid eight memory operations per
832 tap per lane to reload and rewrite it: of its 35 instructions, four loads
833 and four stores were accumulator traffic and nothing else.
834
835 Halving the live set is necessary but not sufficient. `int64_t a[2], b[2]`
836 with the lane loop left rolled spills exactly as badly, because at -Os gcc
837 does not unroll a two-iteration loop and a variable index into a local array
838 has to be memory. The lanes are therefore written out as named scalars,
839 which is what actually lets the accumulators live in registers.
840
841 One lane at a time was tried too. It removes the spill just as completely
842 but re-reads the window pair and recomputes the two zlin base addresses
843 four times per tap instead of twice, which costs 3.2% on an x86-64 host
844 that had registers to spare and was never paying the spill. The pair keeps
845 both ends. */
846#define LOAD(k) const int32_t w0 = w[2*(k)]; const int32_t w1 = w[2*(k) + 1]; const int32_t *vz = &zlin[4*i + g - (k)*64]; const int32_t *vy = &zlin[4*i + g - (15 - (k))*64];
847#define S0(k, st) { LOAD(k); \
848 b0 = (int64_t)vz[0]*w1 + (int64_t)vy[0]*w0; \
849 a0 = (int64_t)vz[0]*w0 - (int64_t)vy[0]*w1; \
850 if (st == 1) { \
851 b1 = (int64_t)vz[1]*w1 + (int64_t)vy[1]*w0; \
852 a1 = (int64_t)vz[1]*w0 - (int64_t)vy[1]*w1; } }
853#define S1(k, st) { LOAD(k); \
854 b0 += (int64_t)vz[0]*w1 + (int64_t)vy[0]*w0; \
855 a0 += (int64_t)vz[0]*w0 - (int64_t)vy[0]*w1; \
856 if (st == 1) { \
857 b1 += (int64_t)vz[1]*w1 + (int64_t)vy[1]*w0; \
858 a1 += (int64_t)vz[1]*w0 - (int64_t)vy[1]*w1; } }
859#define S2(k, st) { LOAD(k); \
860 b0 += (int64_t)vz[0]*w1 + (int64_t)vy[0]*w0; \
861 a0 += (int64_t)vy[0]*w1 - (int64_t)vz[0]*w0; \
862 if (st == 1) { \
863 b1 += (int64_t)vz[1]*w1 + (int64_t)vy[1]*w0; \
864 a1 += (int64_t)vy[1]*w1 - (int64_t)vz[1]*w0; } }
865/* Lane pair g writes (a, b) to (15 - i, 17 + i) for g == 0 and to the same
866 pair plus 32 samples for g == 2; within the pair, lane 0 is the left
867 channel and lane 1 is dstr, which is dstl + (nch - 1). Both are address
868 arithmetic on the pair index rather than four copies of the chain.
869
870 In mono the odd lane recomputes the even one into the same address, so the
871 step is 2 and lane 1 never runs at all -- a1 and b1 are dead, not merely
872 redundant, and the zeroing below is what tells the compiler so rather than
873 a value anything reads. */
874#define MP3D_SYNTH_CHAIN(st) \
875 { \
876 int g; \
877 for (g = 0; g < 4; g += 2) \
878 { \
879 mp3d_sample_t *d = dstl + g*16*nch; \
880 int64_t a0, b0, a1 = 0, b1 = 0; \
881 S0(0, st) S2(1, st) S1(2, st) S2(3, st) \
882 S1(4, st) S2(5, st) S1(6, st) S2(7, st) \
883 if (st == 1) \
884 { \
885 d[(15 - i)*nch + 1] = mp3d_scale_pcm(a1); \
886 d[(17 + i)*nch + 1] = mp3d_scale_pcm(b1); \
887 } \
888 d[(15 - i)*nch] = mp3d_scale_pcm(a0); \
889 d[(17 + i)*nch] = mp3d_scale_pcm(b0); \
890 } \
891 }
892
893 zlin[4*i] = xl[18*(31 - i)];
894 zlin[4*i + 2] = xl[1 + 18*(31 - i)];
895 zlin[4*(i + 16)] = xl[1 + 18*(1 + i)];
896 zlin[4*(i - 16) + 2] = xl[18*(1 + i)];
897 if (nch == 2)
898 {
899 zlin[4*i + 1] = xr[18*(31 - i)];
900 zlin[4*i + 3] = xr[1 + 18*(31 - i)];
901 zlin[4*(i + 16) + 1] = xr[1 + 18*(1 + i)];
902 zlin[4*(i - 16) + 3] = xr[18*(1 + i)];
903 }
904
905#if MP3D_HAVE_INT_SIMD
906 if (use_simd)
907 {
908 /* The vector kernel still produces all four lanes at once, so it
909 keeps the array form and the store block that goes with it. */
910 int64_t a[4], b[4];
911 mp3d_synth_taps(zlin, w, i, a, b);
912 if (nch == 2)
913 {
914 dstr[(15 - i)*nch] = mp3d_scale_pcm(a[1]);
915 dstr[(17 + i)*nch] = mp3d_scale_pcm(b[1]);
916 dstr[(47 - i)*nch] = mp3d_scale_pcm(a[3]);
917 dstr[(49 + i)*nch] = mp3d_scale_pcm(b[3]);
918 }
919 dstl[(15 - i)*nch] = mp3d_scale_pcm(a[0]);
920 dstl[(17 + i)*nch] = mp3d_scale_pcm(b[0]);
921 dstl[(47 - i)*nch] = mp3d_scale_pcm(a[2]);
922 dstl[(49 + i)*nch] = mp3d_scale_pcm(b[2]);
923 }
924 else
925#endif
926 /* In mono the odd lanes recompute the even ones into the same
927 addresses, so the step is 2 and lanes 1 and 3 never run. */
928 if (nch == 1)
929 {
931 }
932 else
933 {
935 }
936 w += 16;
937 }
938}
int16_t mp3d_sample_t
Definition minimp3.h:227
#define MP3D_SYNTH_CHAIN(st)
static void mp3d_synth_pair(mp3d_sample_t *pcm, int nch, const int32_t *z) FL_NO_EXCEPT
MP3D_LEAF mp3d_sample_t mp3d_scale_pcm(int64_t sample) FL_NO_EXCEPT
fl::i32 int32_t
Definition stdint.h:219

References FL_NO_EXCEPT, mp3d_scale_pcm(), MP3D_SYNTH_CHAIN, and mp3d_synth_pair().

Referenced by mp3d_synth_granule().

+ Here is the call graph for this function:
+ Here is the caller graph for this function: