FastLED 3.10.6
Loading...
Searching...
No Matches
soft_float.h
Go to the documentation of this file.
1#pragma once
2
3// fl::soft_float -- integer-only IEEE-754 float<->double conversion.
4//
5// On no-FPU targets (Cortex-M0+, ATmega, ...) a plain `static_cast<double>(f)`
6// or `static_cast<float>(d)` anchors libgcc's soft-FP helpers
7// (`__aeabi_f2d`, `__aeabi_d2f`) which transitively pull `__aeabi_dadd /
8// dmul / ddiv / dcmpun / scalbn` and similar. That's ~6 KB of flash on
9// LPC845 / similar 64 KB-flash parts -- ~10% of the budget.
10//
11// This module provides bit-level conversions that perform the IEEE-754
12// widening / narrowing using only integer arithmetic (shift / mask / add).
13// They are correct on every host with IEEE-754 `float` / `double` (which is
14// every supported FastLED target), produce bit-exact results, and emit
15// zero soft-FP symbols even when called from a no-FPU build.
16//
17// Symmetric narrowing-vs-widening contract:
18// - Sign bit copied directly.
19// - +/-zero, +/-inf preserved bit-exactly.
20// - NaN canonicalized to a quiet-NaN with the standard payload
21// (matches what the JSON path already did with its narrowing helper).
22// - Subnormal float -> normal double (widening); double subnormal /
23// small normal -> float subnormal (narrowing).
24//
25// Predecessor: this code lived in `fl/stl/json/types_impl.h` (`json_*`
26// names) for the LPC845 JSON-RPC bring-up (FastLED #3038 / #3076 / #3226).
27// Moved to `fl/math/` as part of FastLED #3022 so non-JSON callers
28// (audio, animation timing, anyone else who needs FP without the libgcc
29// soft-FP tax) can use it directly.
30
31#include "fl/stl/bit_cast.h"
32#include "fl/stl/cstddef.h"
33#include "fl/stl/stdint.h"
34#include "fl/stl/noexcept.h"
35
36namespace fl {
37
38namespace detail {
39
40// Byte-wise reinterpret. Differs from `fl::bit_cast<To, From>` in that it
41// does NOT static-assert on size equality -- it copies
42// `min(sizeof(To), sizeof(From))` bytes and zero-pads the rest. Used here
43// so the `sizeof(double) == sizeof(u32)` (pathological-host) branch can
44// compile alongside the standard 8-byte-double branch without tripping
45// `bit_cast`'s assertion.
46template <typename To, typename From>
47inline To soft_bit_copy(const From& value) FL_NO_EXCEPT {
48 To out = 0;
49 const char* src = fl::bit_cast_ptr<const char>(&value);
50 char* dst = fl::bit_cast_ptr<char>(&out);
51 const fl::size n = sizeof(To) < sizeof(From) ? sizeof(To) : sizeof(From);
52 for (fl::size i = 0; i < n; ++i) {
53 dst[i] = src[i];
54 }
55 return out;
56}
57
59 if (shift <= 0) {
60 return static_cast<u32>(value);
61 }
62 if (shift >= 64) {
63 return 0;
64 }
65 const u64 half = u64(1) << (shift - 1);
66 const u64 mask = (u64(1) << shift) - 1;
67 const u64 truncated = value >> shift;
68 const u64 remainder = value & mask;
69 const bool round_up = remainder > half || (remainder == half && (truncated & 1));
70 return static_cast<u32>(truncated + (round_up ? 1 : 0));
71}
72
74 int bit = -1;
75 while (value != 0) {
76 value >>= 1;
77 ++bit;
78 }
79 return bit;
80}
81
83 u32 sign, u64 magnitude, int binary_exp) FL_NO_EXCEPT {
84 if (magnitude == 0) {
85 return sign;
86 }
87
88 const int top_bit = soft_floor_log2_u64(magnitude);
89 int exponent = top_bit + binary_exp;
90 if (exponent > 127) {
91 return sign | 0x7F800000u;
92 }
93
94 if (exponent >= -126) {
95 const int shift = top_bit - 23;
96 u32 significand = shift > 0
97 ? soft_round_shift_right_u64(magnitude, shift)
98 : static_cast<u32>(magnitude << -shift);
99 if (significand >= 0x01000000u) {
100 significand >>= 1;
101 ++exponent;
102 if (exponent > 127) {
103 return sign | 0x7F800000u;
104 }
105 }
106 return sign | (static_cast<u32>(exponent + 127) << 23) |
107 (significand & 0x007FFFFFu);
108 }
109
110 const int subnormal_shift = -(binary_exp + 149);
111 u32 mantissa = 0;
112 if (subnormal_shift > 0) {
113 mantissa = soft_round_shift_right_u64(magnitude, subnormal_shift);
114 } else {
115 const int left_shift = -subnormal_shift;
116 if (left_shift >= 64 || magnitude > (u64(-1) >> left_shift)) {
117 return sign | 0x7F800000u;
118 }
119 const u64 shifted = magnitude << left_shift;
120 mantissa = shifted > 0x007FFFFFu ? 0x00800000u : static_cast<u32>(shifted);
121 }
122 if (mantissa >= 0x00800000u) {
123 return sign | 0x00800000u;
124 }
125 return sign | mantissa;
126}
127
128} // namespace detail
129
130// Narrow a 64-bit double's bit pattern to the bit pattern of the closest
131// 32-bit float, using only integer arithmetic. Round-to-nearest-even.
133 const u32 sign = (bits >> 63) ? 0x80000000u : 0u;
134 const u32 exp_bits = static_cast<u32>((bits >> 52) & 0x7FFu);
135 const u64 mantissa_bits = bits & 0x000FFFFFFFFFFFFFull;
136 if (exp_bits == 0x7FFu) {
137 return mantissa_bits == 0 ? sign | 0x7F800000u : sign | 0x7FC00000u;
138 }
139 if (exp_bits == 0) {
140 return detail::soft_float_bits_from_scaled_u64(sign, mantissa_bits, -1074);
141 }
143 sign, mantissa_bits | 0x0010000000000000ull,
144 static_cast<int>(exp_bits) - 1075);
145}
146
147// Widen a 32-bit float's bit pattern to the bit pattern of the same value
148// as a 64-bit double, using only integer arithmetic. Lossless: every finite
149// float has an exact double representation, and the round-trip
150// `double_bits_from_float_bits(float_bits_from_double_bits(d))` is the
151// identity for any `d` that was a widened float.
153 const u64 sign = (bits >> 31) ? 0x8000000000000000ull : 0ull;
154 const u32 exp_bits = (bits >> 23) & 0xFFu;
155 const u32 mantissa_bits = bits & 0x007FFFFFu;
156 if (exp_bits == 0xFFu) {
157 return mantissa_bits == 0 ? sign | 0x7FF0000000000000ull
158 : sign | 0x7FF8000000000000ull;
159 }
160 if (exp_bits == 0) {
161 if (mantissa_bits == 0) {
162 return sign;
163 }
164 // Subnormal float -- always a normal double. Walk the implicit bit
165 // up to position 23, adjust exponent accordingly.
166 u32 m = mantissa_bits;
167 int shift = 0;
168 while ((m & 0x00800000u) == 0) {
169 m <<= 1;
170 ++shift;
171 }
172 const int double_exp = 1 - 127 - shift + 1023;
173 const u64 mantissa_d = static_cast<u64>(m & 0x007FFFFFu) << 29;
174 return sign | (static_cast<u64>(double_exp) << 52) | mantissa_d;
175 }
176 const int double_exp = static_cast<int>(exp_bits) - 127 + 1023;
177 const u64 mantissa_d = static_cast<u64>(mantissa_bits) << 29;
178 return sign | (static_cast<u64>(double_exp) << 52) | mantissa_d;
179}
180
181// Convenience: narrow a `double` value to a `float` via the integer codec.
183 if (sizeof(double) == sizeof(u32)) {
185 }
186 const u32 narrowed_bits =
188 return detail::soft_bit_copy<float>(narrowed_bits);
189}
190
191// Convenience: widen a `float` value to a `double` via the integer codec.
193 if (sizeof(double) == sizeof(u32)) {
195 }
196 const u64 widened_bits =
198 return detail::soft_bit_copy<double>(widened_bits);
199}
200
201} // namespace fl
int soft_floor_log2_u64(u64 value) FL_NO_EXCEPT
Definition soft_float.h:73
u32 soft_float_bits_from_scaled_u64(u32 sign, u64 magnitude, int binary_exp) FL_NO_EXCEPT
Definition soft_float.h:82
u32 soft_round_shift_right_u64(u64 value, int shift) FL_NO_EXCEPT
Definition soft_float.h:58
To soft_bit_copy(const From &value) FL_NO_EXCEPT
Definition soft_float.h:47
Compile-time linker keep-alive hook for a single fl::Bus.
Definition bus_info.h:57
constexpr int type_rank< T >::value
double soft_float_to_double(float value) FL_NO_EXCEPT
Definition soft_float.h:192
u32 soft_float_bits_from_double_bits(u64 bits) FL_NO_EXCEPT
Definition soft_float.h:132
u64 soft_double_bits_from_float_bits(u32 bits) FL_NO_EXCEPT
Definition soft_float.h:152
InputGamut g FL_NO_EXCEPT
Definition rgbw.h:121
float soft_double_to_float(double value) FL_NO_EXCEPT
Definition soft_float.h:182
To * bit_cast_ptr(void *storage) FL_NO_EXCEPT
Definition bit_cast.h:60
constexpr enable_if< is_fixed_point< T >::value, int >::type sign(T x) FL_NO_EXCEPT
Base definition for an LED controller.
Definition crgb.hpp:179
fl::u64 u64
Definition stdint.h:220