malachite_float/float/mod.rs
1// Copyright © 2026 Mikhail Hogrefe
2//
3// This file is part of Malachite.
4//
5// Malachite is free software: you can redistribute it and/or modify it under the terms of the GNU
6// Lesser General Public License (LGPL) as published by the Free Software Foundation; either version
7// 3 of the License, or (at your option) any later version. See <https://www.gnu.org/licenses/>.
8
9#[cfg(feature = "test_build")]
10use crate::InnerFloat::Finite;
11use alloc::string::String;
12use core::cmp::Ordering::{self, *};
13use core::ops::Deref;
14#[cfg(feature = "test_build")]
15use malachite_base::num::arithmetic::traits::DivisibleByPowerOf2;
16use malachite_base::num::basic::integers::PrimitiveInt;
17#[cfg(feature = "test_build")]
18use malachite_base::num::logic::traits::SignificantBits;
19use malachite_nz::natural::Natural;
20use malachite_nz::platform::Limb;
21
22/// A floating-point number.
23///
24/// `Float`s are not yet feature-complete, but the functions that are implemented are thoroughly
25/// tested and documented.
26///
27/// `Float`s are similar to the primitive floats defined by the IEEE 754 standard. They include NaN,
28/// $\infty$ and $-\infty$, and positive and negative zero. There is only one NaN; there is no
29/// concept of a NaN payload.
30///
31/// All the finite `Float`s are dyadic rationals (rational numbers whose denominator is a power of
32/// 2). A finite `Float` consists of several fields:
33/// - a sign, which denotes whether the `Float` is positive or negative;
34/// - a significand, which is a [`Natural`] number whose value is equal to the `Float`'s absolute
35/// value multiplied by a power of 2;
36/// - an exponent, which is one more than the floor of the base-2 logarithm of the `Float`'s
37/// absolute value;
38/// - and finally, a precision, which is greater than zero and indicates the number of significant
39/// bits. It is common to think of a `Float` as an approximation of some real number, and the
40/// precision indicates how good the approximation is intended to be.
41///
42/// `Float`s inherit some odd behavior from the IEEE 754 standard regarding comparison. A `NaN` is
43/// not equal to any `Float`, including itself. Positive and negative zero compare as equal, despite
44/// being two distinct values. Additionally, (and this is not IEEE 754's fault), `Float`s with
45/// different precisions compare as equal if they represent the same numeric value.
46///
47/// In many cases, the above behavior is unsatisfactory, so the [`ComparableFloat`] and
48/// [`ComparableFloat`] wrappers are provided. See their documentation for a description of their
49/// comparison behavior.
50///
51/// In documentation, we will use the '$=$' sign to mean that two `Float`s are identical, writing
52/// things like $-\text{NaN}=\text{NaN}$ and $-(0.0) = -0.0$.
53///
54/// The `Float` type is designed to be very similar to the `mpfr_t` type in
55/// [MPFR](https://www.mpfr.org/mpfr-current/mpfr.html#Nomenclature-and-Types), and all Malachite
56/// functions produce exactly the same result as their counterparts in MPFR, unless otherwise noted.
57///
58/// Here are the structural difference between `Float` and `mpfr_t`:
59/// - `Float` can only represent a single `NaN` value, with no sign or payload.
60/// - Only finite, nonzero `Float`s have a significand, precision, and exponent. For other `Float`s,
61/// these concepts are undefined. In particular, unlike `mpfr_t` zeros, `Float` zeros do not have
62/// a precision.
63/// - The types of `mpfr_t` components are configuration- and platform-dependent. The types of
64/// `Float` components are platform-independent, although the `Limb` type is
65/// configuration-dependent: it is `u64` by default, but may be changed to `u32` using the
66/// `--32_bit_limbs` compiler flag. The type of the exponent is always `i32` and the type of the
67/// precision is always `u64`. The `Limb` type only has a visible effect on the functions that
68/// extract the raw significand. All other functions have the same interface when compiled with
69/// either `Limb` type.
70///
71/// `Float`s whose precision is 64 bits or less can be represented without any memory allocation.
72/// (Unless Malachite is compiled with `32_bit_limbs`, in which case the limit is 32).
73#[derive(Clone)]
74#[cfg_attr(feature = "serde", derive(Deserialize, Serialize))]
75#[cfg_attr(feature = "serde", serde(try_from = "SerdeFloat", into = "SerdeFloat"))]
76pub struct Float(pub(crate) InnerFloat);
77
78// A `Float` is serialized as the string `ComparableFloat`'s `Display` writes in base 16, for
79// example `0x1.8#2`. Going through a string rather than the fields is what `Natural` and `Integer`
80// do too, and here it also keeps the encoding independent of `Limb`'s width: the stored significand
81// is padded out to a whole number of limbs, so its digits would differ between 32- and 64-bit
82// builds, while the digits of the value itself do not. Reading it back parses, so a deserialized
83// `Float` cannot violate the invariants that `is_valid` checks.
84#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
85#[cfg_attr(feature = "serde", serde(transparent))]
86pub(crate) struct SerdeFloat(String);
87
88// We want to limit the visibility of the `NaN`, `Zero`, `Infinity`, and `Finite` constructors to
89// within this crate. To do this, we wrap the `InnerFloat` enum in a struct that gets compiled away.
90#[derive(Clone)]
91pub(crate) enum InnerFloat {
92 NaN,
93 Infinity {
94 sign: bool,
95 },
96 Zero {
97 sign: bool,
98 },
99 Finite {
100 sign: bool,
101 exponent: i32,
102 precision: u64,
103 significand: Natural,
104 },
105}
106
107#[inline]
108pub(crate) fn significand_bits(significand: &Natural) -> u64 {
109 significand.limb_count() << Limb::LOG_WIDTH
110}
111
112// Given the `(Float, Ordering)` pair from a computation that rounded toward negative infinity (the
113// `Float` is the rounded-down value and the `Ordering` compares it to the exact result, as the
114// `*_prec_round` functions return), returns the `(floor, ceiling)` pair of `Float`s bracketing the
115// exact result. When the value is inexact the ceiling is the next `Float` above the floor, so it is
116// obtained by incrementing the floor rather than recomputing — much cheaper when the value comes
117// from, for example, a transcendental function.
118pub(crate) fn floor_and_ceiling((floor, o): (Float, Ordering)) -> (Float, Float) {
119 let mut ceiling = floor.clone();
120 if o != Equal {
121 ceiling.increment();
122 }
123 (floor, ceiling)
124}
125
126// `Limb::WIDTH`-derived bit counts, shared across the crate so each is written out only once.
127pub(crate) const WIDTH_MINUS_1: u64 = Limb::WIDTH - 1;
128pub(crate) const TWICE_WIDTH: u64 = Limb::WIDTH << 1;
129
130impl Float {
131 /// The maximum raw exponent of any [`Float`], equal to $2^{30}-1$, or $1,073,741,823$. This is
132 /// one more than the maximum scientific exponent. If we write a [`Float`] as $\pm m2^e$, with
133 /// $1\leq m<2$ and $e$ an integer, we must have $e\leq 2^{30}-2$. If the result of a
134 /// calculation would produce a [`Float`] with an exponent larger than this, then $\pm\infty$,
135 /// the maximum finite float of the specified precision, or the minimum finite float of the
136 /// specified pecision is returned instead, depending on the rounding mode.
137 pub const MAX_EXPONENT: i32 = 0x3fff_ffff;
138 /// The minimum raw exponent of any [`Float`], equal to $-(2^{30}-1)$, or $-1,073,741,823$. This
139 /// is one more than the minimum scientific exponent. If we write a [`Float`] as $\pm m2^e$,
140 /// with $1\leq m<2$ and $e$ an integer, we must have $e\geq -2^{30}$. If the result of a
141 /// calculation would produce a [`Float`] with an exponent smaller than this, then $\pm0.0$, the
142 /// minimum positive finite [`Float`], or the maximum negative finite [`Float`] is returned
143 /// instead, depending on the rounding mode.
144 pub const MIN_EXPONENT: i32 = -Self::MAX_EXPONENT;
145 // Exponent bounds derived from `MIN_EXPONENT`/`MAX_EXPONENT`, written out once and shared by
146 // the exponent-range checks throughout the crate.
147 pub(crate) const MIN_EXPONENT_MINUS_1: i32 = Self::MIN_EXPONENT - 1;
148 pub(crate) const MIN_EXPONENT_PLUS_2: i32 = Self::MIN_EXPONENT + 2;
149 pub(crate) const MIN_EXPONENT_I64: i64 = Self::MIN_EXPONENT as i64;
150 pub(crate) const MIN_EXPONENT_MINUS_2_I64: i64 = (Self::MIN_EXPONENT - 2) as i64;
151 pub(crate) const MAX_EXPONENT_I64: i64 = Self::MAX_EXPONENT as i64;
152 pub(crate) const MAX_EXPONENT_U64: u64 = Self::MAX_EXPONENT as u64;
153
154 #[cfg(feature = "test_build")]
155 pub fn is_valid(&self) -> bool {
156 match self {
157 Self(Finite {
158 precision,
159 significand,
160 exponent,
161 ..
162 }) => {
163 if *precision == 0
164 || !significand.is_valid()
165 || *exponent > Self::MAX_EXPONENT
166 || *exponent < Self::MIN_EXPONENT
167 {
168 return false;
169 }
170 let bits = significand.significant_bits();
171 bits != 0
172 && bits.divisible_by_power_of_2(Limb::LOG_WIDTH)
173 && *precision <= bits
174 && bits - precision < Limb::WIDTH
175 && significand.divisible_by_power_of_2(bits - precision)
176 }
177 _ => true,
178 }
179 }
180}
181
182/// `ComparableFloat` is a wrapper around a [`Float`], taking the [`Float`] by value.
183///
184/// `CompatableFloat` has different comparison behavior than [`Float`]. See the [`Float`]
185/// documentation for its comparison behavior, which is largely derived from the IEEE 754
186/// specification; the `ComparableFloat` behavior, on the other hand, is more mathematically
187/// well-behaved, and respects the principle that equality should be the finest equivalence
188/// relation: that is, that two equal objects should not be different in any way.
189///
190/// To be more specific: when a [`Float`] is wrapped in a `ComparableFloat`,
191/// - `NaN` is not equal to any other [`Float`], but equal to itself;
192/// - Positive and negative zero are not equal to each other;
193/// - Ordering is total. Negative zero is ordered to be smaller than positive zero, and `NaN` is
194/// arbitrarily ordered to be between the two zeros;
195/// - Two [`Float`]s with different precisions but representing the same value are unequal, and the
196/// one with the greater precision is ordered to be larger;
197/// - The hashing function is compatible with equality.
198///
199/// The analogous wrapper for primitive floats is
200/// [`NiceFloat`](malachite_base::num::float::NiceFloat). However,
201/// [`NiceFloat`](malachite_base::num::float::NiceFloat) also facilitates better string conversion,
202/// something that isn't necessary for [`Float`]s
203///
204/// `ComparableFloat` owns its float. This is useful in many cases, for example if you want to use
205/// [`Float`]s as keys in a hash map. In other situations, it is better to use
206/// [`ComparableFloatRef`], which only has a reference to its float.
207// Serialized as its inner `Float`, that is as the same hexadecimal string, since the wrapper adds
208// no data of its own. That the string carries a precision is what makes the round trip preserve
209// everything `ComparableFloat` compares by.
210#[derive(Clone)]
211#[cfg_attr(feature = "serde", derive(Deserialize, Serialize))]
212#[cfg_attr(feature = "serde", serde(transparent))]
213pub struct ComparableFloat(pub Float);
214
215/// `ComparableFloatRef` is a wrapper around a [`Float`], taking the [`Float`] be reference.
216///
217/// See the [`ComparableFloat`] documentation for details.
218#[derive(Clone)]
219pub struct ComparableFloatRef<'a>(pub &'a Float);
220
221impl ComparableFloat {
222 pub const fn as_ref(&self) -> ComparableFloatRef<'_> {
223 ComparableFloatRef(&self.0)
224 }
225}
226
227impl Deref for ComparableFloat {
228 type Target = Float;
229
230 /// Allows a [`ComparableFloat`] to dereference to a [`Float`].
231 ///
232 /// ```
233 /// use malachite_base::num::basic::traits::One;
234 /// use malachite_float::{ComparableFloat, Float};
235 ///
236 /// let x = ComparableFloat(Float::ONE);
237 /// assert_eq!(*x, Float::ONE);
238 /// ```
239 fn deref(&self) -> &Float {
240 &self.0
241 }
242}
243
244impl Deref for ComparableFloatRef<'_> {
245 type Target = Float;
246
247 /// Allows a [`ComparableFloatRef`] to dereference to a [`Float`].
248 ///
249 /// ```
250 /// use malachite_base::num::basic::traits::One;
251 /// use malachite_float::{ComparableFloatRef, Float};
252 ///
253 /// let x = Float::ONE;
254 /// let y = ComparableFloatRef(&x);
255 /// assert_eq!(*y, Float::ONE);
256 /// ```
257 fn deref(&self) -> &Float {
258 self.0
259 }
260}
261
262/// Traits for arithmetic.
263pub mod arithmetic;
264#[macro_use]
265/// Basic traits for working with [`Float`]s.
266pub mod basic;
267/// Traits for comparing [`Float`]s for equality or order.
268pub mod comparison;
269/// Functions that produce [`Float`] approximations of mathematical constants, using a given
270/// precision and rounding mode.
271pub mod constants;
272/// Traits for converting to and from [`Float`]s, including converting [`Float`]s to and from
273/// strings.
274pub mod conversion;
275/// Iterators that generate [`Float`]s without repetition.
276pub mod exhaustive;
277#[cfg(feature = "random")]
278/// Iterators that generate [`Float`]s randomly.
279pub mod random;