1use crate::decimal::{Decimal, Fraction};
43
44mod arith;
45mod dec;
46
47pub use crate::float::arith::Integral;
48
49#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
57pub enum Format {
58 Half,
60 BFloat16,
63 Single,
65 Double,
67 X87Extended,
70 Quad,
73 DoubleDouble,
86 Decimal32,
93 Decimal64,
95 Decimal128,
97}
98
99const fn not_ieee() -> ! {
105 panic!(
106 "the double-double format is a pair of doubles and the decimal formats count in tens, so \
107 neither has a binary precision, exponent range or significand field to ask about"
108 )
109}
110
111impl Format {
112 #[must_use]
115 pub const fn name(self) -> &'static str {
116 match self {
117 Format::Half => "f16",
118 Format::BFloat16 => "bf16",
119 Format::Single => "f32",
120 Format::Double => "f64",
121 Format::X87Extended => "f80",
122 Format::Quad => "f128",
123 Format::DoubleDouble => "ppc-f128",
124 Format::Decimal32 => "d32",
125 Format::Decimal64 => "d64",
126 Format::Decimal128 => "d128",
127 }
128 }
129
130 #[must_use]
132 pub fn from_name(name: &str) -> Option<Self> {
133 Some(match name {
134 "f16" => Format::Half,
135 "bf16" => Format::BFloat16,
136 "f32" => Format::Single,
137 "f64" => Format::Double,
138 "f80" => Format::X87Extended,
139 "f128" => Format::Quad,
140 "ppc-f128" => Format::DoubleDouble,
141 "d32" => Format::Decimal32,
142 "d64" => Format::Decimal64,
143 "d128" => Format::Decimal128,
144 _ => return None,
145 })
146 }
147
148 #[must_use]
158 pub const fn is_ieee(self) -> bool {
159 !matches!(
160 self,
161 Format::DoubleDouble | Format::Decimal32 | Format::Decimal64 | Format::Decimal128
162 )
163 }
164
165 #[must_use]
175 pub const fn decimal(self) -> Option<crate::dfp::Width> {
176 match self {
177 Format::Decimal32 => Some(crate::dfp::Width::D32),
178 Format::Decimal64 => Some(crate::dfp::Width::D64),
179 Format::Decimal128 => Some(crate::dfp::Width::D128),
180 _ => None,
181 }
182 }
183
184 #[must_use]
190 pub const fn precision(self) -> u32 {
191 match self {
192 Format::Half => 11,
193 Format::BFloat16 => 8,
194 Format::Single => 24,
195 Format::Double => 53,
196 Format::X87Extended => 64,
197 Format::Quad => 113,
198 Format::DoubleDouble | Format::Decimal32 | Format::Decimal64 | Format::Decimal128 => {
199 not_ieee()
200 }
201 }
202 }
203
204 #[must_use]
210 pub const fn max_exponent(self) -> i32 {
211 match self {
212 Format::Half => 15,
213 Format::BFloat16 | Format::Single => 127,
214 Format::Double => 1023,
215 Format::X87Extended | Format::Quad => 16383,
216 Format::DoubleDouble | Format::Decimal32 | Format::Decimal64 | Format::Decimal128 => {
217 not_ieee()
218 }
219 }
220 }
221
222 #[must_use]
228 pub const fn min_exponent(self) -> i32 {
229 1 - self.max_exponent()
230 }
231
232 #[must_use]
239 pub const fn width(self) -> u32 {
240 match self {
241 Format::Half | Format::BFloat16 => 16,
242 Format::Single | Format::Decimal32 => 32,
243 Format::Double | Format::Decimal64 => 64,
244 Format::X87Extended => 80,
245 Format::Quad | Format::DoubleDouble | Format::Decimal128 => 128,
246 }
247 }
248
249 #[must_use]
255 pub const fn has_explicit_integer_bit(self) -> bool {
256 match self {
257 Format::X87Extended => true,
258 Format::Half | Format::BFloat16 | Format::Single | Format::Double | Format::Quad => {
259 false
260 }
261 Format::DoubleDouble | Format::Decimal32 | Format::Decimal64 | Format::Decimal128 => {
262 not_ieee()
263 }
264 }
265 }
266
267 const fn exponent_bits(self) -> u32 {
269 self.width() - self.significand_bits() - 1
270 }
271
272 const fn significand_bits(self) -> u32 {
274 if self.has_explicit_integer_bit() { self.precision() } else { self.precision() - 1 }
275 }
276
277 const fn max_decimal_exponent(self) -> i32 {
283 (self.max_exponent() + 1) * 30103 / 100000 + 2
284 }
285
286 const fn min_decimal_exponent(self) -> i32 {
288 (self.min_exponent() - self.precision() as i32) * 30103 / 100000 - 2
289 }
290}
291
292#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
298pub struct Status(u8);
299
300impl Status {
301 pub const NONE: Status = Status(0);
303 pub const INEXACT: Status = Status(1);
305 pub const OVERFLOW: Status = Status(2);
307 pub const UNDERFLOW: Status = Status(4);
309 pub const INVALID: Status = Status(8);
311 pub const DIVIDE_BY_ZERO: Status = Status(16);
313
314 #[inline]
316 #[must_use]
317 pub const fn has(self, other: Status) -> bool {
318 self.0 & other.0 == other.0
319 }
320
321 #[inline]
323 #[must_use]
324 pub const fn with(self, other: Status) -> Status {
325 Status(self.0 | other.0)
326 }
327
328 #[inline]
330 #[must_use]
331 pub const fn is_none(self) -> bool {
332 self.0 == 0
333 }
334}
335
336#[derive(Debug, Clone, Copy, PartialEq, Eq)]
341pub enum ParseError {
342 NoDigits,
344 NoExponentDigits,
346 Invalid,
348}
349
350#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
352enum Category {
353 Zero,
354 Finite,
355 Infinite,
356 Nan,
357}
358
359#[derive(Debug, Clone, Copy, PartialEq, Eq)]
364pub struct Float {
365 format: Format,
366 category: Category,
367 sign: bool,
368 exponent: i32,
369 significand: u128,
370}
371
372const fn ieee(format: Format) -> Format {
378 if format.is_ieee() { format } else { not_ieee() }
379}
380
381const fn carried(format: Format) -> Format {
384 if format.is_ieee() || format.decimal().is_some() { format } else { not_ieee() }
385}
386
387impl Float {
388 #[must_use]
394 pub const fn zero(format: Format, sign: bool) -> Float {
395 Float {
396 format: carried(format),
397 category: Category::Zero,
398 sign,
399 exponent: 0,
400 significand: 0,
401 }
402 }
403
404 #[must_use]
410 pub const fn infinity(format: Format, sign: bool) -> Float {
411 Float {
412 format: carried(format),
413 category: Category::Infinite,
414 sign,
415 exponent: 0,
416 significand: 0,
417 }
418 }
419
420 #[must_use]
432 pub const fn smallest_normal(format: Format, sign: bool) -> Float {
433 Float {
434 format: ieee(format),
435 category: Category::Finite,
436 sign,
437 exponent: format.min_exponent(),
438 significand: 1u128 << (format.precision() - 1),
439 }
440 }
441
442 #[must_use]
458 pub const fn nan_with(format: Format, sign: bool, quiet: bool, payload: u128) -> Float {
459 if format.decimal().is_some() {
460 let exponent = if quiet { 0 } else { 1 };
461 return Float { format, category: Category::Nan, sign, exponent, significand: 0 };
462 }
463 let format = ieee(format);
464 let mut significand = payload & (Float::quiet_bit(format) - 1);
465 if quiet {
466 significand |= Float::quiet_bit(format);
467 } else if significand == 0 {
468 significand = Float::quiet_bit(format) >> 1;
469 }
470 Float {
471 format,
472 category: Category::Nan,
473 sign,
474 exponent: 0,
475 significand: significand | Float::leading_bit(format),
476 }
477 }
478
479 const fn quiet_bit(format: Format) -> u128 {
482 1u128 << (format.precision() - 2)
483 }
484
485 const fn leading_bit(format: Format) -> u128 {
488 if format.has_explicit_integer_bit() { 1u128 << (format.precision() - 1) } else { 0 }
489 }
490
491 #[must_use]
493 pub const fn format(self) -> Format {
494 self.format
495 }
496
497 #[must_use]
499 pub const fn is_negative(self) -> bool {
500 self.sign
501 }
502
503 #[must_use]
505 pub const fn is_zero(self) -> bool {
506 matches!(self.category, Category::Zero)
507 }
508
509 #[must_use]
511 pub const fn is_infinite(self) -> bool {
512 matches!(self.category, Category::Infinite)
513 }
514
515 #[must_use]
517 pub const fn is_finite(self) -> bool {
518 matches!(self.category, Category::Zero | Category::Finite)
519 }
520
521 #[must_use]
526 pub fn is_normal(self) -> bool {
527 if let Some(width) = self.format.decimal() {
528 return self.decimal_is_normal(width);
529 }
530 matches!(self.category, Category::Finite)
531 && self.significand >> (self.format.precision() - 1) != 0
532 }
533
534 pub fn parse(text: &str, format: Format) -> Result<(Float, Status), ParseError> {
552 if let Some(width) = format.decimal() {
553 let (bits, status) = crate::dfp::parse(text, width)?;
554 return Ok((Float::decimal_from_bits(format, width, bits), status));
555 }
556 let format = ieee(format);
557 let bytes = text.as_bytes();
558 let (sign, rest) = match bytes.first() {
559 Some(b'-') => (true, &bytes[1..]),
560 Some(b'+') => (false, &bytes[1..]),
561 _ => (false, bytes),
562 };
563 if rest.len() > 1 && rest[0] == b'0' && rest[1] | 32 == b'x' {
564 hexadecimal(&rest[2..], sign, format)
565 } else {
566 decimal(rest, sign, format)
567 }
568 }
569
570 #[must_use]
575 pub fn to_bits(self) -> u128 {
576 let format = self.format;
577 if let Some(width) = format.decimal() {
578 return self.decimal_bits(width);
579 }
580 let significand_mask = (1u128 << format.significand_bits()) - 1;
581 let (exponent_field, significand_field) = match self.category {
582 Category::Zero => (0, 0),
583 Category::Infinite => (
584 (1u128 << format.exponent_bits()) - 1,
585 if format.has_explicit_integer_bit() {
586 1u128 << (format.precision() - 1)
587 } else {
588 0
589 },
590 ),
591 Category::Nan => ((1u128 << format.exponent_bits()) - 1, self.significand),
594 Category::Finite => {
595 let subnormal = self.significand >> (format.precision() - 1) == 0;
596 let field =
597 if subnormal { 0 } else { (self.exponent + format.max_exponent()) as u128 };
598 (field, self.significand & significand_mask)
599 }
600 };
601 let sign = u128::from(self.sign) << (format.width() - 1);
602 sign | (exponent_field << format.significand_bits()) | significand_field
603 }
604
605 #[must_use]
616 pub fn from_bits(format: Format, bits: u128) -> Float {
617 if let Some(width) = format.decimal() {
618 return Float::decimal_from_bits(format, width, bits);
619 }
620 let format = ieee(format);
621 let significand_bits = format.significand_bits();
622 let sign = (bits >> (format.width() - 1)) & 1 == 1;
623 let exponent_field =
624 ((bits >> significand_bits) & ((1u128 << format.exponent_bits()) - 1)) as i32;
625 let stored = bits & ((1u128 << significand_bits) - 1);
626 if exponent_field == (1 << format.exponent_bits()) - 1 {
627 let fraction = stored & ((1u128 << (format.precision() - 1)) - 1);
630 if fraction == 0 {
631 return Float::infinity(format, sign);
632 }
633 return Float {
634 format,
635 category: Category::Nan,
636 sign,
637 exponent: 0,
638 significand: stored,
639 };
640 }
641 let implicit = if format.has_explicit_integer_bit() || exponent_field == 0 {
642 0
643 } else {
644 1u128 << (format.precision() - 1)
645 };
646 let significand = stored | implicit;
647 if significand == 0 {
648 return Float::zero(format, sign);
649 }
650 let exponent = if exponent_field == 0 {
651 format.min_exponent()
652 } else {
653 exponent_field - format.max_exponent()
654 };
655 Float { format, category: Category::Finite, sign, exponent, significand }
656 }
657
658 #[must_use]
675 pub fn to_hex(self) -> String {
676 if self.format.decimal().is_some() {
677 return self.decimal_spelling();
678 }
679 let sign = if self.sign { "-" } else { "" };
680 match self.category {
681 Category::Nan => format!("{sign}nan"),
682 Category::Infinite => format!("{sign}0x1p+{}", self.format.max_exponent() + 1),
683 Category::Zero => format!("{sign}0x0p+0"),
684 Category::Finite => {
685 let mut significand = self.significand;
686 let mut exponent = self.exponent - (self.format.precision() as i32 - 1);
687 while significand & 0xf == 0 {
688 significand >>= 4;
689 exponent += 4;
690 }
691 format!("{sign}0x{significand:x}p{exponent:+}")
692 }
693 }
694 }
695}
696
697fn decimal(bytes: &[u8], sign: bool, format: Format) -> Result<(Float, Status), ParseError> {
699 let mut digits = Vec::new();
700 let mut integer_digits = 0i32;
701 let mut seen_point = false;
702 let mut seen_digit = false;
703 let mut index = 0;
704 while index < bytes.len() {
705 match bytes[index] {
706 byte @ b'0'..=b'9' => {
707 digits.push(byte - b'0');
708 if !seen_point {
709 integer_digits += 1;
710 }
711 seen_digit = true;
712 }
713 b'\'' => {}
714 b'.' if !seen_point => seen_point = true,
715 b'e' | b'E' => break,
716 _ => return Err(ParseError::Invalid),
717 }
718 index += 1;
719 }
720 if !seen_digit {
721 return Err(ParseError::NoDigits);
722 }
723 let mut point = integer_digits;
724 if index < bytes.len() {
725 point = point.saturating_add(exponent_of(&bytes[index + 1..])?);
726 }
727 Ok(convert(Decimal::new(digits, point), sign, format))
728}
729
730fn hexadecimal(bytes: &[u8], sign: bool, format: Format) -> Result<(Float, Status), ParseError> {
732 let mut significand: u128 = 0;
733 let mut exponent = 0i32;
734 let mut sticky = false;
735 let mut seen_point = false;
736 let mut seen_digit = false;
737 let mut index = 0;
738 while index < bytes.len() {
739 let byte = bytes[index];
740 let digit = match byte {
741 b'0'..=b'9' => byte - b'0',
742 b'a'..=b'f' => byte - b'a' + 10,
743 b'A'..=b'F' => byte - b'A' + 10,
744 b'\'' => {
745 index += 1;
746 continue;
747 }
748 b'.' if !seen_point => {
749 seen_point = true;
750 index += 1;
751 continue;
752 }
753 b'p' | b'P' => break,
754 _ => return Err(ParseError::Invalid),
755 };
756 seen_digit = true;
757 if significand.leading_zeros() >= 4 {
758 significand = (significand << 4) | u128::from(digit);
759 if seen_point {
760 exponent -= 4;
761 }
762 } else {
763 sticky |= digit != 0;
766 if !seen_point {
767 exponent += 4;
768 }
769 }
770 index += 1;
771 }
772 if !seen_digit {
773 return Err(ParseError::NoDigits);
774 }
775 if index < bytes.len() {
776 exponent = exponent.saturating_add(exponent_of(&bytes[index + 1..])?);
777 }
778 Ok(round(significand, exponent, sticky, sign, format))
779}
780
781fn exponent_of(bytes: &[u8]) -> Result<i32, ParseError> {
783 let (negative, digits) = match bytes.first() {
784 Some(b'-') => (true, &bytes[1..]),
785 Some(b'+') => (false, &bytes[1..]),
786 _ => (false, bytes),
787 };
788 if digits.is_empty() {
789 return Err(ParseError::NoExponentDigits);
790 }
791 let mut value = 0i32;
792 for &byte in digits {
793 if byte == b'\'' {
794 continue;
795 }
796 if !byte.is_ascii_digit() {
797 return Err(ParseError::Invalid);
798 }
799 value = value.saturating_mul(10).saturating_add(i32::from(byte - b'0'));
802 }
803 Ok(if negative { -value } else { value })
804}
805
806fn convert(mut value: Decimal, sign: bool, format: Format) -> (Float, Status) {
808 if value.is_zero() {
809 return (Float::zero(format, sign), Status::NONE);
810 }
811 if value.point() > format.max_decimal_exponent() {
812 return (Float::infinity(format, sign), Status::OVERFLOW.with(Status::INEXACT));
813 }
814 if value.point() < format.min_decimal_exponent() {
815 return (Float::zero(format, sign), Status::UNDERFLOW.with(Status::INEXACT));
816 }
817
818 let mut exponent = 0i32;
822 loop {
823 let point = value.point();
824 if point > 1 || (point == 1 && value.first_digit() >= 2) {
825 let step = binary_digits(point - 1).clamp(1, 60);
826 value.shift(-step);
827 exponent += step;
828 } else if point < 1 {
829 let step = (1 + binary_digits(-point)).clamp(1, 60);
830 value.shift(step);
831 exponent -= step;
832 } else {
833 break;
834 }
835 }
836
837 let precision = format.precision() as i32;
840 let scale = (exponent - precision + 1).max(format.min_exponent() - precision + 1);
841 value.shift(exponent - scale);
842 let (integer, fraction) = value.round_to_u128();
843 let rounded = match fraction {
844 Fraction::Zero | Fraction::BelowHalf => integer,
845 Fraction::Half => integer + (integer & 1),
846 Fraction::AboveHalf => integer + 1,
847 };
848 finish(rounded, scale, fraction != Fraction::Zero, sign, format)
849}
850
851const fn binary_digits(decimal: i32) -> i32 {
853 decimal * 33219 / 10000
854}
855
856fn round(
859 significand: u128,
860 exponent: i32,
861 sticky: bool,
862 sign: bool,
863 format: Format,
864) -> (Float, Status) {
865 if significand == 0 {
866 return (Float::zero(format, sign), Status::NONE);
867 }
868 let precision = format.precision() as i32;
869 let leading = (128 - significand.leading_zeros()) as i32;
870 let scale = (exponent + leading - precision).max(format.min_exponent() - precision + 1);
871 let mut sticky = sticky;
872 let (integer, half) = if scale <= exponent {
873 (significand << (exponent - scale), false)
874 } else {
875 let drop = (scale - exponent) as u32;
876 if drop >= 128 {
877 sticky = true;
878 (0, false)
879 } else {
880 let half = (significand >> (drop - 1)) & 1 == 1;
881 sticky |= drop > 1 && significand & ((1u128 << (drop - 1)) - 1) != 0;
882 (significand >> drop, half)
883 }
884 };
885 let rounded = if half && (sticky || integer & 1 == 1) { integer + 1 } else { integer };
886 finish(rounded, scale, half || sticky, sign, format)
887}
888
889fn finish(
892 significand: u128,
893 scale: i32,
894 inexact: bool,
895 sign: bool,
896 format: Format,
897) -> (Float, Status) {
898 let precision = format.precision();
899 let mut significand = significand;
900 let mut scale = scale;
901 if significand >> precision != 0 {
902 significand >>= 1;
904 scale += 1;
905 }
906 let mut status = if inexact { Status::INEXACT } else { Status::NONE };
907 if significand == 0 {
908 return (Float::zero(format, sign), status.with(Status::UNDERFLOW));
909 }
910 let exponent = scale + precision as i32 - 1;
911 if exponent > format.max_exponent() {
912 return (
913 Float::infinity(format, sign),
914 status.with(Status::OVERFLOW).with(Status::INEXACT),
915 );
916 }
917 let normal = significand >> (precision - 1) != 0;
918 if !normal && inexact {
919 status = status.with(Status::UNDERFLOW);
920 }
921 let exponent = if normal { exponent } else { format.min_exponent() };
922 (Float { format, category: Category::Finite, sign, exponent, significand }, status)
923}
924
925#[cfg(test)]
926mod tests {
927 use super::*;
928
929 fn double(text: &str) -> u128 {
931 Float::parse(text, Format::Double).expect("a number").0.to_bits()
932 }
933
934 fn single(text: &str) -> u128 {
936 Float::parse(text, Format::Single).expect("a number").0.to_bits()
937 }
938
939 #[test]
940 fn the_ordinary_numbers_land_where_the_host_would_put_them() {
941 for text in ["0", "1", "2", "0.5", "1.5", "3.14159", "2.718281828459045", "100", "1e10"] {
942 let host = text.parse::<f64>().expect("a number Rust reads too");
943 assert_eq!(double(text), u128::from(host.to_bits()), "{text}");
944 }
945 }
946
947 #[test]
948 fn a_number_that_needs_the_last_bit_rounded_gets_it_right() {
949 let hard = [
952 "0.1",
953 "0.3",
954 "2.2250738585072011e-308",
955 "2.2250738585072014e-308",
956 "1.7976931348623157e308",
957 "4.9406564584124654e-324",
958 "5e-324",
959 "8.98846567431158e307",
960 "9007199254740993",
961 "123456789012345678901234567890",
962 "1.000000000000000000000000000000000000000000000000000000000000000001",
963 "7.8459735791271921e65",
964 "3.518437208883201171875e13",
965 "0.500000000000000166533453693773481063544750213623046875",
966 ];
967 for text in hard {
968 let host = text.parse::<f64>().expect("a number Rust reads too");
969 assert_eq!(double(text), u128::from(host.to_bits()), "{text}");
970 }
971 }
972
973 #[test]
974 fn the_number_that_takes_seven_hundred_and_sixty_seven_digits() {
975 let text = concat!(
978 "2.47032822920623272088284396434110686182529901307162382",
979 "35378852574870103599108683372845652890455735483022221802",
980 "58573249056416711547735232764105795166208503595426876755",
981 "62317084535693494535245273750735013572761315046354601316",
982 "12127849863326369238975694273040488011871029093711789936",
983 "42245692702737764465109076580131048946378905599180391359",
984 "70011386455512221706120629864144453927884519445934871524",
985 "63344875888932891414823975864211858166195965106373837732",
986 "34435703331457550505022232309998195892058070506176382679",
987 "16323484472119097902806154870514036458498974142754747141",
988 "39683784321102080606305920253373777969877864922227306716",
989 "01324339457879181214233820577228206278891620001855078759",
990 "16278352090142077553206262229158550205643778244387017277",
991 "94459649305087139089301871550805125768938177360937844105",
992 "63661045147381814281647890691181239104545396303476425117",
993 "7562185422741845851144691421326303120484712594187004993e-324"
994 );
995 let host = text.parse::<f64>().expect("a number Rust reads too");
996 assert_eq!(double(text), u128::from(host.to_bits()));
997 }
998
999 #[test]
1000 fn a_sweep_of_random_numbers_agrees_with_rust_in_every_bit() {
1001 let mut state = 0x2545_f491_4f6c_dd1du64;
1005 for _ in 0..4000 {
1006 state = state.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
1007 let digits = state >> 11;
1008 let exponent = (state % 600) as i32 - 300;
1009 let text = format!("{digits}e{exponent}");
1010 let host = text.parse::<f64>().expect("a number Rust reads too");
1011 assert_eq!(double(&text), u128::from(host.to_bits()), "{text}");
1012 let host = text.parse::<f32>().expect("a number Rust reads too");
1013 assert_eq!(single(&text), u128::from(host.to_bits()), "{text} as a float");
1014 }
1015 }
1016
1017 #[test]
1018 fn the_ends_of_the_range_are_an_infinity_and_a_zero() {
1019 let (value, status) = Float::parse("1e400", Format::Double).expect("a number");
1020 assert!(value.is_infinite() && status.has(Status::OVERFLOW));
1021 let (value, status) = Float::parse("1e-400", Format::Double).expect("a number");
1022 assert!(value.is_zero() && status.has(Status::UNDERFLOW) && status.has(Status::INEXACT));
1023 let (value, status) = Float::parse("1.7976931348623157e308", Format::Double).expect("one");
1025 assert!(value.is_finite() && !status.has(Status::OVERFLOW));
1026 let (value, _) = Float::parse("1.8e308", Format::Double).expect("a number");
1027 assert!(value.is_infinite());
1028 assert_eq!(double("2.4e-324"), u128::from((0f64).to_bits()));
1030 assert_eq!(double("2.5e-324"), 1);
1031 }
1032
1033 #[test]
1034 fn a_number_that_is_exactly_what_was_written_says_so() {
1035 assert!(Float::parse("1", Format::Double).expect("a number").1.is_none());
1036 assert!(Float::parse("0.5", Format::Double).expect("a number").1.is_none());
1037 assert!(Float::parse("0.1", Format::Double).expect("a number").1.has(Status::INEXACT));
1038 let (_, status) = Float::parse("1e-320", Format::Double).expect("a number");
1040 assert!(status.has(Status::INEXACT) && status.has(Status::UNDERFLOW));
1041 }
1042
1043 #[test]
1044 fn a_hexadecimal_constant_is_exact_and_needs_no_scaling() {
1045 assert_eq!(double("0x1p0"), u128::from((1f64).to_bits()));
1046 assert_eq!(double("0x1.8p1"), u128::from((3f64).to_bits()));
1047 assert_eq!(double("0x1p-1074"), 1);
1048 assert_eq!(double("0xa.bp-4"), u128::from((0.66796875f64).to_bits()));
1049 assert_eq!(double("0X1.FFFFFFFFFFFFFP+1023"), u128::from(f64::MAX.to_bits()));
1050 assert!(Float::parse("0x1p0", Format::Double).expect("a number").1.is_none());
1051 let (_, status) = Float::parse("0x1.00000000000008p0", Format::Double).expect("a number");
1053 assert!(status.has(Status::INEXACT));
1054 assert_eq!(double("0x1.00000000000008p0"), u128::from((1f64).to_bits()));
1055 assert_eq!(double("0x1.00000000000018p0"), u128::from((1f64).to_bits() + 2));
1056 }
1057
1058 #[test]
1059 fn digit_separators_are_not_part_of_the_number() {
1060 assert_eq!(double("1'000.000'1"), double("1000.0001"));
1061 assert_eq!(double("0x1'0p0"), double("16.0"));
1062 assert_eq!(double("1e1'0"), double("1e10"));
1063 }
1064
1065 #[test]
1066 fn a_spelling_that_is_not_a_number_says_which_way_it_is_wrong() {
1067 assert_eq!(Float::parse("", Format::Double), Err(ParseError::NoDigits));
1068 assert_eq!(Float::parse(".", Format::Double), Err(ParseError::NoDigits));
1069 assert_eq!(Float::parse("1e", Format::Double), Err(ParseError::NoExponentDigits));
1070 assert_eq!(Float::parse("1e+", Format::Double), Err(ParseError::NoExponentDigits));
1071 assert_eq!(Float::parse("0x1p", Format::Double), Err(ParseError::NoExponentDigits));
1072 assert_eq!(Float::parse("0xp1", Format::Double), Err(ParseError::NoDigits));
1073 assert_eq!(Float::parse("1x0", Format::Double), Err(ParseError::Invalid));
1074 }
1075
1076 #[test]
1077 fn a_sign_is_accepted_although_a_c_constant_never_has_one() {
1078 let (value, _) = Float::parse("-1.5", Format::Double).expect("a number");
1079 assert!(value.is_negative());
1080 assert_eq!(value.to_bits(), u128::from((-1.5f64).to_bits()));
1081 let (value, _) = Float::parse("-0.0", Format::Double).expect("a number");
1082 assert!(value.is_zero() && value.is_negative());
1083 assert_eq!(value.to_bits(), u128::from((-0.0f64).to_bits()));
1084 }
1085
1086 #[test]
1087 fn every_format_says_how_wide_its_fields_are() {
1088 for format in [
1089 Format::Half,
1090 Format::BFloat16,
1091 Format::Single,
1092 Format::Double,
1093 Format::X87Extended,
1094 Format::Quad,
1095 ] {
1096 assert_eq!(
1097 format.exponent_bits() + format.significand_bits() + 1,
1098 format.width(),
1099 "{format:?}"
1100 );
1101 assert_eq!(format.min_exponent(), 1 - format.max_exponent());
1102 }
1103 assert_eq!(Format::Half.exponent_bits(), 5);
1104 assert_eq!(Format::BFloat16.exponent_bits(), 8);
1105 assert_eq!(Format::Single.exponent_bits(), 8);
1106 assert_eq!(Format::Double.exponent_bits(), 11);
1107 assert_eq!(Format::X87Extended.exponent_bits(), 15);
1108 assert_eq!(Format::Quad.exponent_bits(), 15);
1109 }
1110
1111 #[test]
1112 fn a_number_survives_a_trip_through_its_encoding() {
1113 for format in [
1114 Format::Half,
1115 Format::BFloat16,
1116 Format::Single,
1117 Format::Double,
1118 Format::X87Extended,
1119 Format::Quad,
1120 ] {
1121 for text in ["0", "-0", "1", "-1.5", "3.14159", "1e-5", "65504", "0x1p-20"] {
1122 let (value, _) = Float::parse(text, format).expect("a number");
1123 let bits = value.to_bits();
1124 assert_eq!(Float::from_bits(format, bits).to_bits(), bits, "{text} in {format:?}");
1125 }
1126 assert_eq!(
1127 Float::from_bits(format, Float::infinity(format, false).to_bits()).to_bits(),
1128 Float::infinity(format, false).to_bits()
1129 );
1130 }
1131 }
1132
1133 #[test]
1134 fn a_hexadecimal_spelling_reads_back_as_the_number_it_came_from() {
1135 for format in [
1136 Format::Half,
1137 Format::BFloat16,
1138 Format::Single,
1139 Format::Double,
1140 Format::X87Extended,
1141 Format::Quad,
1142 ] {
1143 for text in [
1144 "0", "-0", "1", "-1", "0.5", "-1.5", "3.14159", "1e-5", "0x1p-20", "0.1", "255",
1145 "1e30",
1146 ] {
1147 let (value, _) = Float::parse(text, format).expect("a number");
1148 let spelling = value.to_hex();
1149 let (again, status) = Float::parse(&spelling, format).expect("a number");
1150 assert_eq!(again.to_bits(), value.to_bits(), "{text} as {spelling} in {format:?}");
1151 let rounded = status.has(Status::INEXACT) || status.has(Status::OVERFLOW);
1154 assert_eq!(rounded, !value.is_finite(), "{spelling} in {format:?}");
1155 }
1156 let tiny = Float::from_bits(format, 1);
1158 let (again, _) = Float::parse(&tiny.to_hex(), format).expect("a number");
1159 assert_eq!(again.to_bits(), tiny.to_bits(), "the smallest subnormal in {format:?}");
1160 let huge = Float::infinity(format, true);
1162 let (again, status) = Float::parse(&huge.to_hex(), format).expect("a number");
1163 assert!(again.is_infinite() && again.is_negative(), "{format:?}");
1164 assert!(status.has(Status::OVERFLOW));
1165 }
1166 }
1167
1168 #[test]
1169 fn a_round_number_gets_a_short_spelling() {
1170 let hex = |text: &str| Float::parse(text, Format::Double).expect("a number").0.to_hex();
1171 assert_eq!(hex("1"), "0x1p+0");
1172 assert_eq!(hex("-1"), "-0x1p+0");
1173 assert_eq!(hex("0"), "0x0p+0");
1174 assert_eq!(hex("-0"), "-0x0p+0");
1175 assert_eq!(hex("2"), "0x1p+1");
1176 assert_eq!(hex("0.5"), "0x1p-1");
1177 assert_eq!(hex("0.1"), "0x1999999999999ap-56");
1178 }
1179
1180 #[test]
1181 fn the_narrow_formats_round_where_they_are_supposed_to() {
1182 let (value, status) = Float::parse("65504", Format::Half).expect("a number");
1186 assert!(value.is_finite() && status.is_none());
1187 assert_eq!(value.to_bits(), 0x7bff);
1188 let (value, _) = Float::parse("65536", Format::Half).expect("a number");
1189 assert!(value.is_infinite());
1190 assert_eq!(Float::parse("1", Format::Half).expect("one").0.to_bits(), 0x3c00);
1191 assert_eq!(Float::parse("1", Format::BFloat16).expect("one").0.to_bits(), 0x3f80);
1192 assert_eq!(Float::parse("1e30", Format::BFloat16).expect("big").0.to_bits(), 0x714a);
1193 assert_eq!(Float::parse("0x1p-24", Format::Half).expect("tiny").0.to_bits(), 1);
1195 assert!(Float::parse("0x1p-26", Format::Half).expect("tinier").0.is_zero());
1196 }
1197
1198 #[test]
1199 fn the_x87_format_stores_the_bit_the_others_leave_implied() {
1200 let one = Float::parse("1", Format::X87Extended).expect("one").0;
1203 assert_eq!(one.to_bits(), 0x3fff_8000_0000_0000_0000);
1204 assert_eq!(
1205 Float::parse("2", Format::X87Extended).expect("two").0.to_bits(),
1206 0x4000_8000_0000_0000_0000
1207 );
1208 let (value, status) = Float::parse("9007199254740993", Format::X87Extended).expect("one");
1210 assert!(status.is_none());
1211 assert_eq!(value.to_bits(), 0x4034_8000_0000_0000_0400);
1212 assert_eq!(
1215 Float::parse("0.1", Format::X87Extended).expect("a tenth").0.to_bits(),
1216 0x3ffb_cccc_cccc_cccc_cccd
1217 );
1218 assert_eq!(Float::parse("1e-4950", Format::X87Extended).expect("tiny").0.to_bits(), 3);
1221 }
1222
1223 #[test]
1224 fn the_quad_format_has_a_hundred_and_thirteen_bits_of_it() {
1225 assert_eq!(
1226 Float::parse("1", Format::Quad).expect("one").0.to_bits(),
1227 0x3fff_0000_0000_0000_0000_0000_0000_0000
1228 );
1229 assert_eq!(
1231 Float::parse("0.1", Format::Quad).expect("a tenth").0.to_bits(),
1232 0x3ffb_9999_9999_9999_9999_9999_9999_999a
1233 );
1234 assert_eq!(
1236 Float::parse("3.14159", Format::Quad).expect("pi, roughly").0.to_bits(),
1237 0x4000_921f_9f01_b866_e43a_a79b_badc_0981
1238 );
1239 let (value, status) = Float::parse("1e5000", Format::Quad).expect("a number");
1240 assert!(value.is_infinite() && status.has(Status::OVERFLOW));
1241 let (value, _) = Float::parse("1e-5000", Format::Quad).expect("a number");
1242 assert!(value.is_zero());
1243 }
1244
1245 #[test]
1248 fn a_nan_with_a_payload_has_the_bits_gcc_gives_it() {
1249 let double = |quiet, payload| Float::nan_with(Format::Double, false, quiet, payload);
1250 assert_eq!(double(true, 0).to_bits(), 0x7ff8_0000_0000_0000, "__builtin_nan(\"\")");
1251 assert_eq!(double(true, 1).to_bits(), 0x7ff8_0000_0000_0001, "__builtin_nan(\"0x1\")");
1252 assert_eq!(double(true, 8).to_bits(), 0x7ff8_0000_0000_0008, "__builtin_nan(\"010\")");
1253 assert_eq!(double(false, 0).to_bits(), 0x7ff4_0000_0000_0000, "__builtin_nans(\"\")");
1256 assert_eq!(double(false, 1).to_bits(), 0x7ff0_0000_0000_0001, "__builtin_nans(\"0x1\")");
1257 assert_eq!(double(true, 0xf_ffff_ffff_ffff).to_bits(), 0x7fff_ffff_ffff_ffff);
1259 assert_eq!(double(true, 1 << 52).to_bits(), 0x7ff8_0000_0000_0000);
1260 assert_eq!(
1261 Float::nan_with(Format::Single, false, true, 1).to_bits(),
1262 0x7fc0_0001,
1263 "__builtin_nanf(\"0x1\")"
1264 );
1265 assert_eq!(
1266 Float::nan_with(Format::Single, false, false, 0).to_bits(),
1267 0x7fa0_0000,
1268 "__builtin_nansf(\"\")"
1269 );
1270 assert_eq!(
1273 Float::nan_with(Format::X87Extended, false, true, 1).to_bits(),
1274 0x7fff_c000_0000_0000_0001,
1275 "__builtin_nanl(\"0x1\") on x86"
1276 );
1277 assert_eq!(
1278 Float::nan_with(Format::X87Extended, false, false, 0).to_bits(),
1279 0x7fff_a000_0000_0000_0000,
1280 "__builtin_nansl(\"\") on x86"
1281 );
1282 }
1283
1284 #[test]
1286 fn a_payload_comes_back_out_of_the_encoding_it_went_into() {
1287 for format in [Format::Half, Format::Single, Format::Double, Format::X87Extended] {
1288 for (quiet, payload) in [(true, 0), (true, 1), (false, 3), (true, 5)] {
1289 let nan = Float::nan_with(format, false, quiet, payload);
1290 assert!(nan.is_nan(), "{format:?}");
1291 assert_eq!(Float::from_bits(format, nan.to_bits()), nan, "{format:?} {payload}");
1292 }
1293 let nan = Float::nan_with(format, true, true, 7);
1295 assert!(nan.is_negative() && nan.negated().negated() == nan, "{format:?}");
1296 }
1297 }
1298
1299 #[test]
1302 fn the_smallest_normal_is_the_number_below_which_nothing_is_normal() {
1303 assert_eq!(
1304 Float::smallest_normal(Format::Single, false).to_bits(),
1305 u128::from(f32::MIN_POSITIVE.to_bits())
1306 );
1307 assert_eq!(
1308 Float::smallest_normal(Format::Double, false).to_bits(),
1309 u128::from(f64::MIN_POSITIVE.to_bits())
1310 );
1311 assert_eq!(
1314 Float::smallest_normal(Format::X87Extended, false).to_bits(),
1315 (1u128 << 64) | (1u128 << 63)
1316 );
1317 for format in [Format::Half, Format::BFloat16, Format::Single, Format::Double] {
1318 let normal = Float::smallest_normal(format, false);
1319 assert!(normal.is_finite() && !normal.is_zero(), "{format:?}");
1320 assert_eq!(Float::from_bits(format, normal.to_bits()), normal, "{format:?}");
1321 let below = Float::from_bits(format, normal.to_bits() - 1);
1324 assert_eq!(below.compare(normal), Some(std::cmp::Ordering::Less), "{format:?}");
1325 let negative = Float::smallest_normal(format, true);
1327 assert!(negative.is_negative() && negative.negated() == normal, "{format:?}");
1328 }
1329 }
1330
1331 const EVERY_FORMAT: [Format; 10] = [
1333 Format::Half,
1334 Format::BFloat16,
1335 Format::Single,
1336 Format::Double,
1337 Format::X87Extended,
1338 Format::Quad,
1339 Format::DoubleDouble,
1340 Format::Decimal32,
1341 Format::Decimal64,
1342 Format::Decimal128,
1343 ];
1344
1345 #[test]
1346 fn the_double_double_and_the_decimals_are_the_formats_that_are_not_binary_ieee() {
1347 for format in EVERY_FORMAT {
1348 let binary = format != Format::DoubleDouble && format.decimal().is_none();
1349 assert_eq!(format.is_ieee(), binary, "{format:?}");
1350 }
1351 }
1352
1353 #[test]
1354 fn every_format_has_a_name_that_reads_back_as_itself() {
1355 for format in EVERY_FORMAT {
1358 assert_eq!(Format::from_name(format.name()), Some(format), "{format:?}");
1359 }
1360 assert_eq!(Format::from_name("f128"), Some(Format::Quad));
1361 assert_eq!(Format::from_name("ppc-f128"), Some(Format::DoubleDouble));
1362 assert_eq!(Format::from_name("f256"), None);
1363 }
1364
1365 #[test]
1366 fn a_width_is_the_one_question_the_double_double_answers() {
1367 assert_eq!(Format::DoubleDouble.width(), 128);
1371 assert_eq!(Format::Quad.width(), Format::DoubleDouble.width());
1372 assert_ne!(Format::Quad, Format::DoubleDouble);
1373 }
1374
1375 #[test]
1376 #[should_panic(expected = "pair of doubles")]
1377 fn asking_a_double_double_for_a_precision_says_why_there_is_not_one() {
1378 let _ = Format::DoubleDouble.precision();
1379 }
1380
1381 #[test]
1382 #[should_panic(expected = "pair of doubles")]
1383 fn a_double_double_cannot_be_parsed_into() {
1384 let _ = Float::parse("1.0", Format::DoubleDouble);
1387 }
1388
1389 #[test]
1390 #[should_panic(expected = "pair of doubles")]
1391 fn a_double_double_cannot_be_read_out_of_its_bits_either() {
1392 let _ = Float::from_bits(Format::DoubleDouble, 0);
1393 }
1394
1395 #[test]
1396 #[should_panic(expected = "pair of doubles")]
1397 fn not_even_a_double_double_zero_can_be_made() {
1398 let _ = Float::zero(Format::DoubleDouble, false);
1402 }
1403}