From 964986a53d55d9714009476adb1577b4c4a66680 Mon Sep 17 00:00:00 2001 From: Gennaro Prota Date: Mon, 31 Aug 2026 16:00:20 +0200 Subject: [PATCH] Read back infinities and NaNs from text archives An infinity or a NaN were written in the form produced by the standard library (e.g. "inf" or "nan"), but extraction of a floating point number then refused both: it took digits and little else, in libstdc++ and in the Microsoft library alike. So the library wrote text and XML archives which it could not read back. Fix that by serializing with special tags, so that we don't depend on what the standard library writes, and reading back both the tags and the formats produced by the various standard libraries (so old archives read correctly). Fixes #386. --- .../boost/archive/basic_text_iprimitive.hpp | 170 +++++++++++++++++- .../boost/archive/basic_text_oprimitive.hpp | 29 +++ test/Jamfile.v2 | 2 + test/test_non_finite_floats.cpp | 101 +++++++++++ test/test_non_finite_format.cpp | 104 +++++++++++ 5 files changed, 404 insertions(+), 2 deletions(-) create mode 100644 test/test_non_finite_floats.cpp create mode 100644 test/test_non_finite_format.cpp diff --git a/include/boost/archive/basic_text_iprimitive.hpp b/include/boost/archive/basic_text_iprimitive.hpp index 4ab2734cd..11385df7f 100644 --- a/include/boost/archive/basic_text_iprimitive.hpp +++ b/include/boost/archive/basic_text_iprimitive.hpp @@ -24,9 +24,13 @@ // in such cases. So we can't use basic_ostream but rather // use two template parameters +#include +#include #include #include #include // size_t +#include +#include #include #if defined(BOOST_NO_STDC_NAMESPACE) @@ -39,6 +43,7 @@ namespace std{ #endif #include +#include #include #include @@ -85,9 +90,103 @@ class BOOST_SYMBOL_VISIBLE basic_text_iprimitive { > locale_saver; #endif + // Whether a value of T can be an infinity or a NaN, and so may reach + // us written as letters rather than as digits. template - void load(T & t) - { + struct has_non_finite { + typedef typename mpl::bool_< + std::numeric_limits::has_infinity + || std::numeric_limits::has_quiet_NaN + >::type type; + }; + + // Takes a leading sign, if there is one, and says whether it was a + // minus. + bool take_minus_sign(){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + if(NULL == sb){ + return false; + } + is >> std::ws; + const typename traits_type::int_type c = sb->sgetc(); + if(traits_type::eq_int_type(c, traits_type::to_int_type(char_type('-'))) + || traits_type::eq_int_type(c, traits_type::to_int_type(char_type('+')))){ + return traits_type::eq_int_type( + sb->sbumpc(), traits_type::to_int_type(char_type('-')) + ); + } + return false; + } + + // Only the letters are taken, and not everything up to the next space, + // because an XML archive ends a value with a tag. The terminator is + // left where it is for the same reason. Lowered as it goes, since it is + // ASCII either way. + std::string take_letters(){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + std::string token; + for(;;){ + const typename traits_type::int_type c = sb->sgetc(); + if(traits_type::eq_int_type(c, traits_type::eof())){ + break; + } + const char_type letter = traits_type::to_char_type(c); + if(! ((char_type('a') <= letter && letter <= char_type('z')) + || (char_type('A') <= letter && letter <= char_type('Z')))){ + break; + } + token += char(char(letter) | 0x20); + sb->sbumpc(); + } + return token; + } + + // Before 2015, the Microsoft library wrote an infinity as "1.#INF", a NaN + // as "1.#QNAN" and an indeterminate as "-1.#IND". The extraction takes + // the leading "1." for the value and stops at the '#', so an archive + // written then loads as 1 and the rest is left to derail the next read. + // Nothing else puts a '#' after a number, so seeing one settles it. + template + void take_legacy_non_finite(T & t){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + if(NULL == sb + || ! traits_type::eq_int_type( + sb->sgetc(), traits_type::to_int_type(char_type('#')) + )){ + return; + } + sb->sbumpc(); + const std::string token = take_letters(); + if("inf" == token && std::numeric_limits::has_infinity){ + t = std::numeric_limits::infinity(); + } + else if(("qnan" == token || "snan" == token || "ind" == token) + && std::numeric_limits::has_quiet_NaN){ + t = std::numeric_limits::quiet_NaN(); + } + else{ + const std::string message = + detail::stream_position_message(is.rdbuf()); + boost::serialization::throw_exception( + archive_exception( + archive_exception::input_stream_error, + message.c_str() + ) + ); + } + } + + template + void load_impl(T & t, boost::mpl::bool_ &){ if(is >> t) return; const std::string message = @@ -100,6 +199,73 @@ class BOOST_SYMBOL_VISIBLE basic_text_iprimitive { ); } + // An infinity is written as "inf" and a NaN as "nan", because that is + // what the stream writes, and the extraction of a floating point number + // then refuses both, so an archive the library wrote itself would not + // load. Read the letters here rather than change what is written, so + // that archives already in existence start loading (issue #386). + template + void load_impl(T & t, boost::mpl::bool_ &){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + const bool negative = take_minus_sign(); + if(is >> t){ + take_legacy_non_finite(t); + if(negative){ + t = -t; + } + return; + } + is.clear(); + + std::basic_streambuf * const sb = is.rdbuf(); + const std::string token = take_letters(); + // A NaN may carry a parenthesised payload, as in the "nan(ind)" the + // Microsoft library writes, which has to come away with it. + if("nan" == token + && traits_type::eq_int_type( + sb->sgetc(), traits_type::to_int_type(char_type('(')) + ) + ){ + while(! traits_type::eq_int_type( + sb->sbumpc(), traits_type::to_int_type(char_type(')')) + )){ + if(traits_type::eq_int_type(sb->sgetc(), traits_type::eof())){ + break; + } + } + } + + if(("inf" == token || "infinity" == token) + && std::numeric_limits::has_infinity){ + t = std::numeric_limits::infinity(); + } + else if("nan" == token && std::numeric_limits::has_quiet_NaN){ + t = std::numeric_limits::quiet_NaN(); + } + else{ + const std::string message = + detail::stream_position_message(is.rdbuf()); + boost::serialization::throw_exception( + archive_exception( + archive_exception::input_stream_error, + message.c_str() + ) + ); + } + if(negative){ + t = -t; + } + } + + template + void load(T & t) + { + typename has_non_finite::type tag; + load_impl(t, tag); + } + void load(char & t) { short int i; diff --git a/include/boost/archive/basic_text_oprimitive.hpp b/include/boost/archive/basic_text_oprimitive.hpp index acd3cc019..6c589ea48 100644 --- a/include/boost/archive/basic_text_oprimitive.hpp +++ b/include/boost/archive/basic_text_oprimitive.hpp @@ -25,6 +25,7 @@ // use two template parameters #include +#include #include #include // size_t @@ -151,6 +152,31 @@ class BOOST_SYMBOL_VISIBLE basic_text_oprimitive >::type type; }; + // An infinity and a NaN go out in a spelling the library fixes itself, + // rather than in whatever the stream would produce. The latter would + // differ between implementations; e.g. the Microsoft library wrote + // "1.#INF" and "1.#QNAN" before 2015. The read side accepts the older + // spellings too, so nothing written before this stops loading. + template + bool save_non_finite(const T & t){ + if(std::numeric_limits::has_infinity){ + if(t == std::numeric_limits::infinity()){ + put("inf"); + return true; + } + if(t == -std::numeric_limits::infinity()){ + put("-inf"); + return true; + } + } + // Nothing equals a NaN, itself included. + if(std::numeric_limits::has_quiet_NaN && t != t){ + put("nan"); + return true; + } + return false; + } + template void save_impl(const T &t, boost::mpl::bool_ &){ // must be a user mistake - can't serialize un-initialized data @@ -159,6 +185,9 @@ class BOOST_SYMBOL_VISIBLE basic_text_oprimitive archive_exception(archive_exception::output_stream_error) ); } + if(save_non_finite(t)){ + return; + } // The formulae for the number of decimal digits required is given in // http://www2.open-std.org/JTC1/SC22/WG21/docs/papers/2005/n1822.pdf // which is derived from Kahan's paper: diff --git a/test/Jamfile.v2 b/test/Jamfile.v2 index fca08fbd6..0982af3df 100644 --- a/test/Jamfile.v2 +++ b/test/Jamfile.v2 @@ -96,6 +96,7 @@ test-suite "serialization" : [ test-bsl-run_files test_non_default_ctor ] [ test-bsl-run_files test_non_default_ctor2 ] [ test-bsl-run_files test_null_ptr ] + [ test-bsl-run_files test_non_finite_floats ] [ test-bsl-run_files test_nvp : A ] [ test-bsl-run_files test_object ] [ test-bsl-run_files test_primitive ] @@ -161,6 +162,7 @@ if ! $(BOOST_ARCHIVE_LIST) { [ test-bsl-run test_duplicate_type_registration ] [ test-bsl-run test_strong_typedef_move ] [ test-bsl-run test_stream_error_position ] + [ test-bsl-run test_non_finite_format ] [ test-bsl-run test_private_ctor ] [ test-bsl-run test_reset_object_address : A ] [ test-bsl-run test_void_cast ] diff --git a/test/test_non_finite_floats.cpp b/test/test_non_finite_floats.cpp new file mode 100644 index 000000000..1720801b4 --- /dev/null +++ b/test/test_non_finite_floats.cpp @@ -0,0 +1,101 @@ +/////////1/////////2/////////3/////////4/////////5/////////6/////////7/////////8 +// test_non_finite_floats.cpp + +// Copyright 2026 Gennaro Prota. +// Distributed under the Boost Software License, Version 1.0. +// (See accompanying file LICENSE_1_0.txt or copy at +// http://www.boost.org/LICENSE_1_0.txt) + +// See http://www.boost.org for updates, documentation, and revision history. + +// An infinity is written as "inf" and a NaN as "nan", because that is what +// the stream writes, and extraction of a floating point number then refuses +// both: it accepts digits and little else, in libstdc++ and in the Microsoft +// library alike. So the library used to write text archives which it could +// not read back. + +// Reported by nim65s in +// https://github.com/boostorg/serialization/issues/386. Thanks! + +#include +#include +#include +#include + +#include +#if defined(BOOST_NO_STDC_NAMESPACE) +namespace std{ + using ::remove; +} +#endif + +#include "test_tools.hpp" + +#include + +template +struct values { + T positive_infinity; + T negative_infinity; + T not_a_number; + T ordinary; + + values() : + positive_infinity(std::numeric_limits::infinity()), + negative_infinity(-std::numeric_limits::infinity()), + not_a_number(std::numeric_limits::quiet_NaN()), + ordinary(T(24.567)) + {} + + // Deliberately not the constructor above: this one is what the load + // fills in, and it must start from something finite so that a load which + // quietly does nothing cannot pass. + values(int) : + positive_infinity(0), + negative_infinity(0), + not_a_number(0), + ordinary(0) + {} + + template + void serialize(Archive & ar, const unsigned int /* version */){ + ar & boost::serialization::make_nvp("pos_inf", positive_infinity); + ar & boost::serialization::make_nvp("neg_inf", negative_infinity); + ar & boost::serialization::make_nvp("nan", not_a_number); + ar & boost::serialization::make_nvp("ordinary", ordinary); + } +}; + +template +void test_type(){ + const char * testfile = boost::archive::tmpnam(NULL); + BOOST_REQUIRE(NULL != testfile); + + const values written; + { + test_ostream os(testfile, TEST_STREAM_FLAGS); + test_oarchive oa(os, TEST_ARCHIVE_FLAGS); + oa << boost::serialization::make_nvp("values", written); + } + + values read(0); + { + test_istream is(testfile, TEST_STREAM_FLAGS); + test_iarchive ia(is, TEST_ARCHIVE_FLAGS); + ia >> boost::serialization::make_nvp("values", read); + } + + BOOST_CHECK(read.positive_infinity == std::numeric_limits::infinity()); + BOOST_CHECK(read.negative_infinity == -std::numeric_limits::infinity()); + // A NaN is equal to nothing, itself included, so that is the test. + BOOST_CHECK(read.not_a_number != read.not_a_number); + BOOST_CHECK(read.ordinary == written.ordinary); + + std::remove(testfile); +} + +int test_main(int /* argc */, char * /* argv */ []){ + test_type(); + test_type(); + return EXIT_SUCCESS; +} diff --git a/test/test_non_finite_format.cpp b/test/test_non_finite_format.cpp new file mode 100644 index 000000000..3daf485c7 --- /dev/null +++ b/test/test_non_finite_format.cpp @@ -0,0 +1,104 @@ +/////////1/////////2/////////3/////////4/////////5/////////6/////////7/////////8 +// test_non_finite_format.cpp + +// Copyright 2026 Gennaro Prota. +// Distributed under the Boost Software License, Version 1.0. +// (See accompanying file LICENSE_1_0.txt or copy at +// http://www.boost.org/LICENSE_1_0.txt) + +// See http://www.boost.org for updates, documentation, and revision history. + +// The library writes an infinity and a NaN as a special tag, so that an +// archive does not depend on which standard library produced it. + +// Suggested by Robert Ramey in +// https://github.com/boostorg/serialization/pull/387. Thanks! + +#include +#include +#include + +#include + +#include +#include + +template +static std::string written(T value){ + std::ostringstream os; + { + boost::archive::text_oarchive oa(os); + oa << value; + } + const std::string all = os.str(); + // Everything after the last space of the header is the value itself. + const std::string::size_type at = all.rfind(' '); + BOOST_TEST(std::string::npos != at); + std::string token = all.substr(at + 1); + while(! token.empty() + && ('\n' == token.back() || '\r' == token.back())){ + token.erase(token.size() - 1); + } + return token; +} + +// Whatever the spelling, it has to come back as the same value. +template +static void reads_back_as(const std::string & token, T expected){ + // Take a real header, then put the spelling under test where the value + // goes. A zero is written as "0.00000000e+00", so the value cannot be + // found by looking for a digit; the last space of the header is the mark. + std::ostringstream os; + { + boost::archive::text_oarchive oa(os); + const T zero = T(0); + oa << zero; + } + std::string archive = os.str(); + const std::string::size_type at = archive.rfind(' '); + BOOST_TEST(std::string::npos != at); + archive.erase(at + 1); + archive += token; + archive += "\n"; + + std::istringstream is(archive); + boost::archive::text_iarchive ia(is); + T back = T(1); + ia >> back; + if(expected == expected){ + BOOST_TEST(back == expected); + } + else{ + BOOST_TEST(back != back); // a NaN was asked for + } +} + +template +static void test_type(){ + BOOST_TEST_EQ(written(std::numeric_limits::infinity()), std::string("inf")); + BOOST_TEST_EQ(written(-std::numeric_limits::infinity()), std::string("-inf")); + BOOST_TEST_EQ(written(std::numeric_limits::quiet_NaN()), std::string("nan")); + + // Spellings older archives may carry are still accepted. + reads_back_as("inf", std::numeric_limits::infinity()); + reads_back_as("infinity", std::numeric_limits::infinity()); + reads_back_as("INF", std::numeric_limits::infinity()); + reads_back_as("-inf", -std::numeric_limits::infinity()); + reads_back_as("nan", std::numeric_limits::quiet_NaN()); + reads_back_as("nan(ind)", std::numeric_limits::quiet_NaN()); + + // What the Microsoft library wrote before 2015. These begin with digits, + // so the extraction takes the "1." for the value and succeeds; without + // the rest being read they would load as 1 rather than fail outright. + reads_back_as("1.#INF", std::numeric_limits::infinity()); + reads_back_as("-1.#INF", -std::numeric_limits::infinity()); + reads_back_as("1.#QNAN", std::numeric_limits::quiet_NaN()); + reads_back_as("-1.#IND", std::numeric_limits::quiet_NaN()); +} + +int +main(){ + test_type(); + test_type(); + return boost::report_errors(); +}