diff --git a/include/boost/archive/basic_text_iprimitive.hpp b/include/boost/archive/basic_text_iprimitive.hpp index 4ab2734cd..11385df7f 100644 --- a/include/boost/archive/basic_text_iprimitive.hpp +++ b/include/boost/archive/basic_text_iprimitive.hpp @@ -24,9 +24,13 @@ // in such cases. So we can't use basic_ostream but rather // use two template parameters +#include +#include #include #include #include // size_t +#include +#include #include #if defined(BOOST_NO_STDC_NAMESPACE) @@ -39,6 +43,7 @@ namespace std{ #endif #include +#include #include #include @@ -85,9 +90,103 @@ class BOOST_SYMBOL_VISIBLE basic_text_iprimitive { > locale_saver; #endif + // Whether a value of T can be an infinity or a NaN, and so may reach + // us written as letters rather than as digits. template - void load(T & t) - { + struct has_non_finite { + typedef typename mpl::bool_< + std::numeric_limits::has_infinity + || std::numeric_limits::has_quiet_NaN + >::type type; + }; + + // Takes a leading sign, if there is one, and says whether it was a + // minus. + bool take_minus_sign(){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + if(NULL == sb){ + return false; + } + is >> std::ws; + const typename traits_type::int_type c = sb->sgetc(); + if(traits_type::eq_int_type(c, traits_type::to_int_type(char_type('-'))) + || traits_type::eq_int_type(c, traits_type::to_int_type(char_type('+')))){ + return traits_type::eq_int_type( + sb->sbumpc(), traits_type::to_int_type(char_type('-')) + ); + } + return false; + } + + // Only the letters are taken, and not everything up to the next space, + // because an XML archive ends a value with a tag. The terminator is + // left where it is for the same reason. Lowered as it goes, since it is + // ASCII either way. + std::string take_letters(){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + std::string token; + for(;;){ + const typename traits_type::int_type c = sb->sgetc(); + if(traits_type::eq_int_type(c, traits_type::eof())){ + break; + } + const char_type letter = traits_type::to_char_type(c); + if(! ((char_type('a') <= letter && letter <= char_type('z')) + || (char_type('A') <= letter && letter <= char_type('Z')))){ + break; + } + token += char(char(letter) | 0x20); + sb->sbumpc(); + } + return token; + } + + // Before 2015, the Microsoft library wrote an infinity as "1.#INF", a NaN + // as "1.#QNAN" and an indeterminate as "-1.#IND". The extraction takes + // the leading "1." for the value and stops at the '#', so an archive + // written then loads as 1 and the rest is left to derail the next read. + // Nothing else puts a '#' after a number, so seeing one settles it. + template + void take_legacy_non_finite(T & t){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + std::basic_streambuf * const sb = is.rdbuf(); + if(NULL == sb + || ! traits_type::eq_int_type( + sb->sgetc(), traits_type::to_int_type(char_type('#')) + )){ + return; + } + sb->sbumpc(); + const std::string token = take_letters(); + if("inf" == token && std::numeric_limits::has_infinity){ + t = std::numeric_limits::infinity(); + } + else if(("qnan" == token || "snan" == token || "ind" == token) + && std::numeric_limits::has_quiet_NaN){ + t = std::numeric_limits::quiet_NaN(); + } + else{ + const std::string message = + detail::stream_position_message(is.rdbuf()); + boost::serialization::throw_exception( + archive_exception( + archive_exception::input_stream_error, + message.c_str() + ) + ); + } + } + + template + void load_impl(T & t, boost::mpl::bool_ &){ if(is >> t) return; const std::string message = @@ -100,6 +199,73 @@ class BOOST_SYMBOL_VISIBLE basic_text_iprimitive { ); } + // An infinity is written as "inf" and a NaN as "nan", because that is + // what the stream writes, and the extraction of a floating point number + // then refuses both, so an archive the library wrote itself would not + // load. Read the letters here rather than change what is written, so + // that archives already in existence start loading (issue #386). + template + void load_impl(T & t, boost::mpl::bool_ &){ + typedef typename IStream::traits_type traits_type; + typedef typename IStream::char_type char_type; + + const bool negative = take_minus_sign(); + if(is >> t){ + take_legacy_non_finite(t); + if(negative){ + t = -t; + } + return; + } + is.clear(); + + std::basic_streambuf * const sb = is.rdbuf(); + const std::string token = take_letters(); + // A NaN may carry a parenthesised payload, as in the "nan(ind)" the + // Microsoft library writes, which has to come away with it. + if("nan" == token + && traits_type::eq_int_type( + sb->sgetc(), traits_type::to_int_type(char_type('(')) + ) + ){ + while(! traits_type::eq_int_type( + sb->sbumpc(), traits_type::to_int_type(char_type(')')) + )){ + if(traits_type::eq_int_type(sb->sgetc(), traits_type::eof())){ + break; + } + } + } + + if(("inf" == token || "infinity" == token) + && std::numeric_limits::has_infinity){ + t = std::numeric_limits::infinity(); + } + else if("nan" == token && std::numeric_limits::has_quiet_NaN){ + t = std::numeric_limits::quiet_NaN(); + } + else{ + const std::string message = + detail::stream_position_message(is.rdbuf()); + boost::serialization::throw_exception( + archive_exception( + archive_exception::input_stream_error, + message.c_str() + ) + ); + } + if(negative){ + t = -t; + } + } + + template + void load(T & t) + { + typename has_non_finite::type tag; + load_impl(t, tag); + } + void load(char & t) { short int i; diff --git a/include/boost/archive/basic_text_oprimitive.hpp b/include/boost/archive/basic_text_oprimitive.hpp index acd3cc019..6c589ea48 100644 --- a/include/boost/archive/basic_text_oprimitive.hpp +++ b/include/boost/archive/basic_text_oprimitive.hpp @@ -25,6 +25,7 @@ // use two template parameters #include +#include #include #include // size_t @@ -151,6 +152,31 @@ class BOOST_SYMBOL_VISIBLE basic_text_oprimitive >::type type; }; + // An infinity and a NaN go out in a spelling the library fixes itself, + // rather than in whatever the stream would produce. The latter would + // differ between implementations; e.g. the Microsoft library wrote + // "1.#INF" and "1.#QNAN" before 2015. The read side accepts the older + // spellings too, so nothing written before this stops loading. + template + bool save_non_finite(const T & t){ + if(std::numeric_limits::has_infinity){ + if(t == std::numeric_limits::infinity()){ + put("inf"); + return true; + } + if(t == -std::numeric_limits::infinity()){ + put("-inf"); + return true; + } + } + // Nothing equals a NaN, itself included. + if(std::numeric_limits::has_quiet_NaN && t != t){ + put("nan"); + return true; + } + return false; + } + template void save_impl(const T &t, boost::mpl::bool_ &){ // must be a user mistake - can't serialize un-initialized data @@ -159,6 +185,9 @@ class BOOST_SYMBOL_VISIBLE basic_text_oprimitive archive_exception(archive_exception::output_stream_error) ); } + if(save_non_finite(t)){ + return; + } // The formulae for the number of decimal digits required is given in // http://www2.open-std.org/JTC1/SC22/WG21/docs/papers/2005/n1822.pdf // which is derived from Kahan's paper: diff --git a/test/Jamfile.v2 b/test/Jamfile.v2 index fca08fbd6..0982af3df 100644 --- a/test/Jamfile.v2 +++ b/test/Jamfile.v2 @@ -96,6 +96,7 @@ test-suite "serialization" : [ test-bsl-run_files test_non_default_ctor ] [ test-bsl-run_files test_non_default_ctor2 ] [ test-bsl-run_files test_null_ptr ] + [ test-bsl-run_files test_non_finite_floats ] [ test-bsl-run_files test_nvp : A ] [ test-bsl-run_files test_object ] [ test-bsl-run_files test_primitive ] @@ -161,6 +162,7 @@ if ! $(BOOST_ARCHIVE_LIST) { [ test-bsl-run test_duplicate_type_registration ] [ test-bsl-run test_strong_typedef_move ] [ test-bsl-run test_stream_error_position ] + [ test-bsl-run test_non_finite_format ] [ test-bsl-run test_private_ctor ] [ test-bsl-run test_reset_object_address : A ] [ test-bsl-run test_void_cast ] diff --git a/test/test_non_finite_floats.cpp b/test/test_non_finite_floats.cpp new file mode 100644 index 000000000..1720801b4 --- /dev/null +++ b/test/test_non_finite_floats.cpp @@ -0,0 +1,101 @@ +/////////1/////////2/////////3/////////4/////////5/////////6/////////7/////////8 +// test_non_finite_floats.cpp + +// Copyright 2026 Gennaro Prota. +// Distributed under the Boost Software License, Version 1.0. +// (See accompanying file LICENSE_1_0.txt or copy at +// http://www.boost.org/LICENSE_1_0.txt) + +// See http://www.boost.org for updates, documentation, and revision history. + +// An infinity is written as "inf" and a NaN as "nan", because that is what +// the stream writes, and extraction of a floating point number then refuses +// both: it accepts digits and little else, in libstdc++ and in the Microsoft +// library alike. So the library used to write text archives which it could +// not read back. + +// Reported by nim65s in +// https://github.com/boostorg/serialization/issues/386. Thanks! + +#include +#include +#include +#include + +#include +#if defined(BOOST_NO_STDC_NAMESPACE) +namespace std{ + using ::remove; +} +#endif + +#include "test_tools.hpp" + +#include + +template +struct values { + T positive_infinity; + T negative_infinity; + T not_a_number; + T ordinary; + + values() : + positive_infinity(std::numeric_limits::infinity()), + negative_infinity(-std::numeric_limits::infinity()), + not_a_number(std::numeric_limits::quiet_NaN()), + ordinary(T(24.567)) + {} + + // Deliberately not the constructor above: this one is what the load + // fills in, and it must start from something finite so that a load which + // quietly does nothing cannot pass. + values(int) : + positive_infinity(0), + negative_infinity(0), + not_a_number(0), + ordinary(0) + {} + + template + void serialize(Archive & ar, const unsigned int /* version */){ + ar & boost::serialization::make_nvp("pos_inf", positive_infinity); + ar & boost::serialization::make_nvp("neg_inf", negative_infinity); + ar & boost::serialization::make_nvp("nan", not_a_number); + ar & boost::serialization::make_nvp("ordinary", ordinary); + } +}; + +template +void test_type(){ + const char * testfile = boost::archive::tmpnam(NULL); + BOOST_REQUIRE(NULL != testfile); + + const values written; + { + test_ostream os(testfile, TEST_STREAM_FLAGS); + test_oarchive oa(os, TEST_ARCHIVE_FLAGS); + oa << boost::serialization::make_nvp("values", written); + } + + values read(0); + { + test_istream is(testfile, TEST_STREAM_FLAGS); + test_iarchive ia(is, TEST_ARCHIVE_FLAGS); + ia >> boost::serialization::make_nvp("values", read); + } + + BOOST_CHECK(read.positive_infinity == std::numeric_limits::infinity()); + BOOST_CHECK(read.negative_infinity == -std::numeric_limits::infinity()); + // A NaN is equal to nothing, itself included, so that is the test. + BOOST_CHECK(read.not_a_number != read.not_a_number); + BOOST_CHECK(read.ordinary == written.ordinary); + + std::remove(testfile); +} + +int test_main(int /* argc */, char * /* argv */ []){ + test_type(); + test_type(); + return EXIT_SUCCESS; +} diff --git a/test/test_non_finite_format.cpp b/test/test_non_finite_format.cpp new file mode 100644 index 000000000..3daf485c7 --- /dev/null +++ b/test/test_non_finite_format.cpp @@ -0,0 +1,104 @@ +/////////1/////////2/////////3/////////4/////////5/////////6/////////7/////////8 +// test_non_finite_format.cpp + +// Copyright 2026 Gennaro Prota. +// Distributed under the Boost Software License, Version 1.0. +// (See accompanying file LICENSE_1_0.txt or copy at +// http://www.boost.org/LICENSE_1_0.txt) + +// See http://www.boost.org for updates, documentation, and revision history. + +// The library writes an infinity and a NaN as a special tag, so that an +// archive does not depend on which standard library produced it. + +// Suggested by Robert Ramey in +// https://github.com/boostorg/serialization/pull/387. Thanks! + +#include +#include +#include + +#include + +#include +#include + +template +static std::string written(T value){ + std::ostringstream os; + { + boost::archive::text_oarchive oa(os); + oa << value; + } + const std::string all = os.str(); + // Everything after the last space of the header is the value itself. + const std::string::size_type at = all.rfind(' '); + BOOST_TEST(std::string::npos != at); + std::string token = all.substr(at + 1); + while(! token.empty() + && ('\n' == token.back() || '\r' == token.back())){ + token.erase(token.size() - 1); + } + return token; +} + +// Whatever the spelling, it has to come back as the same value. +template +static void reads_back_as(const std::string & token, T expected){ + // Take a real header, then put the spelling under test where the value + // goes. A zero is written as "0.00000000e+00", so the value cannot be + // found by looking for a digit; the last space of the header is the mark. + std::ostringstream os; + { + boost::archive::text_oarchive oa(os); + const T zero = T(0); + oa << zero; + } + std::string archive = os.str(); + const std::string::size_type at = archive.rfind(' '); + BOOST_TEST(std::string::npos != at); + archive.erase(at + 1); + archive += token; + archive += "\n"; + + std::istringstream is(archive); + boost::archive::text_iarchive ia(is); + T back = T(1); + ia >> back; + if(expected == expected){ + BOOST_TEST(back == expected); + } + else{ + BOOST_TEST(back != back); // a NaN was asked for + } +} + +template +static void test_type(){ + BOOST_TEST_EQ(written(std::numeric_limits::infinity()), std::string("inf")); + BOOST_TEST_EQ(written(-std::numeric_limits::infinity()), std::string("-inf")); + BOOST_TEST_EQ(written(std::numeric_limits::quiet_NaN()), std::string("nan")); + + // Spellings older archives may carry are still accepted. + reads_back_as("inf", std::numeric_limits::infinity()); + reads_back_as("infinity", std::numeric_limits::infinity()); + reads_back_as("INF", std::numeric_limits::infinity()); + reads_back_as("-inf", -std::numeric_limits::infinity()); + reads_back_as("nan", std::numeric_limits::quiet_NaN()); + reads_back_as("nan(ind)", std::numeric_limits::quiet_NaN()); + + // What the Microsoft library wrote before 2015. These begin with digits, + // so the extraction takes the "1." for the value and succeeds; without + // the rest being read they would load as 1 rather than fail outright. + reads_back_as("1.#INF", std::numeric_limits::infinity()); + reads_back_as("-1.#INF", -std::numeric_limits::infinity()); + reads_back_as("1.#QNAN", std::numeric_limits::quiet_NaN()); + reads_back_as("-1.#IND", std::numeric_limits::quiet_NaN()); +} + +int +main(){ + test_type(); + test_type(); + return boost::report_errors(); +}