From 61ff830256e8070c672897ae37effd0c4ed0d8a6 Mon Sep 17 00:00:00 2001 From: Juan Cruz Viotti Date: Wed, 23 Sep 2026 12:39:25 -0300 Subject: [PATCH 1/2] Upgrade Sourcemeta dependencies Signed-off-by: Juan Cruz Viotti --- DEPENDENCIES | 4 +- vendor/blaze/CMakeLists.txt | 18 +- vendor/blaze/DEPENDENCIES | 2 +- vendor/blaze/config.cmake.in | 10 +- vendor/blaze/schemas/canonical-draft3.json | 3 - vendor/blaze/src/alterschema/CMakeLists.txt | 2 - vendor/blaze/src/alterschema/alterschema.cc | 9 +- vendor/blaze/src/bundle/CMakeLists.txt | 16 - vendor/blaze/src/bundle/bundle.cc | 755 ------------------ vendor/blaze/src/bundle/helpers.h | 105 --- .../bundle/include/sourcemeta/blaze/bundle.h | 264 ------ .../blaze/src/canonicalizer/canonicalize.cc | 87 ++ vendor/blaze/src/codegen/CMakeLists.txt | 2 - vendor/blaze/src/codegen/codegen.cc | 8 +- vendor/blaze/src/compiler/CMakeLists.txt | 2 - vendor/blaze/src/compiler/compile.cc | 10 +- vendor/blaze/src/configuration/CMakeLists.txt | 1 - vendor/blaze/src/configuration/fetch.cc | 8 +- vendor/blaze/src/convert/CMakeLists.txt | 3 +- vendor/blaze/src/convert/convert.cc | 92 ++- vendor/blaze/src/convert/helpers.h | 208 +++-- .../include/sourcemeta/blaze/convert.h | 8 +- .../include/sourcemeta/blaze/convert_error.h | 81 ++ vendor/blaze/src/convert/rule.h | 3 +- .../src/convert/rules/definitions_to_defs.h | 3 +- .../convert/rules/dependencies_to_dependent.h | 110 +++ .../rules/draft_official_dialect_with_https.h | 3 +- ..._official_dialect_without_empty_fragment.h | 4 +- .../src/convert/rules/empty_object_as_true.h | 3 +- .../blaze/src/convert/rules/enum_to_const.h | 3 +- .../src/convert/rules/metaschema_vocabulary.h | 99 --- ...ern_official_dialect_with_empty_fragment.h | 33 + .../rules/prefix_promoted_2020_12_keywords.h | 3 +- .../prefix_promoted_draft_2019_09_keywords.h | 3 +- .../rules/prefix_promoted_draft_4_keywords.h | 3 +- .../rules/prefix_promoted_draft_6_keywords.h | 3 +- .../rules/prefix_promoted_draft_7_keywords.h | 3 +- .../rules/upgrade_2019_09_to_2020_12.h | 113 ++- .../rules/upgrade_dialect_override_cleanup.h | 16 +- .../rules/upgrade_draft_3_to_draft_4.h | 3 +- .../rules/upgrade_draft_4_to_draft_6.h | 19 +- .../rules/upgrade_draft_6_to_draft_7.h | 3 +- .../rules/upgrade_draft_7_to_draft_2019_09.h | 8 +- vendor/blaze/src/dependencies/CMakeLists.txt | 14 + vendor/blaze/src/dependencies/dependencies.cc | 170 ++++ .../include/sourcemeta/blaze/dependencies.h | 100 +++ .../editor/include/sourcemeta/blaze/editor.h | 6 +- .../core/src/core/jsonschema/CMakeLists.txt | 2 +- vendor/core/src/core/jsonschema/bundle.cc | 614 ++++++++++++++ .../include/sourcemeta/core/jsonschema.h | 194 ++++- .../sourcemeta/core/jsonschema_frame.h | 5 +- vendor/core/src/core/jsonschema/jsonschema.cc | 5 + vendor/core/src/core/openapi/CMakeLists.txt | 3 +- vendor/core/src/core/openapi/discriminator.h | 205 +++++ vendor/core/src/core/openapi/document.h | 85 +- vendor/core/src/core/openapi/frame.cc | 164 ++-- vendor/core/src/core/openapi/helpers.h | 93 ++- .../include/sourcemeta/core/openapi_error.h | 32 + vendor/core/src/core/openapi/security.h | 5 +- 59 files changed, 2250 insertions(+), 1583 deletions(-) delete mode 100644 vendor/blaze/src/bundle/CMakeLists.txt delete mode 100644 vendor/blaze/src/bundle/bundle.cc delete mode 100644 vendor/blaze/src/bundle/helpers.h delete mode 100644 vendor/blaze/src/bundle/include/sourcemeta/blaze/bundle.h create mode 100644 vendor/blaze/src/convert/rules/dependencies_to_dependent.h delete mode 100644 vendor/blaze/src/convert/rules/metaschema_vocabulary.h create mode 100644 vendor/blaze/src/convert/rules/modern_official_dialect_with_empty_fragment.h create mode 100644 vendor/blaze/src/dependencies/CMakeLists.txt create mode 100644 vendor/blaze/src/dependencies/dependencies.cc create mode 100644 vendor/blaze/src/dependencies/include/sourcemeta/blaze/dependencies.h create mode 100644 vendor/core/src/core/jsonschema/bundle.cc create mode 100644 vendor/core/src/core/openapi/discriminator.h diff --git a/DEPENDENCIES b/DEPENDENCIES index 783278f6..c742172f 100644 --- a/DEPENDENCIES +++ b/DEPENDENCIES @@ -1,4 +1,4 @@ vendorpull https://github.com/sourcemeta/vendorpull 1dcbac42809cf87cb5b045106b863e17ad84ba02 -core https://github.com/sourcemeta/core 9c3323a59a4f8113e269cfb713c85ba34cf06d96 -blaze https://github.com/sourcemeta/blaze d8d3b28a989b49672fa141b5d37792551999d532 +core https://github.com/sourcemeta/core 4066fbfdb587d9d30b026e987a02fb9ecfc569bd +blaze https://github.com/sourcemeta/blaze e680badab795e9cdf30752f7dcfe77be708e05d7 bootstrap https://github.com/twbs/bootstrap 1a6fdfae6be09b09eaced8f0e442ca6f7680a61e diff --git a/vendor/blaze/CMakeLists.txt b/vendor/blaze/CMakeLists.txt index b4fc417b..0c07dcff 100644 --- a/vendor/blaze/CMakeLists.txt +++ b/vendor/blaze/CMakeLists.txt @@ -15,7 +15,7 @@ option(BLAZE_CODEGEN "Build the Blaze codegen library" ON) option(BLAZE_CANONICALIZER "Build the Blaze canonicalizer library" ON) option(BLAZE_CONVERT "Build the Blaze convert library" ON) option(BLAZE_EDITOR "Build the Blaze editor schema compatibility library" ON) -option(BLAZE_BUNDLE "Build the Blaze bundle library" ON) +option(BLAZE_DEPENDENCIES "Build the Blaze dependencies library" ON) option(BLAZE_TESTS "Build the Blaze tests" OFF) option(BLAZE_BENCHMARK "Build the Blaze benchmarks" OFF) option(BLAZE_CONTRIB "Build the Blaze contrib programs" OFF) @@ -55,10 +55,6 @@ elseif(BLAZE_UNDEFINED_SANITIZER) sourcemeta_sanitizer(TYPE undefined) endif() -if(BLAZE_BUNDLE) - add_subdirectory(src/bundle) -endif() - if(BLAZE_COMPILER) add_subdirectory(src/compiler) endif() @@ -99,6 +95,10 @@ if(BLAZE_EDITOR) add_subdirectory(src/editor) endif() +if(BLAZE_DEPENDENCIES) + add_subdirectory(src/dependencies) +endif() + if(BLAZE_CONTRIB) add_subdirectory(contrib) endif() @@ -145,10 +145,6 @@ endif() if(BLAZE_TESTS) enable_testing() - if(BLAZE_BUNDLE) - add_subdirectory(test/bundle) - endif() - if(BLAZE_COMPILER) add_subdirectory(test/compiler) endif() @@ -189,6 +185,10 @@ if(BLAZE_TESTS) add_subdirectory(test/editor) endif() + if(BLAZE_DEPENDENCIES) + add_subdirectory(test/dependencies) + endif() + if(PROJECT_IS_TOP_LEVEL) # Otherwise we need the child project to link # against the sanitizers too. diff --git a/vendor/blaze/DEPENDENCIES b/vendor/blaze/DEPENDENCIES index a78168aa..6d6e1dba 100644 --- a/vendor/blaze/DEPENDENCIES +++ b/vendor/blaze/DEPENDENCIES @@ -1,3 +1,3 @@ vendorpull https://github.com/sourcemeta/vendorpull 1dcbac42809cf87cb5b045106b863e17ad84ba02 -core https://github.com/sourcemeta/core 9c3323a59a4f8113e269cfb713c85ba34cf06d96 +core https://github.com/sourcemeta/core 4066fbfdb587d9d30b026e987a02fb9ecfc569bd jsonschema-test-suite https://github.com/json-schema-org/JSON-Schema-Test-Suite 6648e8194c69697b2e1a15fe76a06a480b183a51 diff --git a/vendor/blaze/config.cmake.in b/vendor/blaze/config.cmake.in index c21ab40c..000bfb47 100644 --- a/vendor/blaze/config.cmake.in +++ b/vendor/blaze/config.cmake.in @@ -14,7 +14,7 @@ if(NOT BLAZE_COMPONENTS) list(APPEND BLAZE_COMPONENTS canonicalizer) list(APPEND BLAZE_COMPONENTS convert) list(APPEND BLAZE_COMPONENTS editor) - list(APPEND BLAZE_COMPONENTS bundle) + list(APPEND BLAZE_COMPONENTS dependencies) endif() include(CMakeFindDependencyMacro) @@ -42,13 +42,11 @@ endif() foreach(component ${BLAZE_COMPONENTS}) if(component STREQUAL "compiler") - include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_bundle.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_compiler.cmake") elseif(component STREQUAL "evaluator") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") elseif(component STREQUAL "test") - include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_bundle.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_compiler.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_test.cmake") @@ -56,7 +54,6 @@ foreach(component ${BLAZE_COMPONENTS}) include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_output.cmake") elseif(component STREQUAL "configuration") - include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_bundle.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_configuration.cmake") elseif(component STREQUAL "alterschema") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") @@ -64,7 +61,6 @@ foreach(component ${BLAZE_COMPONENTS}) include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_output.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_alterschema.cmake") elseif(component STREQUAL "codegen") - include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_bundle.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_evaluator.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_compiler.cmake") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_output.cmake") @@ -76,8 +72,8 @@ foreach(component ${BLAZE_COMPONENTS}) include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_convert.cmake") elseif(component STREQUAL "editor") include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_editor.cmake") - elseif(component STREQUAL "bundle") - include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_bundle.cmake") + elseif(component STREQUAL "dependencies") + include("${CMAKE_CURRENT_LIST_DIR}/sourcemeta_blaze_dependencies.cmake") else() message(FATAL_ERROR "Unknown Blaze component: ${component}") endif() diff --git a/vendor/blaze/schemas/canonical-draft3.json b/vendor/blaze/schemas/canonical-draft3.json index 208fea44..f973823c 100644 --- a/vendor/blaze/schemas/canonical-draft3.json +++ b/vendor/blaze/schemas/canonical-draft3.json @@ -37,9 +37,6 @@ "$schema": { "type": "string" }, - "id": { - "type": "string" - }, "definitions": { "$comment": "TODO: `definitions` is permitted on any subschema as an interim measure, because the canonicaliser preserves it wherever the author wrote it. The desired shape is a single flat `definitions` at the root holding every subschema as a node, with `$ref`s as the edges between them, after which this should be tightened to the root only", "type": "object", diff --git a/vendor/blaze/src/alterschema/CMakeLists.txt b/vendor/blaze/src/alterschema/CMakeLists.txt index 055dcf51..c0f017b8 100644 --- a/vendor/blaze/src/alterschema/CMakeLists.txt +++ b/vendor/blaze/src/alterschema/CMakeLists.txt @@ -116,8 +116,6 @@ target_link_libraries(sourcemeta_blaze_alterschema PUBLIC sourcemeta::core::jsonschema) target_link_libraries(sourcemeta_blaze_alterschema PUBLIC sourcemeta::blaze::compiler) -target_link_libraries(sourcemeta_blaze_alterschema PRIVATE - sourcemeta::blaze::bundle) target_link_libraries(sourcemeta_blaze_alterschema PRIVATE sourcemeta::blaze::evaluator) target_link_libraries(sourcemeta_blaze_alterschema PRIVATE diff --git a/vendor/blaze/src/alterschema/alterschema.cc b/vendor/blaze/src/alterschema/alterschema.cc index 289137da..e27e0eb6 100644 --- a/vendor/blaze/src/alterschema/alterschema.cc +++ b/vendor/blaze/src/alterschema/alterschema.cc @@ -1,5 +1,4 @@ #include -#include #include #include #include @@ -189,8 +188,12 @@ inline auto compile_embedded_subschema(const JSON &root, const Pointer container{container_name}; auto document{root}; - bundle(document, walker, resolver, BundleMode::References, default_dialect, - "", container, paths, default_base); + SchemaBundleOptions options; + options.mode = SchemaBundleOptions::Mode::References; + options.default_container = container; + options.paths = paths; + options.default_base = default_base; + schema_bundle(document, walker, resolver, default_dialect, "", options); // Bundling embeds every remote schema into the container, outside of every // schema that the frame located, so each of them is framed as a schema too diff --git a/vendor/blaze/src/bundle/CMakeLists.txt b/vendor/blaze/src/bundle/CMakeLists.txt deleted file mode 100644 index 545b9dcb..00000000 --- a/vendor/blaze/src/bundle/CMakeLists.txt +++ /dev/null @@ -1,16 +0,0 @@ -sourcemeta_library(NAMESPACE sourcemeta PROJECT blaze NAME bundle - FOLDER "Blaze/Bundle" - SOURCES bundle.cc helpers.h) - -if(BLAZE_INSTALL) - sourcemeta_library_install(NAMESPACE sourcemeta PROJECT blaze NAME bundle) -endif() - -target_link_libraries(sourcemeta_blaze_bundle PUBLIC - sourcemeta::core::json) -target_link_libraries(sourcemeta_blaze_bundle PUBLIC - sourcemeta::core::jsonpointer) -target_link_libraries(sourcemeta_blaze_bundle PUBLIC - sourcemeta::core::jsonschema) -target_link_libraries(sourcemeta_blaze_bundle PRIVATE - sourcemeta::core::uri) diff --git a/vendor/blaze/src/bundle/bundle.cc b/vendor/blaze/src/bundle/bundle.cc deleted file mode 100644 index 7512fd62..00000000 --- a/vendor/blaze/src/bundle/bundle.cc +++ /dev/null @@ -1,755 +0,0 @@ -#include - -#include - -#include "helpers.h" - -#include // assert -#include // std::uint64_t -#include // std::reference_wrapper -#include // std::optional -#include // std::string -#include // std::tuple -#include // std::unordered_map -#include // std::unordered_set -#include // std::move, std::pair -#include // std::vector - -namespace { - -auto is_skippable_metaschema_reference( - const sourcemeta::blaze::BundleMode mode, - const sourcemeta::core::WeakPointer &pointer, - const std::string &destination) -> bool { - assert(!pointer.empty()); - assert(pointer.back().is_property()); - if (pointer.back().to_property() != "$schema") { - return false; - } - - return mode == sourcemeta::blaze::BundleMode::References || - sourcemeta::core::schema_is_official(destination); -} - -// Every frame that bundling constructs spends from the same limit, as how -// many frames it ends up needing is a function of what the resolver hands -// back rather than of the schema the caller passed in. A frame that ran past -// what was left of the limit threw rather than returned, so what it holds is -// always within it -auto charge(std::uint64_t &remaining, - const sourcemeta::core::SchemaFrame &frame) -> void { - assert(frame.location_count() <= remaining); - remaining -= frame.location_count(); -} - -auto dependencies_internal( - const sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const sourcemeta::blaze::DependencyCallback &callback, - std::string_view default_dialect, std::string_view default_id, - const sourcemeta::core::SchemaFrame::Paths &paths, - std::unordered_set &visited, std::uint64_t &remaining) - -> void { - sourcemeta::core::SchemaFrame frame{ - sourcemeta::core::SchemaFrame::Mode::References, - schema, - walker, - resolver, - default_dialect, - default_id, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - paths, - "", - remaining}; - charge(remaining, frame); - const auto &origin{frame.root()}; - - std::vector< - std::tuple> - found; - - frame.for_each_unresolved_reference([&](const auto &pointer, - const auto &reference) -> void { - // We don't want to report official schemas, as we can expect - // virtually all implementations to understand them out of the box - if (is_skippable_metaschema_reference( - sourcemeta::blaze::BundleMode::NonOfficialMetaschemas, pointer, - reference.destination)) { - return; - } - - if (reference.base.empty()) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - - // To not infinitely loop on circular references - if (visited.contains(std::string{reference.base})) { - return; - } - - // If we can't find the destination but there is a base and we can - // find the base, then we are facing an unresolved fragment - if (frame.traverse(reference.base).has_value()) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - - assert(!reference.base.empty()); - const auto &identifier{reference.base}; - auto remote{resolver(identifier)}; - if (!remote.has_value()) { - throw sourcemeta::core::SchemaResolutionError( - identifier, "Could not resolve the reference to an external schema"); - } - - if (!remote.value().is_object() && !remote.value().is_boolean()) { - throw sourcemeta::core::SchemaReferenceError( - identifier, sourcemeta::core::to_pointer(pointer), - "The JSON document is not a valid JSON Schema"); - } - - try { - const sourcemeta::core::SchemaFrame remote_frame{ - sourcemeta::core::SchemaFrame::Mode::Root, - remote.value(), - walker, - resolver, - default_dialect, - "", - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - "", - remaining}; - charge(remaining, remote_frame); - } catch (const sourcemeta::core::SchemaUnknownBaseDialectError &) { - throw sourcemeta::core::SchemaReferenceError( - identifier, sourcemeta::core::to_pointer(pointer), - "The JSON document is not a valid JSON Schema"); - } - - callback(origin, pointer, identifier, remote.value()); - visited.emplace(identifier); - - // Official schemas can only reference other official schemas, so - // recursing into them can never surface further dependencies - if (sourcemeta::core::schema_is_official(identifier)) { - return; - } - - found.emplace_back(std::move(remote).value(), - sourcemeta::core::JSON::String{identifier}); - }); - - for (const auto &entry : found) { - dependencies_internal(std::get<0>(entry), walker, resolver, callback, - default_dialect, std::get<1>(entry), - {sourcemeta::core::EMPTY_WEAK_POINTER}, visited, - remaining); - } -} - -auto embed_schema(sourcemeta::core::JSON &root, - const sourcemeta::core::Pointer &container, - const std::string_view identifier, - sourcemeta::core::JSON &&target) -> void { - auto *current{&root}; - for (const auto &token : container) { - if (token.is_property()) { - current->assign_if_missing(token.to_property(), - sourcemeta::core::JSON::make_object()); - current = ¤t->at(token.to_property()); - } else { - assert(current->is_array() && current->size() >= token.to_index()); - current = ¤t->at(token.to_index()); - } - } - - if (!current->is_object()) { - throw sourcemeta::core::SchemaError( - "Could not bundle to a container path that is not an object"); - } - - std::string key{identifier}; - // Ensure we get a definitions entry that does not exist - while (current->defines(key)) { - key += "/x"; - } - - current->assign(key, std::move(target)); -} - -auto elevate_embedded_resources( - sourcemeta::core::JSON &remote, sourcemeta::core::JSON &root, - const sourcemeta::core::Pointer &container, - const sourcemeta::core::SchemaBaseDialect remote_dialect, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - std::string_view default_dialect, - std::unordered_map &bundled, - std::uint64_t &remaining) -> void { - const auto keyword{sourcemeta::blaze::definitions_keyword(remote_dialect)}; - const sourcemeta::core::JSON::String keyword_string{keyword}; - if (keyword.empty() || !remote.is_object() || - !remote.defines(keyword_string) || - !remote.at(keyword_string).is_object()) { - return; - } - - auto &defs{remote.at(keyword_string)}; - const auto remote_dialect_uri{ - sourcemeta::blaze::declared_dialect(remote, default_dialect)}; - - // Navigate to the root container once, as it doesn't change per entry - const sourcemeta::core::JSON *root_container{&root}; - bool container_exists{true}; - for (const auto &token : container) { - if (!token.is_property() || !root_container->is_object() || - !root_container->defines(token.to_property())) { - container_exists = false; - break; - } - - root_container = &root_container->at(token.to_property()); - } - - std::vector> to_extract; - std::vector to_remove; - for (const auto &entry : defs.as_object()) { - const auto &key{entry.first}; - const auto &value{entry.second}; - // Only an entry that declares an absolute identifier matching its key can - // ever be elevated, and framing rejects the fragment-only identifiers that - // older drafts use for anchors. Rule those out before paying for a frame - if (!value.is_object()) { - continue; - } - const auto *declared_id{value.try_at("$id")}; - if (declared_id == nullptr) { - declared_id = value.try_at("id"); - } - if (declared_id == nullptr || !declared_id->is_string() || - declared_id->to_string() != key || - !sourcemeta::core::URI{declared_id->to_string()}.is_absolute()) { - continue; - } - - // The remote's dialect is what an entry that declares none inherits, so - // hand it to the frame as the default rather than falling back after - sourcemeta::core::SchemaFrame entry_frame{ - sourcemeta::core::SchemaFrame::Mode::Root, - value, - walker, - resolver, - remote_dialect_uri, - "", - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - "", - remaining}; - charge(remaining, entry_frame); - const auto &identifier{entry_frame.root()}; - if (identifier.empty() || identifier != key || - !sourcemeta::core::URI{identifier}.is_absolute()) { - continue; - } - - const sourcemeta::core::JSON::String identifier_string{identifier}; - const auto defines_dialect{value.defines("$schema")}; - if (bundled.contains(identifier_string)) { - if (container_exists && root_container->is_object()) { - for (const auto &root_entry : root_container->as_object()) { - if (!root_entry.first.starts_with(identifier_string)) { - continue; - } - - // Same reasoning as above: rule out what cannot match, and what - // framing would reject, before paying for a frame - if (!root_entry.second.is_object()) { - continue; - } - const auto *stored_declared_id{root_entry.second.try_at("$id")}; - if (stored_declared_id == nullptr) { - stored_declared_id = root_entry.second.try_at("id"); - } - if (stored_declared_id == nullptr || - !stored_declared_id->is_string() || - stored_declared_id->to_string() != identifier_string || - !sourcemeta::core::URI{stored_declared_id->to_string()} - .is_absolute()) { - continue; - } - - sourcemeta::core::SchemaFrame stored_frame{ - sourcemeta::core::SchemaFrame::Mode::Root, - root_entry.second, - walker, - resolver, - remote_dialect_uri, - "", - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - "", - remaining}; - charge(remaining, stored_frame); - const auto &stored_id{stored_frame.root()}; - if (stored_id != identifier_string) { - continue; - } - - if (defines_dialect) { - if (root_entry.second != value) { - throw sourcemeta::core::SchemaError( - "Conflicting embedded resources with the same identifier"); - } - } else { - // The stored copy of the resource got its dialect stamped on - // extraction, so compare against a candidate that is stamped in - // the same way - auto candidate{value}; - candidate.assign("$schema", sourcemeta::core::JSON{ - sourcemeta::blaze::declared_dialect( - value, remote_dialect_uri)}); - if (root_entry.second != candidate) { - throw sourcemeta::core::SchemaError( - "Conflicting embedded resources with the same identifier"); - } - } - - break; - } - } - - to_remove.emplace_back(key); - } else { - to_extract.emplace_back(key, !defines_dialect); - bundled.emplace(identifier_string, identifier_string); - } - } - - for (const auto &[key, needs_dialect] : to_extract) { - auto value{std::move(defs.at(key))}; - defs.erase(key); - // Otherwise the elevated resource would be re-interpreted under the - // dialect of the schema it gets embedded into, which can differ from - // the dialect it inherited from the remote it was elevated out of - if (needs_dialect) { - value.assign("$schema", - sourcemeta::core::JSON{sourcemeta::blaze::declared_dialect( - value, remote_dialect_uri)}); - } - - embed_schema(root, container, key, std::move(value)); - } - - for (const auto &key : to_remove) { - defs.erase(key); - } - - if (defs.empty()) { - remote.erase(sourcemeta::core::JSON::String{keyword}); - } -} - -auto bundle_schema(sourcemeta::core::JSON &root, - const sourcemeta::core::Pointer &container, - sourcemeta::core::JSON &subschema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const sourcemeta::blaze::BundleMode mode, - std::string_view default_dialect, - std::string_view default_id, - const sourcemeta::core::SchemaFrame::Paths &paths, - std::string_view default_base, - std::unordered_map &bundled, - std::uint64_t &remaining, const std::size_t depth = 0) - -> void { - // Create a fresh frame for each schema we analyze to avoid key collisions - // between different schemas that have references at the same pointer paths - static const sourcemeta::core::SchemaFrame::Paths NESTED_PATHS{ - sourcemeta::core::EMPTY_WEAK_POINTER}; - const sourcemeta::core::SchemaFrame frame{ - sourcemeta::core::SchemaFrame::Mode::References, subschema, walker, - resolver, default_dialect, default_id, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - // We only want to frame in "wrapper" mode for the top level object, which - // is also the only one that the base the caller retrieved it from applies - // to, as every remote carries the identity it was resolved by - depth == 0 ? paths : NESTED_PATHS, - depth == 0 ? default_base : std::string_view{}, remaining}; - charge(remaining, frame); - - std::vector> - deferred; - std::vector< - std::pair> - ref_rewrites; - - frame.for_each_unresolved_reference([&](const auto &pointer, - const auto &reference) -> void { - // We don't want to bundle official schemas, as we can expect - // virtually all implementations to understand them out of the box. - // Depending on the bundling strategy, we may skip meta-schemas entirely - if (is_skippable_metaschema_reference(mode, pointer, - reference.destination)) { - return; - } - - // If we can't find the destination but there is a base and we can - // find base, then we are facing an unresolved fragment - if (!reference.base.empty() && frame.traverse(reference.base).has_value()) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - - if (reference.base.empty()) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - - assert(!reference.base.empty()); - const sourcemeta::core::JSON::String identifier{reference.base}; - - if (bundled.contains(identifier)) { - const auto &mapped_id{bundled.at(identifier)}; - if (mapped_id != identifier) { - sourcemeta::core::URI rewrite_uri{mapped_id}; - if (reference.fragment.has_value()) { - rewrite_uri.fragment(reference.fragment.value()); - } - - ref_rewrites.emplace_back(sourcemeta::core::to_pointer(pointer), - rewrite_uri.recompose()); - } - - return; - } - - auto resolved{resolver(identifier)}; - if (!resolved.has_value()) { - if (frame.traverse(identifier).has_value()) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - - throw sourcemeta::core::SchemaResolutionError( - identifier, "Could not resolve the reference to an external schema"); - } - - // Bundling rewrites the schema before embedding it, so it needs a copy - // it owns rather than whatever the resolver chose to hand back - auto remote{std::move(resolved).to_owned()}; - if (!remote.is_object() && !remote.is_boolean()) { - throw sourcemeta::core::SchemaReferenceError( - identifier, sourcemeta::core::to_pointer(pointer), - "The JSON document is not a valid JSON Schema"); - } - - std::optional remote_root_frame; - try { - remote_root_frame.emplace( - sourcemeta::core::SchemaFrame::Mode::Root, remote, walker, resolver, - default_dialect, "", - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - sourcemeta::core::SchemaFrame::Paths{ - sourcemeta::core::EMPTY_WEAK_POINTER}, - "", remaining); - charge(remaining, remote_root_frame.value()); - } catch (const sourcemeta::core::SchemaUnknownBaseDialectError &) { - throw sourcemeta::core::SchemaReferenceError( - identifier, sourcemeta::core::to_pointer(pointer), - "The JSON document is not a valid JSON Schema"); - } - - const auto remote_base_dialect{ - remote_root_frame->root_location().value().get().base_dialect}; - auto remote_id = remote_root_frame->root(); - - // If the reference has a fragment, verify it exists in the remote - // schema - if (reference.fragment.has_value()) { - // A pointer fragment names a place of the document, and the document can - // answer for that on its own without paying to frame it - const auto fragment_pointer{sourcemeta::core::fragment_to_pointer( - sourcemeta::core::URI{reference.destination})}; - bool exists{fragment_pointer.has_value() && - sourcemeta::core::try_get(remote, fragment_pointer.value()) != - nullptr}; - - // An anchor is not a place of the document, and the drafts that spell - // identifiers as `id` let one look just like a pointer, so a miss above - // still has to ask the frame. Only the anchors of the remote matter - // here, rather than every pointer of it - if (!exists) { - const sourcemeta::core::SchemaFrame remote_frame{ - sourcemeta::core::SchemaFrame::Mode::Locations, - remote, - walker, - resolver, - default_dialect, - identifier, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - "", - remaining}; - charge(remaining, remote_frame); - exists = remote_frame.traverse(reference.destination).has_value(); - } - - if (!exists) { - throw sourcemeta::core::SchemaReferenceError( - reference.destination, sourcemeta::core::to_pointer(pointer), - "Could not resolve schema reference"); - } - } - - sourcemeta::core::JSON::String effective_id{ - remote_id.empty() ? sourcemeta::core::JSON::String{identifier} - : sourcemeta::core::JSON::String{remote_id}}; - - if (remote.is_object()) { - // Otherwise the embedded resource would be re-interpreted under the - // dialect of the schema it gets embedded into, which can differ from - // the default dialect that the remote was resolved with - if (!remote.defines("$schema")) { - remote.assign("$schema", sourcemeta::core::JSON{ - sourcemeta::blaze::declared_dialect( - remote, default_dialect)}); - } - - sourcemeta::core::schema_reidentify(remote, effective_id, - remote_base_dialect); - } - - if (effective_id != identifier) { - sourcemeta::core::URI rewrite_uri{effective_id}; - if (reference.fragment.has_value()) { - rewrite_uri.fragment(reference.fragment.value()); - } - - ref_rewrites.emplace_back(sourcemeta::core::to_pointer(pointer), - rewrite_uri.recompose()); - } - - bundled.emplace(identifier, effective_id); - bundled.emplace(effective_id, effective_id); - deferred.emplace_back(std::move(remote), std::move(effective_id), - remote_base_dialect); - }); - - for (auto &[rewrite_pointer, rewrite_value] : ref_rewrites) { - sourcemeta::core::set(subschema, rewrite_pointer, - sourcemeta::core::JSON{rewrite_value}); - } - - for (auto &[remote, effective_id, remote_dialect] : deferred) { - bundle_schema(root, container, remote, walker, resolver, mode, - default_dialect, effective_id, paths, default_base, bundled, - remaining, depth + 1); - elevate_embedded_resources(remote, root, container, remote_dialect, walker, - resolver, default_dialect, bundled, remaining); - embed_schema(root, container, effective_id, std::move(remote)); - } -} - -} // namespace - -namespace sourcemeta::blaze { - -auto dependencies(const sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const DependencyCallback &callback, - std::string_view default_dialect, std::string_view default_id, - const sourcemeta::core::SchemaFrame::Paths &paths, - const std::uint64_t max_locations) -> void { - std::unordered_set visited; - auto remaining{max_locations}; - try { - dependencies_internal(schema, walker, resolver, callback, default_dialect, - default_id, paths, visited, remaining); - } catch (const sourcemeta::core::SchemaFrameLimitError &) { - // Every frame spends from what is left rather than from the whole, so the - // one that ran out reports what it was handed. The caller set the limit - // for the operation, so that is what the operation reports back - throw sourcemeta::core::SchemaFrameLimitError{max_locations}; - } -} - -// TODO: Refactor this function to internally rely on the `.dependencies()` -// function -static auto bundle_internal( - sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, const BundleMode mode, - std::string_view default_dialect, std::string_view default_id, - const std::optional &default_container, - const sourcemeta::core::SchemaFrame::Paths &paths, - std::string_view default_base, std::uint64_t &remaining) -> void { - // Pre-scan the schema to find any already-embedded schemas and mark them - // as bundled to avoid re-embedding them. This includes the root schema itself - // and any schemas already embedded within it - std::unordered_map - bundled; - sourcemeta::core::SchemaFrame initial_frame{ - sourcemeta::core::SchemaFrame::Mode::Locations, - schema, - walker, - resolver, - default_dialect, - default_id, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - paths, - default_base, - remaining}; - charge(remaining, initial_frame); - initial_frame.for_each_resource_uri([&bundled](const auto &uri) -> void { - bundled.emplace(sourcemeta::core::JSON::String{uri}, - sourcemeta::core::JSON::String{uri}); - }); - if (default_container.has_value()) { - // This is undefined behavior - assert(!default_container.value().empty()); - bundle_schema(schema, default_container.value(), schema, walker, resolver, - mode, default_dialect, default_id, paths, default_base, - bundled, remaining); - return; - } - - // If the schema identifier is implicit, add it to the top-level of the - // bundled schema. Otherwise, potential relative references based on this - // implicit base URI will likely not resolve unless end users happen to - // know that this implicit base URI is. Note that boolean schemas cannot - // declare identifiers, so we leave those untouched - if (!default_id.empty() && schema.is_object()) { - // Deliberately framed without a default identifier, so that the root - // comes back empty exactly when the schema declares none of its own - sourcemeta::core::SchemaFrame declared_frame{ - sourcemeta::core::SchemaFrame::Mode::Root, - schema, - walker, - resolver, - default_dialect, - "", - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - "", - remaining}; - charge(remaining, declared_frame); - if (declared_frame.root().empty()) { - schema_reidentify(schema, default_id, resolver, default_dialect); - } - } - - std::optional schema_root_frame; - try { - schema_root_frame.emplace( - sourcemeta::core::SchemaFrame::Mode::Root, schema, walker, resolver, - default_dialect, default_id, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - sourcemeta::core::SchemaFrame::Paths{ - sourcemeta::core::EMPTY_WEAK_POINTER}, - "", remaining); - charge(remaining, schema_root_frame.value()); - } catch (const sourcemeta::core::SchemaUnknownBaseDialectError &) { - throw sourcemeta::core::SchemaError( - "Could not determine how to perform bundling in this dialect"); - } - - const auto schema_base_dialect{ - schema_root_frame->root_location().value().get().base_dialect}; - - const auto container_keyword{definitions_keyword(schema_base_dialect)}; - if (container_keyword.empty()) { - sourcemeta::core::SchemaFrame frame{ - sourcemeta::core::SchemaFrame::Mode::References, - schema, - walker, - resolver, - default_dialect, - default_id, - sourcemeta::core::SchemaFrame::IdentifierMode::Additional, - {sourcemeta::core::EMPTY_WEAK_POINTER}, - default_base, - remaining}; - charge(remaining, frame); - if (frame.standalone()) { - return; - } - - throw sourcemeta::core::SchemaError( - "Could not determine how to perform bundling in this dialect"); - } - - if (ref_overrides_adjacent_keywords(schema_base_dialect) && - schema.is_object() && schema.defines("$ref")) { - if (schema.size() == 1) { - const auto is_draft3{ - schema_base_dialect == - sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3 || - schema_base_dialect == - sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER}; - auto branches{sourcemeta::core::JSON::make_array()}; - branches.push_back(schema); - schema.at("$ref").into(std::move(branches)); - schema.rename("$ref", is_draft3 ? "extends" : "allOf"); - } else { - throw sourcemeta::core::SchemaError( - "Cannot bundle a JSON Schema Draft 7 or older with a top-level " - "`$ref` (which overrides sibling keywords) without introducing " - "undefined behavior"); - } - } - - bundle_schema(schema, {sourcemeta::core::JSON::String{container_keyword}}, - schema, walker, resolver, mode, default_dialect, default_id, - paths, default_base, bundled, remaining); -} - -auto bundle(sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const BundleMode mode, std::string_view default_dialect, - std::string_view default_id, - const std::optional &default_container, - const sourcemeta::core::SchemaFrame::Paths &paths, - std::string_view default_base, const std::uint64_t max_locations) - -> void { - auto remaining{max_locations}; - try { - bundle_internal(schema, walker, resolver, mode, default_dialect, default_id, - default_container, paths, default_base, remaining); - } catch (const sourcemeta::core::SchemaFrameLimitError &) { - // Every frame spends from what is left rather than from the whole, so the - // one that ran out reports what it was handed. The caller set the limit - // for the operation, so that is what the operation reports back - throw sourcemeta::core::SchemaFrameLimitError{max_locations}; - } -} - -auto bundle(const sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const BundleMode mode, std::string_view default_dialect, - std::string_view default_id, - const std::optional &default_container, - const sourcemeta::core::SchemaFrame::Paths &paths, - std::string_view default_base, const std::uint64_t max_locations) - -> sourcemeta::core::JSON { - sourcemeta::core::JSON copy = schema; - bundle(copy, walker, resolver, mode, default_dialect, default_id, - default_container, paths, default_base, max_locations); - return copy; -} - -} // namespace sourcemeta::blaze diff --git a/vendor/blaze/src/bundle/helpers.h b/vendor/blaze/src/bundle/helpers.h deleted file mode 100644 index 9eccfa75..00000000 --- a/vendor/blaze/src/bundle/helpers.h +++ /dev/null @@ -1,105 +0,0 @@ -#ifndef SOURCEMETA_BLAZE_BUNDLE_HELPERS_H -#define SOURCEMETA_BLAZE_BUNDLE_HELPERS_H - -#include - -#include - -#include // assert -#include // std::string_view - -namespace sourcemeta::blaze { - -inline auto id_keyword(const sourcemeta::core::SchemaBaseDialect base_dialect) - -> std::string_view { - switch (base_dialect) { - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2020_12: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2020_12_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2019_09: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2019_09_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6_HYPER: - return "$id"; - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_2_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_1_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_0_HYPER: - return "id"; - } - - assert(false); - return "$id"; -} - -inline auto -definitions_keyword(const sourcemeta::core::SchemaBaseDialect base_dialect) - -> std::string_view { - switch (base_dialect) { - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2020_12: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2020_12_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2019_09: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2019_09_HYPER: - return "$defs"; - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER: - return "definitions"; - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_2_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_1_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_0_HYPER: - return ""; - } - - assert(false); - return "$defs"; -} - -// In older drafts, the presence of `$ref` would override any sibling keywords -// See -// https://json-schema.org/draft-07/draft-handrews-json-schema-01#rfc.section.8.3 -inline auto ref_overrides_adjacent_keywords( - const sourcemeta::core::SchemaBaseDialect base_dialect) -> bool { - switch (base_dialect) { - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_7_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_6_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_2_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_1_HYPER: - case sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_0_HYPER: - return true; - default: - return false; - } -} - -// The dialect a schema declares, falling back to the given default -inline auto declared_dialect(const sourcemeta::core::JSON &schema, - const std::string_view default_dialect) - -> std::string_view { - if (!schema.is_object()) { - return default_dialect; - } - - const auto *dialect{schema.try_at("$schema")}; - return (dialect != nullptr && dialect->is_string()) ? dialect->to_string() - : default_dialect; -} - -} // namespace sourcemeta::blaze - -#endif diff --git a/vendor/blaze/src/bundle/include/sourcemeta/blaze/bundle.h b/vendor/blaze/src/bundle/include/sourcemeta/blaze/bundle.h deleted file mode 100644 index d67de7b2..00000000 --- a/vendor/blaze/src/bundle/include/sourcemeta/blaze/bundle.h +++ /dev/null @@ -1,264 +0,0 @@ -#ifndef SOURCEMETA_BLAZE_BUNDLE_H_ -#define SOURCEMETA_BLAZE_BUNDLE_H_ - -/// @defgroup bundle Bundle -/// @brief Bundle JSON Schemas by inlining their external references. -/// -/// This functionality is included as follows: -/// -/// ```cpp -/// #include -/// ``` - -#ifndef SOURCEMETA_BLAZE_BUNDLE_EXPORT -#include -#endif - -#include -#include - -#include - -#include // std::uint8_t, std::uint64_t -#include // std::function -#include // std::numeric_limits -#include // std::optional, std::nullopt -#include // std::string_view - -namespace sourcemeta::blaze { - -/// @ingroup bundle -/// A callback to get dependency information -/// - Origin URI (empty if none) -/// - Pointer (reference keyword from the origin) -/// - Target URI -/// - Target schema -using DependencyCallback = - std::function; - -/// @ingroup bundle -/// The strategies that the bundling process can follow -enum class BundleMode : std::uint8_t { - /// Embed every external reference, including any non-official - /// meta-schemas that the schema or its dependencies declare, along - /// with the dependencies of those meta-schemas - NonOfficialMetaschemas, - /// Embed every external reference, skipping meta-schema - /// declarations entirely - References -}; - -/// @ingroup bundle -/// -/// This function recursively traverses and reports the external references in a -/// schema. References to official schemas are reported but not traversed into, -/// as official schemas can only reference other official schemas. For example: -/// -/// ```cpp -/// #include -/// #include -/// #include -/// -/// // A custom resolver that knows about an additional schema -/// static auto test_resolver(std::string_view identifier) -/// -> sourcemeta::core::SchemaResolverResult { -/// if (identifier == "https://www.example.com/test") { -/// return sourcemeta::core::parse_json(R"JSON({ -/// "$id": "https://www.example.com/test", -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "type": "string" -/// })JSON"); -/// } else { -/// return sourcemeta::core::schema_resolver(identifier); -/// } -/// } -/// -/// sourcemeta::core::JSON document = -/// sourcemeta::core::parse_json(R"JSON({ -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "items": { "$ref": "https://www.example.com/test" } -/// })JSON"); -/// -/// sourcemeta::blaze::dependencies(document, -/// sourcemeta::core::schema_walker, test_resolver, -/// [](const auto &origin, -/// const auto &pointer, -/// const auto &target, -/// const auto &schema) { -/// // Do something with the information -/// }); -/// ``` -/// -/// How many schemas this ends up analysing follows from what the resolver -/// hands back rather than from the schema the caller passed in, so pass -/// `max_locations` to bound it. Every frame that this constructs spends from -/// that one limit, throwing sourcemeta::core::SchemaFrameLimitError once it -/// runs out. See sourcemeta::core::SchemaFrame for what the unit counts and -/// what it leaves to the caller -SOURCEMETA_BLAZE_BUNDLE_EXPORT -auto dependencies( - const sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const DependencyCallback &callback, std::string_view default_dialect = "", - std::string_view default_id = "", - const sourcemeta::core::SchemaFrame::Paths &paths = - {sourcemeta::core::EMPTY_WEAK_POINTER}, - std::uint64_t max_locations = std::numeric_limits::max()) - -> void; - -/// @ingroup bundle -/// -/// This function bundles a JSON Schema (starting from Draft 4) by embedding -/// every remote reference into the top level schema resource, handling circular -/// dependencies and more. This overload mutates the input schema. For example: -/// -/// ```cpp -/// #include -/// #include -/// #include -/// #include -/// -/// // A custom resolver that knows about an additional schema -/// static auto test_resolver(std::string_view identifier) -/// -> sourcemeta::core::SchemaResolverResult { -/// if (identifier == "https://www.example.com/test") { -/// return sourcemeta::core::parse_json(R"JSON({ -/// "$id": "https://www.example.com/test", -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "type": "string" -/// })JSON"); -/// } else { -/// return sourcemeta::core::schema_resolver(identifier); -/// } -/// } -/// -/// sourcemeta::core::JSON document = -/// sourcemeta::core::parse_json(R"JSON({ -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "items": { "$ref": "https://www.example.com/test" } -/// })JSON"); -/// -/// sourcemeta::blaze::bundle(document, -/// sourcemeta::core::schema_walker, test_resolver, -/// sourcemeta::blaze::BundleMode::NonOfficialMetaschemas); -/// -/// const sourcemeta::core::JSON expected = -/// sourcemeta::core::parse_json(R"JSON({ -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "items": { "$ref": "https://www.example.com/test" }, -/// "$defs": { -/// "https://www.example.com/test": { -/// "$id": "https://www.example.com/test", -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "type": "string" -/// } -/// } -/// })JSON"); -/// -/// assert(document == expected); -/// ``` -/// -/// Pass `default_base` to state the base URI that the document was retrieved -/// from, which a relative reference within any of the given `paths` resolves -/// against. As with sourcemeta::core::SchemaFrame, this does not claim that the -/// document declares an identifier, so bundling never writes it into the -/// document -/// -/// How many schemas this ends up embedding follows from what the resolver -/// hands back rather than from the schema the caller passed in, so pass -/// `max_locations` to bound it. Every frame that bundling constructs spends -/// from that one limit, throwing sourcemeta::core::SchemaFrameLimitError once -/// it runs out, which bounds how many remote schemas this embeds and how deep -/// it recurses along with how much framing it does. Note that a remote is -/// copied out of the resolver before anything charges for it, so the limit -/// bounds how many oversized schemas get copied rather than whether one does -SOURCEMETA_BLAZE_BUNDLE_EXPORT -auto bundle( - sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, const BundleMode mode, - std::string_view default_dialect = "", std::string_view default_id = "", - const std::optional &default_container = - std::nullopt, - const sourcemeta::core::SchemaFrame::Paths &paths = - {sourcemeta::core::EMPTY_WEAK_POINTER}, - std::string_view default_base = "", - std::uint64_t max_locations = std::numeric_limits::max()) - -> void; - -/// @ingroup bundle -/// -/// This function bundles a JSON Schema (starting from Draft 4) by embedding -/// every remote reference into the top level schema resource, handling circular -/// dependencies and more. This overload returns a new schema, without mutating -/// the input schema. For example: -/// -/// ```cpp -/// #include -/// #include -/// #include -/// #include -/// -/// // A custom resolver that knows about an additional schema -/// static auto test_resolver(std::string_view identifier) -/// -> sourcemeta::core::SchemaResolverResult { -/// if (identifier == "https://www.example.com/test") { -/// return sourcemeta::core::parse_json(R"JSON({ -/// "$id": "https://www.example.com/test", -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "type": "string" -/// })JSON"); -/// } else { -/// return sourcemeta::core::schema_resolver(identifier); -/// } -/// } -/// -/// const sourcemeta::core::JSON document = -/// sourcemeta::core::parse_json(R"JSON({ -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "items": { "$ref": "https://www.example.com/test" } -/// })JSON"); -/// -/// const sourcemeta::core::JSON result = -/// sourcemeta::blaze::bundle(document, -/// sourcemeta::core::schema_walker, test_resolver, -/// sourcemeta::blaze::BundleMode::NonOfficialMetaschemas); -/// -/// const sourcemeta::core::JSON expected = -/// sourcemeta::core::parse_json(R"JSON({ -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "items": { "$ref": "https://www.example.com/test" }, -/// "$defs": { -/// "https://www.example.com/test": { -/// "$id": "https://www.example.com/test", -/// "$schema": "https://json-schema.org/draft/2020-12/schema", -/// "type": "string" -/// } -/// } -/// })JSON"); -/// -/// assert(result == expected); -/// ``` -/// -/// As with the mutating overload, pass `max_locations` to bound how much -/// analysis an untrusted schema and whatever the resolver hands back for it -/// may cost -SOURCEMETA_BLAZE_BUNDLE_EXPORT -auto bundle( - const sourcemeta::core::JSON &schema, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, const BundleMode mode, - std::string_view default_dialect = "", std::string_view default_id = "", - const std::optional &default_container = - std::nullopt, - const sourcemeta::core::SchemaFrame::Paths &paths = - {sourcemeta::core::EMPTY_WEAK_POINTER}, - std::string_view default_base = "", - std::uint64_t max_locations = std::numeric_limits::max()) - -> sourcemeta::core::JSON; - -} // namespace sourcemeta::blaze - -#endif diff --git a/vendor/blaze/src/canonicalizer/canonicalize.cc b/vendor/blaze/src/canonicalizer/canonicalize.cc index e5acf67c..08094eab 100644 --- a/vendor/blaze/src/canonicalizer/canonicalize.cc +++ b/vendor/blaze/src/canonicalizer/canonicalize.cc @@ -243,6 +243,87 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, } } +auto is_draft3(const core::SchemaBaseDialect base_dialect) -> bool { + return base_dialect == core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3 || + base_dialect == core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER; +} + +/// Remove every Draft 3 identifier, including anchors and the identifier of +/// the root. Every reference is first resolved while the frame still knows +/// the identifiers, and rewritten into a form that does not depend on any of +/// them: a JSON Pointer into this document when the destination is local, or +/// the absolute URI when it is not. Only then are the identifiers erased, so +/// no reference can lose the base it was resolved against. +/// +/// This only applies to documents that are Draft 3 throughout. A document of +/// another dialect may embed a Draft 3 resource and reference it by its +/// identifier from outside, with a reference this pass must not rewrite, so +/// such documents are left untouched +auto eliminate_identifiers(sourcemeta::core::JSON &schema, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + const std::string_view default_dialect, + const std::string_view default_id) -> void { + if (!schema.is_object()) { + return; + } + + std::vector> reference_changes; + std::vector identifier_owners; + + { + const core::SchemaFrame frame{core::SchemaFrame::Mode::References, + schema, + walker, + resolver, + default_dialect, + default_id, + core::SchemaFrame::IdentifierMode::Fallback}; + + if (frame.any_subschema( + [](const core::SchemaFrame::Location &location) -> bool { + return !is_draft3(location.base_dialect); + })) { + return; + } + + frame.for_each_reference( + [&](const core::SchemaReferenceType, const core::WeakPointer &origin, + const core::SchemaFrame::Reference &reference) -> void { + assert(!origin.empty() && origin.back().is_property()); + if (origin.back().to_property() != "$ref") { + return; + } + + const auto destination{frame.traverse(reference.destination)}; + reference_changes.emplace_back( + core::to_pointer(origin), + destination.has_value() + ? core::to_uri(destination->get().pointer).recompose() + : reference.destination); + }); + + frame.for_each_subschema( + [&](const core::SchemaFrame::Location &location) -> void { + const auto &subschema{core::get(schema, location.pointer)}; + if (subschema.is_object() && subschema.defines("id")) { + identifier_owners.push_back(core::to_pointer(location.pointer)); + } + }); + } + + for (const auto &[pointer, value] : reference_changes) { + core::set(schema, pointer, core::JSON{value}); + } + + // Rewriting a reference cannot turn a subschema into something else + for (const auto &pointer : identifier_owners) { + auto &subschema{core::get(schema, pointer)}; + assert(subschema.is_object()); + subschema.erase("id"); + } +} + #include "helpers.h" #include "rules/additional_items_implicit.h" @@ -514,6 +595,12 @@ auto canonicalize(sourcemeta::core::JSON &schema, rules.push_back(make_rule()); rules.push_back(make_rule()); apply(rules, schema, walker, resolver, default_dialect, default_id); + + // Identifiers are removed last, once no rule will reshape the schema again. + // Removing them first would hand the rules references that are plain JSON + // Pointers, which some of them do not yet preserve when they move or copy + // the subschemas those pointers go through + eliminate_identifiers(schema, walker, resolver, default_dialect, default_id); } } // namespace sourcemeta::blaze diff --git a/vendor/blaze/src/codegen/CMakeLists.txt b/vendor/blaze/src/codegen/CMakeLists.txt index b90ae394..b2a5ee49 100644 --- a/vendor/blaze/src/codegen/CMakeLists.txt +++ b/vendor/blaze/src/codegen/CMakeLists.txt @@ -16,7 +16,5 @@ target_link_libraries(sourcemeta_blaze_codegen PUBLIC sourcemeta::core::json) target_link_libraries(sourcemeta_blaze_codegen PUBLIC sourcemeta::core::jsonschema) -target_link_libraries(sourcemeta_blaze_codegen PRIVATE - sourcemeta::blaze::bundle) target_link_libraries(sourcemeta_blaze_codegen PRIVATE sourcemeta::blaze::canonicalizer) diff --git a/vendor/blaze/src/codegen/codegen.cc b/vendor/blaze/src/codegen/codegen.cc index a74510f5..9053ed65 100644 --- a/vendor/blaze/src/codegen/codegen.cc +++ b/vendor/blaze/src/codegen/codegen.cc @@ -1,4 +1,3 @@ -#include #include #include #include @@ -78,9 +77,10 @@ auto compile(const sourcemeta::core::JSON &input, // (1) Bundle the schema to resolve external references // -------------------------------------------------------------------------- - auto schema{sourcemeta::blaze::bundle( - input, walker, resolver, sourcemeta::blaze::BundleMode::References, - default_dialect, default_id)}; + sourcemeta::core::SchemaBundleOptions bundle_options; + bundle_options.mode = sourcemeta::core::SchemaBundleOptions::Mode::References; + auto schema{sourcemeta::core::schema_bundle( + input, walker, resolver, default_dialect, default_id, bundle_options)}; // -------------------------------------------------------------------------- // (2) Canonicalize the schema for easier analysis diff --git a/vendor/blaze/src/compiler/CMakeLists.txt b/vendor/blaze/src/compiler/CMakeLists.txt index 4ec75763..987106ef 100644 --- a/vendor/blaze/src/compiler/CMakeLists.txt +++ b/vendor/blaze/src/compiler/CMakeLists.txt @@ -26,7 +26,5 @@ target_link_libraries(sourcemeta_blaze_compiler PRIVATE sourcemeta::core::uri) target_link_libraries(sourcemeta_blaze_compiler PUBLIC sourcemeta::core::jsonschema) -target_link_libraries(sourcemeta_blaze_compiler PRIVATE - sourcemeta::blaze::bundle) target_link_libraries(sourcemeta_blaze_compiler PUBLIC sourcemeta::blaze::evaluator) diff --git a/vendor/blaze/src/compiler/compile.cc b/vendor/blaze/src/compiler/compile.cc index 286078ca..fabb4bf7 100644 --- a/vendor/blaze/src/compiler/compile.cc +++ b/vendor/blaze/src/compiler/compile.cc @@ -1,4 +1,3 @@ -#include #include #include #include @@ -833,10 +832,11 @@ auto compile(const sourcemeta::core::JSON &schema, // Make sure the input schema is bundled, otherwise we won't be able to // resolve remote references here. Meta-schemas are not needed, as we // can determine vocabularies through the resolver - const sourcemeta::core::JSON result{sourcemeta::blaze::bundle( - schema, walker, resolver, sourcemeta::blaze::BundleMode::References, - default_dialect, default_id, std::nullopt, - {sourcemeta::core::EMPTY_WEAK_POINTER}, "", max_locations)}; + sourcemeta::core::SchemaBundleOptions bundle_options; + bundle_options.mode = sourcemeta::core::SchemaBundleOptions::Mode::References; + bundle_options.max_locations = max_locations; + const sourcemeta::core::JSON result{sourcemeta::core::schema_bundle( + schema, walker, resolver, default_dialect, default_id, bundle_options)}; sourcemeta::core::SchemaFrame frame{ sourcemeta::core::SchemaFrame::Mode::References, diff --git a/vendor/blaze/src/configuration/CMakeLists.txt b/vendor/blaze/src/configuration/CMakeLists.txt index d3410611..9a9d08c0 100644 --- a/vendor/blaze/src/configuration/CMakeLists.txt +++ b/vendor/blaze/src/configuration/CMakeLists.txt @@ -11,4 +11,3 @@ target_link_libraries(sourcemeta_blaze_configuration PRIVATE sourcemeta::core::u target_link_libraries(sourcemeta_blaze_configuration PRIVATE sourcemeta::core::io) target_link_libraries(sourcemeta_blaze_configuration PRIVATE sourcemeta::core::crypto) target_link_libraries(sourcemeta_blaze_configuration PRIVATE sourcemeta::core::jsonschema) -target_link_libraries(sourcemeta_blaze_configuration PRIVATE sourcemeta::blaze::bundle) diff --git a/vendor/blaze/src/configuration/fetch.cc b/vendor/blaze/src/configuration/fetch.cc index 06a713f9..8c655ee3 100644 --- a/vendor/blaze/src/configuration/fetch.cc +++ b/vendor/blaze/src/configuration/fetch.cc @@ -1,4 +1,3 @@ -#include #include #include @@ -117,10 +116,9 @@ auto fetch_and_write( try { const std::string default_dialect_value{default_dialect.value_or("")}; - sourcemeta::blaze::bundle( - out_schema, sourcemeta::core::schema_walker, resolver, - sourcemeta::blaze::BundleMode::NonOfficialMetaschemas, - default_dialect_value, dependency_uri); + sourcemeta::core::schema_bundle(out_schema, sourcemeta::core::schema_walker, + resolver, default_dialect_value, + dependency_uri); } catch (...) { emit_event(callback, FetchEvent::Type::Error, dependency_uri, dependency_path, index, total, "Failed to bundle schema", diff --git a/vendor/blaze/src/convert/CMakeLists.txt b/vendor/blaze/src/convert/CMakeLists.txt index efa1f6c6..dd2b242c 100644 --- a/vendor/blaze/src/convert/CMakeLists.txt +++ b/vendor/blaze/src/convert/CMakeLists.txt @@ -7,11 +7,12 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT blaze NAME convert # Rules rules/definitions_to_defs.h + rules/dependencies_to_dependent.h rules/draft_official_dialect_with_https.h rules/draft_official_dialect_without_empty_fragment.h rules/empty_object_as_true.h rules/enum_to_const.h - rules/metaschema_vocabulary.h + rules/modern_official_dialect_with_empty_fragment.h rules/prefix_promoted_2020_12_keywords.h rules/prefix_promoted_draft_2019_09_keywords.h rules/prefix_promoted_draft_4_keywords.h diff --git a/vendor/blaze/src/convert/convert.cc b/vendor/blaze/src/convert/convert.cc index b397e10f..97232aac 100644 --- a/vendor/blaze/src/convert/convert.cc +++ b/vendor/blaze/src/convert/convert.cc @@ -31,6 +31,7 @@ using namespace sourcemeta::core; namespace { +#include "helpers.h" #include "rule.h" using Rule = std::tuple, bool>; @@ -42,13 +43,60 @@ template T> std::is_same_v}; } +/// A reference that lands on something other than a schema is not a reference +/// the conversion can carry across dialects, as the document never had one +auto assert_schema_references(const core::SchemaFrame &frame) -> void { + frame.for_each_reference( + [&frame](const core::SchemaReferenceType, const core::WeakPointer &origin, + const core::SchemaFrame::Reference &reference) -> void { + const auto destination{frame.traverse(reference.destination)}; + if (destination.has_value() && + destination.value().get().type == + core::SchemaFrame::LocationType::Pointer) { + throw ConvertInvalidReferenceError{reference.destination, + core::to_pointer(origin)}; + } + }); +} + +/// Conversion renames keywords, while a meta-schema names those same keywords +/// as ordinary data that nothing renames alongside them. Until the two can be +/// told apart, a document that describes itself or that carries the +/// meta-schema something in it declares is refused. A dialect the ladder does +/// not name is refused too, as there are no rules for moving a schema off it +auto assert_convertible_dialects(const core::JSON &schema, + const core::SchemaFrame &frame, + const std::string_view default_id) -> void { + const auto document{frame.traverse(core::EMPTY_WEAK_POINTER)}; + if (document.has_value() && + describes_itself(schema, document.value().get().base_dialect, + default_id)) { + throw ConvertUnsupportedMetaschemaError{schema.at("$schema").to_string(), + core::EMPTY_POINTER}; + } + + frame.for_each_subschema( + [&schema, &frame](const core::SchemaFrame::Location &location) -> void { + auto pointer{core::to_pointer(location.pointer)}; + if (is_metaschema_target(core::get(schema, pointer), frame, + location.pointer)) { + throw ConvertUnsupportedMetaschemaError{location.dialect, + std::move(pointer)}; + } + + if (!names_ladder_dialect(location.dialect)) { + throw ConvertUnsupportedDialectError{location.dialect, + std::move(pointer)}; + } + }); +} + /// Apply the given rules top-down to every subschema until none of them applies auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, const sourcemeta::core::SchemaWalker &walker, const sourcemeta::core::SchemaResolver &resolver, const std::string_view default_dialect, - const std::string_view default_id, const bool is_metaschema) - -> void { + const std::string_view default_id) -> void { assert(!rules.empty()); struct ProcessedRuleHasher { @@ -77,6 +125,7 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, }; std::vector potentially_broken_references; + bool asserted{false}; while (true) { if (!frame.has_value()) { @@ -87,6 +136,12 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, frame.emplace(core::SchemaFrame::Mode::References, schema, walker, resolver, default_dialect, default_id, sourcemeta::core::SchemaFrame::IdentifierMode::Fallback); + + if (!asserted) { + assert_convertible_dialects(schema, frame.value(), default_id); + assert_schema_references(frame.value()); + asserted = true; + } } std::unordered_set visited; @@ -107,9 +162,9 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, frame->vocabularies(location, resolver)}; for (const auto &[rule, reframe_after_transform] : rules) { - const auto outcome{ - rule->condition(current, schema, current_vocabularies, *frame, - location, walker, resolver, is_metaschema)}; + const auto outcome{rule->condition(current, schema, + current_vocabularies, *frame, + location, walker, resolver)}; if (!outcome) { continue; @@ -165,7 +220,11 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, new_location.value().get().relative_pointer}; const auto current_slice{entry_pointer.slice(resource_offset)}; for (const auto &saved_reference : potentially_broken_references) { - if (core::try_get(schema, saved_reference.target_pointer)) { + // A reference only breaks when its destination stops resolving. + // The target sitting at a different pointer than before is not + // enough, as a resource that moved as a whole keeps resolving + // the fragments that its own identifier is the base of + if (frame->traverse(saved_reference.destination).has_value()) { continue; } @@ -216,7 +275,7 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, assert(!rule->condition(current, schema, new_vocabularies, *frame, new_location.value().get(), walker, - resolver, is_metaschema)); + resolver)); std::tuple mark{ entry_pointer, rule->name(), current}; @@ -241,14 +300,13 @@ auto apply(const std::vector &rules, sourcemeta::core::JSON &schema, } } -#include "helpers.h" - #include "rules/definitions_to_defs.h" +#include "rules/dependencies_to_dependent.h" #include "rules/draft_official_dialect_with_https.h" #include "rules/draft_official_dialect_without_empty_fragment.h" #include "rules/empty_object_as_true.h" #include "rules/enum_to_const.h" -#include "rules/metaschema_vocabulary.h" +#include "rules/modern_official_dialect_with_empty_fragment.h" #include "rules/prefix_promoted_2020_12_keywords.h" #include "rules/prefix_promoted_draft_2019_09_keywords.h" #include "rules/prefix_promoted_draft_4_keywords.h" @@ -269,12 +327,12 @@ auto convert(sourcemeta::core::JSON &schema, const sourcemeta::core::SchemaWalker &walker, const sourcemeta::core::SchemaResolver &resolver, const ConvertTarget target, const std::string_view default_dialect, - const std::string_view default_id, const bool is_metaschema) - -> void { + const std::string_view default_id) -> void { std::vector rules; - rules.reserve(18); + rules.reserve(20); rules.push_back(make_rule()); rules.push_back(make_rule()); + rules.push_back(make_rule()); rules.push_back(make_rule()); rules.push_back(make_rule()); @@ -284,21 +342,21 @@ auto convert(sourcemeta::core::JSON &schema, rules.push_back(make_rule()); rules.push_back(make_rule()); rules.push_back(make_rule()); + rules.push_back(make_rule()); } if (target == ConvertTarget::Draft7 || target == ConvertTarget::Draft201909 || target == ConvertTarget::Draft202012) { rules.push_back(make_rule()); rules.push_back(make_rule()); - rules.push_back(make_rule()); } if (target == ConvertTarget::Draft201909 || target == ConvertTarget::Draft202012) { rules.push_back(make_rule()); rules.push_back(make_rule()); - rules.push_back(make_rule()); rules.push_back(make_rule()); + rules.push_back(make_rule()); } if (target == ConvertTarget::Draft202012) { @@ -307,8 +365,8 @@ auto convert(sourcemeta::core::JSON &schema, } rules.push_back(make_rule()); - apply(rules, schema, walker, resolver, default_dialect, default_id, - is_metaschema); + apply(rules, schema, walker, resolver, default_dialect, default_id); + erase_dialect_overrides(schema); } } // namespace sourcemeta::blaze diff --git a/vendor/blaze/src/convert/helpers.h b/vendor/blaze/src/convert/helpers.h index fd29462c..91341123 100644 --- a/vendor/blaze/src/convert/helpers.h +++ b/vendor/blaze/src/convert/helpers.h @@ -36,6 +36,13 @@ inline auto current_dialect_or_override(const sourcemeta::core::JSON &schema) return declared_dialect(schema); } +// The empty fragment does not change which dialect a URI names, and only some +// of the official spellings have a rule of their own to settle them +inline auto without_empty_fragment(const std::string_view uri) + -> std::string_view { + return uri.ends_with('#') ? uri.substr(0, uri.size() - 1) : uri; +} + // A subschema that a `$schema` of the document resolves to is a meta-schema of // that document, no matter where within the document it sits. Every base // dialect asks such a subschema to declare an identifier, which is what keeps @@ -48,6 +55,17 @@ inline auto is_metaschema_target(const sourcemeta::core::JSON &schema, return false; } + // A document that takes its dialect from the caller rather than from a + // `$schema` of its own names no meta-schema anywhere, so what the document + // reads as is the only thing left to ask + const auto document{frame.traverse(sourcemeta::core::EMPTY_WEAK_POINTER)}; + if (document.has_value()) { + const auto target{frame.traverse(document.value().get().dialect)}; + if (target.has_value() && target.value().get().pointer == pointer) { + return true; + } + } + return frame.any_reference( [&frame, &pointer]( const sourcemeta::core::SchemaReferenceType, @@ -64,63 +82,6 @@ inline auto is_metaschema_target(const sourcemeta::core::JSON &schema, }); } -// Whether any subschema that names this one through `$schema` still has work -// of its own left. Such a referrer is read under the dialect this subschema -// defines, so moving this one first would take the referrer off the dialect -// the caller is acting on before its turn ever comes -template -auto has_pending_metaschema_referrer( - const sourcemeta::core::JSON &root, - const sourcemeta::core::SchemaFrame &frame, - const sourcemeta::core::WeakPointer &pointer, const Predicate &pending) - -> bool { - return frame.any_reference( - [&root, &frame, &pointer, &pending]( - const sourcemeta::core::SchemaReferenceType, - const sourcemeta::core::WeakPointer &origin, - const sourcemeta::core::SchemaFrame::Reference &reference) -> bool { - if (origin.empty() || !origin.back().is_property() || - origin.back().to_property() != "$schema") { - return false; - } - - const auto destination{frame.traverse(reference.destination)}; - if (!destination.has_value() || - destination.value().get().pointer != pointer) { - return false; - } - - const auto referrer{sourcemeta::core::to_pointer(origin).initial()}; - const auto referrer_pointer{ - sourcemeta::core::to_weak_pointer(referrer)}; - - // A meta-schema that describes itself is its own referrer, and waiting - // on itself would leave it on the dialect it came in with for good - if (referrer_pointer == pointer) { - return false; - } - - if (pending(sourcemeta::core::get(root, referrer))) { - return true; - } - - // Everything under the referrer is read under the dialect this - // subschema defines too, so work down there counts just as much as - // work on the referrer itself. Another meta-schema is governed by its - // own referrers rather than by this one - return frame.any_subschema_under( - referrer_pointer, - [&root, &frame, &pending]( - const sourcemeta::core::SchemaFrame::Location &entry) -> bool { - const auto &entry_schema{sourcemeta::core::get( - root, sourcemeta::core::to_pointer(entry.pointer))}; - return !is_metaschema_target(entry_schema, frame, - entry.pointer) && - pending(entry_schema); - }); - }); -} - inline auto subschema_at_dialect(const sourcemeta::core::JSON &schema, const sourcemeta::core::SchemaFrame::Location &location, @@ -155,6 +116,103 @@ inline auto dialect_position(const std::string_view dialect) -> std::size_t { return 0; } +// Core reads this keyword as a dialect too, so it may well be a keyword the +// caller wrote. The ladder only ever records one of the dialects it walks +// through, so anything else is not ours to clear +inline auto is_own_dialect_override(const sourcemeta::core::JSON &value) + -> bool { + return value.is_string() && dialect_position(value.to_string()) > 0; +} + +// The spellings the normalising rules settle on all name the same dialect, so +// whether the ladder names one has to be asked of the spelling those rules +// would produce rather than of what the document happens to say +inline auto normalized_official_dialect(const std::string_view dialect) + -> std::string { + std::string result{without_empty_fragment(dialect)}; + if (result.starts_with("https://json-schema.org/draft-")) { + result.erase(4, 1); + } + + return result; +} + +// A dialect the ladder does not name is one the conversion has no rules for, +// whether it belongs to a draft older than the ladder starts at or to a +// meta-schema of the caller's own +inline auto names_ladder_dialect(const std::string_view dialect) -> bool { + const auto candidate{normalized_official_dialect(dialect)}; + return std::ranges::any_of( + LADDER_DIALECTS, [&candidate](const auto &entry) -> bool { + return without_empty_fragment(entry) == candidate; + }); +} + +// Whether an identifier and a dialect name the same thing once both are +// resolved against what the caller said the document is called +inline auto names_the_same_uri(const sourcemeta::core::JSON &schema, + const sourcemeta::core::JSON::StringView keyword, + const std::string_view dialect, + const std::string_view default_id) -> bool { + const auto *identifier{schema.try_at(keyword)}; + if (identifier == nullptr || !identifier->is_string()) { + return false; + } + + if (without_empty_fragment(identifier->to_string()) == + without_empty_fragment(dialect)) { + return true; + } + + // Resolving is what lets an identifier written relative to whatever the + // caller named the document meet a dialect that is spelled out in full. + // A value that does not parse is not for this question to complain about, + // as framing says so in better words a moment later + try { + sourcemeta::core::URI left{identifier->to_string()}; + sourcemeta::core::URI right{std::string{dialect}}; + if (!default_id.empty()) { + const sourcemeta::core::URI base{std::string{default_id}}; + left.resolve_from(base); + right.resolve_from(base); + } + + left.canonicalize(); + right.canonicalize(); + return left.recompose() == right.recompose(); + } catch (const sourcemeta::core::URIParseError &) { + return false; + } catch (const sourcemeta::core::URIError &) { + return false; + } +} + +// A document whose identifier is the very dialect it declares describes +// itself, so it is a meta-schema on the strongest evidence there is. The +// ladder rewrites that `$schema` on the first bump, taking the evidence with +// it, so the question has to be asked before any rule runs +inline auto +describes_itself(const sourcemeta::core::JSON &schema, + const sourcemeta::core::SchemaBaseDialect base_dialect, + const std::string_view default_id) -> bool { + if (!schema.is_object()) { + return false; + } + + const auto *dialect{schema.try_at("$schema")}; + if (dialect == nullptr || !dialect->is_string()) { + return false; + } + + // Draft 3 and Draft 4 carry the identifier in `id` and everything after them + // in `$id`, so the other keyword is ordinary data there. Which one is which + // is the base dialect's answer to give, not something to read off a URI that + // has more than one accepted spelling + return names_the_same_uri( + schema, sourcemeta::core::schema_identifier_keyword(base_dialect), + dialect->to_string(), default_id); +} + inline auto moved_past(const sourcemeta::core::JSON &schema, const std::string_view dialect) -> bool { const auto *override_value{schema.try_at(DIALECT_OVERRIDE_KEYWORD)}; @@ -163,6 +221,38 @@ inline auto moved_past(const sourcemeta::core::JSON &schema, dialect_position(dialect); } +// The marker is state of the ladder rather than of the schema. A resource that +// declares a dialect the conversion does not own can never materialise it into +// a `$schema`, so whatever survives the ladder has to come off before the +// caller ever sees it +inline auto erase_dialect_overrides(sourcemeta::core::JSON &schema) -> void { + if (schema.is_array()) { + for (auto &item : schema.as_array()) { + erase_dialect_overrides(item); + } + + return; + } + + if (!schema.is_object()) { + return; + } + + const auto *marker{schema.try_at(DIALECT_OVERRIDE_KEYWORD)}; + if (marker != nullptr && is_own_dialect_override(*marker)) { + schema.erase(DIALECT_OVERRIDE_KEYWORD); + } + + std::vector keys; + keys.reserve(schema.size()); + for (const auto &entry : schema.as_object()) { + keys.push_back(entry.first); + } + for (const auto &key : keys) { + erase_dialect_overrides(schema.at(key)); + } +} + inline auto drop_dialect_overrides(sourcemeta::core::JSON &schema, const bool is_root, const std::string_view dialect) -> void { @@ -186,7 +276,9 @@ inline auto drop_dialect_overrides(sourcemeta::core::JSON &schema, // its marker. Dropping it would leave the keywords that move brought in // looking like keywords of the dialect it has left behind, and the rules // that reserve those names would prefix them away - if (is_root || !moved_past(schema, dialect)) { + const auto *marker{schema.try_at(DIALECT_OVERRIDE_KEYWORD)}; + if (marker != nullptr && is_own_dialect_override(*marker) && + (is_root || !moved_past(schema, dialect))) { schema.erase(DIALECT_OVERRIDE_KEYWORD); } diff --git a/vendor/blaze/src/convert/include/sourcemeta/blaze/convert.h b/vendor/blaze/src/convert/include/sourcemeta/blaze/convert.h index 66252b0f..0ac219da 100644 --- a/vendor/blaze/src/convert/include/sourcemeta/blaze/convert.h +++ b/vendor/blaze/src/convert/include/sourcemeta/blaze/convert.h @@ -46,7 +46,10 @@ enum class ConvertTarget : std::uint8_t { /// @ingroup convert /// Convert the given schema, in place, to the given dialect. Only upgrades are /// supported, so a schema already on that dialect or a newer one is left as -/// is. For example: +/// is. A document that describes itself or that carries the meta-schema +/// something in it declares raises `ConvertUnsupportedMetaschemaError`, and one +/// that sits on a dialect outside the official ladder raises +/// `ConvertUnsupportedDialectError`. For example: /// /// ```cpp /// #include @@ -67,8 +70,7 @@ auto convert(sourcemeta::core::JSON &schema, const sourcemeta::core::SchemaResolver &resolver, const ConvertTarget target, const std::string_view default_dialect = "", - const std::string_view default_id = "", - const bool is_metaschema = false) -> void; + const std::string_view default_id = "") -> void; } // namespace sourcemeta::blaze diff --git a/vendor/blaze/src/convert/include/sourcemeta/blaze/convert_error.h b/vendor/blaze/src/convert/include/sourcemeta/blaze/convert_error.h index fa790f01..5b83840b 100644 --- a/vendor/blaze/src/convert/include/sourcemeta/blaze/convert_error.h +++ b/vendor/blaze/src/convert/include/sourcemeta/blaze/convert_error.h @@ -21,6 +21,87 @@ namespace sourcemeta::blaze { #pragma warning(disable : 4251 4275) #endif +/// @ingroup convert +/// An error that represents a dialect the conversion cannot move a schema from +class SOURCEMETA_BLAZE_CONVERT_EXPORT ConvertUnsupportedDialectError + : public std::exception { +public: + ConvertUnsupportedDialectError(const std::string_view identifier, + sourcemeta::core::Pointer location) + : identifier_{identifier}, location_{std::move(location)} {} + + [[nodiscard]] auto what() const noexcept -> const char * override { + return "The conversion does not support this dialect"; + } + + [[nodiscard]] auto identifier() const noexcept -> std::string_view { + return this->identifier_; + } + + [[nodiscard]] auto location() const noexcept + -> const sourcemeta::core::Pointer & { + return this->location_; + } + +private: + std::string identifier_; + sourcemeta::core::Pointer location_; +}; + +/// @ingroup convert +/// An error that represents a meta-schema that the conversion cannot move +class SOURCEMETA_BLAZE_CONVERT_EXPORT ConvertUnsupportedMetaschemaError + : public std::exception { +public: + ConvertUnsupportedMetaschemaError(const std::string_view identifier, + sourcemeta::core::Pointer location) + : identifier_{identifier}, location_{std::move(location)} {} + + [[nodiscard]] auto what() const noexcept -> const char * override { + return "The conversion does not support meta-schemas"; + } + + [[nodiscard]] auto identifier() const noexcept -> std::string_view { + return this->identifier_; + } + + [[nodiscard]] auto location() const noexcept + -> const sourcemeta::core::Pointer & { + return this->location_; + } + +private: + std::string identifier_; + sourcemeta::core::Pointer location_; +}; + +/// @ingroup convert +/// An error that represents a schema reference that does not point to a schema +class SOURCEMETA_BLAZE_CONVERT_EXPORT ConvertInvalidReferenceError + : public std::exception { +public: + ConvertInvalidReferenceError(const std::string_view identifier, + sourcemeta::core::Pointer location) + : identifier_{identifier}, location_{std::move(location)} {} + + [[nodiscard]] auto what() const noexcept -> const char * override { + return "The reference does not point to a schema"; + } + + [[nodiscard]] auto identifier() const noexcept -> std::string_view { + return this->identifier_; + } + + [[nodiscard]] auto location() const noexcept + -> const sourcemeta::core::Pointer & { + return this->location_; + } + +private: + std::string identifier_; + sourcemeta::core::Pointer location_; +}; + /// @ingroup convert /// An error that represents a broken schema reference after conversion class SOURCEMETA_BLAZE_CONVERT_EXPORT ConvertBrokenReferenceError diff --git a/vendor/blaze/src/convert/rule.h b/vendor/blaze/src/convert/rule.h index c7b91ff2..f9a1f075 100644 --- a/vendor/blaze/src/convert/rule.h +++ b/vendor/blaze/src/convert/rule.h @@ -33,8 +33,7 @@ class SchemaTransformRule { const sourcemeta::core::SchemaFrame &frame, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, - const bool is_metaschema) const -> bool = 0; + const sourcemeta::core::SchemaResolver &resolver) const -> bool = 0; virtual auto transform(sourcemeta::core::JSON &schema) const -> void = 0; diff --git a/vendor/blaze/src/convert/rules/definitions_to_defs.h b/vendor/blaze/src/convert/rules/definitions_to_defs.h index 42cf0d9c..e7e7f576 100644 --- a/vendor/blaze/src/convert/rules/definitions_to_defs.h +++ b/vendor/blaze/src/convert/rules/definitions_to_defs.h @@ -10,8 +10,7 @@ class DefinitionsToDefs final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains_any( {SchemaVocabularies::Known::JSON_SCHEMA_2020_12_CORE, diff --git a/vendor/blaze/src/convert/rules/dependencies_to_dependent.h b/vendor/blaze/src/convert/rules/dependencies_to_dependent.h new file mode 100644 index 00000000..23f411a9 --- /dev/null +++ b/vendor/blaze/src/convert/rules/dependencies_to_dependent.h @@ -0,0 +1,110 @@ +class DependenciesToDependent final : public SchemaTransformRule { +public: + using reframe_after_transform = std::true_type; + DependenciesToDependent() + : SchemaTransformRule{"dependencies_to_dependent"} {}; + + [[nodiscard]] auto + condition(const sourcemeta::core::JSON &schema, + const sourcemeta::core::JSON &, + const sourcemeta::core::SchemaVocabularies &vocabularies, + const sourcemeta::core::SchemaFrame &, + const sourcemeta::core::SchemaFrame::Location &, + const sourcemeta::core::SchemaWalker &, + const sourcemeta::core::SchemaResolver &) const -> bool override { + ONLY_CONTINUE_IF( + vocabularies.contains_any( + {SchemaVocabularies::Known::JSON_SCHEMA_2020_12_APPLICATOR, + SchemaVocabularies::Known::JSON_SCHEMA_2019_09_APPLICATOR}) && + schema.is_object() && !schema.defines("dependentSchemas") && + !schema.defines("dependentRequired")); + + const auto *dependencies{schema.try_at("dependencies")}; + ONLY_CONTINUE_IF(dependencies && dependencies->is_object()); + + for (const auto &entry : dependencies->as_object()) { + if (!entry.second.is_array() && !entry.second.is_object() && + !entry.second.is_boolean()) { + return false; + } + } + + return true; + } + + auto transform(sourcemeta::core::JSON &schema) const -> void override { + this->renames_.clear(); + auto dependent_required{sourcemeta::core::JSON::make_object()}; + auto dependent_schemas{sourcemeta::core::JSON::make_object()}; + + for (const auto &entry : schema.at("dependencies").as_object()) { + if (entry.second.is_array()) { + dependent_required.assign(entry.first, entry.second); + } else { + dependent_schemas.assign(entry.first, entry.second); + } + } + + // An empty container has nothing to sort into the two keywords that + // replaced this one, and picking either would invent a claim the document + // never made. A `definitions` has a single successor, which is why that + // one is renamed rather than dropped + if (dependent_required.empty() && dependent_schemas.empty()) { + schema.erase("dependencies"); + return; + } + + if (!dependent_required.empty() && !dependent_schemas.empty()) { + for (const auto &entry : dependent_schemas.as_object()) { + this->renames_.emplace_back( + sourcemeta::core::Pointer{"dependencies", entry.first}, + sourcemeta::core::Pointer{"dependentSchemas", entry.first}); + } + for (const auto &entry : dependent_required.as_object()) { + this->renames_.emplace_back( + sourcemeta::core::Pointer{"dependencies", entry.first}, + sourcemeta::core::Pointer{"dependentRequired", entry.first}); + } + schema.try_assign_before("dependentSchemas", dependent_schemas, + "dependencies"); + schema.rename("dependencies", "dependentRequired"); + schema.at("dependentRequired").into(std::move(dependent_required)); + return; + } + + if (!dependent_schemas.empty()) { + this->renames_.emplace_back( + sourcemeta::core::Pointer{"dependencies"}, + sourcemeta::core::Pointer{"dependentSchemas"}); + schema.rename("dependencies", "dependentSchemas"); + schema.at("dependentSchemas").into(std::move(dependent_schemas)); + return; + } + + this->renames_.emplace_back(sourcemeta::core::Pointer{"dependencies"}, + sourcemeta::core::Pointer{"dependentRequired"}); + schema.rename("dependencies", "dependentRequired"); + schema.at("dependentRequired").into(std::move(dependent_required)); + } + + [[nodiscard]] auto rereference(const std::string_view, + const sourcemeta::core::Pointer &, + const sourcemeta::core::Pointer &target, + const sourcemeta::core::Pointer ¤t) const + -> std::optional override { + for (const auto &[old_pointer, new_pointer] : this->renames_) { + const auto result{target.rebase(current.concat(old_pointer), + current.concat(new_pointer))}; + if (result != target) { + return result; + } + } + + return target; + } + +private: + mutable std::vector< + std::pair> + renames_; +}; diff --git a/vendor/blaze/src/convert/rules/draft_official_dialect_with_https.h b/vendor/blaze/src/convert/rules/draft_official_dialect_with_https.h index a9c8110f..f33fbf10 100644 --- a/vendor/blaze/src/convert/rules/draft_official_dialect_with_https.h +++ b/vendor/blaze/src/convert/rules/draft_official_dialect_with_https.h @@ -11,8 +11,7 @@ class DraftOfficialDialectWithHttps final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { using sourcemeta::core::SchemaBaseDialect; ONLY_CONTINUE_IF( location.base_dialect == SchemaBaseDialect::JSON_SCHEMA_DRAFT_7 || diff --git a/vendor/blaze/src/convert/rules/draft_official_dialect_without_empty_fragment.h b/vendor/blaze/src/convert/rules/draft_official_dialect_without_empty_fragment.h index 2f36787c..43abf757 100644 --- a/vendor/blaze/src/convert/rules/draft_official_dialect_without_empty_fragment.h +++ b/vendor/blaze/src/convert/rules/draft_official_dialect_without_empty_fragment.h @@ -11,8 +11,8 @@ class DraftOfficialDialectWithoutEmptyFragment final const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, - const bool) const -> bool override { + const sourcemeta::core::SchemaResolver &) const + -> bool override { ONLY_CONTINUE_IF(schema.is_object()); const auto *schema_keyword{schema.try_at("$schema")}; ONLY_CONTINUE_IF(schema_keyword && schema_keyword->is_string()); diff --git a/vendor/blaze/src/convert/rules/empty_object_as_true.h b/vendor/blaze/src/convert/rules/empty_object_as_true.h index eca2d0ae..34caffa5 100644 --- a/vendor/blaze/src/convert/rules/empty_object_as_true.h +++ b/vendor/blaze/src/convert/rules/empty_object_as_true.h @@ -10,8 +10,7 @@ class EmptyObjectAsTrue final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF(vocabularies.contains_any( {SchemaVocabularies::Known::JSON_SCHEMA_2020_12_CORE, SchemaVocabularies::Known::JSON_SCHEMA_2019_09_CORE, diff --git a/vendor/blaze/src/convert/rules/enum_to_const.h b/vendor/blaze/src/convert/rules/enum_to_const.h index f95a68b4..111025a9 100644 --- a/vendor/blaze/src/convert/rules/enum_to_const.h +++ b/vendor/blaze/src/convert/rules/enum_to_const.h @@ -10,8 +10,7 @@ class EnumToConst final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains_any( {SchemaVocabularies::Known::JSON_SCHEMA_2020_12_VALIDATION, diff --git a/vendor/blaze/src/convert/rules/metaschema_vocabulary.h b/vendor/blaze/src/convert/rules/metaschema_vocabulary.h deleted file mode 100644 index c1e54b30..00000000 --- a/vendor/blaze/src/convert/rules/metaschema_vocabulary.h +++ /dev/null @@ -1,99 +0,0 @@ -class MetaschemaVocabulary final : public SchemaTransformRule { -public: - using reframe_after_transform = std::true_type; - MetaschemaVocabulary() : SchemaTransformRule{"metaschema_vocabulary"} {}; - - [[nodiscard]] auto - condition(const sourcemeta::core::JSON &schema, - const sourcemeta::core::JSON &, - const sourcemeta::core::SchemaVocabularies &vocabularies, - const sourcemeta::core::SchemaFrame &frame, - const sourcemeta::core::SchemaFrame::Location &location, - const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, - const bool is_metaschema) const -> bool override { - ONLY_CONTINUE_IF(schema.is_object() && !schema.defines("$vocabulary")); - - // Whichever dialect the meta-schema ends up on is the one whose - // vocabularies it has to declare, rather than the one the caller asked to - // convert to - this->emits_2020_12_ = vocabularies.contains( - SchemaVocabularies::Known::JSON_SCHEMA_2020_12_CORE); - ONLY_CONTINUE_IF(this->emits_2020_12_ || - vocabularies.contains( - SchemaVocabularies::Known::JSON_SCHEMA_2019_09_CORE)); - - return (is_metaschema && location.pointer.empty()) || - is_metaschema_target(schema, frame, location.pointer); - } - - auto transform(sourcemeta::core::JSON &schema) const -> void override { - if (this->emits_2020_12_) { - synthesize_vocabulary(schema, VOCABULARIES_2020_12); - } else { - synthesize_vocabulary(schema, VOCABULARIES_2019_09); - } - } - -private: - using Vocabulary = std::pair; - - static constexpr std::array VOCABULARIES_2019_09{ - {{"https://json-schema.org/draft/2019-09/vocab/core", true}, - {"https://json-schema.org/draft/2019-09/vocab/applicator", true}, - {"https://json-schema.org/draft/2019-09/vocab/validation", true}, - {"https://json-schema.org/draft/2019-09/vocab/meta-data", true}, - {"https://json-schema.org/draft/2019-09/vocab/format", false}, - {"https://json-schema.org/draft/2019-09/vocab/content", true}}}; - - static constexpr std::array VOCABULARIES_2020_12{ - {{"https://json-schema.org/draft/2020-12/vocab/core", true}, - {"https://json-schema.org/draft/2020-12/vocab/applicator", true}, - {"https://json-schema.org/draft/2020-12/vocab/unevaluated", true}, - {"https://json-schema.org/draft/2020-12/vocab/validation", true}, - {"https://json-schema.org/draft/2020-12/vocab/meta-data", true}, - {"https://json-schema.org/draft/2020-12/vocab/format-annotation", false}, - {"https://json-schema.org/draft/2020-12/vocab/content", true}}}; - - mutable bool emits_2020_12_{false}; - - template - static auto synthesize_vocabulary(sourcemeta::core::JSON &schema, - const std::array &entries) - -> void { - std::string_view anchor; - if (schema.defines("$id")) { - anchor = "$id"; - } else if (schema.defines("$schema")) { - anchor = "$schema"; - } - - const std::string *next_key{nullptr}; - if (!anchor.empty()) { - bool found_anchor{false}; - for (const auto &entry : schema.as_object()) { - if (found_anchor) { - next_key = &entry.first; - break; - } - if (entry.first == anchor) { - found_anchor = true; - } - } - } - - if (next_key != nullptr) { - schema.try_assign_before( - "$vocabulary", sourcemeta::core::JSON::make_object(), *next_key); - } else { - schema.assign_assume_new("$vocabulary", - sourcemeta::core::JSON::make_object()); - } - - auto &vocabularies{schema.at("$vocabulary")}; - for (const auto &[uri, required] : entries) { - vocabularies.assign_assume_new(std::string{uri}, - sourcemeta::core::JSON{required}); - } - } -}; diff --git a/vendor/blaze/src/convert/rules/modern_official_dialect_with_empty_fragment.h b/vendor/blaze/src/convert/rules/modern_official_dialect_with_empty_fragment.h new file mode 100644 index 00000000..e3158b1f --- /dev/null +++ b/vendor/blaze/src/convert/rules/modern_official_dialect_with_empty_fragment.h @@ -0,0 +1,33 @@ +class ModernOfficialDialectWithEmptyFragment final + : public SchemaTransformRule { +public: + using reframe_after_transform = std::true_type; + ModernOfficialDialectWithEmptyFragment() + : SchemaTransformRule{"modern_official_dialect_with_empty_fragment"} {}; + + [[nodiscard]] auto condition(const sourcemeta::core::JSON &schema, + const sourcemeta::core::JSON &, + const sourcemeta::core::SchemaVocabularies &, + const sourcemeta::core::SchemaFrame &, + const sourcemeta::core::SchemaFrame::Location &, + const sourcemeta::core::SchemaWalker &, + const sourcemeta::core::SchemaResolver &) const + -> bool override { + ONLY_CONTINUE_IF(schema.is_object()); + const auto *schema_keyword{schema.try_at("$schema")}; + ONLY_CONTINUE_IF(schema_keyword && schema_keyword->is_string()); + const auto &dialect{schema_keyword->to_string()}; + ONLY_CONTINUE_IF( + dialect == "https://json-schema.org/draft/2019-09/schema#" || + dialect == "https://json-schema.org/draft/2019-09/hyper-schema#" || + dialect == "https://json-schema.org/draft/2020-12/schema#" || + dialect == "https://json-schema.org/draft/2020-12/hyper-schema#"); + return true; + } + + auto transform(sourcemeta::core::JSON &schema) const -> void override { + auto dialect{std::move(schema.at("$schema")).to_string()}; + dialect.pop_back(); + schema.at("$schema").into(sourcemeta::core::JSON{dialect}); + } +}; diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_2020_12_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_2020_12_keywords.h index 3dc230ec..30705e35 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_2020_12_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_2020_12_keywords.h @@ -11,8 +11,7 @@ class PrefixPromoted202012Keywords final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF(vocabularies.contains( SchemaVocabularies::Known::JSON_SCHEMA_2019_09_CORE) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_draft_2019_09_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_draft_2019_09_keywords.h index 78b6dbad..047a43da 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_draft_2019_09_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_draft_2019_09_keywords.h @@ -11,8 +11,7 @@ class PrefixPromoted201909Keywords final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_7) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_draft_4_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_draft_4_keywords.h index 840ab7ea..737da6f1 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_draft_4_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_draft_4_keywords.h @@ -11,8 +11,7 @@ class PrefixPromotedDraft4Keywords final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_3) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h index 90980397..428c3117 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h @@ -11,8 +11,7 @@ class PrefixPromotedDraft6Keywords final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_4) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_draft_7_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_draft_7_keywords.h index ad636b2b..89b9ab41 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_draft_7_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_draft_7_keywords.h @@ -11,8 +11,7 @@ class PrefixPromotedDraft7Keywords final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &, const sourcemeta::core::SchemaFrame::Location &, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_6) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/upgrade_2019_09_to_2020_12.h b/vendor/blaze/src/convert/rules/upgrade_2019_09_to_2020_12.h index 7e83a602..def871b5 100644 --- a/vendor/blaze/src/convert/rules/upgrade_2019_09_to_2020_12.h +++ b/vendor/blaze/src/convert/rules/upgrade_2019_09_to_2020_12.h @@ -4,15 +4,13 @@ class Upgrade201909To202012 final : public SchemaTransformRule { Upgrade201909To202012() : SchemaTransformRule{"upgrade_2019_09_to_2020_12"} {}; - [[nodiscard]] auto - condition(const sourcemeta::core::JSON &schema, - const sourcemeta::core::JSON &root, - const sourcemeta::core::SchemaVocabularies &vocabularies, - const sourcemeta::core::SchemaFrame &frame, - const sourcemeta::core::SchemaFrame::Location &location, - const sourcemeta::core::SchemaWalker &walker, - const sourcemeta::core::SchemaResolver &resolver, const bool) const - -> bool override { + [[nodiscard]] auto condition( + const sourcemeta::core::JSON &schema, const sourcemeta::core::JSON &root, + const sourcemeta::core::SchemaVocabularies &vocabularies, + const sourcemeta::core::SchemaFrame &frame, + const sourcemeta::core::SchemaFrame::Location &location, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver) const -> bool override { this->sanitize_pending_ = false; ONLY_CONTINUE_IF(vocabularies.contains( @@ -32,10 +30,11 @@ class Upgrade201909To202012 final : public SchemaTransformRule { any_descendant_has_pending_pattern(root, frame, location); this->resource_has_recursive_anchor_ = compute_resource_has_recursive_anchor(root, frame, location); - this->anchor_at_resource_root_ = - location.pointer.empty() || - location.type == - sourcemeta::core::SchemaFrame::LocationType::Resource; + this->anchor_at_resource_root_ = is_resource_root(frame, location); + if (needs_dynamic_anchor_name(schema)) { + this->dynamic_anchor_name_ = compute_dynamic_anchor_name(root); + } + this->document_has_unevaluated_items_ = compute_document_has_unevaluated_items(root, frame, walker, resolver); @@ -58,9 +57,11 @@ class Upgrade201909To202012 final : public SchemaTransformRule { this->resource_has_recursive_anchor_ = compute_resource_has_recursive_anchor(root, frame, location); - this->anchor_at_resource_root_ = - location.pointer.empty() || - location.type == sourcemeta::core::SchemaFrame::LocationType::Resource; + this->anchor_at_resource_root_ = is_resource_root(frame, location); + if (needs_dynamic_anchor_name(schema)) { + this->dynamic_anchor_name_ = compute_dynamic_anchor_name(root); + } + this->document_has_unevaluated_items_ = compute_document_has_unevaluated_items(root, frame, walker, resolver); return true; @@ -84,7 +85,8 @@ class Upgrade201909To202012 final : public SchemaTransformRule { if (schema.at("$recursiveAnchor").to_boolean() && this->anchor_at_resource_root_) { schema.rename("$recursiveAnchor", "$dynamicAnchor"); - schema.at("$dynamicAnchor").into(sourcemeta::core::JSON{"meta"}); + schema.at("$dynamicAnchor") + .into(sourcemeta::core::JSON{this->dynamic_anchor_name_}); } else { schema.erase("$recursiveAnchor"); } @@ -93,7 +95,8 @@ class Upgrade201909To202012 final : public SchemaTransformRule { if (schema.defines("$recursiveRef")) { schema.rename("$recursiveRef", "$dynamicRef"); if (this->resource_has_recursive_anchor_) { - schema.at("$dynamicRef").into(sourcemeta::core::JSON{"#meta"}); + schema.at("$dynamicRef") + .into(sourcemeta::core::JSON{"#" + this->dynamic_anchor_name_}); } } @@ -260,10 +263,12 @@ class Upgrade201909To202012 final : public SchemaTransformRule { const auto iter{VOCAB_URI_MAP_2019_09_TO_2020_12.find(entry.first)}; if (iter == VOCAB_URI_MAP_2019_09_TO_2020_12.cend()) { fresh.assign(entry.first, entry.second); + // The unevaluated keywords are defined in terms of the annotations + // the applicator ones produce, so the declaration that survives for + // the applicator vocabulary is the one that speaks for both if (entry.first == APPLICATOR_2020_12_URI && should_inline_unevaluated) { - fresh.assign(std::string{UNEVALUATED_2020_12_URI}, - applicator_2019_09_value.value()); + fresh.assign(std::string{UNEVALUATED_2020_12_URI}, entry.second); } continue; } @@ -295,8 +300,10 @@ class Upgrade201909To202012 final : public SchemaTransformRule { mutable std::vector< std::pair> renames_; + mutable bool resource_has_recursive_anchor_{false}; mutable bool anchor_at_resource_root_{false}; + mutable std::string dynamic_anchor_name_{"meta"}; mutable bool is_inside_contains_wrapper_{false}; mutable bool descendant_has_pending_pattern_{false}; mutable bool document_has_unevaluated_items_{false}; @@ -626,6 +633,72 @@ class Upgrade201909To202012 final : public SchemaTransformRule { } } + // Whichever frame entry the traversal happens to reach first decides the + // entry type, and a resource has several. Asking the frame which resource + // encloses this location answers the same question the same way every time + static auto + is_resource_root(const sourcemeta::core::SchemaFrame &frame, + const sourcemeta::core::SchemaFrame::Location &location) + -> bool { + if (location.pointer.empty()) { + return true; + } + + const auto closest{find_enclosing_resource(frame, location)}; + return closest.has_value() && + closest.value().get().pointer == location.pointer; + } + + // A dynamic anchor is what a dynamic reference binds to across every + // resource in the same scope, so every resource that gets one has to spell + // it the same way. That makes the name a property of the document. Only the + // static anchors it already spells are in the way: another dynamic anchor + // carrying this very name is the point rather than a collision + static auto collect_static_anchors(const sourcemeta::core::JSON &node, + std::set &names) -> void { + if (node.is_array()) { + for (const auto &item : node.as_array()) { + collect_static_anchors(item, names); + } + + return; + } + + if (!node.is_object()) { + return; + } + + const auto *anchor{node.try_at("$anchor")}; + if (anchor != nullptr && anchor->is_string()) { + names.emplace(anchor->to_string()); + } + + for (const auto &entry : node.as_object()) { + collect_static_anchors(entry.second, names); + } + } + + // Naming a dynamic anchor means reading the whole document, so only a + // subschema that is about to carry one or point at one pays for it + static auto needs_dynamic_anchor_name(const sourcemeta::core::JSON &schema) + -> bool { + return schema.is_object() && + schema.defines_any({"$recursiveAnchor", "$recursiveRef"}); + } + + static auto compute_dynamic_anchor_name(const sourcemeta::core::JSON &root) + -> std::string { + std::set in_use; + collect_static_anchors(root, in_use); + + std::string name{"meta"}; + while (in_use.contains(name)) { + name.insert(0, "x-"); + } + + return name; + } + static auto compute_resource_has_recursive_anchor( const sourcemeta::core::JSON &root, const sourcemeta::core::SchemaFrame &frame, diff --git a/vendor/blaze/src/convert/rules/upgrade_dialect_override_cleanup.h b/vendor/blaze/src/convert/rules/upgrade_dialect_override_cleanup.h index 138c589f..ba10325b 100644 --- a/vendor/blaze/src/convert/rules/upgrade_dialect_override_cleanup.h +++ b/vendor/blaze/src/convert/rules/upgrade_dialect_override_cleanup.h @@ -4,15 +4,13 @@ class UpgradeDialectOverrideCleanup final : public SchemaTransformRule { UpgradeDialectOverrideCleanup() : SchemaTransformRule{"upgrade_dialect_override_cleanup"} {}; - [[nodiscard]] auto - condition(const sourcemeta::core::JSON &schema, - const sourcemeta::core::JSON &root, - const sourcemeta::core::SchemaVocabularies &, - const sourcemeta::core::SchemaFrame &frame, - const sourcemeta::core::SchemaFrame::Location &location, - const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &resolver, const bool) const - -> bool override { + [[nodiscard]] auto condition( + const sourcemeta::core::JSON &schema, const sourcemeta::core::JSON &root, + const sourcemeta::core::SchemaVocabularies &, + const sourcemeta::core::SchemaFrame &frame, + const sourcemeta::core::SchemaFrame::Location &location, + const sourcemeta::core::SchemaWalker &, + const sourcemeta::core::SchemaResolver &resolver) const -> bool override { ONLY_CONTINUE_IF(location.pointer.empty() && schema.is_object()); this->redundant_.clear(); diff --git a/vendor/blaze/src/convert/rules/upgrade_draft_3_to_draft_4.h b/vendor/blaze/src/convert/rules/upgrade_draft_3_to_draft_4.h index 454da9bc..cdeda88e 100644 --- a/vendor/blaze/src/convert/rules/upgrade_draft_3_to_draft_4.h +++ b/vendor/blaze/src/convert/rules/upgrade_draft_3_to_draft_4.h @@ -11,8 +11,7 @@ class UpgradeDraft3ToDraft4 final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &frame, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_3) && schema.is_object()); diff --git a/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h b/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h index d967c8e0..a9ea1133 100644 --- a/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h +++ b/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h @@ -11,8 +11,7 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &frame, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { this->sanitize_pending_ = false; ONLY_CONTINUE_IF( @@ -35,15 +34,6 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { ONLY_CONTINUE_IF(sanitization_branch || other_branch || root_via_default_dialect); - // A meta-schema the document embeds decides the dialect its referrers are - // read under, so moving it before them would take them off this dialect - // before their turn - if (is_metaschema_target(schema, frame, location.pointer) && - has_pending_metaschema_referrer(root, frame, location.pointer, - has_pending_draft_4_pattern)) { - return false; - } - if (!sanitization_branch && other_branch && enclosing_resource_has_pending_sanitization(location, root, frame)) { return false; @@ -52,17 +42,12 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { if (!sanitization_branch) { if (frame.any_subschema_under( location.pointer, - [&root, - &frame](const sourcemeta::core::SchemaFrame::Location &entry) + [&root](const sourcemeta::core::SchemaFrame::Location &entry) -> bool { const auto entry_pointer{ sourcemeta::core::to_pointer(entry.pointer)}; const auto &entry_schema{ sourcemeta::core::get(root, entry_pointer)}; - if (is_metaschema_target(entry_schema, frame, entry.pointer)) { - return false; - } - if (entry_schema.is_object() && entry_schema.defines("$ref")) { return false; } diff --git a/vendor/blaze/src/convert/rules/upgrade_draft_6_to_draft_7.h b/vendor/blaze/src/convert/rules/upgrade_draft_6_to_draft_7.h index 92e7b831..98d204a6 100644 --- a/vendor/blaze/src/convert/rules/upgrade_draft_6_to_draft_7.h +++ b/vendor/blaze/src/convert/rules/upgrade_draft_6_to_draft_7.h @@ -11,8 +11,7 @@ class UpgradeDraft6ToDraft7 final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &frame, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_6) && subschema_at_dialect(schema, location, DRAFT_6_URL)); diff --git a/vendor/blaze/src/convert/rules/upgrade_draft_7_to_draft_2019_09.h b/vendor/blaze/src/convert/rules/upgrade_draft_7_to_draft_2019_09.h index b3cb872d..101fe6a1 100644 --- a/vendor/blaze/src/convert/rules/upgrade_draft_7_to_draft_2019_09.h +++ b/vendor/blaze/src/convert/rules/upgrade_draft_7_to_draft_2019_09.h @@ -11,8 +11,7 @@ class UpgradeDraft7To201909 final : public SchemaTransformRule { const sourcemeta::core::SchemaFrame &frame, const sourcemeta::core::SchemaFrame::Location &location, const sourcemeta::core::SchemaWalker &, - const sourcemeta::core::SchemaResolver &, const bool) const - -> bool override { + const sourcemeta::core::SchemaResolver &) const -> bool override { ONLY_CONTINUE_IF( vocabularies.contains(SchemaVocabularies::Known::JSON_SCHEMA_DRAFT_7) && schema.is_object()); @@ -30,6 +29,7 @@ class UpgradeDraft7To201909 final : public SchemaTransformRule { sourcemeta::core::to_pointer(entry.pointer)}; const auto &entry_schema{ sourcemeta::core::get(root, entry_pointer)}; + return has_descendant_pending_pattern(entry_schema, entry.dialect); })) { @@ -263,6 +263,10 @@ class UpgradeDraft7To201909 final : public SchemaTransformRule { } } + // An empty container has nothing to sort into the two keywords that + // replaced this one, and picking either would invent a claim the document + // never made. A `definitions` has a single successor, which is why that + // one is renamed rather than dropped if (dependent_required.empty() && dependent_schemas.empty()) { schema.erase("dependencies"); return; diff --git a/vendor/blaze/src/dependencies/CMakeLists.txt b/vendor/blaze/src/dependencies/CMakeLists.txt new file mode 100644 index 00000000..03fe8517 --- /dev/null +++ b/vendor/blaze/src/dependencies/CMakeLists.txt @@ -0,0 +1,14 @@ +sourcemeta_library(NAMESPACE sourcemeta PROJECT blaze NAME dependencies + FOLDER "Blaze/Dependencies" + SOURCES dependencies.cc) + +if(BLAZE_INSTALL) + sourcemeta_library_install(NAMESPACE sourcemeta PROJECT blaze NAME dependencies) +endif() + +target_link_libraries(sourcemeta_blaze_dependencies PUBLIC + sourcemeta::core::json) +target_link_libraries(sourcemeta_blaze_dependencies PUBLIC + sourcemeta::core::jsonpointer) +target_link_libraries(sourcemeta_blaze_dependencies PUBLIC + sourcemeta::core::jsonschema) diff --git a/vendor/blaze/src/dependencies/dependencies.cc b/vendor/blaze/src/dependencies/dependencies.cc new file mode 100644 index 00000000..09f53b4a --- /dev/null +++ b/vendor/blaze/src/dependencies/dependencies.cc @@ -0,0 +1,170 @@ +#include + +#include + +#include // assert +#include // std::uint64_t +#include // std::string +#include // std::tuple +#include // std::unordered_set +#include // std::move +#include // std::vector + +namespace { + +auto is_skippable_metaschema_reference( + const sourcemeta::core::WeakPointer &pointer, + const std::string &destination) -> bool { + assert(!pointer.empty()); + assert(pointer.back().is_property()); + if (pointer.back().to_property() != "$schema") { + return false; + } + + return sourcemeta::core::schema_is_official(destination); +} + +// Every frame that this constructs spends from the same limit, as how many +// frames it ends up needing is a function of what the resolver hands back +// rather than of the schema the caller passed in. A frame that ran past what +// was left of the limit threw rather than returned, so what it holds is +// always within it +auto charge(std::uint64_t &remaining, + const sourcemeta::core::SchemaFrame &frame) -> void { + assert(frame.location_count() <= remaining); + remaining -= frame.location_count(); +} + +auto dependencies_internal( + const sourcemeta::core::JSON &schema, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + const sourcemeta::blaze::DependencyCallback &callback, + std::string_view default_dialect, std::string_view default_id, + const sourcemeta::core::SchemaFrame::Paths &paths, + std::unordered_set &visited, std::uint64_t &remaining) + -> void { + sourcemeta::core::SchemaFrame frame{ + sourcemeta::core::SchemaFrame::Mode::References, + schema, + walker, + resolver, + default_dialect, + default_id, + sourcemeta::core::SchemaFrame::IdentifierMode::Additional, + paths, + "", + remaining}; + charge(remaining, frame); + const auto &origin{frame.root()}; + + std::vector< + std::tuple> + found; + + frame.for_each_unresolved_reference([&](const auto &pointer, + const auto &reference) -> void { + // We don't want to report official schemas, as we can expect + // virtually all implementations to understand them out of the box + if (is_skippable_metaschema_reference(pointer, reference.destination)) { + return; + } + + if (reference.base.empty()) { + throw sourcemeta::core::SchemaReferenceError( + reference.destination, sourcemeta::core::to_pointer(pointer), + "Could not resolve schema reference"); + } + + // To not infinitely loop on circular references + if (visited.contains(std::string{reference.base})) { + return; + } + + // If we can't find the destination but there is a base and we can + // find the base, then we are facing an unresolved fragment + if (frame.traverse(reference.base).has_value()) { + throw sourcemeta::core::SchemaReferenceError( + reference.destination, sourcemeta::core::to_pointer(pointer), + "Could not resolve schema reference"); + } + + assert(!reference.base.empty()); + const auto &identifier{reference.base}; + auto remote{resolver(identifier)}; + if (!remote.has_value()) { + throw sourcemeta::core::SchemaResolutionError( + identifier, "Could not resolve the reference to an external schema"); + } + + if (!remote.value().is_object() && !remote.value().is_boolean()) { + throw sourcemeta::core::SchemaReferenceError( + identifier, sourcemeta::core::to_pointer(pointer), + "The JSON document is not a valid JSON Schema"); + } + + try { + const sourcemeta::core::SchemaFrame remote_frame{ + sourcemeta::core::SchemaFrame::Mode::Root, + remote.value(), + walker, + resolver, + default_dialect, + "", + sourcemeta::core::SchemaFrame::IdentifierMode::Additional, + {sourcemeta::core::EMPTY_WEAK_POINTER}, + "", + remaining}; + charge(remaining, remote_frame); + } catch (const sourcemeta::core::SchemaUnknownBaseDialectError &) { + throw sourcemeta::core::SchemaReferenceError( + identifier, sourcemeta::core::to_pointer(pointer), + "The JSON document is not a valid JSON Schema"); + } + + callback(origin, pointer, identifier, remote.value()); + visited.emplace(identifier); + + // Official schemas can only reference other official schemas, so + // recursing into them can never surface further dependencies + if (sourcemeta::core::schema_is_official(identifier)) { + return; + } + + found.emplace_back(std::move(remote).value(), + sourcemeta::core::JSON::String{identifier}); + }); + + for (const auto &entry : found) { + dependencies_internal(std::get<0>(entry), walker, resolver, callback, + default_dialect, std::get<1>(entry), + {sourcemeta::core::EMPTY_WEAK_POINTER}, visited, + remaining); + } +} + +} // namespace + +namespace sourcemeta::blaze { + +auto dependencies(const sourcemeta::core::JSON &schema, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + const DependencyCallback &callback, + std::string_view default_dialect, std::string_view default_id, + const sourcemeta::core::SchemaFrame::Paths &paths, + const std::uint64_t max_locations) -> void { + std::unordered_set visited; + auto remaining{max_locations}; + try { + dependencies_internal(schema, walker, resolver, callback, default_dialect, + default_id, paths, visited, remaining); + } catch (const sourcemeta::core::SchemaFrameLimitError &) { + // Every frame spends from what is left rather than from the whole, so the + // one that ran out reports what it was handed. The caller set the limit + // for the operation, so that is what the operation reports back + throw sourcemeta::core::SchemaFrameLimitError{max_locations}; + } +} + +} // namespace sourcemeta::blaze diff --git a/vendor/blaze/src/dependencies/include/sourcemeta/blaze/dependencies.h b/vendor/blaze/src/dependencies/include/sourcemeta/blaze/dependencies.h new file mode 100644 index 00000000..8fb7da0f --- /dev/null +++ b/vendor/blaze/src/dependencies/include/sourcemeta/blaze/dependencies.h @@ -0,0 +1,100 @@ +#ifndef SOURCEMETA_BLAZE_DEPENDENCIES_H_ +#define SOURCEMETA_BLAZE_DEPENDENCIES_H_ + +/// @defgroup dependencies Dependencies +/// @brief Report the external references of JSON Schemas. +/// +/// This functionality is included as follows: +/// +/// ```cpp +/// #include +/// ``` + +#ifndef SOURCEMETA_BLAZE_DEPENDENCIES_EXPORT +#include +#endif + +#include +#include + +#include + +#include // std::uint64_t +#include // std::function +#include // std::numeric_limits +#include // std::string_view + +namespace sourcemeta::blaze { + +/// @ingroup dependencies +/// A callback to get dependency information +/// - Origin URI (empty if none) +/// - Pointer (reference keyword from the origin) +/// - Target URI +/// - Target schema +using DependencyCallback = + std::function; + +/// @ingroup dependencies +/// +/// This function recursively traverses and reports the external references in a +/// schema. References to official schemas are reported but not traversed into, +/// as official schemas can only reference other official schemas. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// // A custom resolver that knows about an additional schema +/// static auto test_resolver(std::string_view identifier) +/// -> sourcemeta::core::SchemaResolverResult { +/// if (identifier == "https://www.example.com/test") { +/// return sourcemeta::core::parse_json(R"JSON({ +/// "$id": "https://www.example.com/test", +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "type": "string" +/// })JSON"); +/// } else { +/// return sourcemeta::core::schema_resolver(identifier); +/// } +/// } +/// +/// sourcemeta::core::JSON document = +/// sourcemeta::core::parse_json(R"JSON({ +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "items": { "$ref": "https://www.example.com/test" } +/// })JSON"); +/// +/// sourcemeta::blaze::dependencies(document, +/// sourcemeta::core::schema_walker, test_resolver, +/// [](const auto &origin, +/// const auto &pointer, +/// const auto &target, +/// const auto &schema) { +/// // Do something with the information +/// }); +/// ``` +/// +/// How many schemas this ends up analysing follows from what the resolver +/// hands back rather than from the schema the caller passed in, so pass +/// `max_locations` to bound it. Every frame that this constructs spends from +/// that one limit, throwing sourcemeta::core::SchemaFrameLimitError once it +/// runs out. See sourcemeta::core::SchemaFrame for what the unit counts and +/// what it leaves to the caller +SOURCEMETA_BLAZE_DEPENDENCIES_EXPORT +auto dependencies( + const sourcemeta::core::JSON &schema, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + const DependencyCallback &callback, std::string_view default_dialect = "", + std::string_view default_id = "", + const sourcemeta::core::SchemaFrame::Paths &paths = + {sourcemeta::core::EMPTY_WEAK_POINTER}, + std::uint64_t max_locations = std::numeric_limits::max()) + -> void; + +} // namespace sourcemeta::blaze + +#endif diff --git a/vendor/blaze/src/editor/include/sourcemeta/blaze/editor.h b/vendor/blaze/src/editor/include/sourcemeta/blaze/editor.h index 74fbfef1..58dce3ec 100644 --- a/vendor/blaze/src/editor/include/sourcemeta/blaze/editor.h +++ b/vendor/blaze/src/editor/include/sourcemeta/blaze/editor.h @@ -36,7 +36,6 @@ namespace sourcemeta::blaze { /// ```cpp /// #include /// #include -/// #include /// #include /// /// // A custom resolver that knows about the referenced schema @@ -59,9 +58,8 @@ namespace sourcemeta::blaze { /// "$ref": "another" /// })JSON"); /// -/// sourcemeta::blaze::bundle(schema, -/// sourcemeta::core::schema_walker, test_resolver, -/// sourcemeta::blaze::BundleMode::NonOfficialMetaschemas); +/// sourcemeta::core::schema_bundle(schema, +/// sourcemeta::core::schema_walker, test_resolver); /// sourcemeta::blaze::for_editor(schema, /// sourcemeta::core::schema_walker, test_resolver); /// ``` diff --git a/vendor/core/src/core/jsonschema/CMakeLists.txt b/vendor/core/src/core/jsonschema/CMakeLists.txt index 7929157b..564f03d4 100644 --- a/vendor/core/src/core/jsonschema/CMakeLists.txt +++ b/vendor/core/src/core/jsonschema/CMakeLists.txt @@ -5,7 +5,7 @@ include(./known_resolver.cmake) sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME jsonschema PRIVATE_HEADERS error.h types.h vocabularies.h frame.h SOURCES jsonschema.cc vocabularies.cc known_walker.cc frame.cc format.cc - helpers.h iterator.h + bundle.cc helpers.h iterator.h "${CMAKE_CURRENT_BINARY_DIR}/known_resolver.cc") if(SOURCEMETA_CORE_INSTALL) diff --git a/vendor/core/src/core/jsonschema/bundle.cc b/vendor/core/src/core/jsonschema/bundle.cc new file mode 100644 index 00000000..708bd4d0 --- /dev/null +++ b/vendor/core/src/core/jsonschema/bundle.cc @@ -0,0 +1,614 @@ +#include + +#include + +#include "helpers.h" + +#include // std::ranges::any_of +#include // assert +#include // std::size_t +#include // std::uint64_t +#include // std::cref +#include // std::optional +#include // std::string +#include // std::string_view +#include // std::tuple +#include // std::unordered_map +#include // std::move, std::pair +#include // std::vector + +namespace sourcemeta::core { + +namespace { + +auto is_skippable_metaschema_reference(const SchemaBundleOptions::Mode mode, + const WeakPointer &pointer, + const std::string &destination) -> bool { + assert(!pointer.empty()); + assert(pointer.back().is_property()); + if (pointer.back().to_property() != "$schema") { + return false; + } + + return mode == SchemaBundleOptions::Mode::References || + schema_is_official(destination); +} + +// The dialect a schema declares, falling back to the given default +auto declared_dialect(const JSON &schema, + const std::string_view default_dialect) + -> std::string_view { + if (!schema.is_object()) { + return default_dialect; + } + + const auto *dialect{schema.try_at("$schema")}; + return (dialect != nullptr && dialect->is_string()) ? dialect->to_string() + : default_dialect; +} + +// Every frame that bundling constructs spends from the same limit, as how +// many frames it ends up needing is a function of what the resolver hands +// back rather than of the schema the caller passed in. A frame that ran past +// what was left of the limit threw rather than returned, so what it holds is +// always within it +auto charge(std::uint64_t &remaining, const SchemaFrame &frame) -> void { + assert(frame.location_count() <= remaining); + remaining -= frame.location_count(); +} + +auto embed_schema(JSON &root, const Pointer &container, + const std::string_view identifier, JSON &&target, + const SchemaBundleOptions::Callback &callback) -> void { + auto *current{&root}; + for (const auto &token : container) { + if (token.is_property()) { + current->assign_if_missing(token.to_property(), JSON::make_object()); + current = ¤t->at(token.to_property()); + } else { + assert(current->is_array() && current->size() >= token.to_index()); + current = ¤t->at(token.to_index()); + } + } + + if (!current->is_object()) { + throw SchemaError("Could not bundle to a container path that is not an " + "object"); + } + + std::string key{identifier}; + // Ensure we get a definitions entry that does not exist + while (current->defines(key)) { + key += "/x"; + } + + current->assign(key, std::move(target)); + + if (callback) { + auto location{to_weak_pointer(container)}; + location.push_back(std::cref(key)); + callback(identifier, location); + } +} + +auto elevate_embedded_resources( + JSON &remote, JSON &root, const Pointer &container, + const SchemaBaseDialect remote_dialect, const SchemaWalker &walker, + const SchemaResolver &resolver, std::string_view default_dialect, + std::unordered_map &bundled, + std::uint64_t &remaining, const SchemaBundleOptions::Callback &callback) + -> void { + const auto keyword{definitions_keyword(remote_dialect)}; + const JSON::String keyword_string{keyword}; + if (keyword.empty() || !remote.is_object() || + !remote.defines(keyword_string) || + !remote.at(keyword_string).is_object()) { + return; + } + + auto &defs{remote.at(keyword_string)}; + const auto remote_dialect_uri{declared_dialect(remote, default_dialect)}; + + // Navigate to the root container once, as it doesn't change per entry + const JSON *root_container{&root}; + bool container_exists{true}; + for (const auto &token : container) { + if (!token.is_property() || !root_container->is_object() || + !root_container->defines(token.to_property())) { + container_exists = false; + break; + } + + root_container = &root_container->at(token.to_property()); + } + + std::vector> to_extract; + std::vector to_remove; + for (const auto &entry : defs.as_object()) { + const auto &key{entry.first}; + const auto &value{entry.second}; + // Only an entry that declares an absolute identifier matching its key can + // ever be elevated, and framing rejects the fragment-only identifiers that + // older drafts use for anchors. Rule those out before paying for a frame + if (!value.is_object()) { + continue; + } + const auto *declared_id{value.try_at("$id")}; + if (declared_id == nullptr) { + declared_id = value.try_at("id"); + } + if (declared_id == nullptr || !declared_id->is_string() || + declared_id->to_string() != key || + !URI{declared_id->to_string()}.is_absolute()) { + continue; + } + + // The remote's dialect is what an entry that declares none inherits, so + // hand it to the frame as the default rather than falling back after + SchemaFrame entry_frame{SchemaFrame::Mode::Root, + value, + walker, + resolver, + remote_dialect_uri, + "", + SchemaFrame::IdentifierMode::Additional, + {EMPTY_WEAK_POINTER}, + "", + remaining}; + charge(remaining, entry_frame); + const auto &identifier{entry_frame.root()}; + if (identifier.empty() || identifier != key || + !URI{identifier}.is_absolute()) { + continue; + } + + const JSON::String identifier_string{identifier}; + const auto defines_dialect{value.defines("$schema")}; + if (bundled.contains(identifier_string)) { + if (container_exists && root_container->is_object()) { + for (const auto &root_entry : root_container->as_object()) { + if (!root_entry.first.starts_with(identifier_string)) { + continue; + } + + // Same reasoning as above: rule out what cannot match, and what + // framing would reject, before paying for a frame + if (!root_entry.second.is_object()) { + continue; + } + const auto *stored_declared_id{root_entry.second.try_at("$id")}; + if (stored_declared_id == nullptr) { + stored_declared_id = root_entry.second.try_at("id"); + } + if (stored_declared_id == nullptr || + !stored_declared_id->is_string() || + stored_declared_id->to_string() != identifier_string || + !URI{stored_declared_id->to_string()}.is_absolute()) { + continue; + } + + SchemaFrame stored_frame{SchemaFrame::Mode::Root, + root_entry.second, + walker, + resolver, + remote_dialect_uri, + "", + SchemaFrame::IdentifierMode::Additional, + {EMPTY_WEAK_POINTER}, + "", + remaining}; + charge(remaining, stored_frame); + const auto &stored_id{stored_frame.root()}; + if (stored_id != identifier_string) { + continue; + } + + if (defines_dialect) { + if (root_entry.second != value) { + throw SchemaError( + "Conflicting embedded resources with the same identifier"); + } + } else { + // The stored copy of the resource got its dialect stamped on + // extraction, so compare against a candidate that is stamped in + // the same way + auto candidate{value}; + candidate.assign("$schema", + JSON{declared_dialect(value, remote_dialect_uri)}); + if (root_entry.second != candidate) { + throw SchemaError( + "Conflicting embedded resources with the same identifier"); + } + } + + break; + } + } + + to_remove.emplace_back(key); + } else { + to_extract.emplace_back(key, !defines_dialect); + bundled.emplace(identifier_string, identifier_string); + } + } + + for (const auto &[key, needs_dialect] : to_extract) { + auto value{std::move(defs.at(key))}; + defs.erase(key); + // Otherwise the elevated resource would be re-interpreted under the + // dialect of the schema it gets embedded into, which can differ from + // the dialect it inherited from the remote it was elevated out of + if (needs_dialect) { + value.assign("$schema", + JSON{declared_dialect(value, remote_dialect_uri)}); + } + + embed_schema(root, container, key, std::move(value), callback); + } + + for (const auto &key : to_remove) { + defs.erase(key); + } + + if (defs.empty()) { + remote.erase(JSON::String{keyword}); + } +} + +auto embed_references( + JSON &root, const Pointer &container, JSON &subschema, + const SchemaWalker &walker, const SchemaResolver &resolver, + const SchemaBundleOptions::Mode mode, std::string_view default_dialect, + std::string_view default_id, const SchemaFrame::Paths &paths, + std::string_view default_base, + std::unordered_map &bundled, + std::uint64_t &remaining, const SchemaBundleOptions::Callback &callback, + const std::size_t depth = 0) -> void { + // Create a fresh frame for each schema we analyze to avoid key collisions + // between different schemas that have references at the same pointer paths + static const SchemaFrame::Paths NESTED_PATHS{EMPTY_WEAK_POINTER}; + const SchemaFrame frame{ + SchemaFrame::Mode::References, subschema, walker, resolver, + default_dialect, default_id, SchemaFrame::IdentifierMode::Additional, + // We only want to frame in "wrapper" mode for the top + // level object, which is also the only one that the + // base the caller retrieved it from applies to, as + // every remote carries the identity it was resolved by + depth == 0 ? paths : NESTED_PATHS, + depth == 0 ? default_base : std::string_view{}, remaining}; + charge(remaining, frame); + + std::vector> deferred; + std::vector> ref_rewrites; + + frame.for_each_unresolved_reference([&](const auto &pointer, + const auto &reference) -> void { + // We don't want to bundle official schemas, as we can expect + // virtually all implementations to understand them out of the box. + // Depending on the bundling strategy, we may skip meta-schemas entirely + if (is_skippable_metaschema_reference(mode, pointer, + reference.destination)) { + return; + } + + // If we can't find the destination but there is a base and we can + // find base, then we are facing an unresolved fragment + if (!reference.base.empty() && frame.traverse(reference.base).has_value()) { + throw SchemaReferenceError(reference.destination, to_pointer(pointer), + "Could not resolve schema reference"); + } + + if (reference.base.empty()) { + throw SchemaReferenceError(reference.destination, to_pointer(pointer), + "Could not resolve schema reference"); + } + + assert(!reference.base.empty()); + const JSON::String identifier{reference.base}; + + if (bundled.contains(identifier)) { + const auto &mapped_id{bundled.at(identifier)}; + if (mapped_id != identifier) { + URI rewrite_uri{mapped_id}; + if (reference.fragment.has_value()) { + rewrite_uri.fragment(reference.fragment.value()); + } + + ref_rewrites.emplace_back(to_pointer(pointer), rewrite_uri.recompose()); + } + + return; + } + + auto resolved{resolver(identifier)}; + if (!resolved.has_value()) { + if (frame.traverse(identifier).has_value()) { + throw SchemaReferenceError(reference.destination, to_pointer(pointer), + "Could not resolve schema reference"); + } + + throw SchemaResolutionError( + identifier, "Could not resolve the reference to an external schema"); + } + + // Bundling rewrites the schema before embedding it, so it needs a copy + // it owns rather than whatever the resolver chose to hand back + auto remote{std::move(resolved).to_owned()}; + if (!remote.is_object() && !remote.is_boolean()) { + throw SchemaReferenceError(identifier, to_pointer(pointer), + "The JSON document is not a valid JSON " + "Schema"); + } + + std::optional remote_root_frame; + try { + remote_root_frame.emplace( + SchemaFrame::Mode::Root, remote, walker, resolver, default_dialect, + "", SchemaFrame::IdentifierMode::Additional, + SchemaFrame::Paths{EMPTY_WEAK_POINTER}, "", remaining); + charge(remaining, remote_root_frame.value()); + } catch (const SchemaUnknownBaseDialectError &) { + throw SchemaReferenceError(identifier, to_pointer(pointer), + "The JSON document is not a valid JSON " + "Schema"); + } + + const auto remote_base_dialect{ + remote_root_frame->root_location().value().get().base_dialect}; + auto remote_id = remote_root_frame->root(); + + // If the reference has a fragment, verify it exists in the remote + // schema + if (reference.fragment.has_value()) { + // A pointer fragment names a place of the document, and the document can + // answer for that on its own without paying to frame it + const auto fragment_pointer{ + fragment_to_pointer(URI{reference.destination})}; + bool exists{fragment_pointer.has_value() && + try_get(remote, fragment_pointer.value()) != nullptr}; + + // An anchor is not a place of the document, and the drafts that spell + // identifiers as `id` let one look just like a pointer, so a miss above + // still has to ask the frame. Only the anchors of the remote matter + // here, rather than every pointer of it + if (!exists) { + const SchemaFrame remote_frame{SchemaFrame::Mode::Locations, + remote, + walker, + resolver, + default_dialect, + identifier, + SchemaFrame::IdentifierMode::Additional, + {EMPTY_WEAK_POINTER}, + "", + remaining}; + charge(remaining, remote_frame); + exists = remote_frame.traverse(reference.destination).has_value(); + } + + if (!exists) { + throw SchemaReferenceError(reference.destination, to_pointer(pointer), + "Could not resolve schema reference"); + } + } + + JSON::String effective_id{remote_id.empty() ? JSON::String{identifier} + : JSON::String{remote_id}}; + + if (remote.is_object()) { + // Otherwise the embedded resource would be re-interpreted under the + // dialect of the schema it gets embedded into, which can differ from + // the default dialect that the remote was resolved with + if (!remote.defines("$schema")) { + remote.assign("$schema", + JSON{declared_dialect(remote, default_dialect)}); + } + + schema_reidentify(remote, effective_id, remote_base_dialect); + } + + if (effective_id != identifier) { + URI rewrite_uri{effective_id}; + if (reference.fragment.has_value()) { + rewrite_uri.fragment(reference.fragment.value()); + } + + ref_rewrites.emplace_back(to_pointer(pointer), rewrite_uri.recompose()); + } + + bundled.emplace(identifier, effective_id); + bundled.emplace(effective_id, effective_id); + deferred.emplace_back(std::move(remote), std::move(effective_id), + remote_base_dialect); + }); + + for (auto &[rewrite_pointer, rewrite_value] : ref_rewrites) { + set(subschema, rewrite_pointer, JSON{rewrite_value}); + } + + for (auto &[remote, effective_id, remote_dialect] : deferred) { + embed_references(root, container, remote, walker, resolver, mode, + default_dialect, effective_id, paths, default_base, + bundled, remaining, callback, depth + 1); + elevate_embedded_resources(remote, root, container, remote_dialect, walker, + resolver, default_dialect, bundled, remaining, + callback); + embed_schema(root, container, effective_id, std::move(remote), callback); + } +} + +auto bundle_internal(JSON &schema, const SchemaWalker &walker, + const SchemaResolver &resolver, + const SchemaBundleOptions::Mode mode, + std::string_view default_dialect, + std::string_view default_id, + const std::optional &default_container, + const SchemaFrame::Paths &paths, + std::string_view default_base, std::uint64_t &remaining, + const SchemaBundleOptions::Callback &callback) -> void { + // Pre-scan the schema to find any already-embedded schemas and mark them + // as bundled to avoid re-embedding them. This includes the root schema itself + // and any schemas already embedded within it + std::unordered_map bundled; + SchemaFrame initial_frame{SchemaFrame::Mode::Locations, + schema, + walker, + resolver, + default_dialect, + default_id, + SchemaFrame::IdentifierMode::Additional, + paths, + default_base, + remaining}; + charge(remaining, initial_frame); + initial_frame.for_each_resource_uri([&bundled](const auto &uri) -> void { + bundled.emplace(JSON::String{uri}, JSON::String{uri}); + }); + if (default_container.has_value()) { + // This is undefined behavior + assert(!default_container.value().empty()); + // Whatever bundling embeds has to land somewhere that framing reaches + // again, or a later pass cannot see it and embeds a second copy. So a + // container has to be a keyword that the dialect reserves for schema + // definitions, declared on a schema that the given paths cover. A wrapper + // format keeps its container outside every schema it frames, where JSON + // Schema has nothing to say about where things may go + const auto container{to_weak_pointer(default_container.value())}; + if (std::ranges::any_of(paths, [&container](const auto &path) -> bool { + return container.starts_with(path); + })) { + const auto parent_pointer{default_container.value().initial()}; + const auto parent{ + initial_frame.traverse(to_weak_pointer(parent_pointer))}; + if (!parent.has_value() || + (parent.value().get().type != SchemaFrame::LocationType::Resource && + parent.value().get().type != SchemaFrame::LocationType::Subschema) || + !default_container.value().back().is_property() || + walker(default_container.value().back().to_property(), + initial_frame.vocabularies(parent.value().get(), resolver)) + .type != SchemaKeywordType::LocationMembers) { + throw SchemaError("Could not bundle to a container that the dialect " + "does not reserve for schema definitions"); + } + } + + embed_references(schema, default_container.value(), schema, walker, + resolver, mode, default_dialect, default_id, paths, + default_base, bundled, remaining, callback); + return; + } + + // If the schema identifier is implicit, add it to the top-level of the + // bundled schema. Otherwise, potential relative references based on this + // implicit base URI will likely not resolve unless end users happen to + // know that this implicit base URI is. Note that boolean schemas cannot + // declare identifiers, so we leave those untouched + if (!default_id.empty() && schema.is_object()) { + // Deliberately framed without a default identifier, so that the root + // comes back empty exactly when the schema declares none of its own + SchemaFrame declared_frame{SchemaFrame::Mode::Root, + schema, + walker, + resolver, + default_dialect, + "", + SchemaFrame::IdentifierMode::Additional, + {EMPTY_WEAK_POINTER}, + "", + remaining}; + charge(remaining, declared_frame); + if (declared_frame.root().empty()) { + schema_reidentify(schema, default_id, resolver, default_dialect); + } + } + + std::optional schema_root_frame; + try { + schema_root_frame.emplace( + SchemaFrame::Mode::Root, schema, walker, resolver, default_dialect, + default_id, SchemaFrame::IdentifierMode::Additional, + SchemaFrame::Paths{EMPTY_WEAK_POINTER}, "", remaining); + charge(remaining, schema_root_frame.value()); + } catch (const SchemaUnknownBaseDialectError &) { + throw SchemaError("Could not determine how to perform bundling in this " + "dialect"); + } + + const auto schema_base_dialect{ + schema_root_frame->root_location().value().get().base_dialect}; + + const auto container_keyword{definitions_keyword(schema_base_dialect)}; + if (container_keyword.empty()) { + SchemaFrame frame{SchemaFrame::Mode::References, + schema, + walker, + resolver, + default_dialect, + default_id, + SchemaFrame::IdentifierMode::Additional, + {EMPTY_WEAK_POINTER}, + default_base, + remaining}; + charge(remaining, frame); + if (frame.standalone()) { + return; + } + + throw SchemaError("Could not determine how to perform bundling in this " + "dialect"); + } + + if (ref_overrides_adjacent_keywords(schema_base_dialect) && + schema.is_object() && schema.defines("$ref")) { + if (schema.size() == 1) { + const auto is_draft3{ + schema_base_dialect == SchemaBaseDialect::JSON_SCHEMA_DRAFT_3 || + schema_base_dialect == SchemaBaseDialect::JSON_SCHEMA_DRAFT_3_HYPER}; + auto branches{JSON::make_array()}; + branches.push_back(schema); + schema.at("$ref").into(std::move(branches)); + schema.rename("$ref", is_draft3 ? "extends" : "allOf"); + } else { + throw SchemaError( + "Cannot bundle a JSON Schema Draft 7 or older with a top-level " + "`$ref` (which overrides sibling keywords) without introducing " + "undefined behavior"); + } + } + + embed_references(schema, {JSON::String{container_keyword}}, schema, walker, + resolver, mode, default_dialect, default_id, paths, + default_base, bundled, remaining, callback); +} + +} // namespace + +auto schema_bundle(JSON &schema, const SchemaWalker &walker, + const SchemaResolver &resolver, + std::string_view default_dialect, + std::string_view default_id, + const SchemaBundleOptions &options) -> void { + auto remaining{options.max_locations}; + try { + bundle_internal(schema, walker, resolver, options.mode, default_dialect, + default_id, options.default_container, options.paths, + options.default_base, remaining, options.callback); + } catch (const SchemaFrameLimitError &) { + // Every frame spends from what is left rather than from the whole, so the + // one that ran out reports what it was handed. The caller set the limit + // for the operation, so that is what the operation reports back + throw SchemaFrameLimitError{options.max_locations}; + } +} + +auto schema_bundle(const JSON &schema, const SchemaWalker &walker, + const SchemaResolver &resolver, + std::string_view default_dialect, + std::string_view default_id, + const SchemaBundleOptions &options) -> JSON { + JSON copy = schema; + schema_bundle(copy, walker, resolver, default_dialect, default_id, options); + return copy; +} + +} // namespace sourcemeta::core diff --git a/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema.h b/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema.h index 7937cb99..f849619c 100644 --- a/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema.h +++ b/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema.h @@ -14,6 +14,9 @@ #include // NOLINTEND(misc-include-cleaner) +#include // std::uint8_t, std::uint64_t +#include // std::function +#include // std::numeric_limits #include // std::optional, std::nullopt #include // std::string_view @@ -25,10 +28,10 @@ /// It offers the building blocks that any operation on a schema needs, /// independently of what that operation is: identification, dialect and /// vocabulary detection, resolution of remote schemas, keyword classification -/// across dialects, and framing a schema into the locations and references it -/// declares. Evaluation is only one of the consumers of these utilities, -/// alongside bundling, linting, transformation, code generation, and -/// documentation tooling. +/// across dialects, framing a schema into the locations and references it +/// declares, and bundling a schema into a self-contained document. Evaluation +/// is only one of the consumers of these utilities, alongside linting, +/// transformation, code generation, and documentation tooling. /// /// This functionality is included as follows: /// @@ -92,6 +95,24 @@ auto schema_reidentify(sourcemeta::core::JSON &schema, const SchemaResolver &resolver, std::string_view default_dialect = "") -> void; +/// @ingroup jsonschema +/// +/// The keyword that carries a schema identifier in the given base dialect. +/// For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// assert(sourcemeta::core::schema_identifier_keyword( +/// sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_2020_12) == "$id"); +/// assert(sourcemeta::core::schema_identifier_keyword( +/// sourcemeta::core::SchemaBaseDialect::JSON_SCHEMA_DRAFT_4) == "id"); +/// ``` +SOURCEMETA_CORE_JSONSCHEMA_EXPORT +auto schema_identifier_keyword(const SchemaBaseDialect base_dialect) + -> std::string_view; + /// @ingroup jsonschema /// /// A shortcut to sourcemeta::core::schema_reidentify if you know the base @@ -137,6 +158,171 @@ SOURCEMETA_CORE_JSONSCHEMA_EXPORT auto schema_format(sourcemeta::core::JSON &schema, const SchemaFrame &frame) -> void; +/// @ingroup jsonschema +/// Everything bundling takes beyond the schema and how to read it +struct SchemaBundleOptions { + /// A callback to report which schema got embedded and where, as the + /// identifier that schema answers to and a pointer from the root of the + /// schema being bundled. A schema that declares one of its own answers to + /// that rather than to whichever URI it was resolved by, which are not + /// always the same. The two are given separately because bundling picks a + /// key that is free rather than one that matches, so the last token of that + /// pointer is not always the identifier either + using Callback = std::function; + + /// The strategies that the bundling process can follow + enum class Mode : std::uint8_t { + /// Embed every external reference, including any non-official + /// meta-schemas that the schema or its dependencies declare, along + /// with the dependencies of those meta-schemas + NonOfficialMetaschemas, + /// Embed every external reference, skipping meta-schema + /// declarations entirely + References + }; + + /// The strategy to follow + Mode mode{Mode::NonOfficialMetaschemas}; + /// Where to embed what bundling pulls in + std::optional default_container; + /// The paths to bundle within a schema wrapper + SchemaFrame::Paths paths{sourcemeta::core::EMPTY_WEAK_POINTER}; + /// The base URI that the document was retrieved from, which a relative + /// reference within any of the given paths resolves against. As with + /// sourcemeta::core::SchemaFrame, this does not claim that the document + /// declares an identifier, so bundling never writes it into the document + std::string_view default_base; + /// The maximum number of frame locations that analysis may register. How + /// many schemas bundling ends up embedding follows from what the resolver + /// hands back rather than from the schema the caller passed in, and every + /// frame that bundling constructs spends from this one limit, throwing + /// sourcemeta::core::SchemaFrameLimitError once it runs out. Note that a + /// remote is copied out of the resolver before anything charges for it, so + /// this bounds how many oversized schemas get copied rather than whether + /// one does + std::uint64_t max_locations{std::numeric_limits::max()}; + /// A callback to report where each schema got embedded, which is the only + /// way to know what a later call has to frame when bundling into a + /// container that the dialect does not otherwise traverse + Callback callback; +}; + +/// @ingroup jsonschema +/// +/// This function bundles a JSON Schema (starting from Draft 4) by embedding +/// every remote reference into the top level schema resource, handling circular +/// dependencies and more. This overload mutates the input schema. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// // A custom resolver that knows about an additional schema +/// static auto test_resolver(std::string_view identifier) +/// -> sourcemeta::core::SchemaResolverResult { +/// if (identifier == "https://www.example.com/test") { +/// return sourcemeta::core::parse_json(R"JSON({ +/// "$id": "https://www.example.com/test", +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "type": "string" +/// })JSON"); +/// } else { +/// return sourcemeta::core::schema_resolver(identifier); +/// } +/// } +/// +/// sourcemeta::core::JSON document = +/// sourcemeta::core::parse_json(R"JSON({ +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "items": { "$ref": "https://www.example.com/test" } +/// })JSON"); +/// +/// sourcemeta::core::schema_bundle(document, +/// sourcemeta::core::schema_walker, test_resolver); +/// +/// const sourcemeta::core::JSON expected = +/// sourcemeta::core::parse_json(R"JSON({ +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "items": { "$ref": "https://www.example.com/test" }, +/// "$defs": { +/// "https://www.example.com/test": { +/// "$id": "https://www.example.com/test", +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "type": "string" +/// } +/// } +/// })JSON"); +/// +/// assert(document == expected); +/// ``` +SOURCEMETA_CORE_JSONSCHEMA_EXPORT +auto schema_bundle(sourcemeta::core::JSON &schema, const SchemaWalker &walker, + const SchemaResolver &resolver, + std::string_view default_dialect = "", + std::string_view default_id = "", + const SchemaBundleOptions &options = {}) -> void; + +/// @ingroup jsonschema +/// +/// This function bundles a JSON Schema (starting from Draft 4) by embedding +/// every remote reference into the top level schema resource, handling circular +/// dependencies and more. This overload returns a new schema, without mutating +/// the input schema. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// // A custom resolver that knows about an additional schema +/// static auto test_resolver(std::string_view identifier) +/// -> sourcemeta::core::SchemaResolverResult { +/// if (identifier == "https://www.example.com/test") { +/// return sourcemeta::core::parse_json(R"JSON({ +/// "$id": "https://www.example.com/test", +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "type": "string" +/// })JSON"); +/// } else { +/// return sourcemeta::core::schema_resolver(identifier); +/// } +/// } +/// +/// const sourcemeta::core::JSON document = +/// sourcemeta::core::parse_json(R"JSON({ +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "items": { "$ref": "https://www.example.com/test" } +/// })JSON"); +/// +/// const sourcemeta::core::JSON result = +/// sourcemeta::core::schema_bundle(document, +/// sourcemeta::core::schema_walker, test_resolver); +/// +/// const sourcemeta::core::JSON expected = +/// sourcemeta::core::parse_json(R"JSON({ +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "items": { "$ref": "https://www.example.com/test" }, +/// "$defs": { +/// "https://www.example.com/test": { +/// "$id": "https://www.example.com/test", +/// "$schema": "https://json-schema.org/draft/2020-12/schema", +/// "type": "string" +/// } +/// } +/// })JSON"); +/// +/// assert(result == expected); +/// ``` +SOURCEMETA_CORE_JSONSCHEMA_EXPORT +auto schema_bundle(const sourcemeta::core::JSON &schema, + const SchemaWalker &walker, const SchemaResolver &resolver, + std::string_view default_dialect = "", + std::string_view default_id = "", + const SchemaBundleOptions &options = {}) + -> sourcemeta::core::JSON; + } // namespace sourcemeta::core #endif diff --git a/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema_frame.h b/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema_frame.h index 7882f198..18aad623 100644 --- a/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema_frame.h +++ b/vendor/core/src/core/jsonschema/include/sourcemeta/core/jsonschema_frame.h @@ -107,7 +107,10 @@ class SOURCEMETA_CORE_JSONSCHEMA_EXPORT SchemaFrame { // location entry to point to if it is an external unresolved reference /// The absolute URI that the reference resolves to sourcemeta::core::JSON::String destination; - /// The base that the reference resolved against, empty when there is none + /// The part of the destination that comes before its fragment, which is + /// empty when the destination is a fragment alone. This is where the + /// reference leads rather than what it was resolved against, which is the + /// base of whichever location holds it std::string_view base; /// The fragment of the destination, if it declares one std::optional fragment; diff --git a/vendor/core/src/core/jsonschema/jsonschema.cc b/vendor/core/src/core/jsonschema/jsonschema.cc index e0e6651c..45b5241b 100644 --- a/vendor/core/src/core/jsonschema/jsonschema.cc +++ b/vendor/core/src/core/jsonschema/jsonschema.cc @@ -210,6 +210,11 @@ auto sourcemeta::core::schema_reidentify(sourcemeta::core::JSON &schema, schema_reidentify(schema, new_identifier, resolved_base_dialect.value()); } +auto sourcemeta::core::schema_identifier_keyword( + const SchemaBaseDialect base_dialect) -> std::string_view { + return sourcemeta::core::id_keyword(base_dialect).name; +} + auto sourcemeta::core::schema_reidentify(sourcemeta::core::JSON &schema, std::string_view new_identifier, const SchemaBaseDialect base_dialect) diff --git a/vendor/core/src/core/openapi/CMakeLists.txt b/vendor/core/src/core/openapi/CMakeLists.txt index 3527459d..e5ed135c 100644 --- a/vendor/core/src/core/openapi/CMakeLists.txt +++ b/vendor/core/src/core/openapi/CMakeLists.txt @@ -2,7 +2,8 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME openapi PRIVATE_HEADERS error.h SOURCES helpers.h reference.h example.h content.h link.h response.h parameter.h request_body.h security.h path_item.h paths.h - components.h external_documentation.h info.h server.h tag.h + components.h discriminator.h external_documentation.h info.h + server.h tag.h document.h frame.cc version.cc) if(SOURCEMETA_CORE_INSTALL) diff --git a/vendor/core/src/core/openapi/discriminator.h b/vendor/core/src/core/openapi/discriminator.h new file mode 100644 index 00000000..38d6b6bd --- /dev/null +++ b/vendor/core/src/core/openapi/discriminator.h @@ -0,0 +1,205 @@ +#ifndef SOURCEMETA_CORE_OPENAPI_DISCRIMINATOR_H_ +#define SOURCEMETA_CORE_OPENAPI_DISCRIMINATOR_H_ + +#include + +#include "components.h" +#include "helpers.h" + +#include // std::optional, std::nullopt +#include // std::set +#include // std::string_view +#include // std::move +#include // std::vector + +namespace sourcemeta::core { + +constexpr auto OPENAPI_HASH_DISCRIMINATOR{ + JSON::Object::hash("discriminator"sv)}; +constexpr auto OPENAPI_HASH_MAPPING{JSON::Object::hash("mapping"sv)}; +constexpr auto OPENAPI_HASH_DEFAULT_MAPPING{ + JSON::Object::hash("defaultMapping"sv)}; + +/// Where a Discriminator Object names a schema, by the name of a component or +/// by URI. OpenAPI Specification 3.1.1, Section 4.3 lists the URI form of a +/// `mapping` among the fields that connect the documents of a description, and +/// Section 4.3.3 lists the name form among the connections it makes by name, +/// so either way one of these is a place the description reaches for +struct OpenAPIDiscriminator { + /// Where the mapping value sits, as a pointer from the root of the document + Pointer origin; + /// Where it points, resolved and canonicalised + JSON::String destination; + /// What it resolved against, which is the nearest identifier an enclosing + /// schema declares for the URI form, and the description itself for the + /// name form + JSON::String scope; +}; + +// OpenAPI Specification 3.1.1, Section 4.8.25: a `mapping` entry "maps a +// specific property value to either a different schema component name, or to a +// schema identified by a URI". Only the latter is a reference, as Section +// 4.3.3 lists the name form among the connections a description makes by name, +// and Section 4.8.25 settles which of the two a value is: +// +// The behavior of a `mapping` value that is both a valid schema name and a +// valid relative URI reference is implementation-defined, but it is +// RECOMMENDED that it be treated as a schema name. To ensure that an +// ambiguous value (e.g. `"foo"`) is treated as a relative URI reference by +// all implementations, authors MUST prefix it with the `"."` path segment +// +// OpenAPI Specification 3.2.1, Section 4.25.3 says the same of the default a +// Discriminator Object may declare beside the map, which is what holds the two +// of them to the one rule: "The behavior of a `mapping` value or +// `defaultMapping` value that is both a valid schema name and a valid relative +// URI reference is implementation-defined" +// +// A schema name is what the Components Object takes as a key, and the path +// segment an author writes to force the other reading is no such key. Whether +// a value holds up as one is asked of the value alone rather than of what the +// description declares, as Section 4.8.25 calls `"foo"` ambiguous without +// regard to whether a component goes by that name +// Keep what one of those names, where it names a schema by URI rather than by +// the name a component goes by. Section 4.8.25 types a mapping as holding +// "schema names or URI references", so a value that holds up as neither is one +// nothing can read, which is what every other field this specification types +// as a URI is held to as well +inline auto openapi_record_discriminator( + std::vector &result, const JSON::String &base, + Pointer origin, const JSON &value, const JSON::String &scope) -> void { + if (!value.is_string()) { + throw OpenAPIError{ + base, std::move(origin), + "A Discriminator Object mapping must name a schema with a string"}; + } + + // Section 4.3.3 lists the name form among the connections a description + // makes by name rather than by URI, and has one resolve "from the entry + // document, rather than the current document". So a name stands for the + // schema that the Components Object of the description holds under it, + // wherever the Discriminator Object naming it happens to sit + if (openapi_is_component_key(value.to_string())) { + Pointer named; + named.push_back(JSON::String{"components"}); + named.push_back(JSON::String{"schemas"}); + named.push_back(JSON::String{value.to_string()}); + result.push_back({.origin = std::move(origin), + .destination = openapi_location_uri(base, named), + .scope = base}); + return; + } + + if (!URI::is_uri_reference(value.to_string())) { + throw OpenAPIError{base, std::move(origin), + "A Discriminator Object mapping must name a schema by " + "the name of a component or by a URI reference"}; + } + + const auto destination{openapi_resolve_uri(value.to_string(), scope)}; + if (!destination.has_value()) { + return; + } + + result.push_back({.origin = std::move(origin), + .destination = destination.value().recompose(), + .scope = scope}); +} + +// Every schema that a Discriminator Object of the document names. +// Section 4.6 has a relative reference of a Schema Object resolve against +// "the nearest parent `$id`", which is the base that framing the schemas +// settles for wherever the Discriminator Object sits +inline auto +openapi_discriminators(const JSON &document, const SchemaFrame &schemas, + const JSON::String &base, const SchemaWalker &walker, + const SchemaResolver &resolver) + -> std::vector { + std::vector result; + // A schema that declares an identifier of its own is registered under every + // base it can be reached by, and what it holds is the one thing whichever + // way it is reached + std::set seen; + schemas.for_each_subschema([&document, &schemas, &base, &result, &seen, + &walker, + &resolver](const auto &location) -> void { + const auto *schema{try_get(document, location.pointer)}; + if (schema == nullptr || !schema->is_object()) { + return; + } + + const auto *discriminator{ + schema->try_at("discriminator", OPENAPI_HASH_DISCRIMINATOR)}; + if (discriminator == nullptr || !discriminator->is_object()) { + return; + } + + auto origin{to_pointer(location.pointer)}; + if (!seen.insert(to_string(origin)).second) { + return; + } + + // Section 4.8.24.2 lists `discriminator` among the keywords that the + // dialect this specification publishes is made of, and Section 4.8.24.5 + // has a Schema Object read under whichever dialect it declares. So a + // schema written against one that leaves the keyword out holds no + // Discriminator Object at all, however the member happens to be spelled, + // and what a keyword amounts to where it sits is the walker's to say + const auto &vocabularies{schemas.vocabularies(location, resolver)}; + if (walker("discriminator", vocabularies).type == + SchemaKeywordType::Unknown) { + return; + } + + const JSON::String scope{location.base}; + origin.push_back(JSON::String{"discriminator"}); + + const auto *mapping{discriminator->try_at("mapping", OPENAPI_HASH_MAPPING)}; + if (mapping != nullptr && mapping->is_object()) { + const auto mapped{origin.concat(JSON::String{"mapping"})}; + for (const auto &entry : mapping->as_object()) { + openapi_record_discriminator(result, base, mapped.concat(entry.first), + entry.second, scope); + } + } + + // OpenAPI Specification 3.2.1, Section 4.25 gives a Discriminator Object a + // default of its own, which is "the schema name or URI reference to a + // schema" just as every entry of the map beside it is. Only the dialect of + // that revision defines the field, which is what settles whether there is + // one to read rather than what the document says of itself. + // + // Section 4.25.1 goes on to require one wherever the discriminating + // property is optional, which is a demand on what the schema holding it + // says of its own properties. Reading that far into a Schema Object is + // the business of whatever understands JSON Schema, so it is left there + if (!vocabularies.contains(SchemaVocabularies::Known::OPENAPI_3_2_BASE)) { + return; + } + + const auto *fallback{ + discriminator->try_at("defaultMapping", OPENAPI_HASH_DEFAULT_MAPPING)}; + if (fallback != nullptr) { + openapi_record_discriminator( + result, base, origin.concat(JSON::String{"defaultMapping"}), + *fallback, scope); + } + }); + + return result; +} + +// Whether the schemas of the document hold what a mapping names. Section +// 4.8.25 has one name "a schema identified by a URI", and framing a document +// records the places of it that are no schema of their own as well, so landing +// on one of those is landing on nothing this was after +inline auto +openapi_discriminator_lands(const SchemaFrame &schemas, + const OpenAPIDiscriminator &discriminator) -> bool { + const auto location{schemas.traverse(discriminator.destination)}; + return location.has_value() && + location.value().get().type != SchemaFrame::LocationType::Pointer; +} + +} // namespace sourcemeta::core + +#endif diff --git a/vendor/core/src/core/openapi/document.h b/vendor/core/src/core/openapi/document.h index c24eaf88..1f64766e 100644 --- a/vendor/core/src/core/openapi/document.h +++ b/vendor/core/src/core/openapi/document.h @@ -23,6 +23,8 @@ #include "tag.h" #include // std::array +#include // std::uint64_t +#include // std::numeric_limits #include // std::optional #include // std::string_view #include // std::move, std::swap, std::unreachable @@ -79,7 +81,8 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) // What a document's `$self` establishes as its base, or nothing when it // establishes none. OpenAPI Specification 3.2.1, Section 4.1: the field // "provides the self-assigned URI of this document, which also serves as its -// base URI in accordance with RFC3986 Section 5.1.1", and Section 4.7.1.1: "If +// base URI in accordance with RFC3986 Section 5.1.1", and Section 4.1.2.2.1: +// "If // `$self` is a relative URI reference, it is resolved against the next // possible base URI source before being used". That next source is whatever // base is in force here, which is the retrieval URI for the entry document and @@ -216,20 +219,7 @@ inline auto openapi_follow_internal_reference(const URI &target, inline auto openapi_reference_target(const JSON::StringView reference, const OpenAPIWalk &walk) -> std::optional { - try { - URI target{JSON::String{reference}}; - if (!walk.base.empty()) { - target.resolve_from(URI{walk.base}); - } - - // Canonicalising here is what makes two spellings of one place one place, - // both to the set that remembers where the walk has been and to a caller - // comparing a destination against a location - target.canonicalize(); - return target; - } catch (const URIParseError &) { - return std::nullopt; - } + return openapi_resolve_uri(reference, walk.base); } // Reading whatever a reference landed on, which is the same work wherever the @@ -238,7 +228,6 @@ inline auto openapi_reference_target(const JSON::StringView reference, // is not a reference the frame writes down inline auto openapi_follow_target(const URI &target, const Pointer &origin, const OpenAPIObjectKind expected, - const bool demands_its_own_kind, OpenAPIWalk &walk) -> void { const auto identifier{target.recompose_without_fragment()}; const auto names_a_fragment{target.fragment().has_value()}; @@ -252,7 +241,7 @@ inline auto openapi_follow_target(const URI &target, const Pointer &origin, return; } - if (demands_its_own_kind && !names_a_fragment && walk.document != nullptr && + if (!names_a_fragment && walk.document != nullptr && openapi_is_document(*walk.document)) { throw OpenAPIError{walk.base, origin, "This reference must name a document that holds only " @@ -284,21 +273,16 @@ inline auto openapi_follow_reference(const JSON::StringView reference, OpenAPIReference{.original = JSON::String{reference}, .destination = target.value().recompose()}); - // Section 4.8.9, of a Path Item Object's `$ref`: "the referenced structure - // MUST be in the form of a Path Item Object", and Section 4.8.20, of a Link - // Object's `operationRef`: it "MUST point to an Operation Object". A - // document that declares a root `openapi` field is an OpenAPI Description - // and neither of those, so a reference from one of those two positions that - // names such a document whole has landed on the wrong thing. Section 4.3.1's - // detection settles how a document is read, which is a separate question - // from whether a reference was allowed to point at it. Section 4.8.23 holds - // a Reference Object to nothing but the form of a URI, and those two - // positions are the only ones a `$ref` reaches either kind from, so what - // the demand really follows is the position rather than the kind - openapi_follow_target(target.value(), origin, expected, - expected == OpenAPIObjectKind::PathItem || - expected == OpenAPIObjectKind::Operation, - walk); + // OpenAPI Specification 3.2.1, Section 4.1.2: "all documents in an OAD MUST + // have either an OpenAPI Object or a Schema Object at the root". A Schema + // Object is what a Schema Object reference names, which never reaches here, + // so every document that a reference of the shell may name holds an OpenAPI + // Object. That is not one of the kinds any position here expects to find, + // so a reference that names such a document whole has landed on the wrong + // thing whatever kind it expected. Section 4.8.9 and Section 4.8.20 say as + // much of the two positions they speak of, and the rest follows from what a + // document may hold rather than from what those two sections single out + openapi_follow_target(target.value(), origin, expected, walk); } // OpenAPI Specification 3.1.1, Section 4.8.1: "This is the root object of the @@ -373,7 +357,7 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) if (established.has_value()) { walk.base = std::move(established.value()); - // Section 4.7.1: "To ensure interoperability, references MUST use the + // Section 4.1.1: "To ensure interoperability, references MUST use the // target document's `$self` URI if the `$self` field is present". So // this is the URI the document answers to, and one that names it by // where it was retrieved from instead names another document, which @@ -476,6 +460,41 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) } } +// Everything the checks need in order to start from nothing, which is a walk +// of the given document keyed by the given base. A 3.2 document may name +// itself, so what the walk ends up keyed by is what it reports rather than +// what it was handed +inline auto openapi_analyse(const JSON &document, JSON::String base, + const std::uint64_t max_locations = + std::numeric_limits::max()) + -> OpenAPIWalk { + OpenAPIWalk walk{.base = std::move(base), + .document = &document, + .operation_ids = {}, + .visited = {}, + .locations = {}, + .references = {}, + .parameters = {}, + .path_items = {}, + .operation_records = {}, + .callbacks = {}, + .endpoints = {}, + .servers = {}, + .security = {}, + .security_schemes = {}, + .tags = {}, + .tag_parents = {}, + .tag_names = {}, + .operation_id_links = {}, + .version = OpenAPIVersion::OPENAPI_3_1, + .dialect = {}, + .info = {}, + .remaining = max_locations, + .limit = max_locations}; + openapi_check_document(document, walk); + return walk; +} + } // namespace sourcemeta::core #endif diff --git a/vendor/core/src/core/openapi/frame.cc b/vendor/core/src/core/openapi/frame.cc index 109d4c5c..d202a20a 100644 --- a/vendor/core/src/core/openapi/frame.cc +++ b/vendor/core/src/core/openapi/frame.cc @@ -1,13 +1,11 @@ #include -#include -#include - +#include "discriminator.h" #include "document.h" #include "helpers.h" #include "info.h" -#include // std::ranges::find +#include // std::ranges::all_of, std::ranges::find #include // assert #include // std::size_t #include // std::map @@ -21,15 +19,9 @@ namespace { using namespace std::string_view_literals; -// The base a location is keyed by is the key up to its fragment, which is why -// nothing repeats it on the entry itself. A parent is given as one of these -// keys rather than as a bare pointer, so that following it is a lookup in the -// same map rather than a key the reader has to rebuild -auto document_of(const sourcemeta::core::JSON::String &uri) - -> sourcemeta::core::JSON::String { - return sourcemeta::core::JSON::String{sourcemeta::core::take_until(uri, '#')}; -} - +// A parent is given as one of the keys a location is held under rather than as +// a bare pointer, so that following it is a lookup in the same map rather than +// a key the reader has to rebuild auto parent_of(const std::map &locations, const sourcemeta::core::JSON::String &uri, @@ -39,7 +31,7 @@ auto parent_of(const std::map sourcemeta::core::JSON::String { - if (input.empty()) { - return {}; - } - - std::optional base; - try { - base.emplace(input); - } catch (const sourcemeta::core::URIParseError &) { - base.reset(); - } - - // RFC 3986 Section 5.2.1: "only the scheme component is required to be - // present in a base URI". Anything without one cannot resolve a reference - // RFC 3986 Section 5.2.2 resolves a reference against a base's scheme, - // authority, path and query, and never against its fragment, so a fragment - // is no part of what a base is. Keeping one would also put two of them in - // every location this frame reports - if (base.has_value() && base.value().scheme().has_value()) { - base.value().canonicalize(); - const auto result{base.value().recompose_without_fragment()}; - if (result.has_value()) { - return result.value(); - } - } - - throw sourcemeta::core::OpenAPIError{ - sourcemeta::core::EMPTY_POINTER, - "The OpenAPI Description base must be a URI with a scheme"}; -} - // A Reference Object, and a Path Item Object that declares a `$ref`, stand in // for what they lead to. OpenAPI Specification 3.1.1, Section 4.8.9 has `$ref` // "allow for a referenced definition of this path item" and leaves what a @@ -521,35 +475,6 @@ auto project(const sourcemeta::core::OpenAPIWalk &walk) return result; } -auto analyse(const sourcemeta::core::JSON &document, - sourcemeta::core::JSON::String base) - -> sourcemeta::core::OpenAPIWalk { - sourcemeta::core::OpenAPIWalk walk{ - .base = std::move(base), - .document = &document, - .operation_ids = {}, - .visited = {}, - .locations = {}, - .references = {}, - .parameters = {}, - .path_items = {}, - .operation_records = {}, - .callbacks = {}, - .endpoints = {}, - .servers = {}, - .security = {}, - .security_schemes = {}, - .tags = {}, - .tag_parents = {}, - .tag_names = {}, - .operation_id_links = {}, - .version = sourcemeta::core::OpenAPIVersion::OPENAPI_3_1, - .dialect = {}, - .info = {}}; - sourcemeta::core::openapi_check_document(document, walk); - return walk; -} - } // namespace namespace sourcemeta::core { @@ -563,6 +488,9 @@ struct OpenAPIFrame::Internal { std::map locations; std::map references; std::vector operations; + // What a Discriminator Object names by URI, which is a reference the schemas + // hold rather than one the shell around them does + std::vector discriminators; // Reading inside a Schema Object is the business of whatever understands // JSON Schema, so this is that pass over every Schema Object position at // once. It is declared last so that it is destroyed first, as it holds @@ -577,7 +505,9 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, const std::string_view default_base, const std::uint64_t max_locations) : internal_{std::make_unique()} { - auto walk{analyse(document, canonical_base(default_base))}; + auto walk{sourcemeta::core::openapi_analyse( + document, sourcemeta::core::openapi_canonical_base(default_base), + max_locations)}; this->internal_->version = walk.version; this->internal_->info = walk.info; // What the caller passed in is where the entry document was retrieved from, @@ -599,12 +529,13 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, // Projecting reads the whole walk, so nothing is taken out of it until after this->internal_->operations = project(walk); + const auto walk_locations{walk.locations.size()}; this->internal_->locations = std::move(walk.locations); this->internal_->references = std::move(walk.references); // Every Schema Object position of the document at once, rather than one // pass each, so that a schema referring to another resolves against a frame - // that holds both. Section 4.8.24.1 scopes `jsonSchemaDialect` to "all + // that holds both. Section 4.8.24.5 scopes `jsonSchemaDialect` to "all // Schema Objects contained within an OAS document", and a document has one // base, so what those positions have in common is the whole of what this // pass needs to be told @@ -613,8 +544,8 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, // What sits inside a Schema Object is JSON Schema's to make sense of, so a // reference that names a place in there has the description read one of its - // own Objects out of a schema. Appendix G of OAS 3.2, and Section 3.2 of - // 3.1, leave what to do about a place read as two kinds of thing to the + // own Objects out of a schema. Appendix G of OAS 3.2, and Section 4.3.2 + // of 3.1, leave what to do about a place read as two kinds of thing to the // implementation and allow saying so, which is what this does. Framing the // schemas could not proceed regardless, as it is given each of these // positions to frame and they must not sit within one another @@ -643,17 +574,45 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, } this->internal_->schema_resolver = resolver; - this->internal_->schemas = std::make_unique( - SchemaFrame::Mode::References, document, walker, resolver, - root->second.dialect, "", SchemaFrame::IdentifierMode::Additional, - this->internal_->schema_paths, this->internal_->base, max_locations); + // What the shell of a description goes by and what its schemas go by are + // places of the one description, so they spend from the one allowance. The + // walk above has already spent its share, and it never spends more than it + // was handed + assert(walk_locations <= max_locations); + try { + this->internal_->schemas = std::make_unique( + SchemaFrame::Mode::References, document, walker, resolver, + root->second.dialect, "", SchemaFrame::IdentifierMode::Additional, + this->internal_->schema_paths, this->internal_->base, + max_locations - walk_locations); + } catch (const SchemaFrameLimitError &) { + // Framing the schemas was handed what was left rather than the whole, so + // the allowance it reports is not the one the caller set + throw OpenAPIFrameLimitError{max_locations}; + } + + // OpenAPI Specification 3.1.1, Section 4.3 lists the URI form of a + // Discriminator Object `mapping` among the fields that connect the documents + // of a description, and Appendix G of 3.2 keeps it among the connections a + // description makes. It is the one of them that a schema frame does not + // read, as the keyword it sits under belongs to the dialect this + // specification publishes rather than to JSON Schema + this->internal_->discriminators = + openapi_discriminators(document, *(this->internal_->schemas), + this->internal_->base, walker, resolver); + const auto every_mapping_lands{ + std::ranges::all_of(this->internal_->discriminators, + [this](const auto &discriminator) -> bool { + return openapi_discriminator_lands( + *(this->internal_->schemas), discriminator); + })}; // What a Schema Object references is as much a part of the description as // what the shell around it does, so a description whose schemas reach for // something nobody holds is one that is missing a part of itself just the // same - this->internal_->standalone = - every_reference_lands && this->internal_->schemas->standalone(); + this->internal_->standalone = every_reference_lands && every_mapping_lands && + this->internal_->schemas->standalone(); // Section 4.3.3 has resolving a Link Object `operationId` require "parsing // all referenced documents prior to determining an `operationId` to be @@ -766,6 +725,25 @@ auto OpenAPIFrame::to_json() const -> JSON { } result.assign_assume_new("operations", std::move(operations)); + + // Only a description whose schemas name one carries these, so a description + // that names none reports nothing rather than an empty list + if (!this->internal_->discriminators.empty()) { + auto discriminators{JSON::make_array()}; + for (const auto &discriminator : this->internal_->discriminators) { + auto entry{JSON::make_object()}; + entry.assign_assume_new("pointer", JSON{to_string(discriminator.origin)}); + entry.assign_assume_new("destination", JSON{discriminator.destination}); + entry.assign_assume_new("scope", JSON{discriminator.scope}); + entry.assign_assume_new("dangling", + JSON{!openapi_discriminator_lands( + *(this->internal_->schemas), discriminator)}); + discriminators.push_back(std::move(entry)); + } + + result.assign_assume_new("discriminators", std::move(discriminators)); + } + return result; } diff --git a/vendor/core/src/core/openapi/helpers.h b/vendor/core/src/core/openapi/helpers.h index c3de0623..73ce1d9a 100644 --- a/vendor/core/src/core/openapi/helpers.h +++ b/vendor/core/src/core/openapi/helpers.h @@ -3,13 +3,15 @@ #include +#include #include #include // std::ranges::find #include // std::array #include // std::size_t -#include // std::uint8_t +#include // std::uint8_t, std::uint64_t #include // std::initializer_list +#include // std::numeric_limits #include // std::map #include // std::optional #include // std::set @@ -291,7 +293,7 @@ struct OpenAPIWalk { // the kind expected as well as by the place, because one place reached as // two kinds is a conflict, and reading it once as whichever reference was // walked first would let the order the description is written in decide what - // it is. Section 3.2 of OAS 3.2 names this hazard and says the behaviour + // it is. Appendix G of OAS 3.2 names this hazard and says the behaviour // "MAY be treated as an error if detected", so reading it as each kind in // turn surfaces the conflict rather than hiding it std::set> visited; @@ -359,6 +361,11 @@ struct OpenAPIWalk { /// What the entry document says about the API, kept so that reading it once /// serves both the walk and the caller OpenAPIInfo info; + /// What recording a location may still spend, and what the caller allowed in + /// the first place, which is what running out reports rather than whatever + /// was left of it by then + std::uint64_t remaining{std::numeric_limits::max()}; + std::uint64_t limit{std::numeric_limits::max()}; }; // Where a position that stands in for another leads, following as far as the @@ -389,8 +396,7 @@ inline auto openapi_reference_target(JSON::StringView reference, -> std::optional; inline auto openapi_follow_target(const URI &target, const Pointer &origin, - OpenAPIObjectKind expected, - bool demands_its_own_kind, OpenAPIWalk &walk) + OpenAPIObjectKind expected, OpenAPIWalk &walk) -> void; inline auto openapi_resolve_position(const OpenAPIWalk &walk, @@ -488,10 +494,77 @@ inline auto openapi_location_uri(const JSON::String &base, return result; } +// Resolve a URI reference against a base and canonicalise what it comes to, +// which is what makes two spellings of one place one place, both to the set +// that remembers where a walk has been and to a caller comparing a destination +// against a location. Without a base there is nothing to resolve against, +// though what comes back is canonicalised either way +inline auto openapi_resolve_uri(const JSON::StringView reference, + const JSON::String &base) + -> std::optional { + try { + URI target{JSON::String{reference}}; + if (!base.empty()) { + target.resolve_from(URI{base}); + } + + target.canonicalize(); + return target; + } catch (const URIParseError &) { + return std::nullopt; + } +} + +// The document a location key names, which is that key up to its fragment, +// which is why nothing repeats it on the entry a key leads to +inline auto openapi_document_uri(const JSON::String &uri) -> JSON::String { + return JSON::String{take_until(uri, '#')}; +} + +// OpenAPI Specification 3.1.1, Section 4.6 determines a document's base URI +// "in accordance with RFC3986 Section 5.1.2 - 5.1.4", a range that starts at +// 5.1.2 and so leaves out 5.1.1, "Base URI Embedded in Content". A 3.1 +// document therefore has no way of declaring its own base, and what remains is +// 5.1.3, "Base URI from the Retrieval URI", which only the caller can supply. +// Section 4.6 says as much: implementations "SHOULD allow users to provide +// documents with their intended retrieval URIs" +inline auto openapi_canonical_base(const std::string_view input) + -> JSON::String { + if (input.empty()) { + return {}; + } + + std::optional base; + try { + base.emplace(input); + } catch (const URIParseError &) { + base.reset(); + } + + // RFC 3986 Section 5.2.1: "only the scheme component is required to be + // present in a base URI". Anything without one cannot resolve a reference + // RFC 3986 Section 5.2.2 resolves a reference against a base's scheme, + // authority, path and query, and never against its fragment, so a fragment + // is no part of what a base is. Keeping one would also put two of them in + // every location this frame reports + if (base.has_value() && base.value().scheme().has_value()) { + base.value().canonicalize(); + const auto result{base.value().recompose_without_fragment()}; + if (result.has_value()) { + return result.value(); + } + } + + throw OpenAPIError{EMPTY_POINTER, + "The OpenAPI Description base must be a URI with a " + "scheme"}; +} + // Every Object gets one of these. The nearest recorded ancestor is the parent, // which holds because an Object is always recorded before anything inside it // -// Appendix G of OAS 3.2, and Section 3.2 of 3.1, say of one place read as two +// Appendix G of OAS 3.2, and Section 4.3.2 of 3.1, say of one place read as +// two // kinds of Object: // // the resulting behavior is implementation defined, and MAY be treated as @@ -509,6 +582,16 @@ inline auto openapi_record(OpenAPIWalk &walk, const Pointer &pointer, -> void { auto uri{openapi_location_uri(walk.base, pointer)}; const auto known{walk.locations.find(uri)}; + // Reading one place twice records it once, so what an allowance is spent on + // is the places a description holds rather than the times it is read + if (known == walk.locations.cend()) { + if (walk.remaining == 0) { + throw OpenAPIFrameLimitError{walk.limit}; + } + + walk.remaining -= 1; + } + if (known != walk.locations.cend() && known->second.type != kind) { throw OpenAPIError{walk.base, pointer, "This place is read as more than one kind of Object"}; diff --git a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h index fa517fda..9f4cbc51 100644 --- a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h +++ b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h @@ -8,6 +8,7 @@ #include #include +#include // std::uint64_t #include // std::exception #include // std::string #include // std::string_view @@ -79,6 +80,37 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIError : public std::exception { const char *message_; }; +/// @ingroup openapi +/// An error that represents framing that ran past what the caller allowed it +/// to register. For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// const sourcemeta::core::OpenAPIFrameLimitError error{100}; +/// assert(error.limit() == 100); +/// ``` +class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrameLimitError + : public std::exception { +public: + /// Create a framing limit error + OpenAPIFrameLimitError(const std::uint64_t limit) : limit_{limit} {} + + [[nodiscard]] auto what() const noexcept -> const char * override { + return "The OpenAPI Description exceeds the maximum number of locations " + "that framing may register"; + } + + /// The maximum number of locations that framing was allowed to register + [[nodiscard]] auto limit() const noexcept -> std::uint64_t { + return this->limit_; + } + +private: + std::uint64_t limit_; +}; + #if defined(_MSC_VER) #pragma warning(pop) #endif diff --git a/vendor/core/src/core/openapi/security.h b/vendor/core/src/core/openapi/security.h index 241b7776..3df4a23a 100644 --- a/vendor/core/src/core/openapi/security.h +++ b/vendor/core/src/core/openapi/security.h @@ -433,10 +433,9 @@ inline auto openapi_check_security_scheme_name(const JSON::StringView name, } // Naming a whole OpenAPI Description is naming something that is not a - // Security Scheme Object, which is the same demand a Path Item Object's - // `$ref` makes of what it points at + // Security Scheme Object openapi_follow_target(target.value(), origin, - OpenAPIObjectKind::SecurityScheme, true, walk); + OpenAPIObjectKind::SecurityScheme, walk); } inline auto openapi_check_security_requirement(const JSON &value, From 755fa2bf90cd1c2fadcd6fdf1a93f60299a53217 Mon Sep 17 00:00:00 2001 From: Juan Cruz Viotti Date: Wed, 23 Sep 2026 12:41:37 -0300 Subject: [PATCH 2/2] Fix Signed-off-by: Juan Cruz Viotti --- config.cmake.in | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/config.cmake.in b/config.cmake.in index 06272dec..0f05b050 100644 --- a/config.cmake.in +++ b/config.cmake.in @@ -10,7 +10,7 @@ endif() include(CMakeFindDependencyMacro) find_dependency(Core COMPONENTS json uri jsonpointer jsonschema numeric regex io) -find_dependency(Blaze COMPONENTS bundle alterschema canonicalizer) +find_dependency(Blaze COMPONENTS alterschema canonicalizer) foreach(component ${JSONBINPACK_COMPONENTS}) if(component STREQUAL "runtime")