From 095b1a1018aae12e6eab40c85d9cbb012810e738 Mon Sep 17 00:00:00 2001 From: Juan Cruz Viotti Date: Tue, 29 Sep 2026 11:45:09 -0300 Subject: [PATCH] Upgrade Sourcemeta dependencies Signed-off-by: Juan Cruz Viotti --- DEPENDENCIES | 4 +- vendor/blaze/DEPENDENCIES | 2 +- .../rules/prefix_promoted_draft_6_keywords.h | 4 +- .../rules/upgrade_draft_4_to_draft_6.h | 20 +- vendor/blaze/src/evaluator/CMakeLists.txt | 2 +- .../sourcemeta/blaze/evaluator_string_set.h | 141 -- .../sourcemeta/blaze/evaluator_value.h | 4 +- vendor/core/CMakeLists.txt | 1 + vendor/core/DEPENDENCIES | 4 +- vendor/core/cmake/FindPCRE2.cmake | 8 +- vendor/core/cmake/FindSLJIT.cmake | 44 + .../src/core/crypto/crypto_aes_cbc_hmac.cc | 33 +- .../core/src/core/crypto/crypto_ecdh_other.cc | 5 +- vendor/core/src/core/crypto/crypto_helpers.h | 39 +- .../core/src/core/crypto/crypto_hkdf_loop.h | 14 +- vendor/core/src/core/crypto/crypto_kdf.h | 5 +- .../src/core/crypto/crypto_rsa_oaep_other.cc | 16 +- .../core/src/core/crypto/crypto_sign_other.cc | 87 +- vendor/core/src/core/crypto/crypto_system.h | 10 +- .../src/core/crypto/crypto_verify_other.cc | 15 +- .../core/diff/include/sourcemeta/core/diff.h | 45 + vendor/core/src/core/diff/stringify.h | 94 +- vendor/core/src/core/json/CMakeLists.txt | 5 +- .../core/json/include/sourcemeta/core/json.h | 1 + .../include/sourcemeta/core/json_object.h | 20 +- .../sourcemeta/core/json_property_set.h | 166 ++ .../core/src/core/json/json_property_set.cc | 74 + vendor/core/src/core/jsonschema/bundle.cc | 89 +- vendor/core/src/core/oauth/CMakeLists.txt | 4 +- .../oauth/include/sourcemeta/core/oauth.h | 1 + .../include/sourcemeta/core/oauth_duration.h | 43 + vendor/core/src/core/oauth/oauth_duration.cc | 39 + vendor/core/src/core/oauth/oauth_json.h | 25 +- .../core/src/core/oidc/oidc_registration.cc | 12 +- vendor/core/src/core/openapi/CMakeLists.txt | 2 +- vendor/core/src/core/openapi/bundle.cc | 1962 +++++++++++++++++ vendor/core/src/core/openapi/components.h | 37 +- vendor/core/src/core/openapi/content.h | 35 +- vendor/core/src/core/openapi/discriminator.h | 39 +- vendor/core/src/core/openapi/document.h | 255 ++- vendor/core/src/core/openapi/example.h | 8 +- vendor/core/src/core/openapi/format.cc | 530 +++++ vendor/core/src/core/openapi/frame.cc | 586 +++-- vendor/core/src/core/openapi/helpers.h | 332 +-- .../openapi/include/sourcemeta/core/openapi.h | 756 ++++++- .../include/sourcemeta/core/openapi_error.h | 149 ++ vendor/core/src/core/openapi/parameter.h | 92 +- vendor/core/src/core/openapi/path_item.h | 10 +- vendor/core/src/core/openapi/paths.h | 10 +- vendor/core/src/core/openapi/response.h | 20 +- vendor/core/src/core/openapi/security.h | 23 +- vendor/core/src/core/openapi/server.h | 212 +- vendor/core/src/core/openapi/tag.h | 2 +- vendor/core/src/core/punycode/punycode.cc | 21 +- vendor/core/src/core/regex/CMakeLists.txt | 5 + vendor/core/src/core/uri/filesystem.cc | 15 +- .../core/yaml/include/sourcemeta/core/yaml.h | 95 +- .../include/sourcemeta/core/yaml_roundtrip.h | 27 + vendor/core/src/core/yaml/lexer.h | 100 +- vendor/core/src/core/yaml/parser.h | 293 ++- vendor/core/src/core/yaml/stringify.h | 611 +++-- vendor/core/src/core/yaml/yaml.cc | 113 +- .../core/vendor/openapi-test-suite-3-2.mask | 28 + .../vendor/openapi-test-suite-3-2/LICENSE | 201 ++ .../fail/encoding-enc-item-exclusion.yaml | 13 + .../fail/encoding-enc-prefix-exclusion.yaml | 13 + .../tests/schema/fail/example-examples.yaml | 17 + .../fail/example-object-old-exclusions.yaml | 10 + .../fail/example-object-old-vs-data.yaml | 10 + .../fail/example-object-old-vs-ser.yaml | 10 + .../fail/example-object-ser-exclusions.yaml | 10 + .../fail/header-object-allowReserved.yaml | 12 + .../tests/schema/fail/header-object-name.yaml | 12 + .../schema/fail/invalid_schema_types.yaml | 12 + .../fail/media-type-enc-item-exclusion.yaml | 11 + .../fail/media-type-enc-prefix-exclusion.yaml | 11 + .../tests/schema/fail/no_containers.yaml | 7 + .../tests/schema/fail/openapi-version.yaml | 5 + ...eration-object-query-with-querystring.yaml | 20 + .../operation-object-two-querystrings.yaml | 20 + ...rameter-object-content-not-with-style.yaml | 14 + ...parameter-object-cookie-allowReserved.yaml | 12 + ...parameter-object-header-allowReserved.yaml | 11 + .../fail/parameter-object-header-name.yaml | 10 + .../fail/parameter-object-path-name.yaml | 10 + ...er-object-querystring-not-with-schema.yaml | 11 + ...ject-conflicting-additional-operation.yaml | 64 + ...th-item-object-query-with-querystring.yaml | 19 + .../path-item-object-two-querystrings.yaml | 20 + .../tests/schema/fail/server_enum_empty.yaml | 15 + .../tests/schema/fail/servers.yaml | 11 + .../tests/schema/fail/unknown_container.yaml | 8 + .../tests/schema/fail/xml-attr-exclusion.yaml | 11 + .../schema/fail/xml-wrapped-exclusion.yaml | 11 + .../schema/pass/callback-object-examples.yaml | 32 + .../tests/schema/pass/comp_pathitems.yaml | 6 + .../pass/components-object-example.yaml | 71 + .../schema/pass/example-object-examples.yaml | 70 + .../schema/pass/header-object-examples.yaml | 25 + .../schema/pass/info-object-example.yaml | 20 + .../tests/schema/pass/info_summary.yaml | 6 + .../schema/pass/json_schema_dialect.yaml | 15 + .../tests/schema/pass/license_identifier.yaml | 9 + .../schema/pass/link-object-examples.yaml | 66 + .../schema/pass/media-type-examples.yaml | 178 ++ .../tests/schema/pass/mega.yaml | 62 + .../tests/schema/pass/minimal_comp.yaml | 5 + .../tests/schema/pass/minimal_hooks.yaml | 5 + .../tests/schema/pass/minimal_paths.yaml | 5 + .../tests/schema/pass/non-oauth-scopes.yaml | 19 + .../schema/pass/operation-object-example.yaml | 47 + ...eter-object-cookie-form-allowReserved.yaml | 18 + .../pass/parameter-object-examples.yaml | 79 + .../parameter-object-path-allowReserved.yaml | 12 + .../parameter-object-query-allowReserved.yaml | 11 + .../schema/pass/path-item-object-example.yaml | 78 + .../pass/path_item_servers_parameters.yaml | 114 + .../tests/schema/pass/path_no_response.yaml | 7 + .../schema/pass/path_var_empty_pathitem.yaml | 6 + .../schema/pass/paths-object-example.yaml | 20 + .../schema/pass/request-body-examples.yaml | 37 + .../schema/pass/response-object-examples.yaml | 45 + ...ema-object-deprecated-example-keyword.yaml | 17 + .../tests/schema/pass/schema.yaml | 55 + .../pass/security-scheme-object-examples.yaml | 69 + .../tests/schema/pass/servers.yaml | 26 + .../schema/pass/specification-extensions.yaml | 6 + .../tests/schema/pass/style-defaults.yaml | 105 + .../tests/schema/pass/tag-object-example.yaml | 25 + .../tests/schema/pass/valid_schema_types.yaml | 14 + .../tests/schema/pass/webhook-example.yaml | 35 + .../core/vendor/pcre2/src/pcre2_jit_compile.c | 56 +- .../vendor/{pcre2/deps => }/sljit/LICENSE | 0 .../allocator_src/sljitExecAllocatorApple.c | 0 .../allocator_src/sljitExecAllocatorCore.c | 0 .../allocator_src/sljitExecAllocatorFreeBSD.c | 0 .../allocator_src/sljitExecAllocatorPosix.c | 0 .../allocator_src/sljitExecAllocatorWindows.c | 0 .../sljitProtExecAllocatorNetBSD.c | 0 .../sljitProtExecAllocatorPosix.c | 0 .../allocator_src/sljitWXExecAllocatorPosix.c | 0 .../sljitWXExecAllocatorWindows.c | 0 .../deps => }/sljit/sljit_src/sljitConfig.h | 0 .../sljit/sljit_src/sljitConfigCPU.h | 0 .../sljit/sljit_src/sljitConfigInternal.h | 8 +- .../deps => }/sljit/sljit_src/sljitLir.c | 0 .../deps => }/sljit/sljit_src/sljitLir.h | 0 .../sljit/sljit_src/sljitNativeALPHA_64.c | 0 .../sljit/sljit_src/sljitNativeARM_32.c | 0 .../sljit/sljit_src/sljitNativeARM_64.c | 0 .../sljit/sljit_src/sljitNativeARM_T2_32.c | 0 .../sljit/sljit_src/sljitNativeLOONGARCH_64.c | 2 +- .../sljit/sljit_src/sljitNativeMIPS_32.c | 10 - .../sljit/sljit_src/sljitNativeMIPS_64.c | 0 .../sljit/sljit_src/sljitNativeMIPS_common.c | 92 +- .../sljit/sljit_src/sljitNativePPC_32.c | 0 .../sljit/sljit_src/sljitNativePPC_64.c | 0 .../sljit/sljit_src/sljitNativePPC_common.c | 0 .../sljit/sljit_src/sljitNativeRISCV_32.c | 0 .../sljit/sljit_src/sljitNativeRISCV_64.c | 0 .../sljit/sljit_src/sljitNativeRISCV_common.c | 0 .../sljit/sljit_src/sljitNativeS390X.c | 0 .../sljit/sljit_src/sljitNativeX86_32.c | 0 .../sljit/sljit_src/sljitNativeX86_64.c | 0 .../sljit/sljit_src/sljitNativeX86_common.c | 0 .../sljit/sljit_src/sljitSerialize.c | 0 .../deps => }/sljit/sljit_src/sljitUtils.c | 0 167 files changed, 8511 insertions(+), 1142 deletions(-) delete mode 100644 vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_string_set.h create mode 100644 vendor/core/cmake/FindSLJIT.cmake create mode 100644 vendor/core/src/core/json/include/sourcemeta/core/json_property_set.h create mode 100644 vendor/core/src/core/json/json_property_set.cc create mode 100644 vendor/core/src/core/oauth/include/sourcemeta/core/oauth_duration.h create mode 100644 vendor/core/src/core/oauth/oauth_duration.cc create mode 100644 vendor/core/src/core/openapi/bundle.cc create mode 100644 vendor/core/src/core/openapi/format.cc create mode 100644 vendor/core/vendor/openapi-test-suite-3-2.mask create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/LICENSE create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/encoding-enc-item-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/encoding-enc-prefix-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/example-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/example-object-old-exclusions.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/example-object-old-vs-data.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/example-object-old-vs-ser.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/example-object-ser-exclusions.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/header-object-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/header-object-name.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/invalid_schema_types.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/media-type-enc-item-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/media-type-enc-prefix-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/no_containers.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/openapi-version.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/operation-object-query-with-querystring.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/operation-object-two-querystrings.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-content-not-with-style.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-cookie-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-header-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-header-name.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-path-name.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/parameter-object-querystring-not-with-schema.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/path-item-object-conflicting-additional-operation.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/path-item-object-query-with-querystring.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/path-item-object-two-querystrings.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/server_enum_empty.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/servers.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/unknown_container.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/xml-attr-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/fail/xml-wrapped-exclusion.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/callback-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/comp_pathitems.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/components-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/example-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/header-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/info-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/info_summary.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/json_schema_dialect.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/license_identifier.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/link-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/media-type-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/mega.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/minimal_comp.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/minimal_hooks.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/minimal_paths.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/non-oauth-scopes.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/operation-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/parameter-object-cookie-form-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/parameter-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/parameter-object-path-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/parameter-object-query-allowReserved.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/path-item-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/path_item_servers_parameters.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/path_no_response.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/path_var_empty_pathitem.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/paths-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/request-body-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/response-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/schema-object-deprecated-example-keyword.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/schema.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/security-scheme-object-examples.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/servers.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/specification-extensions.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/style-defaults.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/tag-object-example.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/valid_schema_types.yaml create mode 100644 vendor/core/vendor/openapi-test-suite-3-2/tests/schema/pass/webhook-example.yaml rename vendor/core/vendor/{pcre2/deps => }/sljit/LICENSE (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitExecAllocatorApple.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitExecAllocatorCore.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitExecAllocatorFreeBSD.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitExecAllocatorPosix.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitExecAllocatorWindows.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitProtExecAllocatorNetBSD.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitProtExecAllocatorPosix.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitWXExecAllocatorPosix.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/allocator_src/sljitWXExecAllocatorWindows.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitConfig.h (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitConfigCPU.h (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitConfigInternal.h (99%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitLir.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitLir.h (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeALPHA_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeARM_32.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeARM_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeARM_T2_32.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeLOONGARCH_64.c (99%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeMIPS_32.c (97%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeMIPS_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeMIPS_common.c (98%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativePPC_32.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativePPC_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativePPC_common.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeRISCV_32.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeRISCV_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeRISCV_common.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeS390X.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeX86_32.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeX86_64.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitNativeX86_common.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitSerialize.c (100%) rename vendor/core/vendor/{pcre2/deps => }/sljit/sljit_src/sljitUtils.c (100%) diff --git a/DEPENDENCIES b/DEPENDENCIES index c742172fd..95dc77e49 100644 --- a/DEPENDENCIES +++ b/DEPENDENCIES @@ -1,4 +1,4 @@ vendorpull https://github.com/sourcemeta/vendorpull 1dcbac42809cf87cb5b045106b863e17ad84ba02 -core https://github.com/sourcemeta/core 4066fbfdb587d9d30b026e987a02fb9ecfc569bd -blaze https://github.com/sourcemeta/blaze e680badab795e9cdf30752f7dcfe77be708e05d7 +core https://github.com/sourcemeta/core 614dd74eb014f50b4b83d81db3a39de43ffe83d0 +blaze https://github.com/sourcemeta/blaze 87b62693c610f7a07efb63d1a8baf22e99571aa9 bootstrap https://github.com/twbs/bootstrap 1a6fdfae6be09b09eaced8f0e442ca6f7680a61e diff --git a/vendor/blaze/DEPENDENCIES b/vendor/blaze/DEPENDENCIES index 6d6e1dba5..5723a721b 100644 --- a/vendor/blaze/DEPENDENCIES +++ b/vendor/blaze/DEPENDENCIES @@ -1,3 +1,3 @@ vendorpull https://github.com/sourcemeta/vendorpull 1dcbac42809cf87cb5b045106b863e17ad84ba02 -core https://github.com/sourcemeta/core 4066fbfdb587d9d30b026e987a02fb9ecfc569bd +core https://github.com/sourcemeta/core 22e6e9521ed93ef268839bf03e41287f482839bb jsonschema-test-suite https://github.com/json-schema-org/JSON-Schema-Test-Suite 6648e8194c69697b2e1a15fe76a06a480b183a51 diff --git a/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h b/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h index 428c31178..7b5576f00 100644 --- a/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h +++ b/vendor/blaze/src/convert/rules/prefix_promoted_draft_6_keywords.h @@ -62,8 +62,8 @@ class PrefixPromotedDraft6Keywords final : public SchemaTransformRule { private: // NOLINTNEXTLINE(cert-err58-cpp,bugprone-throwing-static-initialization) - static inline const std::array KEYWORDS{ - {"const", "contains", "propertyNames", "examples"}}; + static inline const std::array KEYWORDS{ + {"$id", "const", "contains", "propertyNames", "examples"}}; mutable std::unordered_map renames_; }; diff --git a/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h b/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h index a9ea1133e..ae8edb9e0 100644 --- a/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h +++ b/vendor/blaze/src/convert/rules/upgrade_draft_4_to_draft_6.h @@ -79,8 +79,7 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { } } - if (schema.defines("id") && schema.at("id").is_string() && - !schema.defines("$id")) { + if (schema.defines("id") && schema.at("id").is_string()) { schema.rename("id", "$id"); } @@ -137,8 +136,7 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { return true; } - if (subschema.defines("id") && subschema.at("id").is_string() && - !subschema.defines("$id")) { + if (subschema.defines("id") && subschema.at("id").is_string()) { const auto fragment{extract_id_fragment(subschema.at("id"))}; if (!fragment.has_value() || fragment.value().empty() || is_strict_plain_name(fragment.value())) { @@ -162,7 +160,19 @@ class UpgradeDraft4ToDraft6 final : public SchemaTransformRule { } } - return false; + return has_stray_identifier(subschema); + } + + // Draft 4 does not know `$id`, so one written there is inert data that + // Draft 6 would read as the identifier, and it has to be shadowed before + // `id` takes that name. It is also the one Draft 6 addition this rule + // produces itself, so unlike every other promoted keyword its presence only + // means work is pending while the subschema is still on Draft 4 or older + static auto has_stray_identifier(const sourcemeta::core::JSON &subschema) + -> bool { + return subschema.defines("$id") && + dialect_position(declared_dialect(subschema)) <= + dialect_position(DRAFT_4_URL); } static auto is_strict_plain_name_first_char(const char character) -> bool { diff --git a/vendor/blaze/src/evaluator/CMakeLists.txt b/vendor/blaze/src/evaluator/CMakeLists.txt index 2b04a12f8..99f1dd142 100644 --- a/vendor/blaze/src/evaluator/CMakeLists.txt +++ b/vendor/blaze/src/evaluator/CMakeLists.txt @@ -1,6 +1,6 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT blaze NAME evaluator FOLDER "Blaze/Evaluator" - PRIVATE_HEADERS error.h value.h instruction.h string_set.h dispatch.h + PRIVATE_HEADERS error.h value.h instruction.h dispatch.h SOURCES evaluator_json.cc evaluator_describe.cc) if(BLAZE_INSTALL) diff --git a/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_string_set.h b/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_string_set.h deleted file mode 100644 index af77b6d68..000000000 --- a/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_string_set.h +++ /dev/null @@ -1,141 +0,0 @@ -#ifndef SOURCEMETA_BLAZE_EVALUATOR_STRING_SET_H -#define SOURCEMETA_BLAZE_EVALUATOR_STRING_SET_H - -#ifndef SOURCEMETA_BLAZE_EVALUATOR_EXPORT -#include -#endif - -#include - -#include // std::ranges::sort -#include // std::optional -#include // std::pair, std::move -#include // std::vector - -namespace sourcemeta::blaze { - -/// @ingroup evaluator -class SOURCEMETA_BLAZE_EVALUATOR_EXPORT StringSet { -public: - StringSet() = default; - - using string_type = sourcemeta::core::JSON::String; - using hash_type = sourcemeta::core::JSON::Object::hash_type; - using value_type = std::pair; - using underlying_type = std::vector; - using size_type = underlying_type::size_type; - using difference_type = underlying_type::difference_type; - using const_iterator = underlying_type::const_iterator; - - [[nodiscard]] auto contains(const string_type &value, - const hash_type hash) const -> bool { - if (this->hasher_.is_perfect(hash)) { - // A perfect hash captures the key bytes but not its length, so two keys - // that only differ in trailing length hash the same and the size is - // confirmed too - for (const auto &entry : this->data_) { - if (entry.second == hash && entry.first.size() == value.size()) { - return true; - } - } - } else { - for (const auto &entry : this->data_) { - if (entry.second == hash && entry.first == value) { - return true; - } - } - } - - return false; - } - [[nodiscard]] auto contains(const string_type &value) const -> bool { - return this->contains(value, this->hasher_(value)); - } - - [[nodiscard]] auto at(const size_type index) const noexcept - -> const value_type & { - return this->data_[index]; - } - - auto insert(const string_type &value) -> void { - const auto hash{this->hasher_(value)}; - if (!this->contains(value, hash)) { - this->data_.emplace_back(value, hash); - std::ranges::sort(this->data_, - [](const auto &left, const auto &right) -> bool { - return left.first < right.first; - }); - } - } - auto insert(string_type &&value) -> void { - const auto hash{this->hasher_(value)}; - if (!this->contains(value, hash)) { - this->data_.emplace_back(std::move(value), hash); - std::ranges::sort(this->data_, - [](const auto &left, const auto &right) -> bool { - return left.first < right.first; - }); - } - } - - [[nodiscard]] auto empty() const noexcept -> bool { - return this->data_.empty(); - } - [[nodiscard]] auto size() const noexcept -> size_type { - return this->data_.size(); - } - - [[nodiscard]] auto begin() const -> const_iterator { - return this->data_.begin(); - } - [[nodiscard]] auto end() const -> const_iterator { return this->data_.end(); } - [[nodiscard]] auto cbegin() const -> const_iterator { - return this->data_.cbegin(); - } - [[nodiscard]] auto cend() const -> const_iterator { - return this->data_.cend(); - } - - [[nodiscard]] auto to_json() const -> sourcemeta::core::JSON { - return sourcemeta::core::to_json(this->data_, [](const auto &item) -> auto { - return sourcemeta::core::to_json(item.first); - }); - } - - static auto from_json(const sourcemeta::core::JSON &value) - -> std::optional { - if (!value.is_array()) { - return std::nullopt; - } - - StringSet result; - for (const auto &item : value.as_array()) { - auto subvalue{ - sourcemeta::core::from_json(item)}; - if (!subvalue.has_value()) { - return std::nullopt; - } - - result.insert(std::move(subvalue).value()); - } - - return result; - } - -private: -// Exporting symbols that depends on the standard C++ library is considered -// safe. -// https://learn.microsoft.com/en-us/cpp/error-messages/compiler-warnings/compiler-warning-level-2-c4275?view=msvc-170&redirectedfrom=MSDN -#if defined(_MSC_VER) -#pragma warning(disable : 4251 4275) -#endif - underlying_type data_; -#if defined(_MSC_VER) -#pragma warning(default : 4251 4275) -#endif - sourcemeta::core::PropertyHashJSON hasher_; -}; - -} // namespace sourcemeta::blaze - -#endif diff --git a/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_value.h b/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_value.h index 38e59f7a5..54dbe2120 100644 --- a/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_value.h +++ b/vendor/blaze/src/evaluator/include/sourcemeta/blaze/evaluator_value.h @@ -5,8 +5,6 @@ #include #include -#include - #include // std::uint8_t #include // std::optional #include // std::string @@ -55,7 +53,7 @@ using ValueStrings = std::vector; /// @ingroup evaluator /// Represents a compiler step string set of values -using ValueStringSet = StringSet; +using ValueStringSet = sourcemeta::core::JSONPropertySet; /// @ingroup evaluator /// Represents a compiler step JSON types value as a bitmask diff --git a/vendor/core/CMakeLists.txt b/vendor/core/CMakeLists.txt index 3d0e39f6e..8ae4dd906 100644 --- a/vendor/core/CMakeLists.txt +++ b/vendor/core/CMakeLists.txt @@ -184,6 +184,7 @@ if(SOURCEMETA_CORE_CRYPTO) endif() if(SOURCEMETA_CORE_REGEX) + find_package(SLJIT REQUIRED) find_package(PCRE2 REQUIRED) add_subdirectory(src/core/regex) endif() diff --git a/vendor/core/DEPENDENCIES b/vendor/core/DEPENDENCIES index 9c86c5aa4..3fd54ece8 100644 --- a/vendor/core/DEPENDENCIES +++ b/vendor/core/DEPENDENCIES @@ -5,7 +5,8 @@ yaml-test-suite https://github.com/yaml/yaml-test-suite data-2022-01-17 uritemplate-test https://github.com/uri-templates/uritemplate-test 1eb27ab4462b9e5819dc47db99044f5fd1fa9bc7 pyca-cryptography https://github.com/pyca/cryptography 9747d06e83764e7f1ea4c04daf134cb8f861700b wycheproof https://github.com/C2SP/wycheproof 6d7cccd0fcb1917368579adeeac10fe802f1b521 -pcre2 https://github.com/PCRE2Project/pcre2 pcre2-10.48 +pcre2 https://github.com/PCRE2Project/pcre2 pcre2-10.49 +sljit https://github.com/zherczeg/sljit 3908d4c1d46764b7f86e172411e28cec3d0d601c unicodetools https://github.com/unicode-org/unicodetools final-17.0-20250910 jose-cookbook https://github.com/ietf-jose/cookbook 13692b68bfc18b99557a5b1ed311fd5077bfff04 w3c-json-ld https://github.com/w3c/json-ld-api 8654ac22b6cf4f441d2fee915ae634d36b5a8067 @@ -22,6 +23,7 @@ jsonschema-draft1 https://github.com/json-schema-org/json-schema-spec 2072feec9f jsonschema-draft0 https://github.com/json-schema-org/json-schema-spec 7ea575aef8d5c0183acbe6ff65b4c98ee9c236ec openapi https://github.com/OAI/OpenAPI-Specification 74906beddddab9e555337031b2a8d8e9338c4972 openapi-test-suite-3-1 https://github.com/OAI/OpenAPI-Specification 8df69dd6b8c12f50c99e28a864e500aadf3394e5 +openapi-test-suite-3-2 https://github.com/OAI/OpenAPI-Specification bb95c80a93c42749ed79b11a4bb4a5f4cf8eafc0 referencing-suite https://github.com/python-jsonschema/referencing-suite 61c4cc202b1e96ed5adcaf4842a595f68d659212 iana-oauth/parameters.csv https://www.iana.org/assignments/oauth-parameters/parameters.csv acfe19a091a15279587e761a399704e886447eec906b19a41528cb37dc2afc35 iana-oauth/extensions-error.csv https://www.iana.org/assignments/oauth-parameters/extensions-error.csv b49d3e1c9904667551170d12d0943ae9613751f42581f6438a4714c774d64838 diff --git a/vendor/core/cmake/FindPCRE2.cmake b/vendor/core/cmake/FindPCRE2.cmake index 1a834de0a..b3e01f667 100644 --- a/vendor/core/cmake/FindPCRE2.cmake +++ b/vendor/core/cmake/FindPCRE2.cmake @@ -93,6 +93,12 @@ if(NOT PCRE2_FOUND) add_library(pcre2 OBJECT ${PCRE2_SOURCES}) sourcemeta_add_default_options(PRIVATE pcre2) + # The code generator that the just-in-time compiler drives is a dependency + # of its own, which the caller is expected to have satisfied by now + # Public, as the translation unit that drives the code generator has to see + # the configuration it was built with to agree on its structure layouts + target_link_libraries(pcre2 PUBLIC SLJIT::sljit) + if(SOURCEMETA_COMPILER_LLVM OR SOURCEMETA_COMPILER_GCC) target_compile_options(pcre2 PRIVATE -Wno-implicit-int-conversion) target_compile_options(pcre2 PRIVATE -Wno-sign-conversion) @@ -126,8 +132,6 @@ if(NOT PCRE2_FOUND) target_compile_definitions(pcre2 PUBLIC PCRE2_CODE_UNIT_WIDTH=8) target_compile_definitions(pcre2 PRIVATE SUPPORT_PCRE2_8=1) target_compile_definitions(pcre2 PRIVATE SUPPORT_UNICODE=1) - # The just-in-time compiler brings its own code generator in as part of one - # of its translation units, so there is no second library to build for it target_compile_definitions(pcre2 PRIVATE SUPPORT_JIT=1) # Declarations of an import from a shared library of our own build of this # one, which no longer exists, would be unresolvable on Windows diff --git a/vendor/core/cmake/FindSLJIT.cmake b/vendor/core/cmake/FindSLJIT.cmake new file mode 100644 index 000000000..82bd5cde4 --- /dev/null +++ b/vendor/core/cmake/FindSLJIT.cmake @@ -0,0 +1,44 @@ +if(NOT SLJIT_FOUND) + set(SLJIT_DIR "${PROJECT_SOURCE_DIR}/vendor/sljit") + set(SLJIT_SOURCE_DIR "${SLJIT_DIR}/sljit_src") + + # This library ships one source that includes every backend and allocator + # that the target architecture and platform select + # Merged into the library that uses it, so that no archive, no header and + # no CMake package of our own build of it reaches an installed consumer + add_library(sljit OBJECT "${SLJIT_SOURCE_DIR}/sljitLir.c") + sourcemeta_add_default_options(PRIVATE sljit) + + # The compiler structure that this library hands out grows extra members + # under any of the tracing, argument checking, or debugging options, so + # every translation unit that reaches for the header has to be told the + # same configuration that the library itself was built with + target_compile_definitions(sljit PUBLIC SLJIT_CONFIG_AUTO=1) + target_compile_definitions(sljit PUBLIC SLJIT_VERBOSE=0) + target_compile_definitions(sljit PUBLIC SLJIT_DEBUG=0) + + if(SOURCEMETA_COMPILER_LLVM OR SOURCEMETA_COMPILER_GCC) + # Generated code accumulates into a trailing single-element array that is + # over-allocated and written well past its first element, so the strictest + # interpretation of what counts as a trailing flexible array would treat + # every byte this library emits as running off the end of the object + target_compile_options(sljit PRIVATE -fstrict-flex-arrays=0) + endif() + + if(SOURCEMETA_COMPILER_LLVM) + # The immediate byte of a vector lane instruction is only read back on the + # paths that set it, which the compiler cannot correlate + target_compile_options(sljit PRIVATE -Wno-conditional-uninitialized) + endif() + + if(SOURCEMETA_COMPILER_MSVC) + target_compile_options(sljit PRIVATE /wd4701) + endif() + + target_include_directories(sljit PUBLIC + "$") + + add_library(SLJIT::sljit ALIAS sljit) + + set(SLJIT_FOUND ON) +endif() diff --git a/vendor/core/src/core/crypto/crypto_aes_cbc_hmac.cc b/vendor/core/src/core/crypto/crypto_aes_cbc_hmac.cc index 2dfd0514d..4dd9b15d3 100644 --- a/vendor/core/src/core/crypto/crypto_aes_cbc_hmac.cc +++ b/vendor/core/src/core/crypto/crypto_aes_cbc_hmac.cc @@ -9,7 +9,7 @@ #include // std::optional, std::nullopt #include // std::string #include // std::string_view -#include // std::move, std::unreachable +#include // std::move namespace sourcemeta::core { @@ -58,25 +58,20 @@ auto authentication_tag(const std::size_t tag_length, message.append(iv); message.append(ciphertext); message.append(length_block); - switch (tag_length) { - case 16: { - const auto digest{hmac_sha256_digest(mac_key, message)}; - return std::string{reinterpret_cast(digest.data()), - tag_length}; - } - case 24: { - const auto digest{hmac_sha384_digest(mac_key, message)}; - return std::string{reinterpret_cast(digest.data()), - tag_length}; - } - case 32: { - const auto digest{hmac_sha512_digest(mac_key, message)}; - return std::string{reinterpret_cast(digest.data()), - tag_length}; - } - default: - std::unreachable(); + if (tag_length == 16) { + const auto digest{hmac_sha256_digest(mac_key, message)}; + return std::string{reinterpret_cast(digest.data()), + tag_length}; } + + if (tag_length == 24) { + const auto digest{hmac_sha384_digest(mac_key, message)}; + return std::string{reinterpret_cast(digest.data()), + tag_length}; + } + + const auto digest{hmac_sha512_digest(mac_key, message)}; + return std::string{reinterpret_cast(digest.data()), tag_length}; } } // namespace diff --git a/vendor/core/src/core/crypto/crypto_ecdh_other.cc b/vendor/core/src/core/crypto/crypto_ecdh_other.cc index f6f17c93c..8266d6428 100644 --- a/vendor/core/src/core/crypto/crypto_ecdh_other.cc +++ b/vendor/core/src/core/crypto/crypto_ecdh_other.cc @@ -6,7 +6,6 @@ #include // std::optional, std::nullopt #include // std::string -#include // std::unreachable // A from-scratch ECDH primitive for the reference backend, over the shared // constant-time elliptic curve arithmetic. The shared secret is the affine x @@ -23,10 +22,10 @@ auto to_curve_parameters(const EllipticCurve curve) -> EllipticCurveParameters { case EllipticCurve::P384: return curve_p384(); case EllipticCurve::P521: - return curve_p521(); + break; } - std::unreachable(); + return curve_p521(); } } // namespace diff --git a/vendor/core/src/core/crypto/crypto_helpers.h b/vendor/core/src/core/crypto/crypto_helpers.h index 62d7a69e5..8078085a6 100644 --- a/vendor/core/src/core/crypto/crypto_helpers.h +++ b/vendor/core/src/core/crypto/crypto_helpers.h @@ -12,7 +12,6 @@ #include // std::uint8_t #include // std::string #include // std::string_view -#include // std::unreachable namespace sourcemeta::core { @@ -107,10 +106,10 @@ inline auto curve_field_bytes(const EllipticCurve curve) noexcept case EllipticCurve::P384: return 48; case EllipticCurve::P521: - return 66; + break; } - std::unreachable(); + return 66; } // The group order of each NIST prime curve as big-endian octets (FIPS 186-4 @@ -126,13 +125,13 @@ inline auto curve_order_bytes(const EllipticCurve curve) -> std::string { "c7634d81f4372ddf581a0db248b0a77aecec196accc52973") .value(); case EllipticCurve::P521: - return hex_to_bytes("01fffffffffffffffffffffffffffffffffffffffffffffff" - "ffffffffffffffffffa51868783bf2f966b7fcc0148f709a5" - "d03bb5c9b8899c47aebb6fb71e91386409") - .value(); + break; } - std::unreachable(); + return hex_to_bytes("01fffffffffffffffffffffffffffffffffffffffffffffff" + "ffffffffffffffffffa51868783bf2f966b7fcc0148f709a5" + "d03bb5c9b8899c47aebb6fb71e91386409") + .value(); } // Whether an elliptic curve private scalar lies in the valid range [1, n) @@ -179,10 +178,10 @@ inline auto eddsa_public_key_bytes(const EdwardsCurve curve) noexcept case EdwardsCurve::Ed25519: return 32; case EdwardsCurve::Ed448: - return 57; + break; } - std::unreachable(); + return 57; } inline auto digest_message(const SignatureHashFunction hash, @@ -196,13 +195,12 @@ inline auto digest_message(const SignatureHashFunction hash, const auto digest{sha384_digest(message)}; return {reinterpret_cast(digest.data()), digest.size()}; } - case SignatureHashFunction::SHA512: { - const auto digest{sha512_digest(message)}; - return {reinterpret_cast(digest.data()), digest.size()}; - } + case SignatureHashFunction::SHA512: + break; } - std::unreachable(); + const auto digest{sha512_digest(message)}; + return {reinterpret_cast(digest.data()), digest.size()}; } // The same digest returned in wiping storage, for the secret-bearing hashing of @@ -222,14 +220,13 @@ inline auto secure_digest_message(const SignatureHashFunction hash, const SecureBufferScope digest_scope{digest.data(), digest.size()}; return {reinterpret_cast(digest.data()), digest.size()}; } - case SignatureHashFunction::SHA512: { - auto digest{sha512_digest(message)}; - const SecureBufferScope digest_scope{digest.data(), digest.size()}; - return {reinterpret_cast(digest.data()), digest.size()}; - } + case SignatureHashFunction::SHA512: + break; } - std::unreachable(); + auto digest{sha512_digest(message)}; + const SecureBufferScope digest_scope{digest.data(), digest.size()}; + return {reinterpret_cast(digest.data()), digest.size()}; } } // namespace sourcemeta::core diff --git a/vendor/core/src/core/crypto/crypto_hkdf_loop.h b/vendor/core/src/core/crypto/crypto_hkdf_loop.h index c41676e2b..eaa329e5d 100644 --- a/vendor/core/src/core/crypto/crypto_hkdf_loop.h +++ b/vendor/core/src/core/crypto/crypto_hkdf_loop.h @@ -14,7 +14,6 @@ #include // std::uint8_t #include // std::string #include // std::string_view -#include // std::unreachable namespace sourcemeta::core { @@ -56,15 +55,14 @@ hkdf_loop_hmac(const KDFHash hash, const std::string_view key, secure_zero(digest.data(), digest.size()); return digest.size(); } - case KDFHash::SHA512: { - auto digest{hmac_sha512_digest(key, message)}; - std::copy_n(digest.begin(), digest.size(), output.begin()); - secure_zero(digest.data(), digest.size()); - return digest.size(); - } + case KDFHash::SHA512: + break; } - std::unreachable(); + auto digest{hmac_sha512_digest(key, message)}; + std::copy_n(digest.begin(), digest.size(), output.begin()); + secure_zero(digest.data(), digest.size()); + return digest.size(); } // RFC 5869 Section 2.2: PRK = HMAC-Hash(salt, IKM), noting that the salt keys diff --git a/vendor/core/src/core/crypto/crypto_kdf.h b/vendor/core/src/core/crypto/crypto_kdf.h index 54ea74740..4e00e88e3 100644 --- a/vendor/core/src/core/crypto/crypto_kdf.h +++ b/vendor/core/src/core/crypto/crypto_kdf.h @@ -4,7 +4,6 @@ #include // std::size_t #include // std::uint8_t #include // std::string_view -#include // std::unreachable namespace sourcemeta::core { @@ -19,10 +18,10 @@ inline auto kdf_digest_bytes(const KDFHash hash) noexcept -> std::size_t { case KDFHash::SHA384: return 48; case KDFHash::SHA512: - return 64; + break; } - std::unreachable(); + return 64; } // The widest digest any of the above produces, so a caller can hold one on the diff --git a/vendor/core/src/core/crypto/crypto_rsa_oaep_other.cc b/vendor/core/src/core/crypto/crypto_rsa_oaep_other.cc index d9fc8adc6..50a920c92 100644 --- a/vendor/core/src/core/crypto/crypto_rsa_oaep_other.cc +++ b/vendor/core/src/core/crypto/crypto_rsa_oaep_other.cc @@ -15,7 +15,6 @@ #include // std::span #include // std::string #include // std::string_view -#include // std::unreachable // A from-scratch RSA-OAEP (RFC 8017 Section 7.1) for the reference backend, // over the shared big integer arithmetic. The decode is not constant-time, @@ -32,10 +31,10 @@ auto hash_length(const RSAOAEPHash hash) -> std::size_t { case RSAOAEPHash::SHA1: return 20; case RSAOAEPHash::SHA256: - return 32; + break; } - std::unreachable(); + return 32; } auto oaep_hash(const RSAOAEPHash hash, const std::string_view input) @@ -46,14 +45,13 @@ auto oaep_hash(const RSAOAEPHash hash, const std::string_view input) return std::string{reinterpret_cast(digest.data()), digest.size()}; } - case RSAOAEPHash::SHA256: { - const auto digest{sha256_digest(input)}; - return std::string{reinterpret_cast(digest.data()), - digest.size()}; - } + case RSAOAEPHash::SHA256: + break; } - std::unreachable(); + const auto digest{sha256_digest(input)}; + return std::string{reinterpret_cast(digest.data()), + digest.size()}; } // The mask generation function (RFC 8017 Appendix B.2.1) diff --git a/vendor/core/src/core/crypto/crypto_sign_other.cc b/vendor/core/src/core/crypto/crypto_sign_other.cc index 049764dbe..4e4dd2a9c 100644 --- a/vendor/core/src/core/crypto/crypto_sign_other.cc +++ b/vendor/core/src/core/crypto/crypto_sign_other.cc @@ -20,7 +20,7 @@ #include // std::span #include // std::string #include // std::string_view -#include // std::move, std::unreachable +#include // std::move, std::pair namespace sourcemeta::core { namespace { @@ -46,11 +46,11 @@ auto digest_info_prefix(const SignatureHashFunction hash) -> std::string_view { return {reinterpret_cast(DIGEST_INFO_SHA384.data()), DIGEST_INFO_SHA384.size()}; case SignatureHashFunction::SHA512: - return {reinterpret_cast(DIGEST_INFO_SHA512.data()), - DIGEST_INFO_SHA512.size()}; + break; } - std::unreachable(); + return {reinterpret_cast(DIGEST_INFO_SHA512.data()), + DIGEST_INFO_SHA512.size()}; } // EMSA-PKCS1-v1_5 encoding (RFC 8017 Section 9.2) @@ -143,10 +143,10 @@ auto to_curve_parameters(const EllipticCurve curve) -> EllipticCurveParameters { case EllipticCurve::P384: return curve_p384(); case EllipticCurve::P521: - return curve_p521(); + break; } - std::unreachable(); + return curve_p521(); } struct HashSizes { @@ -161,10 +161,10 @@ auto hash_sizes(const SignatureHashFunction hash) -> HashSizes { case SignatureHashFunction::SHA384: return {.block_bytes = 128, .output_bytes = 48}; case SignatureHashFunction::SHA512: - return {.block_bytes = 128, .output_bytes = 64}; + break; } - std::unreachable(); + return {.block_bytes = 128, .output_bytes = 64}; } // HMAC (RFC 2104) keyed on the signature hash function, the primitive that the @@ -606,34 +606,32 @@ auto make_private_key(const std::string_view pem) -> std::optional { .coordinate_x = std::move(point.first), .coordinate_y = std::move(point.second)}}; } - case PKCS8KeyKind::Edwards: { - const auto seed{der_read(parsed->key)}; - if (!seed.has_value() || seed->tag != 0x04 || - seed->content.size() != - eddsa_public_key_bytes(parsed->edwards_curve)) { - return std::nullopt; - } + case PKCS8KeyKind::Edwards: + break; + } - return PrivateKey{ - new PrivateKey::Internal{.kind = PrivateKey::Type::Edwards, - .modulus = {}, - .public_exponent = {}, - .private_exponent = {}, - .prime1 = {}, - .prime2 = {}, - .exponent1 = {}, - .exponent2 = {}, - .coefficient = {}, - .scalar = {}, - .elliptic_curve = {}, - .edwards_seed = std::string{seed->content}, - .edwards_curve = parsed->edwards_curve, - .coordinate_x = {}, - .coordinate_y = {}}}; - } + const auto seed{der_read(parsed->key)}; + if (!seed.has_value() || seed->tag != 0x04 || + seed->content.size() != eddsa_public_key_bytes(parsed->edwards_curve)) { + return std::nullopt; } - std::unreachable(); + return PrivateKey{ + new PrivateKey::Internal{.kind = PrivateKey::Type::Edwards, + .modulus = {}, + .public_exponent = {}, + .private_exponent = {}, + .prime1 = {}, + .prime2 = {}, + .exponent1 = {}, + .exponent2 = {}, + .coefficient = {}, + .scalar = {}, + .elliptic_curve = {}, + .edwards_seed = std::string{seed->content}, + .edwards_curve = parsed->edwards_curve, + .coordinate_x = {}, + .coordinate_y = {}}}; } auto make_ec_private_key(const EllipticCurve curve, @@ -818,10 +816,10 @@ auto eddsa_sign(const PrivateKey &key, const std::string_view message) case EdwardsCurve::Ed25519: return edwards25519_sign(internal->edwards_seed, message); case EdwardsCurve::Ed448: - return edwards448_sign(internal->edwards_seed, message); + break; } - std::unreachable(); + return edwards448_sign(internal->edwards_seed, message); } auto derive_public_key(const PrivateKey &key) -> std::optional { @@ -843,19 +841,18 @@ auto derive_public_key(const PrivateKey &key) -> std::optional { return make_ec_public_key(internal->elliptic_curve, internal->coordinate_x, internal->coordinate_y); - case PrivateKey::Type::Edwards: { - const auto point{internal->edwards_curve == EdwardsCurve::Ed25519 - ? edwards25519_public_key(internal->edwards_seed) - : edwards448_public_key(internal->edwards_seed)}; - if (!point.has_value()) { - return std::nullopt; - } + case PrivateKey::Type::Edwards: + break; + } - return make_eddsa_public_key(internal->edwards_curve, point.value()); - } + const auto point{internal->edwards_curve == EdwardsCurve::Ed25519 + ? edwards25519_public_key(internal->edwards_seed) + : edwards448_public_key(internal->edwards_seed)}; + if (!point.has_value()) { + return std::nullopt; } - std::unreachable(); + return make_eddsa_public_key(internal->edwards_curve, point.value()); } } // namespace sourcemeta::core diff --git a/vendor/core/src/core/crypto/crypto_system.h b/vendor/core/src/core/crypto/crypto_system.h index e36e4971c..f174f9f18 100644 --- a/vendor/core/src/core/crypto/crypto_system.h +++ b/vendor/core/src/core/crypto/crypto_system.h @@ -14,7 +14,7 @@ #include // std::optional, std::nullopt #include // std::string #include // std::string_view -#include // std::pair, std::unreachable +#include // std::pair namespace sourcemeta::core { @@ -46,10 +46,10 @@ inline auto curve_bit_length(const EllipticCurve curve) noexcept case EllipticCurve::P384: return 384; case EllipticCurve::P521: - return 521; + break; } - std::unreachable(); + return 521; } // Identify a curve from its field width, the inverse of the mapping from a @@ -76,10 +76,10 @@ inline auto eddsa_signature_bytes(const EdwardsCurve curve) noexcept case EdwardsCurve::Ed25519: return 64; case EdwardsCurve::Ed448: - return 114; + break; } - std::unreachable(); + return 114; } // Read the modulus and public exponent from a PKCS#1 RSAPublicKey structure diff --git a/vendor/core/src/core/crypto/crypto_verify_other.cc b/vendor/core/src/core/crypto/crypto_verify_other.cc index 618fd1f29..f58990a46 100644 --- a/vendor/core/src/core/crypto/crypto_verify_other.cc +++ b/vendor/core/src/core/crypto/crypto_verify_other.cc @@ -14,7 +14,6 @@ #include // std::optional, std::nullopt #include // std::string #include // std::string_view -#include // std::unreachable namespace sourcemeta::core { @@ -46,11 +45,11 @@ auto digest_info_prefix(const SignatureHashFunction hash) -> std::string_view { return {reinterpret_cast(DIGEST_INFO_SHA384.data()), DIGEST_INFO_SHA384.size()}; case SignatureHashFunction::SHA512: - return {reinterpret_cast(DIGEST_INFO_SHA512.data()), - DIGEST_INFO_SHA512.size()}; + break; } - std::unreachable(); + return {reinterpret_cast(DIGEST_INFO_SHA512.data()), + DIGEST_INFO_SHA512.size()}; } // EMSA-PKCS1-v1_5 encoding (RFC 8017 Section 9.2) @@ -173,10 +172,10 @@ auto to_curve_parameters(const EllipticCurve curve) -> EllipticCurveParameters { case EllipticCurve::P384: return curve_p384(); case EllipticCurve::P521: - return curve_p521(); + break; } - std::unreachable(); + return curve_p521(); } // FIPS 186-4 Section 6.4 step 2, deriving the integer e from the leftmost bits @@ -469,10 +468,10 @@ auto eddsa_verify(const PublicKey &key, const std::string_view message, case EdwardsCurve::Ed25519: return edwards25519_verify(internal->coordinate_x, message, signature); case EdwardsCurve::Ed448: - return edwards448_verify(internal->coordinate_x, message, signature); + break; } - std::unreachable(); + return edwards448_verify(internal->coordinate_x, message, signature); } auto rsa_public_components(const PublicKey &key) diff --git a/vendor/core/src/core/diff/include/sourcemeta/core/diff.h b/vendor/core/src/core/diff/include/sourcemeta/core/diff.h index e88b2abb6..187dd8556 100644 --- a/vendor/core/src/core/diff/include/sourcemeta/core/diff.h +++ b/vendor/core/src/core/diff/include/sourcemeta/core/diff.h @@ -7,6 +7,7 @@ #include // std::size_t #include // std::uint8_t +#include // std::function #include // std::ostream #include // std::string_view #include // std::vector @@ -87,12 +88,44 @@ struct Diff { /// Controls the presentation of a rendered set of differences struct FormatOptions { + /// The semantic role of a line in a unified diff + enum class LineType : std::uint8_t { + /// The header indicating the original document label + HeaderOriginal, + /// The header indicating the modified document label + HeaderModified, + /// A hunk range header + Hunk, + /// An unchanged context line + Context, + /// A deleted line from the original document + Delete, + /// An inserted line into the modified document + Insert, + /// The marker indicating a missing final newline + NoNewline + }; + + /// Custom writer for rendering individual logical unified-diff lines. + /// + /// The callback receives the destination stream, the line's semantic role, + /// its textual prefix (such as `"-"` or `"@@ "`), and its content, both + /// excluding the terminating newline. The callback is responsible for + /// writing only the line body to the stream; Core unconditionally appends + /// the terminating newline. Views passed to the callback are only + /// guaranteed to remain valid for the duration of the callback. + using LineWriter = + std::function; + /// The name given to the original input in the header std::string_view original_label{"a"}; /// The name given to the modified input in the header std::string_view modified_label{"b"}; /// The number of unchanged lines shown around each change std::size_t context{3}; + /// Optional custom line writer for logical unified-diff lines + LineWriter line_writer{}; }; /// The tokens of the original input @@ -158,6 +191,18 @@ auto diff(const std::string_view original, const std::string_view modified, /// " foo\n" /// "-bar\n" /// "+baz\n"); +/// +/// std::ostringstream custom_stream; +/// sourcemeta::core::stringify( +/// result, custom_stream, sourcemeta::core::Diff::Format::Unified, +/// {.line_writer = +/// [](std::ostream &output, +/// const sourcemeta::core::Diff::FormatOptions::LineType type, +/// const std::string_view prefix, +/// const std::string_view content) { +/// output.write(prefix.data(), prefix.size()); +/// output.write(content.data(), content.size()); +/// }}); /// ``` SOURCEMETA_CORE_DIFF_EXPORT auto stringify(const Diff &document, std::ostream &stream, diff --git a/vendor/core/src/core/diff/stringify.h b/vendor/core/src/core/diff/stringify.h index 34168ffe3..832ca8e20 100644 --- a/vendor/core/src/core/diff/stringify.h +++ b/vendor/core/src/core/diff/stringify.h @@ -27,15 +27,41 @@ inline auto write_diff_range(std::ostream &stream, const std::size_t start, } } -inline auto write_diff_line(std::ostream &stream, const char prefix, +inline auto write_diff_output_line(std::ostream &stream, + const Diff::FormatOptions &options, + const Diff::FormatOptions::LineType type, + const std::string_view prefix, + const std::string_view content) -> void { + if (options.line_writer) { + options.line_writer(stream, type, prefix, content); + } else { + write_diff_text(stream, prefix); + write_diff_text(stream, content); + } + stream.put('\n'); +} + +inline auto append_diff_range(std::string &result, const std::size_t start, + const std::size_t count) -> void { + digits_append(result, count == 0 ? start : start + 1); + if (count != 1) { + result.push_back(','); + digits_append(result, count); + } +} + +inline auto write_diff_line(std::ostream &stream, + const Diff::FormatOptions &options, + const Diff::FormatOptions::LineType type, + const std::string_view prefix, const std::vector &lines, const std::size_t index, const bool ends_with_newline) -> void { - stream.put(prefix); - write_diff_text(stream, lines[index]); - stream.put('\n'); + write_diff_output_line(stream, options, type, prefix, lines[index]); if (!ends_with_newline && index + 1 == lines.size()) { - write_diff_text(stream, "\\ No newline at end of file\n"); + write_diff_output_line(stream, options, + Diff::FormatOptions::LineType::NoNewline, "\\ ", + "No newline at end of file"); } } @@ -98,24 +124,40 @@ inline auto stringify_diff_unified(const Diff &document, std::ostream &stream, const auto modified_end{operations[change_end - 1].modified_end + trailing}; if (!wrote_header) { - write_diff_text(stream, "--- "); - write_diff_text(stream, options.original_label); - stream.put('\n'); - write_diff_text(stream, "+++ "); - write_diff_text(stream, options.modified_label); - stream.put('\n'); + write_diff_output_line(stream, options, + Diff::FormatOptions::LineType::HeaderOriginal, + "--- ", options.original_label); + write_diff_output_line(stream, options, + Diff::FormatOptions::LineType::HeaderModified, + "+++ ", options.modified_label); wrote_header = true; } - write_diff_text(stream, "@@ -"); - write_diff_range(stream, original_start, original_end - original_start); - write_diff_text(stream, " +"); - write_diff_range(stream, modified_start, modified_end - modified_start); - write_diff_text(stream, " @@\n"); + if (!options.line_writer) { + write_diff_text(stream, "@@ -"); + write_diff_range(stream, original_start, original_end - original_start); + write_diff_text(stream, " +"); + write_diff_range(stream, modified_start, modified_end - modified_start); + write_diff_text(stream, " @@\n"); + } else { + std::string hunk_content; + hunk_content.reserve(48); + hunk_content.push_back('-'); + append_diff_range(hunk_content, original_start, + original_end - original_start); + hunk_content.append(" +"); + append_diff_range(hunk_content, modified_start, + modified_end - modified_start); + hunk_content.append(" @@"); + write_diff_output_line(stream, options, + Diff::FormatOptions::LineType::Hunk, "@@ ", + hunk_content); + } for (auto line{original_start}; line < operations[change_begin].original_start; ++line) { - write_diff_line(stream, ' ', document.original, line, + write_diff_line(stream, options, Diff::FormatOptions::LineType::Context, + " ", document.original, line, document.original_ends_with_newline); } @@ -125,24 +167,27 @@ inline auto stringify_diff_unified(const Diff &document, std::ostream &stream, case Diff::Operation::Type::Equal: for (auto line{operation.original_start}; line < operation.original_end; ++line) { - write_diff_line(stream, ' ', document.original, line, - document.original_ends_with_newline); + write_diff_line( + stream, options, Diff::FormatOptions::LineType::Context, " ", + document.original, line, document.original_ends_with_newline); } break; case Diff::Operation::Type::Delete: for (auto line{operation.original_start}; line < operation.original_end; ++line) { - write_diff_line(stream, '-', document.original, line, - document.original_ends_with_newline); + write_diff_line( + stream, options, Diff::FormatOptions::LineType::Delete, "-", + document.original, line, document.original_ends_with_newline); } break; case Diff::Operation::Type::Insert: for (auto line{operation.modified_start}; line < operation.modified_end; ++line) { - write_diff_line(stream, '+', document.modified, line, - document.modified_ends_with_newline); + write_diff_line( + stream, options, Diff::FormatOptions::LineType::Insert, "+", + document.modified, line, document.modified_ends_with_newline); } break; @@ -151,7 +196,8 @@ inline auto stringify_diff_unified(const Diff &document, std::ostream &stream, for (auto line{operations[change_end - 1].original_end}; line < original_end; ++line) { - write_diff_line(stream, ' ', document.original, line, + write_diff_line(stream, options, Diff::FormatOptions::LineType::Context, + " ", document.original, line, document.original_ends_with_newline); } diff --git a/vendor/core/src/core/json/CMakeLists.txt b/vendor/core/src/core/json/CMakeLists.txt index 898e9728a..b23094367 100644 --- a/vendor/core/src/core/json/CMakeLists.txt +++ b/vendor/core/src/core/json/CMakeLists.txt @@ -1,6 +1,7 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME json - PRIVATE_HEADERS array.h error.h object.h value.h hash.h auto.h - SOURCES grammar.h parser.h stringify.h json.cc json_value.cc) + PRIVATE_HEADERS array.h error.h object.h value.h hash.h auto.h property_set.h + SOURCES grammar.h parser.h stringify.h json.cc json_value.cc + json_property_set.cc) if(SOURCEMETA_CORE_INSTALL) sourcemeta_library_install(NAMESPACE sourcemeta PROJECT core NAME json) diff --git a/vendor/core/src/core/json/include/sourcemeta/core/json.h b/vendor/core/src/core/json/include/sourcemeta/core/json.h index 726587f5f..c1f06321c 100644 --- a/vendor/core/src/core/json/include/sourcemeta/core/json.h +++ b/vendor/core/src/core/json/include/sourcemeta/core/json.h @@ -8,6 +8,7 @@ // NOLINTBEGIN(misc-include-cleaner) #include #include +#include #include // NOLINTEND(misc-include-cleaner) diff --git a/vendor/core/src/core/json/include/sourcemeta/core/json_object.h b/vendor/core/src/core/json/include/sourcemeta/core/json_object.h index 56491bf00..955b5bf80 100644 --- a/vendor/core/src/core/json/include/sourcemeta/core/json_object.h +++ b/vendor/core/src/core/json/include/sourcemeta/core/json_object.h @@ -578,6 +578,12 @@ template class JSONObject { const auto key_hash{this->hash(key)}; const auto suffix_hash{this->hash(suffix)}; + // The suffix can precede the key, so inserting on the first sight of it + // would add a second entry under a key the object already holds. The whole + // object is scanned for the key instead, remembering where the suffix sat, + // which keeps this to the single pass the caller pays for either way + auto insertion_point{this->data_.end()}; + if (this->HASHER.is_perfect(key_hash)) { for (auto iterator = this->data_.begin(); iterator != this->data_.end(); ++iterator) { @@ -586,9 +592,9 @@ template class JSONObject { iterator->second = value; return key_hash; } - if (iterator->hash == suffix_hash && iterator->first == suffix) { - this->data_.insert(iterator, {key, value, key_hash}); - return key_hash; + if (insertion_point == this->data_.end() && + iterator->hash == suffix_hash && iterator->first == suffix) { + insertion_point = iterator; } } } else { @@ -598,14 +604,14 @@ template class JSONObject { iterator->second = value; return key_hash; } - if (iterator->hash == suffix_hash && iterator->first == suffix) { - this->data_.insert(iterator, {key, value, key_hash}); - return key_hash; + if (insertion_point == this->data_.end() && + iterator->hash == suffix_hash && iterator->first == suffix) { + insertion_point = iterator; } } } - this->data_.push_back({key, value, key_hash}); + this->data_.insert(insertion_point, {key, value, key_hash}); return key_hash; } diff --git a/vendor/core/src/core/json/include/sourcemeta/core/json_property_set.h b/vendor/core/src/core/json/include/sourcemeta/core/json_property_set.h new file mode 100644 index 000000000..e90d363de --- /dev/null +++ b/vendor/core/src/core/json/include/sourcemeta/core/json_property_set.h @@ -0,0 +1,166 @@ +#ifndef SOURCEMETA_CORE_JSON_PROPERTY_SET_H_ +#define SOURCEMETA_CORE_JSON_PROPERTY_SET_H_ + +#ifndef SOURCEMETA_CORE_JSON_EXPORT +#include +#endif + +#include +#include + +#include // assert +#include // std::optional +#include // std::pair +#include // std::vector + +namespace sourcemeta::core { + +// Exporting symbols that depends on the standard C++ library is considered +// safe. +// https://learn.microsoft.com/en-us/cpp/error-messages/compiler-warnings/compiler-warning-level-2-c4275?view=msvc-170&redirectedfrom=MSDN +#if defined(_MSC_VER) +#pragma warning(push) +#pragma warning(disable : 4251 4275) +#endif + +/// @ingroup json +/// A set of JSON object property names, sorted in lexicographic order, that +/// keeps the hash of every name alongside it so that repeated lookups on JSON +/// objects do not have to hash the same names over and over again. For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// sourcemeta::core::JSONPropertySet properties; +/// properties.insert("foo"); +/// properties.insert("bar"); +/// +/// const sourcemeta::core::JSON document = +/// sourcemeta::core::parse_json("{ \"foo\": 1, \"bar\": 2 }"); +/// for (const auto &property : properties) { +/// assert(document.defines(property.first, property.second)); +/// } +/// ``` +class SOURCEMETA_CORE_JSON_EXPORT JSONPropertySet { +public: + JSONPropertySet() = default; + + /// The type of the property names held by the set + using string_type = JSON::String; + /// The type of the hash kept alongside every property name + using hash_type = JSON::Object::hash_type; + /// A property name together with its hash + using value_type = std::pair; + /// The underlying container that holds the property names + using underlying_type = std::vector; + using size_type = underlying_type::size_type; + using difference_type = underlying_type::difference_type; + using const_iterator = underlying_type::const_iterator; + + /// Check whether the set contains a property name whose hash is already + /// known, avoiding the cost of hashing it again + [[nodiscard]] auto contains(const string_type &value, + const hash_type hash) const -> bool { + assert(HASHER(value) == hash); + if (HASHER.is_perfect(hash)) { + // A perfect hash captures the property name bytes but not its length, so + // two names that differ only by trailing NUL bytes hash equal. Comparing + // sizes disambiguates them without the cost of a full string comparison + for (const auto &entry : this->data_) { + if (entry.second == hash && entry.first.size() == value.size()) { + return true; + } + } + } else { + for (const auto &entry : this->data_) { + if (entry.second == hash && entry.first == value) { + return true; + } + } + } + + return false; + } + + /// Check whether the set contains a property name + [[nodiscard]] auto contains(const string_type &value) const -> bool { + return this->contains(value, HASHER(value)); + } + + /// Add a property name whose hash is already known to the set, keeping the + /// set sorted. Returns whether the name was added, so that a caller can tell + /// a name it had not seen before from one the set already held + auto insert(const string_type &value, const hash_type hash) -> bool; + + /// Add a property name whose hash is already known to the set, keeping the + /// set sorted. Returns whether the name was added, so that a caller can tell + /// a name it had not seen before from one the set already held + auto insert(string_type &&value, const hash_type hash) -> bool; + + /// Add a property name to the set, keeping the set sorted. Returns whether + /// the name was added, so that a caller can tell a name it had not seen + /// before from one the set already held + auto insert(const string_type &value) -> bool; + + /// Add a property name to the set, keeping the set sorted. Returns whether + /// the name was added, so that a caller can tell a name it had not seen + /// before from one the set already held + auto insert(string_type &&value) -> bool; + + /// Get a property name and its hash by index + [[nodiscard]] auto at(const size_type index) const noexcept + -> const value_type & { + assert(index < this->data_.size()); + return this->data_[index]; + } + + /// Check whether the set is empty + [[nodiscard]] auto empty() const noexcept -> bool { + return this->data_.empty(); + } + + /// Get the number of property names in the set + [[nodiscard]] auto size() const noexcept -> size_type { + return this->data_.size(); + } + + /// Get a constant begin iterator on the set + [[nodiscard]] auto begin() const noexcept -> const_iterator { + return this->data_.begin(); + } + + /// Get a constant end iterator on the set + [[nodiscard]] auto end() const noexcept -> const_iterator { + return this->data_.end(); + } + + /// Get a constant begin iterator on the set + [[nodiscard]] auto cbegin() const noexcept -> const_iterator { + return this->data_.cbegin(); + } + + /// Get a constant end iterator on the set + [[nodiscard]] auto cend() const noexcept -> const_iterator { + return this->data_.cend(); + } + + /// Serialise the set as a JSON array of property names + [[nodiscard]] auto to_json() const -> JSON; + + /// Reconstruct a set from a JSON array of property names, yielding no result + /// if the given document is not an array of strings + static auto from_json(const JSON &value) -> std::optional; + +private: + static constexpr PropertyHashJSON HASHER{}; + underlying_type data_; +}; + +#if defined(_MSC_VER) +#pragma warning(pop) +#endif + +} // namespace sourcemeta::core + +#endif diff --git a/vendor/core/src/core/json/json_property_set.cc b/vendor/core/src/core/json/json_property_set.cc new file mode 100644 index 000000000..d1b654a0e --- /dev/null +++ b/vendor/core/src/core/json/json_property_set.cc @@ -0,0 +1,74 @@ +#include +#include + +#include // std::ranges::lower_bound +#include // assert +#include // std::optional, std::nullopt +#include // std::move + +namespace sourcemeta::core { + +auto JSONPropertySet::insert(const string_type &value, const hash_type hash) + -> bool { + assert(HASHER(value) == hash); + const auto &entries{this->data_}; + const auto position{ + std::ranges::lower_bound(entries, value, {}, &value_type::first)}; + if (position != entries.cend() && position->first == value) { + return false; + } + + this->data_.emplace(position, value, hash); + return true; +} + +auto JSONPropertySet::insert(string_type &&value, const hash_type hash) + -> bool { + assert(HASHER(value) == hash); + const auto &entries{this->data_}; + const auto position{ + std::ranges::lower_bound(entries, value, {}, &value_type::first)}; + if (position != entries.cend() && position->first == value) { + return false; + } + + this->data_.emplace(position, std::move(value), hash); + return true; +} + +auto JSONPropertySet::insert(const string_type &value) -> bool { + return this->insert(value, HASHER(value)); +} + +auto JSONPropertySet::insert(string_type &&value) -> bool { + const auto hash{HASHER(value)}; + return this->insert(std::move(value), hash); +} + +auto JSONPropertySet::to_json() const -> JSON { + return sourcemeta::core::to_json(this->data_, [](const auto &entry) -> JSON { + return sourcemeta::core::to_json(entry.first); + }); +} + +auto JSONPropertySet::from_json(const JSON &value) + -> std::optional { + if (!value.is_array()) { + return std::nullopt; + } + + JSONPropertySet result; + result.data_.reserve(value.size()); + for (const auto &item : value.as_array()) { + auto subvalue{sourcemeta::core::from_json(item)}; + if (!subvalue.has_value()) { + return std::nullopt; + } + + result.insert(std::move(subvalue).value()); + } + + return result; +} + +} // namespace sourcemeta::core diff --git a/vendor/core/src/core/jsonschema/bundle.cc b/vendor/core/src/core/jsonschema/bundle.cc index 708bd4d09..13332bdc7 100644 --- a/vendor/core/src/core/jsonschema/bundle.cc +++ b/vendor/core/src/core/jsonschema/bundle.cc @@ -34,6 +34,46 @@ auto is_skippable_metaschema_reference(const SchemaBundleOptions::Mode mode, schema_is_official(destination); } +// RFC 3986, section 4.4 calls a reference that leads back to the document +// holding it a same-document reference, whose "most frequent examples [...] are +// relative references that are empty or include only the number sign ('#') +// separator followed by a fragment identifier". Those stay as spelled, as they +// need no base URI beyond the one already in effect. Any other relative +// reference names another document, so bundling restates it as the URI it +// resolves to, which is what lets it keep naming its target once the bundled +// document travels somewhere else +// +// A dynamic reference needs no exception here. JSON Schema 2020-12, section +// 8.2.3.2 has it "resolved against the current URI base" like any other before +// the dynamic scope is consulted, so the absolute form of that resolution is +// what the document should spell, and framing only reports the anchor it ends +// up at in place of that resolution when the two name the same place anyway. +// JSON Schema 2019-09, section 8.2.4.2.1 defines the behavior of +// `$recursiveRef` "only for the value `#`", which the rule below already spares +auto spells_another_document_relatively(const SchemaFrame::Reference &reference) + -> bool { + if (reference.original == reference.destination) { + return false; + } + + const URI original{reference.original}; + return original.is_relative() && !original.is_fragment_only() && + !original.empty(); +} + +// The absolute form of a reference whose target answers to an identifier other +// than the URI it was resolved by +auto rebase_reference(const JSON::String &base, + const std::optional &fragment) + -> JSON::String { + URI result{base}; + if (fragment.has_value()) { + result.fragment(fragment.value()); + } + + return result.recompose(); +} + // The dialect a schema declares, falling back to the given default auto declared_dialect(const JSON &schema, const std::string_view default_dialect) @@ -167,12 +207,11 @@ auto elevate_embedded_resources( if (bundled.contains(identifier_string)) { if (container_exists && root_container->is_object()) { for (const auto &root_entry : root_container->as_object()) { - if (!root_entry.first.starts_with(identifier_string)) { - continue; - } - // Same reasoning as above: rule out what cannot match, and what - // framing would reject, before paying for a frame + // framing would reject, before paying for a frame. What a container + // calls an entry is no guide to what that entry identifies, since a + // caller may hold one under a name of its own choosing, so the + // declared identifier below is what rules an entry in or out if (!root_entry.second.is_object()) { continue; } @@ -309,12 +348,12 @@ auto embed_references( if (bundled.contains(identifier)) { const auto &mapped_id{bundled.at(identifier)}; if (mapped_id != identifier) { - URI rewrite_uri{mapped_id}; - if (reference.fragment.has_value()) { - rewrite_uri.fragment(reference.fragment.value()); - } - - ref_rewrites.emplace_back(to_pointer(pointer), rewrite_uri.recompose()); + ref_rewrites.emplace_back( + to_pointer(pointer), + rebase_reference(mapped_id, reference.fragment)); + } else if (spells_another_document_relatively(reference)) { + ref_rewrites.emplace_back(to_pointer(pointer), + JSON::String{reference.destination}); } return; @@ -408,12 +447,12 @@ auto embed_references( } if (effective_id != identifier) { - URI rewrite_uri{effective_id}; - if (reference.fragment.has_value()) { - rewrite_uri.fragment(reference.fragment.value()); - } - - ref_rewrites.emplace_back(to_pointer(pointer), rewrite_uri.recompose()); + ref_rewrites.emplace_back( + to_pointer(pointer), + rebase_reference(effective_id, reference.fragment)); + } else if (spells_another_document_relatively(reference)) { + ref_rewrites.emplace_back(to_pointer(pointer), + JSON::String{reference.destination}); } bundled.emplace(identifier, effective_id); @@ -422,6 +461,22 @@ auto embed_references( remote_base_dialect); }); + // Whatever the walk above reaches is on its way into the document, so its + // spelling is settled there rather than here. What is left is a reference + // whose target the document already holds, which still has to name it in a + // way that does not depend on where the document came from + frame.for_each_reference( + [&](const auto, const auto &pointer, const auto &reference) -> void { + if (!frame.traverse(reference.destination).has_value()) { + return; + } + + if (spells_another_document_relatively(reference)) { + ref_rewrites.emplace_back(to_pointer(pointer), + JSON::String{reference.destination}); + } + }); + for (auto &[rewrite_pointer, rewrite_value] : ref_rewrites) { set(subschema, rewrite_pointer, JSON{rewrite_value}); } diff --git a/vendor/core/src/core/oauth/CMakeLists.txt b/vendor/core/src/core/oauth/CMakeLists.txt index 8f6bb0c69..320be74f4 100644 --- a/vendor/core/src/core/oauth/CMakeLists.txt +++ b/vendor/core/src/core/oauth/CMakeLists.txt @@ -2,7 +2,7 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME oauth PRIVATE_HEADERS error.h profile.h pkce.h bearer.h random.h authorization.h token.h client_authentication.h transaction.h metadata.h metadata_provider.h token_exchange.h revocation.h introspection.h device.h dpop.h par.h - assertion.h registration.h + assertion.h registration.h duration.h SOURCES oauth_error.cc oauth_pkce.cc oauth_bearer.cc oauth_syntax.h oauth_scope.h oauth_random.cc oauth_authorization.cc oauth_authorization_parse.h @@ -11,7 +11,7 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME oauth oauth_transaction.cc oauth_metadata.cc oauth_metadata_provider.cc oauth_ttl.h oauth_token_exchange.cc oauth_revocation.cc oauth_introspection.cc oauth_device.cc oauth_dpop.cc oauth_par.cc - oauth_assertion.cc oauth_registration.cc) + oauth_assertion.cc oauth_registration.cc oauth_duration.cc) target_link_libraries(sourcemeta_core_oauth PUBLIC sourcemeta::core::json) diff --git a/vendor/core/src/core/oauth/include/sourcemeta/core/oauth.h b/vendor/core/src/core/oauth/include/sourcemeta/core/oauth.h index 4e7f144e9..74416e7e9 100644 --- a/vendor/core/src/core/oauth/include/sourcemeta/core/oauth.h +++ b/vendor/core/src/core/oauth/include/sourcemeta/core/oauth.h @@ -12,6 +12,7 @@ #include #include #include +#include #include #include #include diff --git a/vendor/core/src/core/oauth/include/sourcemeta/core/oauth_duration.h b/vendor/core/src/core/oauth/include/sourcemeta/core/oauth_duration.h new file mode 100644 index 000000000..8e52779c2 --- /dev/null +++ b/vendor/core/src/core/oauth/include/sourcemeta/core/oauth_duration.h @@ -0,0 +1,43 @@ +#ifndef SOURCEMETA_CORE_OAUTH_DURATION_H_ +#define SOURCEMETA_CORE_OAUTH_DURATION_H_ + +#ifndef SOURCEMETA_CORE_OAUTH_EXPORT +#include +#endif + +#include + +#include // std::chrono::seconds +#include // std::optional + +namespace sourcemeta::core { + +/// @ingroup oauth +/// Read an integer member of a JSON object as a duration in seconds, rejecting +/// a negative value as malformed, and one past the range of the duration so a +/// bad lifetime or interval cannot reach a caller narrowed. The OAuth and +/// OpenID Connect documents spell every lifetime and interval this way, so the +/// two families read them through this one function rather than each deciding +/// for itself what a malformed member looks like. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// const auto document{sourcemeta::core::parse_json("{ \"expires_in\": 1800 +/// }")}; const auto hash{sourcemeta::core::JSON::Object::hash("expires_in")}; +/// const auto value{ +/// sourcemeta::core::oauth_json_seconds_member(document, "expires_in", +/// hash)}; +/// assert(value.has_value()); +/// assert(value.value() == std::chrono::seconds{1800}); +/// ``` +SOURCEMETA_CORE_OAUTH_EXPORT +auto oauth_json_seconds_member(const JSON &data, const JSON::StringView name, + const JSON::Object::hash_type hash) + -> std::optional; + +} // namespace sourcemeta::core + +#endif diff --git a/vendor/core/src/core/oauth/oauth_duration.cc b/vendor/core/src/core/oauth/oauth_duration.cc new file mode 100644 index 000000000..a15de34ab --- /dev/null +++ b/vendor/core/src/core/oauth/oauth_duration.cc @@ -0,0 +1,39 @@ +#include + +#include + +#include // std::chrono::seconds +#include // std::numeric_limits +#include // std::optional, std::nullopt + +namespace sourcemeta::core { + +auto oauth_json_seconds_member(const JSON &data, const JSON::StringView name, + const JSON::Object::hash_type hash) + -> std::optional { + if (!data.is_object()) { + return std::nullopt; + } + + const auto *member{data.try_at(name, hash)}; + if (member == nullptr || !member->is_integer() || member->to_integer() < 0) { + return std::nullopt; + } + + // A JSON integer is a signed 64 bit value, whereas a duration is only + // promised 35 bits, so a platform whose duration is the narrower of the two + // has to range check the value before narrowing it. Where the two are the + // same width the comparison cannot fail, and asking for it at compile time + // discards it rather than leaving an unreachable branch behind + if constexpr (std::numeric_limits::max() > + std::numeric_limits::max()) { + if (member->to_integer() > + std::numeric_limits::max()) { + return std::nullopt; + } + } + + return std::chrono::seconds{member->to_integer()}; +} + +} // namespace sourcemeta::core diff --git a/vendor/core/src/core/oauth/oauth_json.h b/vendor/core/src/core/oauth/oauth_json.h index d08f7b6bc..ceb4d35d8 100644 --- a/vendor/core/src/core/oauth/oauth_json.h +++ b/vendor/core/src/core/oauth/oauth_json.h @@ -2,9 +2,8 @@ #define SOURCEMETA_CORE_OAUTH_JSON_H_ #include +#include -#include // std::chrono::seconds -#include // std::numeric_limits #include // std::optional, std::nullopt #include // std::string_view @@ -28,28 +27,6 @@ inline auto oauth_json_string_member(const JSON &data, return std::string_view{member->to_string()}; } -// Read an integer member of a JSON object as a duration in seconds, rejecting a -// negative value as malformed, and one past the range of the duration so a -// bad lifetime or interval cannot flow to a caller as a usable or narrowed -// duration -inline auto oauth_json_seconds_member(const JSON &data, - const JSON::StringView name, - const JSON::Object::hash_type hash) - -> std::optional { - if (!data.is_object()) { - return std::nullopt; - } - - const auto *member{data.try_at(name, hash)}; - if (member == nullptr || !member->is_integer() || member->to_integer() < 0 || - member->to_integer() > - std::numeric_limits::max()) { - return std::nullopt; - } - - return std::chrono::seconds{member->to_integer()}; -} - } // namespace sourcemeta::core #endif diff --git a/vendor/core/src/core/oidc/oidc_registration.cc b/vendor/core/src/core/oidc/oidc_registration.cc index 2932d3695..13dec7fa5 100644 --- a/vendor/core/src/core/oidc/oidc_registration.cc +++ b/vendor/core/src/core/oidc/oidc_registration.cc @@ -6,7 +6,6 @@ #include #include // std::chrono::seconds -#include // std::numeric_limits #include // std::optional, std::nullopt #include // std::span #include // std::string_view @@ -288,15 +287,8 @@ auto OIDCClientMetadata::userinfo_signed_response_alg() const auto OIDCClientMetadata::default_max_age() const -> std::optional { - const auto *member{ - this->oauth_.data().try_at("default_max_age"sv, HASH_DEFAULT_MAX_AGE)}; - if (member == nullptr || !member->is_integer() || member->to_integer() < 0 || - member->to_integer() > - std::numeric_limits::max()) { - return std::nullopt; - } - - return std::chrono::seconds{member->to_integer()}; + return oauth_json_seconds_member(this->oauth_.data(), "default_max_age"sv, + HASH_DEFAULT_MAX_AGE); } auto OIDCClientMetadata::require_auth_time() const -> bool { diff --git a/vendor/core/src/core/openapi/CMakeLists.txt b/vendor/core/src/core/openapi/CMakeLists.txt index e5ed135cf..8be9d8db8 100644 --- a/vendor/core/src/core/openapi/CMakeLists.txt +++ b/vendor/core/src/core/openapi/CMakeLists.txt @@ -4,7 +4,7 @@ sourcemeta_library(NAMESPACE sourcemeta PROJECT core NAME openapi parameter.h request_body.h security.h path_item.h paths.h components.h discriminator.h external_documentation.h info.h server.h tag.h - document.h frame.cc version.cc) + document.h bundle.cc frame.cc format.cc version.cc) if(SOURCEMETA_CORE_INSTALL) sourcemeta_library_install(NAMESPACE sourcemeta PROJECT core NAME openapi) diff --git a/vendor/core/src/core/openapi/bundle.cc b/vendor/core/src/core/openapi/bundle.cc new file mode 100644 index 000000000..492b95370 --- /dev/null +++ b/vendor/core/src/core/openapi/bundle.cc @@ -0,0 +1,1962 @@ +#include + +#include "discriminator.h" +#include "document.h" +#include "helpers.h" + +#include // assert +#include // std::size_t +#include // std::uint64_t +#include // std::map +#include // std::optional +#include // std::set +#include // std::to_string +#include // std::move, std::make_pair, std::pair +#include // std::vector + +namespace { + +// Every walk and every frame that bundling builds spends from the same +// allowance, as how much of it bundling ends up needing is a function of what +// the resolvers hand back rather than of the document the caller passed in. A +// walk reads a document in full before anything charges for it, so what this +// bounds is how many oversized documents are read rather than whether one is +auto charge(std::uint64_t &remaining, const std::size_t locations) -> void { + assert(locations <= remaining); + remaining -= locations; +} + +// A reference that bundling has to make whole: the member that spells it, what +// it names, and what the position it sits in expects to find there +struct OpenAPIPending { + sourcemeta::core::Pointer origin; + sourcemeta::core::JSON::String destination; + sourcemeta::core::OpenAPIObjectKind expected; + // Whether a Discriminator Object mapping is what names it. A `$ref` of a + // Schema Object is one that whatever reads inside a Schema Object follows + // for itself, and Section 4.8.25 makes a mapping an annotation that nothing + // there has any account of, so the two part company wherever the shell + // cannot reach what is named + bool mapping{false}; + // Whether a Security Requirement Object is what names it. OpenAPI + // Specification 3.2.1, Section 4.30 lets a name "be the URI of a Security + // Scheme Object", and that URI is the member the scopes sit under rather than + // a value of its own, so making it whole renames a member rather than writing + // to one. Nothing else this brings in is spelled that way + bool requirement{false}; + // What the reference resolves against, which for everything but a Schema + // Object is the base of the document that makes it. OpenAPI Specification + // 3.1.1, Section 4.6 has a relative reference inside a Schema Object use + // "the nearest parent `$id` as a Base URI" instead + sourcemeta::core::JSON::String scope; +}; + +// What the description reaches for and does not hold. A reference that names +// the document being read and lands nowhere is left alone, as OpenAPI +// Specification 3.1.1, Section 4.8.23 holds a `$ref` to the form of a URI and +// says nothing about it having to resolve, so there is nothing to fetch here +// and nothing this specification lets us report +auto pending(const sourcemeta::core::OpenAPIWalk &walk) + -> std::vector { + std::vector result; + for (const auto &entry : walk.references) { + if (walk.locations.contains(entry.second.destination) || + sourcemeta::core::openapi_within_document(entry.second.destination, + walk.base)) { + continue; + } + + result.push_back({.origin = entry.second.origin, + .destination = entry.second.destination, + .expected = entry.second.expected, + .scope = walk.base}); + } + + // And so does what a Security Requirement Object names by URI, which OpenAPI + // Specification 3.2.1, Section 4.30 admits alongside the name of a component: + // "The name used for each property MUST either correspond to a security + // scheme declared in the Security Schemes under the Components Object, or be + // the URI of a Security Scheme Object". The frame keeps these apart from the + // references above only because a single one of those Objects may name + // several schemes, which is more than one entry keyed by where it sits + for (const auto &entry : walk.security_references) { + if (walk.locations.contains(entry.second.destination) || + sourcemeta::core::openapi_within_document(entry.second.destination, + walk.base)) { + continue; + } + + result.push_back({.origin = entry.second.origin, + .destination = entry.second.destination, + .expected = entry.second.expected, + .requirement = true, + .scope = walk.base}); + } + + return result; +} + +// Where the Object that bundling embeds begins. A target that sits within a +// component of its own document is embedded as that whole component, so that a +// reference to it and a reference deeper into it land on one copy rather than +// on two. OpenAPI Specification 3.1.1, Section 4.8.7 puts every component at +// the same depth, which is what makes the first three tokens the whole test +constexpr std::size_t COMPONENT_DEPTH{3}; + +// Whether a pointer names a component of the document it belongs to, which is +// what the Components Object holds at a fixed depth of its own +auto is_component(const sourcemeta::core::Pointer &pointer) -> bool { + return pointer.size() == COMPONENT_DEPTH && pointer.at(0).is_property() && + pointer.at(0).to_property() == "components" && + pointer.at(1).is_property(); +} + +auto promote(const sourcemeta::core::Pointer &target) + -> sourcemeta::core::Pointer { + if (target.size() < COMPONENT_DEPTH || + !target.starts_with(sourcemeta::core::EMPTY_POINTER, "components")) { + return target; + } + + return target.slice(0, COMPONENT_DEPTH); +} + +// Where the Path Item Object holding an Operation Object sits. OpenAPI +// Specification 3.1.1, Section 4.8.9 puts an Operation Object directly under +// the Path Item Object that holds it, and 3.2.1, Section 4.9 adds +// `additionalOperations`, "A map of additional operations on this path. The +// map key is the HTTP method with the same capitalization that is to be sent +// in the request", which puts a map of its own between the two. So which place +// holds it is what the walk recorded rather than a fixed number of steps up +auto path_item_of(const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::Pointer &operation) + -> sourcemeta::core::Pointer { + auto prefix{operation}; + while (!prefix.empty()) { + prefix = prefix.initial(); + const auto location{remote.locations.find( + sourcemeta::core::openapi_location_uri(remote.base, prefix))}; + if (location != remote.locations.cend() && + location->second.type == + sourcemeta::core::OpenAPIObjectKind::PathItem) { + return prefix; + } + } + + return operation.initial(); +} + +// Which Components Object member an embedded Object goes under. A component +// keeps the member its own document filed it under, which is what that +// document decided the Object is, rather than the member that the kind of the +// reference reaching it would take. The two differ whenever a reference names +// a place within a component rather than the component itself +auto container_of(const sourcemeta::core::Pointer &origin, + const sourcemeta::core::OpenAPIObjectKind expected) + -> sourcemeta::core::JSON::StringView { + if (is_component(origin)) { + return origin.at(1).to_property(); + } + + return sourcemeta::core::openapi_component_container(expected); +} + +// A reference that leads back to the document being bundled into names a place +// that document already holds, so it is written as a fragment of it rather +// than as the URI that document answers to +auto rebase(const sourcemeta::core::JSON::String &destination, + const sourcemeta::core::JSON::String &base) + -> sourcemeta::core::JSON::String { + if (!sourcemeta::core::openapi_within_document(destination, base)) { + return destination; + } + + return destination.substr(base.size()); +} + +// A Schema Object carries the base of the document it was written in, and +// nothing below the root of a document may declare one of its own. So every +// reference inside a schema that moves is written back as whatever it resolved +// to where it came from, which a schema frame of that document is what settles +auto absolutize_schemas(sourcemeta::core::JSON &value, + const sourcemeta::core::JSON &remote_document, + const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::Pointer &origin, + const sourcemeta::core::Pointer &landing, + const sourcemeta::core::JSON::String &dialect, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + std::uint64_t &remaining) -> void { + sourcemeta::core::SchemaFrame::Paths paths; + for (const auto &entry : remote.locations) { + if (entry.second.type != sourcemeta::core::OpenAPIObjectKind::Schema || + !entry.second.pointer.starts_with(origin)) { + continue; + } + + paths.push_back(sourcemeta::core::to_weak_pointer(entry.second.pointer)); + + // OpenAPI Specification 3.1.1, Section 4.8.24.5 scopes `jsonSchemaDialect` + // to "all Schema Objects contained within an OAS document", so a schema + // that says nothing about the dialect it is written against is read under + // whichever one the document holding it declares. Moving it to a document + // that settles on another dialect is what makes it say so itself + if (remote.dialect == dialect) { + continue; + } + + auto &schema{sourcemeta::core::get( + value, (entry.second.pointer).resolve_from(origin))}; + // Section 4.8.24 lets a Schema Object be a boolean, which declares nothing + if (schema.is_object() && !schema.defines("$schema")) { + schema.assign("$schema", sourcemeta::core::JSON{remote.dialect}); + } + } + + if (paths.empty()) { + return; + } + + const sourcemeta::core::SchemaFrame frame{ + sourcemeta::core::SchemaFrame::Mode::References, + remote_document, + walker, + resolver, + remote.dialect, + "", + sourcemeta::core::SchemaFrame::IdentifierMode::Additional, + paths, + remote.base, + remaining}; + charge(remaining, frame.location_count()); + + // What a moved schema names by a mapping is resolved where it came from, + // just as what it names by a reference is + for (const auto &discriminator : sourcemeta::core::openapi_discriminators( + remote_document, frame, remote.base, walker, resolver)) { + if (!discriminator.origin.starts_with(origin)) { + continue; + } + + const auto held{(discriminator.origin).resolve_from(origin)}; + const auto *written{sourcemeta::core::try_get(value, held)}; + if (written == nullptr || !written->is_string()) { + continue; + } + + // Section 4.3.3 resolves a name against the Components Object of the entry + // document wherever the Discriminator Object naming it sits, so moving the + // schema leaves it naming exactly what it named. That holds whether or not + // the name also happens to lead into what moves, so it is settled before + // any route is, or a name would be written out as the route it took + if (sourcemeta::core::openapi_is_component_key(written->to_string())) { + continue; + } + + const auto target{frame.traverse(discriminator.destination)}; + if (target.has_value() && target.value().get().pointer.starts_with( + sourcemeta::core::to_weak_pointer(origin))) { + // What names a place within what moves moves along with it, so long as + // it names it by something that travels. A name does. RFC 6901 reads a + // pointer from the root of a document, so one that leads into what moves + // leads there by a route that moving is exactly what changes, and it is + // written out as the route to wherever this is headed instead. This is + // the same distinction a reference of the schema is held to, and a + // mapping is held to it for the same reason + // + // The URI is held rather than made and read in one breath, as what + // reads its fragment hands back a view into it + const sourcemeta::core::URI destination{discriminator.destination}; + const auto fragment{destination.fragment()}; + if (!sourcemeta::core::openapi_within_document(discriminator.destination, + remote.base) || + !fragment.has_value() || !fragment.value().starts_with('/')) { + continue; + } + + const auto tail{sourcemeta::core::to_pointer(target.value().get().pointer) + .resolve_from(origin)}; + sourcemeta::core::set( + value, held, + sourcemeta::core::JSON{ + sourcemeta::core::to_uri(landing.concat(tail)).recompose()}); + continue; + } + + if (written->to_string() != discriminator.destination) { + sourcemeta::core::set(value, held, + sourcemeta::core::JSON{discriminator.destination}); + } + } + + const auto moved{sourcemeta::core::to_weak_pointer(origin)}; + + // OpenAPI Specification 3.1.1, Section 4.6: "Relative references in Schema + // Objects, including any that appear as `$id` values, use the nearest parent + // `$id` as a Base URI". So a schema that names itself relatively names + // something else once it sits in another document, and writing back what it + // resolved to where it came from is what keeps its identity its own + frame.for_each_resource([&value, &origin, + &moved](const auto &identifier, + const auto &location) -> void { + if (!location.pointer.starts_with(moved)) { + return; + } + + auto &schema{sourcemeta::core::get( + value, + (sourcemeta::core::to_pointer(location.pointer)).resolve_from(origin))}; + // Which keyword carries that identity is the dialect's to say rather than + // whichever one the latest of them happens to use, and a schema that + // declares none is left without one rather than given one it never had + if (schema.is_object() && + schema.defines(sourcemeta::core::schema_identifier_keyword( + location.base_dialect))) { + sourcemeta::core::schema_reidentify(schema, identifier, + location.base_dialect); + } + }); + + // Section 4.8.24 puts an External Documentation Object on a Schema Object, + // and Section 4.8.11 makes its `url` "The URI for the target documentation". + // The frame of the shell never reaches inside a Schema Object, so the one + // URI a schema carries that is neither a reference nor an identifier is + // written back here + frame.for_each_location([&value, &origin, &moved, &frame, &walker, + &resolver](const auto, const auto &, + const auto &location) -> void { + if (!location.pointer.starts_with(moved)) { + return; + } + + // Section 4.8.24 lists `externalDocs` among the keywords that the dialect + // this specification publishes is made of, and Section 4.8.24.5 has a + // Schema Object read under whichever dialect it declares. So a schema + // written against one that leaves the keyword out holds no External + // Documentation Object at all, and what it spells there names nothing that + // moving could break + const auto &vocabularies{frame.vocabularies(location, resolver)}; + if (walker("externalDocs", vocabularies).type == + sourcemeta::core::SchemaKeywordType::Unknown) { + return; + } + + const auto held{ + (sourcemeta::core::to_pointer(location.pointer)) + .resolve_from(origin) + .concat(sourcemeta::core::Pointer{"externalDocs", "url"})}; + const auto *written{sourcemeta::core::try_get(value, held)}; + if (written == nullptr || !written->is_string()) { + return; + } + + const auto absolute{sourcemeta::core::openapi_resolve_uri( + written->to_string(), sourcemeta::core::JSON::String{location.base})}; + if (absolute.has_value()) { + sourcemeta::core::set( + value, held, sourcemeta::core::JSON{absolute.value().recompose()}); + } + }); + + frame.for_each_reference_from( + moved, + [&value, &origin, &moved, &landing, &frame, &remote]( + const auto type, const auto &pointer, const auto &reference) -> void { + // What a dynamic reference names is settled where the schema is + // evaluated rather than where it sits, so a bare name is left exactly + // as the description wrote it. Which document that name is looked for + // in is settled on load though, which JSON Schema Section 8.2.3.2 + // calls out: "Resolved against the current URI base, it produces the + // URI used as the starting point for runtime resolution. This initial + // resolution is safe to perform on schema load". So one that names a + // document of its own is written back like any other reference, and + // only the name it carries goes on being settled later + if (type != sourcemeta::core::SchemaReferenceType::Static && + reference.original.starts_with('#')) { + return; + } + + // A recursive reference names whichever resource it is evaluated + // within rather than a place, and JSON Schema 2019-09 Section + // 8.2.4.2.1 leaves it one value to say that with: "The behavior of + // this keyword is defined only for the value `#`". One of these frames + // as static wherever no anchor is in scope for it, so writing out what + // it resolved to is what would make it say something the keyword does + // not admit, which reading the result back then turns down + if (!pointer.empty() && pointer.back().is_property() && + pointer.back().to_property() == "$recursiveRef") { + return; + } + + // Anything already spelled out as what it resolves to is left exactly + // as the description wrote it + if (reference.original == reference.destination) { + return; + } + + // And so is anything that names a place inside what moves, which is + // what an anchor that a schema declares and names itself by comes to. + // Both of them moving together is what keeps the one naming the other, + // and resolving it against the document it came from is what would + // break that. + // + // A pointer is not like a name that way. RFC 6901 reads one from the + // root of a document, so one that leads into what moves leads there + // by a route that moving is exactly what changes. Such a reference is + // written out as the route to wherever this is headed instead + const auto target{frame.traverse(reference.destination)}; + if (target.has_value() && + target.value().get().pointer.starts_with(moved)) { + // Which root a pointer counts from is what the reference resolved + // against rather than how it is spelled. One that resolved against + // the document is the one moving takes somewhere else. One that + // resolved against an identifier a schema declares for itself counts + // from that schema, which moves along whole, and a name is not a + // pointer at all + if (!sourcemeta::core::openapi_within_document(reference.destination, + remote.base) || + !reference.fragment.has_value() || + !reference.fragment.value().starts_with('/')) { + return; + } + + const auto tail{ + sourcemeta::core::to_pointer(target.value().get().pointer) + .resolve_from(origin)}; + sourcemeta::core::set( + value, + (sourcemeta::core::to_pointer(pointer)).resolve_from(origin), + sourcemeta::core::JSON{ + sourcemeta::core::to_uri(landing.concat(tail)).recompose()}); + return; + } + + sourcemeta::core::set( + value, (sourcemeta::core::to_pointer(pointer)).resolve_from(origin), + sourcemeta::core::JSON{reference.destination}); + }); +} + +// What to call an embedded Object. The name it went by in the document it came +// from is the one a reader would look for, and Section 4.8.7 constrains every +// component key, so a name that does not hold up there is replaced rather than +// carried over. Taking a name the entry document already uses would change +// what an implicit connection resolves to, which is why this only takes a free +// one +// Whatever the Components Object already holds under a member, which is what a +// name has to be free of before anything is written there +auto openapi_component_container_of( + const sourcemeta::core::JSON &document, + const sourcemeta::core::JSON::StringView container) + -> const sourcemeta::core::JSON * { + const auto *components{document.try_at("components")}; + if (components == nullptr || !components->is_object()) { + return nullptr; + } + + const auto *entries{ + components->try_at(sourcemeta::core::JSON::String{container})}; + return entries == nullptr || !entries->is_object() ? nullptr : entries; +} + +// Section 4.8.7 admits nothing but letters, digits, dots, hyphens and +// underscores into a component key, so what does not hold up to that is +// dropped rather than carried over, and a name left with nothing is replaced +auto sanitise(const sourcemeta::core::JSON::StringView candidate) + -> sourcemeta::core::JSON::String { + sourcemeta::core::JSON::String result; + for (const auto character : candidate) { + if (sourcemeta::core::openapi_is_component_key( + sourcemeta::core::JSON::StringView{&character, 1})) { + result.push_back(character); + } + } + + return result.empty() ? sourcemeta::core::JSON::String{"Bundled"} : result; +} + +// Taking a name that the description already gives a meaning to would change +// what an implicit connection resolves to, which is why this only ever takes +// a free one +auto vacant(const sourcemeta::core::JSON &entries, + sourcemeta::core::JSON::String candidate) + -> sourcemeta::core::JSON::String { + if (!entries.defines(candidate)) { + return candidate; + } + + candidate.append("_"); + if (!entries.defines(candidate)) { + return candidate; + } + + // Going on appending would make the name grow by one for every other name + // that already took it, which is a description naming one place many times + // paying for that many times over in the key it ends up with. Counting + // instead keeps the name to a length the number of them can be written in. + // Section 4.8.7 admits digits into a component key just as it admits the + // underscore + const auto taken{candidate}; + // Trying each number in turn would ask the Components Object about every + // name that already took this one, and asking it is a walk of everything it + // holds, so a description naming one place many times over would pay for it + // many times over again. Closing in on a free number asks it far fewer + // times. What this settles on is a free number rather than the lowest free + // one, which a document already holding some of them out of order is what + // makes the two differ. Either is a name nothing else goes by, which is all + // a name bundling invents has to be + std::uint64_t lower{1}; + std::uint64_t upper{2}; + while (entries.defines(taken + std::to_string(upper))) { + lower = upper; + upper *= 2; + } + + // The number above is free and the one below it is taken, and every step + // keeps both of those true, so the number this ends on is free + while (upper - lower > 1) { + const auto middle{lower + ((upper - lower) / 2)}; + if (entries.defines(taken + std::to_string(middle))) { + lower = middle; + } else { + upper = middle; + } + } + + candidate = taken + std::to_string(upper); + + return candidate; +} + +auto component_name(const sourcemeta::core::JSON &document, + const sourcemeta::core::JSON::StringView container, + const sourcemeta::core::JSON::StringView source, + const sourcemeta::core::Pointer &target, + const sourcemeta::core::OpenAPIBundleOptions::Namer &namer) + -> sourcemeta::core::JSON::String { + sourcemeta::core::JSON::String candidate{"Bundled"}; + if (namer) { + candidate = namer(source, container); + } else if (!target.empty()) { + // RFC 6901 Section 4 leaves which of an array and an object a + // reference token addresses to what it is evaluated against, and one of + // digits reads back as an array index, which spells out to a name the + // Components Object takes as it stands + const auto &token{target.back()}; + if (token.is_property()) { + candidate = token.to_property(); + } else if (token.is_index()) { + candidate = + sourcemeta::core::JSON::String{std::to_string(token.to_index())}; + } + } + + const auto *entries{openapi_component_container_of(document, container)}; + candidate = sanitise(candidate); + return entries == nullptr ? candidate + : vacant(*entries, std::move(candidate)); +} + +auto embed(sourcemeta::core::JSON &document, + const sourcemeta::core::JSON::StringView container, + const sourcemeta::core::JSON::String &name, + sourcemeta::core::JSON &&value) -> sourcemeta::core::Pointer { + const sourcemeta::core::JSON::String container_key{container}; + document.assign_if_missing("components", + sourcemeta::core::JSON::make_object()); + auto &components{document.at("components")}; + components.assign_if_missing(container_key, + sourcemeta::core::JSON::make_object()); + components.at(container_key).assign(name, std::move(value)); + + sourcemeta::core::Pointer result; + result.push_back(sourcemeta::core::JSON::String{"components"}); + result.push_back(container_key); + result.push_back(name); + return result; +} + +// A Schema Object is addressed by the identifier it declares rather than by +// where it sits, so the name it goes under is a label rather than something a +// reference resolves through, and the last segment of that identifier is the +// part of it a reader looks for +auto schema_name(const sourcemeta::core::JSON &schemas, + const sourcemeta::core::JSON::StringView identifier, + const sourcemeta::core::OpenAPIBundleOptions::Namer &namer) + -> sourcemeta::core::JSON::String { + return vacant( + schemas, + sanitise(namer ? sourcemeta::core::JSON::StringView{namer(identifier, + "schemas")} + : identifier.substr(identifier.find_last_of('/') + 1))); +} + +// A boolean carries no keyword, so it cannot say who it is, and whatever named +// it by the URI it was found under would go on naming nothing once it sits +// somewhere else. JSON Schema 2020-12 Section 4.3.2 gives each of the two an +// object that says exactly what it says: "true: Always passes validation, as +// if the empty schema `{}`" and "false: Always fails validation, as if the +// schema `{ "not": {} }`". Written that way it can answer to the URI it was +// found under, which leaves every reference that named it naming it still +auto spell_out(sourcemeta::core::JSON &schema) -> void { + if (!schema.is_boolean()) { + return; + } + + const auto permissive{schema.to_boolean()}; + schema = sourcemeta::core::JSON::make_object(); + if (!permissive) { + schema.assign("not", sourcemeta::core::JSON::make_object()); + } +} + +// Bundling what sits inside a Schema Object is JSON Schema's to do, and the +// Components Object holds the one member this specification reserves for +// schemas. Section 4.8.7 constrains the keys of that member, which the +// identifiers that bundling names an embedded schema by do not hold up to, so +// each of them is renamed once it has landed. Nothing resolves through those +// keys, so renaming reaches nothing else +auto bundle_schemas(sourcemeta::core::JSON &document, + const sourcemeta::core::OpenAPIWalk &walk, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + const sourcemeta::core::JSON::String &base, + const std::uint64_t remaining, + const sourcemeta::core::OpenAPIBundleOptions &options) + -> bool { + const auto &callback{options.callback}; + const auto &namer{options.namer}; + sourcemeta::core::SchemaFrame::Paths paths; + for (const auto &entry : walk.locations) { + if (entry.second.type == sourcemeta::core::OpenAPIObjectKind::Schema) { + paths.push_back(sourcemeta::core::to_weak_pointer(entry.second.pointer)); + } + } + + if (paths.empty()) { + return false; + } + + sourcemeta::core::Pointer container; + container.push_back(sourcemeta::core::JSON::String{"components"}); + container.push_back(sourcemeta::core::JSON::String{"schemas"}); + + // What each schema was resolved by, beside where it landed. Bundling picks a + // key that is free rather than one that matches, so reading the identifier + // back off that pointer is only right until two of them collide + std::vector< + std::pair> + landed; + sourcemeta::core::SchemaBundleOptions schemas_options; + // Section 4.8.7 holds the `schemas` member of the Components Object to + // "reusable Schema Objects", and the dialect a schema is written against is + // not one of those. So what a `$schema` names is left where it is rather + // than carried into the description, whether or not this specification's own + // dialect is one that JSON Schema counts as its own + schemas_options.mode = + sourcemeta::core::SchemaBundleOptions::Mode::References; + schemas_options.default_container = container; + schemas_options.paths = paths; + schemas_options.default_base = base; + schemas_options.max_locations = remaining; + schemas_options.callback = + [&landed](const std::string_view identifier, + const sourcemeta::core::WeakPointer &location) -> void { + landed.emplace_back(sourcemeta::core::JSON::String{identifier}, + sourcemeta::core::to_pointer(location)); + }; + + // Section 4.8.24.5 scopes what the OpenAPI Object sets to the Schema Objects + // "contained within an OAS document", and says of the rest: "For standalone + // JSON Schema documents that do not set `$schema` [...] the dialect SHOULD + // be assumed to be the OAS dialect". A document a resolver hands back is one + // of those, so it says so itself before anything reads it under the dialect + // this description happens to have settled on + const sourcemeta::core::JSON::String standalone{ + sourcemeta::core::openapi_dialect(walk.version)}; + const auto standalone_resolver{ + [&resolver, &standalone, &base](const std::string_view identifier) + -> sourcemeta::core::SchemaResolverResult { + auto result{resolver(identifier)}; + // 3.2.1 Section 4.1.2 has every document of a description hold "either + // an OpenAPI Object or a Schema Object at the root", and what answers + // here answers as the second. One that is the first instead would be + // read as a schema, which Appendix G leaves undefined and lets this + // turn down: "If the same JSON/YAML object is parsed multiple times + // and the respective contexts require it to be parsed as different + // Object types, the resulting behavior is implementation defined, and + // MAY be treated as an error if detected". Turning it down is also + // what keeps a document of another revision from arriving this way + if (result.has_value() && + sourcemeta::core::openapi_is_document(result.value())) { + throw sourcemeta::core::OpenAPIReferenceError{ + base, sourcemeta::core::EMPTY_POINTER, + sourcemeta::core::JSON::String{identifier}, + "This reference must name a schema rather than a document that " + "holds an OpenAPI Description"}; + } + + if (!result.has_value() || !result.value().is_object() || + result.value().defines("$schema")) { + return result; + } + + auto owned{std::move(result).to_owned()}; + owned.assign("$schema", sourcemeta::core::JSON{standalone}); + return owned; + }}; + + sourcemeta::core::schema_bundle(document, walker, standalone_resolver, + walk.dialect, "", schemas_options); + if (landed.empty()) { + return false; + } + + auto &schemas{sourcemeta::core::get(document, container)}; + // A resource that travels this way has to say who it is, which is what JSON + // Schema 2020-12 Section 9.3.1 asks of a Compound Schema Document: "Each + // embedded JSON Schema Resource MUST identify itself with a URI using the + // `$id` keyword". A description is not one of those, so nothing binds it + // here, but a reference that named a schema by URI goes on naming nothing + // unless the schema carries that URI. Section 4.8.24 lets a Schema Object be + // a boolean, which carries no keyword at all and so can say nothing, which + // writing it out as the object saying the same thing settles + for (const auto &entry : landed) { + const auto &identifier{entry.first}; + const auto &key{entry.second.back().to_property()}; + auto &schema{schemas.at(key)}; + if (schema.is_boolean()) { + spell_out(schema); + schema.assign("$schema", sourcemeta::core::JSON{standalone}); + schema.assign("$id", sourcemeta::core::JSON{identifier}); + } + + const auto name{schema_name(schemas, identifier, namer)}; + schemas.rename(key, sourcemeta::core::JSON::String{name}); + if (callback) { + callback(identifier, container.concat(name)); + } + } + + return true; +} + +// What the Schema Objects of a description reach for and it does not hold. +// Section 4.3 lists a Schema Object `$ref` among the fields that connect the +// documents of a description, so one of these may name a Schema Object that +// another of those documents declares, which only the shell can reach +auto schema_pending(const sourcemeta::core::JSON &document, + const sourcemeta::core::OpenAPIWalk &walk, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + std::vector &result, + std::uint64_t &remaining) -> void { + sourcemeta::core::SchemaFrame::Paths paths; + for (const auto &entry : walk.locations) { + if (entry.second.type == sourcemeta::core::OpenAPIObjectKind::Schema) { + paths.push_back(sourcemeta::core::to_weak_pointer(entry.second.pointer)); + } + } + + if (paths.empty()) { + return; + } + + const sourcemeta::core::SchemaFrame frame{ + sourcemeta::core::SchemaFrame::Mode::References, + document, + walker, + resolver, + walk.dialect, + "", + sourcemeta::core::SchemaFrame::IdentifierMode::Additional, + paths, + walk.base, + remaining}; + charge(remaining, frame.location_count()); + + frame.for_each_reference([&frame, &walk, + &result](const auto type, const auto &pointer, + const auto &reference) -> void { + // A dynamic reference names an anchor to be settled where it is + // evaluated, and a `$schema` names the dialect a schema is written + // against rather than a part of the description + if (type != sourcemeta::core::SchemaReferenceType::Static || + (!pointer.empty() && pointer.back().is_property() && + pointer.back().to_property() == "$schema")) { + return; + } + + if (frame.traverse(reference.destination).has_value() || + sourcemeta::core::openapi_within_document(reference.destination, + walk.base)) { + return; + } + + // What a reference resolves against is the base of the schema that holds + // it, which is the nearest identifier an enclosing one declares rather + // than anything the reference says about itself + const auto enclosing{frame.traverse(pointer.initial())}; + result.push_back( + {.origin = sourcemeta::core::to_pointer(pointer), + .destination = reference.destination, + .expected = sourcemeta::core::OpenAPIObjectKind::Schema, + .scope = + enclosing.has_value() + ? sourcemeta::core::JSON::String{enclosing.value().get().base} + : walk.base}); + }); + + // And so does what a Discriminator Object names by URI, which the frame + // above does not read because the keyword it sits under belongs to the + // OpenAPI dialect rather than to JSON Schema + for (auto &discriminator : sourcemeta::core::openapi_discriminators( + document, frame, walk.base, walker, resolver)) { + if (sourcemeta::core::openapi_discriminator_lands(frame, discriminator) || + sourcemeta::core::openapi_within_document(discriminator.destination, + walk.base)) { + continue; + } + + result.push_back({.origin = std::move(discriminator.origin), + .destination = std::move(discriminator.destination), + .expected = sourcemeta::core::OpenAPIObjectKind::Schema, + .mapping = true, + .scope = std::move(discriminator.scope)}); + } +} + +// Whether what a reference names is the kind that the position it sits in +// expects. A Schema Object reference is the one that may name a place inside +// an Object rather than an Object this specification types, as Section 4.6 +// reads the fragment as a JSON Pointer and a Schema Object may sit within a +// component of any kind. So what answers for it is the nearest enclosing +// place that framing did record, which has to be a Schema Object itself +auto lands(const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::JSON::String &identifier, + const sourcemeta::core::Pointer &target, + const sourcemeta::core::OpenAPIObjectKind expected) -> bool { + auto prefix{target}; + auto exact{true}; + while (true) { + const auto location{remote.locations.find( + sourcemeta::core::openapi_location_uri(identifier, prefix))}; + if (location != remote.locations.cend()) { + if (location->second.type == expected) { + return true; + } + + // A Reference Object stands in for whatever the position expects, so it + // answers for every kind rather than for one of them. It answers for the + // place it sits at and for nowhere below it though. Section 4.8.23 gives + // it three fields and every one of them holds a string, and of anything + // further it says "This object cannot be extended with additional + // properties, and any properties added SHALL be ignored". So no place + // named within one is an Object of any kind, and reading the Object it + // stands in for as the answer would embed something the reference never + // named + return exact && location->second.type == + sourcemeta::core::OpenAPIObjectKind::Reference; + } + + if (expected != sourcemeta::core::OpenAPIObjectKind::Schema || + prefix.empty()) { + return false; + } + + prefix = prefix.initial(); + exact = false; + } +} + +// The same place, spelled the way the document that holds it spells it. RFC +// 6901 Section 3 makes every reference token a string, and Section 4 leaves +// which of an array and an object it addresses to what it is evaluated +// against, so a +// pointer read out of a URI fragment takes a name made of digits for a place +// in an array. A pointer the walk built knows better, having been there. The +// two then spell one place two ways and compare as two, which every reference +// into a response keyed by a status code would otherwise fall foul of +auto retype(const sourcemeta::core::JSON &document, + const sourcemeta::core::Pointer &pointer) + -> sourcemeta::core::Pointer { + sourcemeta::core::Pointer result; + const auto *current{&document}; + for (const auto &token : pointer) { + if (current != nullptr && current->is_array() && token.is_index()) { + result.push_back(token.to_index()); + current = token.to_index() < current->size() + ? ¤t->at(token.to_index()) + : nullptr; + continue; + } + + auto name{token.is_property() ? token.to_property() + : sourcemeta::core::JSON::String{ + std::to_string(token.to_index())}}; + current = current != nullptr && current->is_object() ? current->try_at(name) + : nullptr; + result.push_back(std::move(name)); + } + + return result; +} + +// Where the Schema Object holding a place named within one sits. A reference +// may name a subschema, which whatever reads JSON Schema reaches through the +// Schema Object holding it rather than a place this specification types. That +// Schema Object carries the base every relative reference under it resolves +// against, so it is what has to travel +auto schema_of(const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::Pointer &target) + -> sourcemeta::core::Pointer { + auto prefix{target}; + while (true) { + const auto location{remote.locations.find( + sourcemeta::core::openapi_location_uri(remote.base, prefix))}; + if (location != remote.locations.cend() && + location->second.type == sourcemeta::core::OpenAPIObjectKind::Schema) { + return prefix; + } + + if (prefix.empty()) { + return target; + } + + prefix = prefix.initial(); + } +} + +// What the Schema Objects of another document answer to. OpenAPI Specification +// 3.2.1, Section 4.1.2.1: "Reference targets are defined by fields including +// the OpenAPI Object's `$self` field and the Schema Object's `$id`, `$anchor`, +// and `$dynamicAnchor` keywords". Neither is a document of its +// own, which is why nothing that goes looking for documents finds them, and +// 3.2.1 Section 4.1.2 leaves none of them to be given up on while the document +// that declares one has been read: "Implementations MUST NOT treat a reference +// as unresolvable before completely parsing all documents provided to the +// implementation as possible parts of the OAD" +auto index_schemas(const sourcemeta::core::JSON &remote_document, + const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::JSON::String &identifier, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &resolver, + std::map> &result, + std::uint64_t &remaining) -> void { + sourcemeta::core::SchemaFrame::Paths paths; + for (const auto &entry : remote.locations) { + if (entry.second.type == sourcemeta::core::OpenAPIObjectKind::Schema) { + paths.push_back(sourcemeta::core::to_weak_pointer(entry.second.pointer)); + } + } + + if (paths.empty()) { + return; + } + + const sourcemeta::core::SchemaFrame frame{ + sourcemeta::core::SchemaFrame::Mode::Locations, + remote_document, + walker, + resolver, + remote.dialect, + "", + sourcemeta::core::SchemaFrame::IdentifierMode::Additional, + paths, + remote.base, + remaining}; + charge(remaining, frame.location_count()); + + // What a document itself answers to is already how it is reached, and taking + // that for a place within it would embed a part of it where the whole was + // named. A document answers to the URI it was retrieved by as well as to the + // one it names itself with, and 3.2.1 Section 4.1.1 keeps the two apart, so a + // Schema Object claiming either of them is a Schema Object claiming a whole + // document + frame.for_each_resource([&identifier, &remote, &result]( + const auto &uri, const auto &location) -> void { + if (uri == remote.base || uri == identifier) { + return; + } + + result.emplace(sourcemeta::core::JSON::String{uri}, + std::make_pair(identifier, sourcemeta::core::to_pointer( + location.pointer))); + }); + + // A dynamic anchor is settled where a schema is evaluated rather than where + // it sits, so only the ones that name a place outright are places to reach + frame.for_each_anchor( + sourcemeta::core::SchemaReferenceType::Static, + [&identifier, &result](const auto &uri, const auto &location) -> void { + result.emplace(sourcemeta::core::JSON::String{uri}, + std::make_pair(identifier, sourcemeta::core::to_pointer( + location.pointer))); + }); +} + +// Which place of which document a reference naming a Schema Object identifier +// leads to, whether it names the whole of one or a place within one by a +// pointer from it +auto declared_target( + const std::map> &identifiers, + const sourcemeta::core::JSON::String &destination) + -> std::optional< + std::pair> { + const auto exact{identifiers.find(destination)}; + if (exact != identifiers.cend()) { + return exact->second; + } + + const auto resource{ + identifiers.find(sourcemeta::core::openapi_document_uri(destination))}; + if (resource == identifiers.cend()) { + return std::nullopt; + } + + // Section 4.6 reads a fragment of this shape as a JSON Pointer, and one that + // counts from an identifier a schema declares counts from where that schema + // sits rather than from the root of the document holding it + const auto tail{sourcemeta::core::fragment_to_pointer( + sourcemeta::core::URI{destination})}; + if (!tail.has_value()) { + return std::nullopt; + } + + return std::make_pair(resource->second.first, + resource->second.second.concat(tail.value())); +} + +// Which Object of the remote document a reference names, as a pointer. OpenAPI +// Specification 3.1.1, Section 4.6: "If the representation of the referenced +// document is JSON or YAML, then the fragment identifier SHOULD be interpreted +// as a JSON-Pointer as per RFC6901", and a fragment shaped like anything else +// names nothing this can embed +auto target_of(const sourcemeta::core::JSON::String &destination) + -> std::optional { + const auto pointer{sourcemeta::core::fragment_to_pointer( + sourcemeta::core::URI{destination})}; + // An empty fragment names the root of the document, which is the document + // itself just as naming no fragment at all is + if (pointer.has_value() && pointer.value().empty()) { + return std::nullopt; + } + + return pointer; +} + +} // namespace + +namespace sourcemeta::core { + +namespace { + +// Nothing below the root of a document may declare a base of its own, so an +// Object taken out of one goes on reading whatever it carries against +// whichever document it ends up in. Writing back what each of those resolved +// to where it came from is what keeps them naming the same places once that +// Object sits somewhere else +auto absolutize(JSON &value, const OpenAPIWalk &remote, const Pointer &origin, + const JSON::String &base) -> void { + for (const auto &entry : remote.references) { + if (entry.second.origin.starts_with(origin)) { + set(value, (entry.second.origin).resolve_from(origin), + JSON{rebase(entry.second.destination, base)}); + } + } + + // OpenAPI Specification 3.2.1, Section 4.30 has a Security Requirement + // Object name a Security Scheme Object by the URI of one, which is the one + // connection of a description spelled as the member that holds the scopes + // rather than as a value, so this renames rather than writes + // + // Every name of one Security Requirement Object is written back at once. + // Taken one at a time, each rename would read an Object some of whose + // members had already moved, and making room at a name means giving up + // whatever sits there, so a member could take the place of one still + // waiting its turn and carry its scopes off with it + std::map holders; + std::map> renames; + for (const auto &entry : remote.security_references) { + if (!entry.second.origin.starts_with(origin)) { + continue; + } + + auto rewritten{rebase(entry.second.destination, base)}; + if (rewritten == entry.second.original) { + continue; + } + + const auto held{entry.second.origin.initial()}; + const auto key{openapi_location_uri(remote.base, held)}; + holders.insert_or_assign(key, held); + renames[key].insert_or_assign(entry.second.original, std::move(rewritten)); + } + + for (const auto &group : renames) { + const auto &held{holders.at(group.first)}; + auto &requirement{get(value, held.resolve_from(origin))}; + auto rebuilt{JSON::make_object()}; + for (const auto &member : requirement.as_object()) { + const auto renamed{group.second.find(member.first)}; + const auto &name{renamed == group.second.cend() ? member.first + : renamed->second}; + // 3.2.1 Section 4.30 says nothing against two names of one Object leading + // to one scheme, and reading such an Object is no trouble. Writing one + // back out is what cannot keep both, as the single name they come to is a + // key that holds one list of scopes rather than two + const auto *taken{rebuilt.try_at(name)}; + if (taken != nullptr && *taken != member.second) { + throw OpenAPIError{remote.base, held.concat(JSON::String{member.first}), + "A Security Requirement Object that names one " + "security scheme twice over cannot keep a list of " + "scopes for each of them"}; + } + + rebuilt.assign(name, member.second); + } + + requirement.into(std::move(rebuilt)); + } + + // And so is every other URI it carries, which the frame does not record as a + // reference because nothing about the description hangs off where it leads + for (const auto &entry : remote.locations) { + if (!entry.second.pointer.starts_with(origin)) { + continue; + } + + const auto relative{(entry.second.pointer).resolve_from(origin)}; + + // Section 4.8.5 makes a Server Object URL a template rather than a URI + // reference, and Section 4.8.5 has a relative one name a place "relative to + // the location where the document containing the Server Object is being + // served", so it resolves as a template against the document it was + // written in + if (entry.second.type == OpenAPIObjectKind::Server) { + const auto held{relative.concat(JSON::String{"url"})}; + const auto *written{try_get(value, held)}; + if (written != nullptr && written->is_string()) { + // 3.2.1 Section 4.5.2.1 works this very case through: a document + // retrieved from one place and naming itself another resolves what it + // says of the API against where it was found rather than against the + // name it gave itself + auto address{openapi_resolve_server_url( + written->to_string(), remote.retrieval, + try_get(value, relative.concat(JSON::String{"variables"})))}; + if (!address.has_value()) { + throw OpenAPIError{ + remote.base, entry.second.pointer.concat(JSON::String{"url"}), + "A Server Object URL template that leaves what it is " + "relative to for its variables to decide cannot be read " + "from another document"}; + } + + set(value, held, JSON{std::move(address.value())}); + } + + continue; + } + + for (const auto &field : openapi_embedded_uri_fields(entry.second.type)) { + const auto held{relative.concat(JSON::String{field})}; + const auto *written{try_get(value, held)}; + if (written == nullptr || !written->is_string()) { + continue; + } + + const auto absolute{ + openapi_reference_target(written->to_string(), remote)}; + if (absolute.has_value()) { + set(value, held, JSON{absolute.value().recompose()}); + } + } + } +} + +// Lift one Object out of the document that declares it and into the entry +// document, writing back everything it carries that would otherwise go on +// resolving against a base that is no longer its own. Where it lands is what +// every reference that reaches it is then written to name +auto adopt(sourcemeta::core::JSON &document, + const sourcemeta::core::JSON &remote_document, + const sourcemeta::core::OpenAPIWalk &remote, + const sourcemeta::core::Pointer &origin, + const sourcemeta::core::JSON::StringView container, + const sourcemeta::core::JSON::String &key, + const sourcemeta::core::JSON::String &base, + const sourcemeta::core::JSON::String &dialect, + const sourcemeta::core::SchemaWalker &walker, + const sourcemeta::core::SchemaResolver &schema_resolver, + const sourcemeta::core::OpenAPIBundleOptions &options, + std::uint64_t &remaining) -> sourcemeta::core::Pointer { + auto value{*try_get(remote_document, origin)}; + absolutize(value, remote, origin, base); + + // Where this is headed is settled before anything it carries is written + // back, as a reference of its own that names a place within it by a pointer + // from the root of a document names that pointer rather than the place, and + // the pointer is what moving changes + const auto name{ + component_name(document, container, key, origin, options.namer)}; + sourcemeta::core::Pointer landing; + landing.push_back(sourcemeta::core::JSON::String{"components"}); + landing.push_back(sourcemeta::core::JSON::String{container}); + landing.push_back(name); + + absolutize_schemas(value, remote_document, remote, origin, landing, dialect, + walker, schema_resolver, remaining); + + const auto landed{embed(document, container, name, std::move(value))}; + if (options.callback) { + options.callback(key, landed); + } + + return landed; +} + +// OpenAPI Specification 3.1.1, Section 4.3.3: "It is RECOMMENDED to consider +// all Operation Objects from all parsed documents when resolving any Link +// Object `operationId`. This requires parsing all referenced documents prior +// to determining an `operationId` to be unresolvable". Every document the +// description spans is one bundling has read, so an identifier that named an +// operation of one of them named something before bundling and has to go on +// naming it afterwards. Nothing points at that operation for its own sake, so +// the Path Item holding it is brought along the way one named by an +// `operationRef` is +auto adopt_operations(JSON &document, const OpenAPIWalk &walk, + const std::map &documents, + const std::map &walks, + std::map &bundled, + const JSON::String &base, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIBundleOptions &options, + std::uint64_t &remaining) -> bool { + bool changed{false}; + for (const auto &entry : walk.operation_id_links) { + if (walk.operation_ids.contains(entry.second)) { + continue; + } + + for (const auto &other : walks) { + const auto named{other.second.operation_ids.find(entry.second)}; + if (named == other.second.operation_ids.cend()) { + continue; + } + + const auto operation{other.second.locations.find(named->second)}; + if (operation == other.second.locations.cend()) { + continue; + } + + // A Path Item Object is what holds an Operation Object, and the Object + // the Components Object has a home for + const auto origin{ + promote(path_item_of(other.second, operation->second.pointer))}; + // Keyed by what the document answers to rather than by where it was + // found, which is what every other place that fills this map uses. + // 3.2.1 Section 4.1.1 has a reference name the former, so keying by the + // latter would embed one Path Item twice over + const auto key{openapi_location_uri(other.second.base, origin)}; + if (bundled.contains(key)) { + break; + } + + bundled.emplace( + key, + adopt(document, documents.at(other.first), other.second, origin, + container_of(origin, OpenAPIObjectKind::PathItem), key, base, + walk.dialect, walker, schema_resolver, options, remaining)); + changed = true; + break; + } + } + + return changed; +} + +// Where a document declares the Tag Object of a given name, which the walk +// records as a place of its own rather than by the name it goes under +auto tag_of(const JSON &document, const OpenAPIWalk &walk, + const JSON::String &name) -> std::optional { + for (const auto &entry : walk.locations) { + if (entry.second.type != OpenAPIObjectKind::Tag) { + continue; + } + + const auto *value{try_get(document, entry.second.pointer)}; + if (value == nullptr || !value->is_object()) { + continue; + } + + const auto *declared{value->try_at("name")}; + if (declared != nullptr && declared->is_string() && + declared->to_string() == name) { + return entry.second.pointer; + } + } + + return std::nullopt; +} + +// OpenAPI Specification 3.2.1, Section 4.22, of a Tag Object's `parent`: "The +// `name` of a tag that this tag is nested under. The named tag MUST exist in +// the API description". A description is every document it spans rather than +// the entry one alone, so that tag may be one another document declares, and +// 3.2.1 Section 4.1 only ever puts a Tag Object at the root of a document. +// Bundling moves what the Components Object holds and leaves every root where +// it is, so a name that the description satisfied has to travel along to go on +// being satisfied by what bundling produces. Every document that holds one is +// one bundling has read, just as for a Link Object `operationId`, as a name is +// not something there is anywhere to go and fetch +auto adopt_tags(JSON &document, const OpenAPIWalk &walk, + const std::map &documents, + const std::map &walks, + const JSON::String &base, const OpenAPIBundleOptions &options) + -> bool { + bool changed{false}; + // What this pass has already brought in. The names the walk holds are the + // ones it read before any of them travelled, and 3.2.1 Section 4.1 has "Each + // tag name in the list MUST be unique", so two tags nested under one that + // only another document declares bring it along once between them + std::set adopted; + for (const auto &entry : walk.tag_parents) { + if (walk.tag_names.contains(entry.second.second) || + adopted.contains(entry.second.second)) { + continue; + } + + for (const auto &other : walks) { + const auto &remote_document{documents.at(other.first)}; + const auto declared{ + tag_of(remote_document, other.second, entry.second.second)}; + if (!declared.has_value()) { + continue; + } + + auto value{*try_get(remote_document, declared.value())}; + absolutize(value, other.second, declared.value(), base); + document.assign_if_missing("tags", JSON::make_array()); + auto &tags{document.at("tags")}; + if (options.callback) { + options.callback( + openapi_location_uri(other.second.base, declared.value()), + Pointer{JSON::String{"tags"}, tags.size()}); + } + + tags.push_back(std::move(value)); + adopted.insert(entry.second.second); + changed = true; + break; + } + } + + return changed; +} + +// Section 4.3 counts "the URI form of the Discriminator Object `mapping` +// field" among the fields that identify the referenced elements of a +// description, so one that names a schema of another document names a part of +// that description. The shell reaches such a schema wherever an OpenAPI +// document holds it. One that stands on its own is left here instead, as a +// mapping is an annotation to whatever reads inside a Schema Object and +// nothing there follows it. It goes under the member Section 4.8.7 reserves +// for schemas, keeping the identifier it answers to, so the mapping goes on +// naming what it always named +auto adopt_mappings( + JSON &document, + const std::map> &deferred, + std::map &adopted, + std::map &identities, const JSON::String &base, + const SchemaWalker &walker, const SchemaResolver &schema_resolver, + const JSON::StringView dialect, const OpenAPIBundleOptions &options, + std::uint64_t &remaining) -> bool { + bool changed{false}; + // What this pass brought in. Whether a mapping that named it still lands is + // for the pass after to say, as a mapping may name a place within a schema + // rather than the whole of it, and such a place is only there to be found + // once the schema holding it is + std::set fresh; + for (const auto &entry : deferred) { + const auto identifier{openapi_document_uri(entry.first)}; + // One schema answers for a mapping once, and bringing it in again would + // not make a mapping that still does not land any likelier to. That it + // still does not means what the schema says of itself is not what the + // mapping asked for, which naming where it went is what is left for. + // + // Which root that is counted from is what the mapping resolved against, + // as Section 4.6 has one inside a Schema Object that declares an + // identifier count from there rather than from the document + const auto previous{adopted.find(identifier)}; + if (previous != adopted.cend()) { + if (fresh.contains(identifier)) { + continue; + } + + for (const auto &pending : entry.second) { + JSON named{pending.scope == base + ? to_uri(previous->second).recompose() + : openapi_location_uri(base, previous->second)}; + const auto *written{try_get(document, pending.origin)}; + if (written != nullptr && *written != named) { + set(document, pending.origin, std::move(named)); + changed = true; + } + } + + continue; + } + + auto resolved{schema_resolver(identifier)}; + if (!resolved.has_value()) { + continue; + } + + // 3.2.1 Section 4.1.2 has a document of a description hold either an + // OpenAPI Object or a Schema Object at its root, and a mapping names the + // second of those. Reading the first as one is what Appendix G leaves + // undefined + if (openapi_is_document(resolved.value())) { + throw OpenAPIReferenceError{ + base, entry.second.front().origin, identifier, + "This mapping must name a schema rather than a document that holds " + "an OpenAPI Description"}; + } + + auto schema{std::move(resolved).to_owned()}; + // JSON Schema 2020-12 Section 4.3: "A JSON Schema MUST be an object or a + // boolean", and nothing that reads one can make sense of anything else. + // What the entry document holds is held to this before it is read, and + // what a resolver hands back is no different + if (!schema.is_object() && !schema.is_boolean()) { + throw OpenAPIReferenceError{ + base, entry.second.front().origin, identifier, + "A Schema Object must be an object or a boolean"}; + } + + // What a schema answers to is its own to say. JSON Schema 2020-12 + // Section 8.2.1 has an identifier a schema declares be "its canonical + // [RFC6596] URI", and the base every relative reference + // under it resolves against, so taking the URI it happened to be fetched + // by over the one it declares would re-aim every one of those. Only a + // schema that declares none is given the one it was found under, which is + // what lets a mapping that named a place within it go on naming that place. + // + // What it declares is a URI reference rather than a URI though, and + // 3.2.1 Section 4.1.2.2 resolves one of those: "The most common base URI + // source that is used in the event of a missing or relative `$self` (in the + // OpenAPI Object) and (for Schema Object) `$id` is the retrieval URI". So + // the URI it was fetched by is what a relative one is read against, which + // is what makes the identifier that comes back one the description can use + std::optional root; + try { + root.emplace(SchemaFrame::Mode::Root, schema, walker, schema_resolver, + JSON::String{dialect}, identifier, + SchemaFrame::IdentifierMode::Additional, + SchemaFrame::Paths{EMPTY_WEAK_POINTER}, identifier, + remaining); + } catch (const SchemaUnknownBaseDialectError &) { + // What a schema is written against is what says how to read it, and + // Section 4.8.24.5 leaves the dialect to whatever the schema names. One + // that names a dialect nothing here can place is one nothing here can + // read, which is a different complaint from naming the wrong kind of + // document altogether + throw OpenAPIReferenceError{ + base, entry.second.front().origin, identifier, + "This mapping must name a schema written against a dialect that is " + "possible to determine"}; + } + + charge(remaining, root.value().location_count()); + const auto &declared{root.value().root()}; + const JSON::String identity{declared.empty() ? identifier + : JSON::String{declared}}; + + // A schema answers to what it declares rather than to the URI it happened + // to be fetched by, so two mappings reaching one resource by two spellings + // reach one schema. Landing it a second time would leave the description + // holding one identifier in two places, which is what whoever frames the + // result turns down + // What a schema declares is its own, and what it was fetched by is the + // description's. Reading one against the other would take a mapping whose + // URI happens to spell what another schema calls itself for a second + // mention of that schema, so the two are kept apart + const auto same{identities.find(identity)}; + if (same != identities.cend()) { + adopted.emplace(identifier, same->second); + continue; + } + // Section 4.8.24.5: "For standalone JSON Schema documents that do not set + // `$schema` [...] the dialect SHOULD be assumed to be the OAS dialect" + spell_out(schema); + if (!schema.defines("$schema")) { + schema.assign("$schema", JSON{dialect}); + } + + schema_reidentify(schema, identity, + root.value().root_location().value().get().base_dialect); + + const auto *schemas{openapi_component_container_of(document, "schemas"sv)}; + const auto name{ + schema_name(schemas == nullptr ? JSON::make_object() : *schemas, + identifier, options.namer)}; + const auto landed{embed(document, "schemas"sv, name, std::move(schema))}; + adopted.emplace(identifier, landed); + identities.emplace(identity, landed); + fresh.insert(identifier); + if (options.callback) { + options.callback(identifier, landed); + } + + changed = true; + } + + return changed; +} + +auto bundle_internal(JSON &document, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIResolver &resolver, + const OpenAPIBundleOptions &options, + std::uint64_t &remaining) -> void { + // Where the document was retrieved from, which is only what it answers to + // until it says otherwise. OpenAPI Specification 3.2.1, Section 4.1 lets one + // name itself with `$self`, "which also serves as its base URI", so what + // every place of this document is named by is what its own analysis settled + // on rather than what the caller handed over + const auto retrieval{openapi_canonical_base(options.default_base)}; + // Where each component that bundling embedded ended up, keyed by the place + // it came from. A description that reaches for one place twice embeds it + // once, which is also what keeps a cycle of documents from going round + std::map bundled; + // A document that more than one reference reaches is read once. What it + // holds cannot change between one reference and the next, and reading it + // again would charge the allowance twice for the same thing + std::map documents; + std::map walks; + // The documents that the shell cannot reach, which only a Schema Object may + // name and which it is left free to go on naming + std::set unavailable; + // What every Schema Object of a document already read answers to, beside the + // document holding it and where in it it sits. A reference naming one of + // these names a place rather than a document, so nothing that goes looking + // for documents would ever find it + std::map> identifiers; + // And of those, the schemas that a Discriminator Object mapping is what + // names, which nothing but this brings in, along with the ones it already did + std::map> deferred; + std::map adopted; + // And where each of them ended up, against the identifier it declares for + // itself rather than the URI it was reached by. JSON Schema 2020-12 Section + // 8.2.1 has the first be "its canonical [RFC6596] URI" and says nothing of + // what is served at the second, so one schema reached by two URIs is told + // from two schemas that happen to spell each other's names + std::map identities; + // The names the description gave before bundling moved anything. Section + // 4.1.2.3 resolves the names a referenced document uses from the entry + // document, and bundling embedding a Security Scheme Object puts a name + // there that the description never had. Letting that answer for a document + // read afterwards would decide, by nothing but how many references away it + // sits, that an operation requires a credential its own document never named + OpenAPIWalk names; + bool named{false}; + + while (true) { + const auto walk{openapi_analyse(document, retrieval, remaining)}; + const auto &base{walk.base}; + if (!named) { + names.security_schemes = walk.security_schemes; + names.tags = walk.tags; + names.tag_names = walk.tag_names; + named = true; + } + + charge(remaining, walk.locations.size()); + auto unresolved{pending(walk)}; + // A Schema Object may name one that another document of the description + // declares, which the shell is what reaches rather than anything that + // reads inside a Schema Object + schema_pending(document, walk, walker, schema_resolver, unresolved, + remaining); + + bool changed{false}; + for (const auto &reference : unresolved) { + // Only the shell knows which documents an OpenAPI Description spans. A + // Schema Object naming something this does not reach is left exactly as + // it was written, for whatever reads inside one to resolve + const auto names_a_schema{reference.expected == + OpenAPIObjectKind::Schema}; + + // Where a reference leads before the document holding it has been read, + // which is all a fragment shaped like a pointer ever needs + const auto pointed{target_of(reference.destination)}; + // Which document it leads to, if it names one at all + const auto holder{openapi_document_uri(reference.destination)}; + + // What another document's Schema Object declares for itself is a name + // for a place of that document, so a reference naming one is a reference + // into it rather than one to go looking for a document of that name. + // + // Whatever the description can reach as a document answers ahead of + // that, as 3.2.1 Section 4.1.1 has a reference name a document by the URI + // that document answers to, and a Schema Object is free to declare an + // identifier that collides with one. So this is asked where a fragment + // names no place to begin with, and otherwise only once the document has + // turned out not to be one + auto declared{names_a_schema && (!pointed.has_value() || + unavailable.contains(holder)) + ? declared_target(identifiers, reference.destination) + : std::nullopt}; + + const auto initial{declared.has_value() + ? std::optional{declared.value().second} + : pointed}; + // A fragment shaped like anything else names an identifier or an anchor + // rather than a place, and which document declares one is only settled + // by framing the documents the description spans. 3.2.1 Section 4.1.2 + // leaves none of those to be given up on before that has happened, so a + // reference of this shape goes on to have its document read rather than + // being left here + const URI destination{reference.destination}; + const auto names_an_anchor{names_a_schema && !initial.has_value() && + destination.fragment().has_value() && + !destination.fragment().value().empty()}; + + if (!initial.has_value() && !names_an_anchor) { + if (names_a_schema) { + if (reference.mapping) { + deferred[reference.destination].push_back(reference); + } + + continue; + } + + throw OpenAPIReferenceError{base, reference.origin, + reference.destination, + "This reference must name a place within " + "the document it points at"}; + } + + const auto identifier{declared.has_value() ? declared.value().first + : holder}; + + if (names_a_schema && !declared.has_value() && + unavailable.contains(identifier)) { + continue; + } + + if (!documents.contains(identifier)) { + auto resolved{resolver(identifier)}; + if (!resolved.has_value()) { + if (names_a_schema) { + unavailable.insert(identifier); + if (reference.mapping) { + deferred[reference.destination].push_back(reference); + } + + continue; + } + + throw OpenAPIResolutionError{ + base, reference.origin, identifier, + "Could not resolve the reference to an external document"}; + } + + auto candidate{std::move(resolved).to_owned()}; + // OpenAPI Specification 3.2.1, Section 4.1.2: "all documents in an + // OAD MUST have either an OpenAPI Object or a Schema Object at the + // root, and MUST be parsed as complete documents". A Schema Object + // at the root is what a Schema Object reference may name, and that + // document is one a JSON Schema implementation reads rather than + // this. No revision of 3.1 carries that sentence, and what it carries + // instead is the choice Section 4.3.1 leaves open, so declining to + // read a document of some other shape is one rule for both revisions + if (!openapi_is_document(candidate)) { + if (names_a_schema) { + unavailable.insert(identifier); + if (reference.mapping) { + deferred[reference.destination].push_back(reference); + } + + continue; + } + + throw OpenAPIReferenceError{base, reference.origin, identifier, + "This reference must name a document " + "that holds an OpenAPI Description"}; + } + + // Nothing in the specification asks for this. It says what a + // description may span and says nothing of the revisions the documents + // it spans declare, so turning one of these down is a choice this + // makes rather than a rule it follows. + // + // What makes the choice is where bundling ends. Section 4.1 has one + // document declare one revision, so everything the result holds has to + // be what that one revision can express, and there is no revision to + // pick that can express both. A 3.2 document holds fields that 3.1 has + // no way of spelling, and 3.2.1 Section 2.1 leaves no room to assume + // the other direction is safe either: "Occasionally, non-backwards + // compatible changes may be made in `minor` versions of the OAS where + // impact is believed to be low relative to the benefit provided". So + // the choice is between turning such a description down and handing + // back a document that says it is one revision while holding what + // another one means. + // + // What the patch component says is no part of this, as 3.2.1 + // Section 2.1 makes a revision the `major`.`minor` pair alone, + // which 3.1.1 says the same of under Section 4.1 + const auto revision{openapi_version(candidate)}; + if (revision.has_value() && revision.value() != walk.version) { + throw OpenAPIReferenceError{ + base, reference.origin, identifier, + "This reference must name a document of the same OpenAPI " + "Specification revision"}; + } + + // The walk keeps a view into the document it read, so the document + // takes its place before anything walks it + const auto &held{ + documents.emplace(identifier, std::move(candidate)).first->second}; + auto analysis{openapi_analyse(held, identifier, remaining, &names)}; + charge(remaining, analysis.locations.size()); + const auto &recorded{ + walks.emplace(identifier, std::move(analysis)).first->second}; + index_schemas(held, recorded, identifier, walker, schema_resolver, + identifiers, remaining); + } + + // A fragment shaped like anything but a pointer names an identifier or + // an anchor rather than a place, which only framing the document that + // declares it settles. 3.2.1 Section 4.1.2 leaves none of those to be + // given up on before that document has been read, so this is asked again + // now that it has been. A fragment that is a pointer already named a + // place of the document just read, and nothing another document declares + // for itself is to take that place from it + if (names_a_schema && !declared.has_value() && !pointed.has_value()) { + declared = declared_target(identifiers, reference.destination); + } + + // A document that holds an OpenAPI Description is never what a reference + // from the shell expects to find, so one naming a whole document has + // landed on the wrong kind of thing + const auto target{declared.has_value() + ? std::optional{declared.value().second} + : target_of(reference.destination)}; + if (!target.has_value()) { + if (names_a_schema) { + if (reference.mapping) { + deferred[reference.destination].push_back(reference); + } + + continue; + } + + throw OpenAPIReferenceError{base, reference.origin, + reference.destination, + "This reference must name a place within " + "the document it points at"}; + } + + const auto &remote_document{documents.at(identifier)}; + const auto &remote{walks.at(identifier)}; + // Spelled as the document that holds it spells it, before anything + // compares it against what the walk of that document recorded + const auto spelled{retype(remote_document, target.value())}; + // What a reference names is held to the kind the position it sits in + // expects, however many references reach the Object that holds it. A + // second one that named nothing would otherwise be written out as a + // place the result does not hold + if (!lands(remote, remote.base, spelled, reference.expected)) { + throw OpenAPIReferenceError{ + base, reference.origin, reference.destination, + "This reference must name an Object of the kind that the " + "position it sits in expects"}; + } + + // An Operation Object is the one kind the Components Object has no home + // for, so what gets embedded is the Path Item Object holding it and the + // reference goes on reaching its operation through that + const auto names_an_operation{reference.expected == + OpenAPIObjectKind::Operation}; + const auto origin{promote(names_an_operation + ? path_item_of(remote, spelled) + : names_a_schema ? schema_of(remote, spelled) + : spelled)}; + const auto container{ + container_of(origin, names_an_operation ? OpenAPIObjectKind::PathItem + : reference.expected)}; + // Every kind that a reference position expects has a home of its own + // once an Operation Object is reached through the Path Item holding it + assert(!container.empty()); + + // Where a document was retrieved from is how a resolver is asked for it, + // and what it answers to is its own to say. OpenAPI Specification 3.2.1, + // Section 4.1.1 lets a document declare the latter and requires the two + // to be told apart: "references MUST use the target document's `$self` + // URI if the `$self` field is present". So every place of it is named by + // the base its own analysis settled on rather than by where it was found + const auto key{openapi_location_uri(remote.base, origin)}; + + if (!bundled.contains(key)) { + bundled.emplace(key, adopt(document, remote_document, remote, origin, + container, key, base, walk.dialect, walker, + schema_resolver, options, remaining)); + } + + // The reference is rewritten as a fragment of the document it now sits + // in, rather than as the URI that document answers to, so that bundling + // leaves behind an output that keeps working wherever it is moved to + const auto landed{spelled.rebase(origin, bundled.at(key))}; + + // What a Security Requirement Object names is the member the scopes sit + // under, so what makes it whole is renaming that member to whatever the + // scheme is called once it sits here, which carries the scopes across + // untouched. 3.2.1 Section 4.30 then reads the result as a component name + // rather than as a URI: "Property names that are identical to a + // component name under the Components Object MUST be treated as a + // component name", and the name was taken free of that Object for it + if (reference.requirement) { + auto &requirement{get(document, reference.origin.initial())}; + const auto &member{reference.origin.back().to_property()}; + JSON::String name{landed.back().to_property()}; + const auto *taken{requirement.try_at(name)}; + if (taken != nullptr && *taken != requirement.at(member)) { + throw OpenAPIError{base, reference.origin, + "A Security Requirement Object that names one " + "security scheme twice over cannot keep a list of " + "scopes for each of them"}; + } + + requirement.rename(member, std::move(name)); + changed = true; + continue; + } + + // A reference that resolves against the document it sits in is written + // as a fragment of it, which is what keeps what bundling produces + // working wherever it is moved to. One that resolves against something + // else, which is a Schema Object that declares an identifier of its own, + // would name a place within that identifier instead, so it cannot be + // written that way + if (reference.scope == base) { + set(document, reference.origin, JSON{to_uri(landed).recompose()}); + changed = true; + continue; + } + + // What such a reference names instead is what the schema it leads to + // says of itself, wherever that schema says anything. JSON Schema + // Section 9.3.1 has a reference to an embedded resource "resolve to a + // schema using the `$id` of an embedded Schema Resource", and an + // identifier holds wherever the description ends up while a place in a + // document does not. Only a schema that declares none leaves the + // document itself as the sole way to name where it leads + const auto *reached{try_get(document, landed)}; + const auto *identity{reached == nullptr || !reached->is_object() + ? nullptr + : reached->try_at("$id")}; + set(document, reference.origin, + identity != nullptr && identity->is_string() + ? JSON{identity->to_string()} + : JSON{openapi_location_uri(base, landed)}); + changed = true; + } + + // A pass that leaves the document as it found it is one that has nothing + // left to bring in, which counts a reference left for a JSON Schema + // implementation as nothing + if (!changed) { + // What the shell reaches is settled before this, as an operation is + // brought in for the sake of a name rather than of a reference and only + // the documents a reference reached are ones to look through + if (adopt_operations(document, walk, documents, walks, bundled, base, + walker, schema_resolver, options, remaining)) { + continue; + } + + // And so is the tag a Tag Object is nested under, which is a name the + // description settles rather than a reference to go and follow + if (adopt_tags(document, walk, documents, walks, base, options)) { + continue; + } + + // And so is what a mapping names, which is settled after the references + // are, as bringing one schema in may be what lets the next be read + if (adopt_mappings(document, deferred, adopted, identities, base, walker, + schema_resolver, openapi_dialect(walk.version), + options, remaining)) { + deferred.clear(); + continue; + } + + // What is left is one document that holds every OpenAPI document the + // description spanned, as the only ones bundling leaves out are the ones + // a Schema Object may name, which hold no Tag Object and no Operation + // Object to name. So the names that Section 4.3.3 has resolve across the + // whole of a description are ones there is now an answer for, and + // leaving a description that has none to be turned down by whoever + // frames it next would be to hand back a bundle that does not describe + // anything + openapi_check_operation_id_links(walk, walk.locations); + openapi_check_tag_parents(walk, walk.locations, true); + // And so are the path parameters a templated path corresponds to, which + // the projection is what settles. What it works out is of no use here, + // as bundling moves what a description holds rather than reporting what + // it exposes, but a description that cannot be projected is one this + // would otherwise hand back for the next reader to turn down + [[maybe_unused]] const auto operations{openapi_project(walk)}; + + // What sits inside a Schema Object is JSON Schema's to bring in, and a + // schema that lands may hold a Discriminator Object naming another, + // which nothing has read yet. So this goes round once more rather than + // being the last thing that happens. + // + // Going round again is safe as well as needed. Every schema that lands + // says who it is, and the one kind that cannot, a boolean, has whatever + // named it written out to name where it went, so a second pass finds + // nothing left outside to ask for rather than landing a second copy + if (bundle_schemas(document, walk, walker, schema_resolver, base, + remaining, options)) { + continue; + } + + return; + } + } +} + +} // namespace + +auto openapi_bundle(JSON &document, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIResolver &resolver, + const OpenAPIBundleOptions &options) -> void { + auto remaining{options.max_locations}; + try { + bundle_internal(document, walker, schema_resolver, resolver, options, + remaining); + } catch (const OpenAPIFrameLimitError &) { + throw OpenAPIBundleLimitError{options.max_locations}; + } catch (const SchemaFrameLimitError &) { + // Every frame spends from what is left rather than from the whole, so the + // one that ran out reports what it was handed. The caller set the + // allowance for the operation, so that is what the operation reports back + throw OpenAPIBundleLimitError{options.max_locations}; + } +} + +auto openapi_bundle(const JSON &document, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIResolver &resolver, + const OpenAPIBundleOptions &options) -> JSON { + JSON copy{document}; + openapi_bundle(copy, walker, schema_resolver, resolver, options); + return copy; +} + +} // namespace sourcemeta::core diff --git a/vendor/core/src/core/openapi/components.h b/vendor/core/src/core/openapi/components.h index 04b68e008..ea18c9c71 100644 --- a/vendor/core/src/core/openapi/components.h +++ b/vendor/core/src/core/openapi/components.h @@ -55,7 +55,7 @@ inline auto openapi_collect_security_schemes(const JSON &document, } for (const auto &entry : schemes->as_object()) { - walk.security_schemes.insert(entry.first); + walk.security_schemes.insert(entry.first, entry.hash); } } @@ -83,6 +83,41 @@ inline auto openapi_is_component_key(const JSON::StringView key) noexcept return !key.empty(); } +// The Components Object member that holds each kind of Object a reference may +// name, which is where bundling puts what it embeds, and nothing for a kind +// that the Components Object has no home for. OpenAPI Specification 3.1.1, +// Section 4.8.7 gives one member per referenceable kind but none for an +// Operation Object, which is reached through the Path Item Object holding it +inline auto openapi_component_container(const OpenAPIObjectKind kind) noexcept + -> JSON::StringView { + switch (kind) { + case OpenAPIObjectKind::Schema: + return "schemas"sv; + case OpenAPIObjectKind::Response: + return "responses"sv; + case OpenAPIObjectKind::Parameter: + return "parameters"sv; + case OpenAPIObjectKind::Example: + return "examples"sv; + case OpenAPIObjectKind::RequestBody: + return "requestBodies"sv; + case OpenAPIObjectKind::Header: + return "headers"sv; + case OpenAPIObjectKind::SecurityScheme: + return "securitySchemes"sv; + case OpenAPIObjectKind::Link: + return "links"sv; + case OpenAPIObjectKind::Callbacks: + return "callbacks"sv; + case OpenAPIObjectKind::PathItem: + return "pathItems"sv; + case OpenAPIObjectKind::MediaType: + return "mediaTypes"sv; + default: + return {}; + } +} + // OpenAPI Specification 3.1.1, Section 4.8.7: "Holds a set of reusable objects // for different aspects of the OAS". What each entry of those maps holds is // not read here, and the Schema Objects under `schemas` are never read at all, diff --git a/vendor/core/src/core/openapi/content.h b/vendor/core/src/core/openapi/content.h index 452838f99..87c723e4c 100644 --- a/vendor/core/src/core/openapi/content.h +++ b/vendor/core/src/core/openapi/content.h @@ -99,11 +99,14 @@ inline auto openapi_check_nested_encoding(const JSON &value, OpenAPIWalk &walk) -> void { const auto *named{value.try_at("encoding", OPENAPI_HASH_ENCODING)}; - // Section 4.14, of `encoding`: "This field MUST NOT be present if + // 3.2.1 Section 4.14, of `encoding`: "This field MUST NOT be present if // `prefixEncoding` or `itemEncoding` are present", and each of those two - // says the same of `encoding` in turn. Section 4.15 defines all three by - // reference to that Object, and the published meta-schema holds the pair to - // the same rule in both places + // says the same of `encoding` in turn. 3.2.1 Section 4.15 defines all three + // by reference to that Object and states no exclusion of its own, so what + // settles the nested pair is the corpus, which files both + // `encoding-enc-prefix-exclusion` and `encoding-enc-item-exclusion` as + // failing, and the published meta-schema, which holds the pair to the same + // rule in both places if (named != nullptr && (value.try_at("prefixEncoding", OPENAPI_HASH_PREFIX_ENCODING) != nullptr || @@ -148,8 +151,9 @@ inline auto openapi_check_encoding(const JSON &value, const Pointer &base, value, OPENAPI_ENCODING_FIELDS_3_1, OPENAPI_ENCODING_FIELDS_3_2, base, "The Encoding Object does not define this field", walk); - // The specification says media type definitions "SHOULD be in compliance - // with RFC6838", which is not a requirement, so only the type is checked + // 3.1.1 Section 3.6 says media type definitions "SHOULD be in compliance + // with RFC6838", which is not a requirement, and 3.2 drops the sentence + // rather than strengthening it, so only the type is checked openapi_check_optional_string( value, base, "contentType"sv, OPENAPI_HASH_CONTENT_TYPE, "The Encoding Object content type must be a string"); @@ -207,8 +211,8 @@ inline auto openapi_check_media_type(const JSON &value, const Pointer &base, "The Media Type Object example and examples are mutually exclusive", "The Media Type Object examples must be an object", walk); - // Section 4.14: "itemSchema | Schema Object", for a sequential media type, - // which is a fifth position where framing hands off to JSON Schema + // 3.2.1 Section 4.14: "itemSchema | Schema Object", for a sequential media + // type, which is a fifth position where framing hands off to JSON Schema const auto *item_schema{value.try_at("itemSchema", OPENAPI_HASH_ITEM_SCHEMA)}; if (item_schema != nullptr) { openapi_expect_schema(*item_schema, openapi_child(base, "itemSchema"sv), @@ -232,8 +236,9 @@ inline auto openapi_check_media_type_or_reference(const JSON &value, } // The map that the Request Body, Response, Parameter and Header Objects all -// key by media type. Section 4.5 says those definitions "SHOULD be in -// compliance with RFC6838", so the keys carry no requirement to enforce. +// key by media type. 3.1.1 Section 3.6 says those definitions "SHOULD be in +// compliance with RFC6838", which carries no requirement to enforce, and 3.2 +// drops the sentence rather than strengthening it. // OpenAPI Specification 3.2.1 widens what the values may be in all four of // those Objects, from "Map[string, Media Type Object]" to "Map[string, Media // Type Object | Reference Object]", which is what its Components Object entry @@ -262,8 +267,11 @@ inline auto openapi_check_header(const JSON &value, const Pointer &base, const auto *schema{value.try_at("schema", OPENAPI_HASH_SCHEMA)}; const auto *content{value.try_at("content", OPENAPI_HASH_CONTENT)}; - // OpenAPI Specification 3.1.1, Section 4.8.21: "The `schema` field and - // `content` field are mutually exclusive", and one of them has to be there + // OpenAPI Specification 3.1.1, Section 4.8.21 has "The Header Object + // follows the structure of the Parameter Object", and none of the changes it + // lists touches either field, so Section 4.8.12's rule reaches here whole: + // "Parameter Objects MUST include either a `content` field or a `schema` + // field, but not both" if (schema != nullptr && content != nullptr) { throw OpenAPIError{ base, "The Header Object schema and content are mutually exclusive"}; @@ -313,7 +321,8 @@ inline auto openapi_check_header(const JSON &value, const Pointer &base, const auto location{openapi_child(base, "content"sv)}; openapi_check_content(*content, location, "The Header Object content must be an object", walk); - // The meta-schema bounds this map at one entry in both directions + // Section 4.8.21: "The map MUST only contain one entry", which an empty + // map answers no better than a crowded one if (content->size() != 1) { throw OpenAPIError{ location, "The Header Object content must hold exactly one entry"}; diff --git a/vendor/core/src/core/openapi/discriminator.h b/vendor/core/src/core/openapi/discriminator.h index 38d6b6bd6..af61860e4 100644 --- a/vendor/core/src/core/openapi/discriminator.h +++ b/vendor/core/src/core/openapi/discriminator.h @@ -19,22 +19,7 @@ constexpr auto OPENAPI_HASH_DISCRIMINATOR{ constexpr auto OPENAPI_HASH_MAPPING{JSON::Object::hash("mapping"sv)}; constexpr auto OPENAPI_HASH_DEFAULT_MAPPING{ JSON::Object::hash("defaultMapping"sv)}; - -/// Where a Discriminator Object names a schema, by the name of a component or -/// by URI. OpenAPI Specification 3.1.1, Section 4.3 lists the URI form of a -/// `mapping` among the fields that connect the documents of a description, and -/// Section 4.3.3 lists the name form among the connections it makes by name, -/// so either way one of these is a place the description reaches for -struct OpenAPIDiscriminator { - /// Where the mapping value sits, as a pointer from the root of the document - Pointer origin; - /// Where it points, resolved and canonicalised - JSON::String destination; - /// What it resolved against, which is the nearest identifier an enclosing - /// schema declares for the URI form, and the description itself for the - /// name form - JSON::String scope; -}; +constexpr auto OPENAPI_HASH_PROPERTY_NAME{JSON::Object::hash("propertyName"sv)}; // OpenAPI Specification 3.1.1, Section 4.8.25: a `mapping` entry "maps a // specific property value to either a different schema component name, or to a @@ -153,6 +138,26 @@ openapi_discriminators(const JSON &document, const SchemaFrame &schemas, const JSON::String scope{location.base}; origin.push_back(JSON::String{"discriminator"}); + // Section 4.8.25 marks this one "**REQUIRED**. The name of the property in + // the payload that will hold the discriminating value", and 3.2.1 Section + // 4.25 says as much, so a Discriminator Object the dialect does define + // carries one or it is no such Object. The dialect 3.2 publishes leaves it + // out of its own list of required fields where the one 3.1 publishes keeps + // it, and Section 4.8 leaves the text authoritative where the two differ + const auto *property_name{ + discriminator->try_at("propertyName", OPENAPI_HASH_PROPERTY_NAME)}; + if (property_name == nullptr) { + throw OpenAPIError{ + base, std::move(origin), + "The Discriminator Object must declare a property name"}; + } + + if (!property_name->is_string()) { + throw OpenAPIError{ + base, origin.concat(JSON::String{"propertyName"}), + "The Discriminator Object property name must be a string"}; + } + const auto *mapping{discriminator->try_at("mapping", OPENAPI_HASH_MAPPING)}; if (mapping != nullptr && mapping->is_object()) { const auto mapped{origin.concat(JSON::String{"mapping"})}; @@ -168,7 +173,7 @@ openapi_discriminators(const JSON &document, const SchemaFrame &schemas, // that revision defines the field, which is what settles whether there is // one to read rather than what the document says of itself. // - // Section 4.25.1 goes on to require one wherever the discriminating + // 3.2.1 Section 4.25.1 goes on to require one wherever the discriminating // property is optional, which is a demand on what the schema holding it // says of its own properties. Reading that far into a Schema Object is // the business of whatever understands JSON Schema, so it is left there diff --git a/vendor/core/src/core/openapi/document.h b/vendor/core/src/core/openapi/document.h index 1f64766ed..7228b6084 100644 --- a/vendor/core/src/core/openapi/document.h +++ b/vendor/core/src/core/openapi/document.h @@ -23,11 +23,14 @@ #include "tag.h" #include // std::array +#include // std::size_t #include // std::uint64_t #include // std::numeric_limits +#include // std::map #include // std::optional #include // std::string_view #include // std::move, std::swap, std::unreachable +#include // std::vector namespace sourcemeta::core { @@ -41,7 +44,15 @@ constexpr auto OPENAPI_DIALECT_3_1{ // URI instead: it "is identified by the URI of the form // `https://spec.openapis.org/oas/3.2/dialect/YYYY-MM-DD` [...] see the list of // current schemas for the specific URI". One date is published for 3.2, and it -// is the one this repository already resolves +// is the one this repository already resolves. +// +// A document that declares 3.2.0 gets this one as well, although that patch +// reads "identified by the URI +// `https://spec.openapis.org/oas/3.1/dialect/base`" and so names the dialect of +// the revision before it. 3.2.1 Section 2.1 makes a revision the +// `major`.`minor` pair alone, so what a later patch of one says is what the +// whole of it says, and the earlier wording is a mistake that patch corrects +// rather than a rule of its own constexpr auto OPENAPI_DIALECT_3_2{ "https://spec.openapis.org/oas/3.2/dialect/2025-09-17"sv}; @@ -81,15 +92,20 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) // What a document's `$self` establishes as its base, or nothing when it // establishes none. OpenAPI Specification 3.2.1, Section 4.1: the field // "provides the self-assigned URI of this document, which also serves as its -// base URI in accordance with RFC3986 Section 5.1.1", and Section 4.1.2.2.1: -// "If -// `$self` is a relative URI reference, it is resolved against the next -// possible base URI source before being used". That next source is whatever -// base is in force here, which is the retrieval URI for the entry document and -// the URI a reference named for any other. RFC 3986 Section 5.2.1 has only the -// scheme required of a base, so a relative `$self` with nothing absolute to -// resolve against establishes nothing, and Section 5.2.2 never resolves -// against a fragment, so one written here is no part of the base either +// base URI in accordance with RFC3986 Section 5.1.1", and 3.2.1 +// Section 4.1.2.2.1: "If `$self` is a relative URI reference, it is resolved +// against the next possible base URI source ([RFC3986] Section 5.1.2 +// [...] 5.1.4) before being used for the resolution of other relative URI +// references". That source is whatever base is in force here, which is the +// retrieval URI for the entry document and the URI a reference named for any +// other. RFC 3986 +// Section 5.2.1 has only the scheme required of a base, so a relative `$self` +// with nothing absolute to resolve against establishes nothing, and +// Section 5.2.2 never resolves against a fragment, so one written here is +// dropped rather than read. The specification's own published schema turns such +// a `$self` down outright, which 3.2.1 Section 4 makes it no place to: "If the +// JSON Schema differs from this section, then this section MUST be considered +// authoritative", and the section it differs from asks only for a URI reference inline auto openapi_document_base(const JSON::StringView self, const OpenAPIWalk &walk) -> std::optional { @@ -271,7 +287,10 @@ inline auto openapi_follow_reference(const JSON::StringView reference, walk.references.insert_or_assign( openapi_location_uri(walk.base, origin.initial()), OpenAPIReference{.original = JSON::String{reference}, - .destination = target.value().recompose()}); + .destination = target.value().recompose(), + .dangling = false, + .expected = expected, + .origin = origin}); // OpenAPI Specification 3.2.1, Section 4.1.2: "all documents in an OAD MUST // have either an OpenAPI Object or a Schema Object at the root". A Schema @@ -281,7 +300,16 @@ inline auto openapi_follow_reference(const JSON::StringView reference, // so a reference that names such a document whole has landed on the wrong // thing whatever kind it expected. Section 4.8.9 and Section 4.8.20 say as // much of the two positions they speak of, and the rest follows from what a - // document may hold rather than from what those two sections single out + // document may hold rather than from what those two sections single out. + // + // No revision of 3.1 says that much, so this holds one of its documents to + // a rule its own text does not carry. What it does carry is a choice: + // Section 4.3.1 of 3.1.1 reads "Implementations MAY support complete-document + // parsing in any of the following ways", one of which is "Detecting a + // document containing a referenceable Object at its root based on the + // expected type of the reference". Reading a whole document as the Object a + // reference wants is what that permits and what this declines, which leaves + // one rule for both revisions rather than a 3.1 that takes what 3.2 forbids openapi_follow_target(target.value(), origin, expected, walk); } @@ -325,8 +353,8 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) document, OPENAPI_ROOT_FIELDS_3_1, OPENAPI_ROOT_FIELDS_3_2, EMPTY_POINTER, "The OpenAPI Object does not define this field", walk); - // Section 4.1: "$self | string | This string MUST be in the form of a URI - // reference as defined by RFC3986 Section 4.1". Only 3.2 defines the + // 3.2.1 Section 4.1: "$self | string | This string MUST be in the form of a + // URI reference as defined by RFC3986 Section 4.1". Only 3.2 defines the // field, and the table above has already turned it down for anything // earlier. What it establishes is the base that every location in this // document is keyed by, so it is settled before anything records one @@ -337,29 +365,30 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) "The OpenAPI Description self identifier must be a string", "The OpenAPI Description self identifier must be a URI reference")}; - // The specification's own published schema for this revision spells the - // field `{"format": "uri-reference", "pattern": "^[^#]*$"}` and says - // why in a comment of its own: + // A fragment written here is admitted, which the specification's own + // published schema for this revision is stricter than. That schema + // spells the field `{"format": "uri-reference", "pattern": "^[^#]*$"}` + // and gives its reason in a comment of its own: // // MUST NOT contain a fragment // - // which RFC 3986 Section 5.1 agrees with, a base URI carrying none. The - // pattern turns down the character rather than a fragment component, so - // an empty one is refused here too - if (reference.find('#') != JSON::StringView::npos) { - throw OpenAPIError{ - walk.base, openapi_child(EMPTY_POINTER, "$self"sv), - "The OpenAPI Description self identifier must not contain a " - "fragment"}; - } - + // but 3.2.1 Section 4 settles which of the two answers for this: "This + // text is the only normative description of the format. A JSON Schema is + // hosted on spec.openapis.org for informational purposes. If the JSON + // Schema differs from this section, then this section MUST be considered + // authoritative". The text asks only for a URI reference, and RFC 3986 + // Section 4.1 admits a fragment in one, so the pattern is a rule the + // normative prose does not carry and is not enforced here. Nothing is + // lost by taking it, as Section 5.2.2 never resolves a reference against + // a fragment, which is why what a fragment names is dropped rather than + // read auto established{openapi_document_base(reference, walk)}; if (established.has_value()) { walk.base = std::move(established.value()); - // Section 4.1.1: "To ensure interoperability, references MUST use the - // target document's `$self` URI if the `$self` field is present". So - // this is the URI the document answers to, and one that names it by + // 3.2.1 Section 4.1.1: "To ensure interoperability, references MUST use + // the target document's `$self` URI if the `$self` field is present". + // So this is the URI the document answers to, and one that names it by // where it was retrieved from instead names another document, which // the same paragraph calls "not interoperable" and NOT RECOMMENDED } @@ -377,12 +406,30 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) // Object", and Section 4.3.3 recommends the entry document for the same // reason it does for security schemes, so both sets of names come from // there and both come before anything else is read - openapi_collect_security_schemes(document, walk); - openapi_collect_tags(document, walk); + // + // A document the description reaches takes those names from the entry + // document and none of its own. Adding its own would let it name a scheme + // that the description it becomes part of does not declare, which is a + // requirement that reads fine here and cannot be met once the Object + // holding it sits in the document that does describe the API + if (!walk.referenced) { + openapi_collect_security_schemes(document, walk); + openapi_collect_tags(document, walk); + } - // Section 3.1: an OpenAPI Description "MUST contain at least one paths - // field, components field, or webhooks field" - if (document.try_at("paths", OPENAPI_HASH_PATHS) == nullptr && + // OpenAPI Specification 3.2.1, Section 4.1 binds every document that holds + // an OpenAPI Object: "In addition to the required fields, at least one of + // the `components`, `paths`, or `webhooks` fields MUST be present". + // + // 3.1.1, Section 3.1 binds the description instead, and names what it is + // made of while doing so: "An OpenAPI Description (OAD) [...] is composed + // of an entry document [...] and any/all of its referenced documents [...] + // and MUST contain at least one `paths` field, `components` field, or + // `webhooks` field". 3.1.0 asked it of a document and 3.1.1 moved the + // subject, so a document of that revision that another one reaches is free + // to hold none of the three as long as the description holds one + if ((!walk.referenced || walk.version == OpenAPIVersion::OPENAPI_3_2) && + document.try_at("paths", OPENAPI_HASH_PATHS) == nullptr && document.try_at("components", OPENAPI_HASH_COMPONENTS) == nullptr && document.try_at("webhooks", OPENAPI_HASH_WEBHOOKS) == nullptr) { throw OpenAPIError{ @@ -412,7 +459,8 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) // Section 4.8.24.1: "To allow use of a different default `$schema` value // for all Schema Objects contained within an OAS document, a // `jsonSchemaDialect` value may be set within the OpenAPI Object. If this - // default is not set, then the OAS dialect schema id MUST be used". What a + // default is not set, then the OAS dialect schema id MUST be used for + // these Schema Objects". What a // Schema Object says about itself overrides this, which is a matter for // whatever reads inside one JSON::String effective_dialect{openapi_dialect(walk.version)}; @@ -460,15 +508,129 @@ inline auto openapi_check_document(const JSON &document, OpenAPIWalk &walk) } } +// OpenAPI Specification 3.2.1, Section 4.22, of a Tag Object's `parent`: "The +// named tag MUST exist in the API description, and circular references between +// parent and child tags MUST NOT be used". A description spans every document +// it references, so a parent naming a tag this one does not declare is only +// missing where nothing is missing, and a cycle is a property of the tags held +// rather than of any one tag +inline auto openapi_check_tag_parents(const OpenAPIWalk &walk, + const OpenAPIFrame::Locations &locations, + const bool whole) -> void { + std::map parents; + for (const auto &[location, edge] : walk.tag_parents) { + if (whole && !walk.tag_names.contains(edge.second)) { + throw openapi_error_at(locations, location, + "The Tag Object parent must name a tag the " + "OpenAPI Description declares", + "parent"sv); + } + + parents.insert_or_assign(edge.first, edge.second); + } + + // Walking upward from each tag terminates at a tag with no parent unless the + // chain comes back round, and a chain longer than the number of edges has + // come back round + for (const auto &[location, edge] : walk.tag_parents) { + auto name{edge.first}; + for (std::size_t step = 0; step <= parents.size(); step += 1) { + const auto next{parents.find(name)}; + if (next == parents.cend()) { + break; + } + + name = next->second; + if (step == parents.size()) { + throw openapi_error_at(locations, location, + "The Tag Object parents must not form a cycle", + "parent"sv); + } + } + } +} + +// OpenAPI Specification 3.1.1, Section 4.8.20: "The identified or reference +// operation MUST be unique, and in the case of an `operationId`, it MUST be +// resolved within the scope of the OpenAPI Description". Section 4.3.3 goes on +// that "This requires parsing all referenced documents prior to determining an +// `operationId` to be unresolvable", so nothing is decided here until every +// document of the description is held at once +inline auto openapi_check_operation_id_links( + const OpenAPIWalk &walk, const OpenAPIFrame::Locations &locations) -> void { + for (const auto &[location, identifier] : walk.operation_id_links) { + // Section 4.8.20 goes on to say that an operation reached through a Path + // Item referenced more than once "cannot be resolved unambiguously", and + // that "in such ambiguous cases, the resulting behavior is + // implementation-defined and MAY result in an error". So naming nothing at + // all is the violation, and naming something twice over is not + if (!walk.operation_ids.contains(identifier)) { + throw openapi_error_at(locations, location, + "The Link Object operation identifier must name " + "an operation the OpenAPI Description declares", + "operationId"sv); + } + } +} + +// OpenAPI Specification 3.1.1, Section 4.8.24 hands a Schema Object over to +// whatever reads JSON Schema, whole and on its own. One sitting within another +// is a place that implementation would be handed twice over, once by itself +// and once as part of something larger, which is not a reading it has any +// account of. Which places those are is settled by the walk as a whole, so +// this cannot be decided while one is still being read +inline auto openapi_check_schema_positions(const OpenAPIWalk &walk) -> void { + for (const auto &location : walk.locations) { + if (location.second.type != OpenAPIObjectKind::Schema) { + continue; + } + + // Which place holds this one is what the walk recorded of each place + // above it rather than anything the order of them suggests. A location is + // held under a key that sorts as a string, and Section 4.8.7 admits both + // `-` and `.` into a component name, either of which falls below the `/` + // that separates a place from what sits within it. So a sibling named + // that way comes between an Object and its own contents, which is why + // this asks after every place above rather than the one just read + auto prefix{location.second.pointer}; + while (!prefix.empty()) { + prefix.pop_back(); + const auto enclosing{ + walk.locations.find(openapi_location_uri(walk.base, prefix))}; + if (enclosing != walk.locations.cend() && + enclosing->second.type == OpenAPIObjectKind::Schema) { + throw OpenAPIError{ + walk.base, location.second.pointer, + "A Schema Object must not sit within another Schema Object"}; + } + } + } +} + +// Every operation the description exposes, worked out from a walk that has +// settled. OpenAPI Specification 3.1.1, Section 4.8.9 has a templated path +// correspond to the path parameters the Path Item Object and its operations +// declare, and which parameters those are is only settled once every Path Item +// the description reaches is at hand, so this is where that is decided +auto openapi_project(const OpenAPIWalk &walk) -> std::vector; + // Everything the checks need in order to start from nothing, which is a walk // of the given document keyed by the given base. A 3.2 document may name // itself, so what the walk ends up keyed by is what it reports rather than // what it was handed +// Section 4.3.3: "For resolving component and tag name connections from a +// referenced (non-entry) document, it is RECOMMENDED that tools resolve from +// the entry document, rather than the current document. This allows Security +// Scheme Objects and Tag Objects to be defined next to the API's deployment +// information [...] and treated as an interface for referenced documents to +// access". A document read on its own has no entry document to resolve from, +// so what one names is settled by whoever reads it as part of a description inline auto openapi_analyse(const JSON &document, JSON::String base, const std::uint64_t max_locations = - std::numeric_limits::max()) - -> OpenAPIWalk { - OpenAPIWalk walk{.base = std::move(base), + std::numeric_limits::max(), + const OpenAPIWalk *entry = nullptr) -> OpenAPIWalk { + OpenAPIWalk walk{.base = base, + .retrieval = std::move(base), .document = &document, .operation_ids = {}, .visited = {}, @@ -482,6 +644,7 @@ inline auto openapi_analyse(const JSON &document, JSON::String base, .servers = {}, .security = {}, .security_schemes = {}, + .security_references = {}, .tags = {}, .tag_parents = {}, .tag_names = {}, @@ -491,7 +654,21 @@ inline auto openapi_analyse(const JSON &document, JSON::String base, .info = {}, .remaining = max_locations, .limit = max_locations}; + // The names of the entry document are in scope before this document's own + // are read, as what it declares itself adds to them rather than replaces + // them + if (entry != nullptr) { + walk.referenced = true; + walk.security_schemes = entry->security_schemes; + // Both of what a tag name settles come from there too, as the names a + // parent may claim and the Tag Objects an operation resolves to are two + // readings of one set rather than two sets + walk.tags = entry->tags; + walk.tag_names = entry->tag_names; + } + openapi_check_document(document, walk); + openapi_check_schema_positions(walk); return walk; } diff --git a/vendor/core/src/core/openapi/example.h b/vendor/core/src/core/openapi/example.h index 9235260cc..9598141e6 100644 --- a/vendor/core/src/core/openapi/example.h +++ b/vendor/core/src/core/openapi/example.h @@ -56,8 +56,8 @@ inline auto openapi_check_example(const JSON &value, const Pointer &base, base, "The Example Object data value and value are mutually exclusive"}; } - // Section 4.19: "serializedValue | string | An example of the serialized - // form of the value [...] If this field is present, `value`, and + // 3.2.1 Section 4.19: "serializedValue | string | An example of the + // serialized form of the value [...] If this field is present, `value`, and // `externalValue` MUST be absent". The `externalValue` field states the // other half of that pair the same way const auto *serialized{ @@ -110,8 +110,8 @@ inline auto openapi_check_example_or_reference(const JSON &value, } // The Parameter, Media Type and Header Objects all carry this pair, and all -// three state that "The `example` field is mutually exclusive of the -// `examples` field" +// three state that "The `example` and `examples` fields are mutually +// exclusive" inline auto openapi_check_examples(const JSON &value, const Pointer &base, const char *exclusive_message, const char *type_message, OpenAPIWalk &walk) diff --git a/vendor/core/src/core/openapi/format.cc b/vendor/core/src/core/openapi/format.cc new file mode 100644 index 000000000..3eaea6dc4 --- /dev/null +++ b/vendor/core/src/core/openapi/format.cc @@ -0,0 +1,530 @@ +#include +#include + +#include "components.h" +#include "content.h" +#include "document.h" +#include "example.h" +#include "external_documentation.h" +#include "helpers.h" +#include "info.h" +#include "link.h" +#include "parameter.h" +#include "path_item.h" +#include "request_body.h" +#include "response.h" +#include "security.h" +#include "server.h" +#include "tag.h" + +#include // std::ranges::find +#include // std::array, std::to_array +#include // assert +#include // std::size_t +#include // std::uint16_t +#include // std::distance +#include // std::numeric_limits +#include // std::span +#include // std::unreachable + +namespace sourcemeta::core { + +namespace { + +// The order in which the fields of each Object are meant to appear, where the +// position of an entry is the rank of the field it names. Each table unions +// what 3.1 and 3.2 define, and what every variant of an Object defines, so +// nothing here branches on the revision a document declares. +// +// An entry of `x-` stands for every Specification Extension rather than for a +// field of that name, and it sits where it does because such a member is almost +// always an annotation. A kind that declares neither identity nor metadata +// therefore ranks it first, which is also where its size would put it +constexpr auto FIELDS_DOCUMENT{std::to_array( + {"openapi", "$self", "jsonSchemaDialect", "info", "x-", "externalDocs", + "security", "servers", "tags", "webhooks", "paths", "components"})}; + +constexpr auto FIELDS_INFO{std::to_array( + {"title", "version", "summary", "description", "termsOfService", "x-", + "contact", "license"})}; + +constexpr auto FIELDS_CONTACT{ + std::to_array({"name", "x-", "url", "email"})}; + +constexpr auto FIELDS_LICENSE{ + std::to_array({"name", "identifier", "x-", "url"})}; + +constexpr auto FIELDS_SERVER{std::to_array( + {"name", "description", "x-", "url", "variables"})}; + +constexpr auto FIELDS_SERVER_VARIABLE{ + std::to_array({"description", "x-", "default", "enum"})}; + +constexpr auto FIELDS_COMPONENTS{std::to_array( + {"x-", "responses", "parameters", "examples", "requestBodies", "mediaTypes", + "headers", "securitySchemes", "links", "callbacks", "pathItems", + "schemas"})}; + +constexpr auto FIELDS_PATH_ITEM{std::to_array( + {"summary", "description", "x-", "$ref", "servers", "parameters", "get", + "query", "head", "post", "put", "patch", "delete", "options", "trace", + "additionalOperations"})}; + +constexpr auto FIELDS_OPERATION{std::to_array( + {"operationId", "summary", "description", "externalDocs", "deprecated", + "x-", "tags", "security", "servers", "parameters", "requestBody", + "responses", "callbacks"})}; + +constexpr auto FIELDS_PARAMETER{std::to_array( + {"name", "in", "description", "required", "deprecated", "example", + "examples", "x-", "style", "explode", "allowEmptyValue", "allowReserved", + "schema", "content"})}; + +constexpr auto FIELDS_REQUEST_BODY{std::to_array( + {"description", "required", "x-", "content"})}; + +constexpr auto FIELDS_RESPONSE{std::to_array( + {"summary", "description", "x-", "headers", "links", "content"})}; + +constexpr auto FIELDS_EXAMPLE{std::to_array( + {"summary", "description", "x-", "externalValue", "value", "dataValue", + "serializedValue"})}; + +constexpr auto FIELDS_HEADER{std::to_array( + {"description", "required", "deprecated", "example", "examples", "x-", + "style", "explode", "schema", "content"})}; + +constexpr auto FIELDS_LINK{std::to_array( + {"operationId", "operationRef", "description", "x-", "parameters", "server", + "requestBody"})}; + +constexpr auto FIELDS_SECURITY_SCHEME{std::to_array( + {"type", "description", "deprecated", "x-", "name", "in", "scheme", + "bearerFormat", "openIdConnectUrl", "oauth2MetadataUrl", "flows"})}; + +constexpr auto FIELDS_OAUTH_FLOWS{std::to_array( + {"x-", "implicit", "password", "clientCredentials", "authorizationCode", + "deviceAuthorization"})}; + +constexpr auto FIELDS_OAUTH_FLOW{std::to_array( + {"x-", "authorizationUrl", "deviceAuthorizationUrl", "tokenUrl", + "refreshUrl", "scopes"})}; + +constexpr auto FIELDS_TAG{ + std::to_array({"name", "parent", "kind", "summary", + "description", "externalDocs", "x-"})}; + +constexpr auto FIELDS_EXTERNAL_DOCS{ + std::to_array({"description", "x-", "url"})}; + +constexpr auto FIELDS_ENCODING{std::to_array( + {"x-", "contentType", "style", "explode", "allowReserved", "headers", + "encoding", "prefixEncoding", "itemEncoding"})}; + +constexpr auto FIELDS_MEDIA_TYPE{std::to_array( + {"example", "examples", "x-", "encoding", "prefixEncoding", "itemEncoding", + "schema", "itemSchema"})}; + +// OpenAPI Specification 3.1.1, Section 4.8.23: "This object cannot be extended +// with additional properties, and any properties added SHALL be ignored". +// Ignoring is what the specification asks for, so framing lets such a member +// through and this is the one table that is not total +constexpr auto FIELDS_REFERENCE{ + std::to_array({"summary", "description", "$ref"})}; + +// The fields of each Object whose value is a map the description's author keys, +// so that what it holds is sorted rather than ranked. A map that the frame +// reports as an Object of its own is absent from these, as it sorts itself +constexpr auto MAPS_DOCUMENT{std::to_array({"webhooks"})}; + +constexpr auto MAPS_SERVER{std::to_array({"variables"})}; + +constexpr auto MAPS_COMPONENTS{std::to_array( + {"schemas", "responses", "parameters", "examples", "requestBodies", + "headers", "securitySchemes", "links", "callbacks", "pathItems", + "mediaTypes"})}; + +constexpr auto MAPS_PATH_ITEM{ + std::to_array({"additionalOperations"})}; + +constexpr auto MAPS_OPERATION{std::to_array({"callbacks"})}; + +constexpr auto MAPS_PARAMETER{ + std::to_array({"examples", "content"})}; + +constexpr auto MAPS_REQUEST_BODY{std::to_array({"content"})}; + +constexpr auto MAPS_RESPONSE{ + std::to_array({"headers", "links", "content"})}; + +constexpr auto MAPS_HEADER{ + std::to_array({"examples", "content"})}; + +constexpr auto MAPS_LINK{std::to_array({"parameters"})}; + +constexpr auto MAPS_OAUTH_FLOW{std::to_array({"scopes"})}; + +constexpr auto MAPS_ENCODING{ + std::to_array({"headers", "encoding"})}; + +constexpr auto MAPS_MEDIA_TYPE{ + std::to_array({"examples", "encoding"})}; + +// A table is total when it ranks every field the specification defines for the +// Object, which is what the arrays framing holds a document to spell out. So a +// document that framed cannot hold a field that reaches no rank. +// +// These two are immediate functions rather than merely constant ones because +// the assertions below are the only callers there will ever be. Asking the +// compiler to hold them to that leaves no runtime symbol for anything to +// wonder why the tests never reach +template +consteval auto +ranks_every_field(const std::array &fields, + const std::array &admitted) + -> bool { + for (const auto &field : admitted) { + if (std::ranges::find(fields, field) == fields.cend()) { + return false; + } + } + + return true; +} + +// And it names nothing else, which is what catches a table misspelling a field +// into a rank that nothing ever reaches +template +consteval auto +defines_every_rank(const std::array &fields, + const Admitted &...admitted) -> bool { + for (const auto &field : fields) { + if (field == OPENAPI_EXTENSION_PREFIX) { + continue; + } + + if (!(... || (std::ranges::find(admitted, field) != admitted.cend()))) { + return false; + } + } + + return true; +} + +static_assert(ranks_every_field(FIELDS_DOCUMENT, OPENAPI_ROOT_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_DOCUMENT, OPENAPI_ROOT_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_DOCUMENT, OPENAPI_ROOT_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_INFO, OPENAPI_INFO_FIELDS)); +static_assert(defines_every_rank(FIELDS_INFO, OPENAPI_INFO_FIELDS)); + +static_assert(ranks_every_field(FIELDS_CONTACT, OPENAPI_CONTACT_FIELDS)); +static_assert(defines_every_rank(FIELDS_CONTACT, OPENAPI_CONTACT_FIELDS)); + +static_assert(ranks_every_field(FIELDS_LICENSE, OPENAPI_LICENSE_FIELDS)); +static_assert(defines_every_rank(FIELDS_LICENSE, OPENAPI_LICENSE_FIELDS)); + +static_assert(ranks_every_field(FIELDS_SERVER, OPENAPI_SERVER_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_SERVER, OPENAPI_SERVER_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_SERVER, OPENAPI_SERVER_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_SERVER_VARIABLE, + OPENAPI_SERVER_VARIABLE_FIELDS)); +static_assert(defines_every_rank(FIELDS_SERVER_VARIABLE, + OPENAPI_SERVER_VARIABLE_FIELDS)); + +static_assert(ranks_every_field(FIELDS_COMPONENTS, + OPENAPI_COMPONENTS_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_COMPONENTS, + OPENAPI_COMPONENTS_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_COMPONENTS, + OPENAPI_COMPONENTS_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_PATH_ITEM, + OPENAPI_PATH_ITEM_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_PATH_ITEM, + OPENAPI_PATH_ITEM_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_PATH_ITEM, + OPENAPI_PATH_ITEM_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_OPERATION, OPENAPI_OPERATION_FIELDS)); +static_assert(defines_every_rank(FIELDS_OPERATION, OPENAPI_OPERATION_FIELDS)); + +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_CONTENT_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_CONTENT_FIELDS_3_2)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_SCHEMA_FIELDS)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_RESERVED_SCHEMA_FIELDS_3_2)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_QUERY_SCHEMA_FIELDS)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_QUERY_CONTENT_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_PARAMETER, + OPENAPI_PARAMETER_QUERY_CONTENT_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_PARAMETER, + OPENAPI_PARAMETER_QUERY_SCHEMA_FIELDS, + OPENAPI_PARAMETER_QUERY_CONTENT_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_REQUEST_BODY, + OPENAPI_REQUEST_BODY_FIELDS)); +static_assert(defines_every_rank(FIELDS_REQUEST_BODY, + OPENAPI_REQUEST_BODY_FIELDS)); + +static_assert(ranks_every_field(FIELDS_RESPONSE, OPENAPI_RESPONSE_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_RESPONSE, OPENAPI_RESPONSE_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_RESPONSE, OPENAPI_RESPONSE_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_EXAMPLE, OPENAPI_EXAMPLE_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_EXAMPLE, OPENAPI_EXAMPLE_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_EXAMPLE, OPENAPI_EXAMPLE_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_HEADER, OPENAPI_HEADER_SCHEMA_FIELDS)); +static_assert(ranks_every_field(FIELDS_HEADER, + OPENAPI_HEADER_CONTENT_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_HEADER, + OPENAPI_HEADER_CONTENT_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_HEADER, OPENAPI_HEADER_SCHEMA_FIELDS, + OPENAPI_HEADER_CONTENT_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_LINK, OPENAPI_LINK_FIELDS)); +static_assert(defines_every_rank(FIELDS_LINK, OPENAPI_LINK_FIELDS)); + +static_assert(ranks_every_field(FIELDS_SECURITY_SCHEME, + OPENAPI_SECURITY_SCHEME_APIKEY_FIELDS_3_2)); +static_assert(ranks_every_field( + FIELDS_SECURITY_SCHEME, OPENAPI_SECURITY_SCHEME_HTTP_BEARER_FIELDS_3_2)); +static_assert(ranks_every_field(FIELDS_SECURITY_SCHEME, + OPENAPI_SECURITY_SCHEME_OAUTH2_FIELDS_3_2)); +static_assert(ranks_every_field(FIELDS_SECURITY_SCHEME, + OPENAPI_SECURITY_SCHEME_OIDC_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_SECURITY_SCHEME, + OPENAPI_SECURITY_SCHEME_APIKEY_FIELDS_3_2, + OPENAPI_SECURITY_SCHEME_HTTP_BEARER_FIELDS_3_2, + OPENAPI_SECURITY_SCHEME_OAUTH2_FIELDS_3_2, + OPENAPI_SECURITY_SCHEME_OIDC_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_OAUTH_FLOWS, + OPENAPI_OAUTH_FLOWS_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_OAUTH_FLOWS, + OPENAPI_OAUTH_FLOWS_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_OAUTH_FLOWS, + OPENAPI_OAUTH_FLOWS_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_OAUTH_FLOW, + OPENAPI_OAUTH_FLOW_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_OAUTH_FLOW, + OPENAPI_OAUTH_FLOW_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_OAUTH_FLOW, + OPENAPI_OAUTH_FLOW_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_TAG, OPENAPI_TAG_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_TAG, OPENAPI_TAG_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_TAG, OPENAPI_TAG_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_EXTERNAL_DOCS, + OPENAPI_EXTERNAL_DOCS_FIELDS)); +static_assert(defines_every_rank(FIELDS_EXTERNAL_DOCS, + OPENAPI_EXTERNAL_DOCS_FIELDS)); + +static_assert(ranks_every_field(FIELDS_ENCODING, OPENAPI_ENCODING_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_ENCODING, OPENAPI_ENCODING_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_ENCODING, OPENAPI_ENCODING_FIELDS_3_2)); + +static_assert(ranks_every_field(FIELDS_MEDIA_TYPE, + OPENAPI_MEDIA_TYPE_FIELDS_3_1)); +static_assert(ranks_every_field(FIELDS_MEDIA_TYPE, + OPENAPI_MEDIA_TYPE_FIELDS_3_2)); +static_assert(defines_every_rank(FIELDS_MEDIA_TYPE, + OPENAPI_MEDIA_TYPE_FIELDS_3_2)); + +// A map a table names has to be a field the same Object ranks, or the lookup +// that sorts it would never find anything +static_assert(ranks_every_field(FIELDS_DOCUMENT, MAPS_DOCUMENT)); +static_assert(ranks_every_field(FIELDS_SERVER, MAPS_SERVER)); +static_assert(ranks_every_field(FIELDS_COMPONENTS, MAPS_COMPONENTS)); +static_assert(ranks_every_field(FIELDS_PATH_ITEM, MAPS_PATH_ITEM)); +static_assert(ranks_every_field(FIELDS_OPERATION, MAPS_OPERATION)); +static_assert(ranks_every_field(FIELDS_PARAMETER, MAPS_PARAMETER)); +static_assert(ranks_every_field(FIELDS_REQUEST_BODY, MAPS_REQUEST_BODY)); +static_assert(ranks_every_field(FIELDS_RESPONSE, MAPS_RESPONSE)); +static_assert(ranks_every_field(FIELDS_HEADER, MAPS_HEADER)); +static_assert(ranks_every_field(FIELDS_LINK, MAPS_LINK)); +static_assert(ranks_every_field(FIELDS_OAUTH_FLOW, MAPS_OAUTH_FLOW)); +static_assert(ranks_every_field(FIELDS_ENCODING, MAPS_ENCODING)); +static_assert(ranks_every_field(FIELDS_MEDIA_TYPE, MAPS_MEDIA_TYPE)); + +// What ordering a kind of Object asks for: how to rank its fields, and which of +// them hold a map whose keys the description's author chose +struct KindRow { + std::span fields; + std::span maps; + bool sorts_own_keys{false}; +}; + +auto row_of(const OpenAPIObjectKind kind) -> KindRow { + switch (kind) { + case OpenAPIObjectKind::Document: + return {.fields = FIELDS_DOCUMENT, .maps = MAPS_DOCUMENT}; + case OpenAPIObjectKind::Info: + return {.fields = FIELDS_INFO, .maps = {}}; + case OpenAPIObjectKind::Contact: + return {.fields = FIELDS_CONTACT, .maps = {}}; + case OpenAPIObjectKind::License: + return {.fields = FIELDS_LICENSE, .maps = {}}; + case OpenAPIObjectKind::Server: + return {.fields = FIELDS_SERVER, .maps = MAPS_SERVER}; + case OpenAPIObjectKind::ServerVariable: + return {.fields = FIELDS_SERVER_VARIABLE, .maps = {}}; + case OpenAPIObjectKind::Components: + return {.fields = FIELDS_COMPONENTS, .maps = MAPS_COMPONENTS}; + case OpenAPIObjectKind::PathItem: + return {.fields = FIELDS_PATH_ITEM, .maps = MAPS_PATH_ITEM}; + case OpenAPIObjectKind::Operation: + return {.fields = FIELDS_OPERATION, .maps = MAPS_OPERATION}; + case OpenAPIObjectKind::Parameter: + return {.fields = FIELDS_PARAMETER, .maps = MAPS_PARAMETER}; + case OpenAPIObjectKind::RequestBody: + return {.fields = FIELDS_REQUEST_BODY, .maps = MAPS_REQUEST_BODY}; + case OpenAPIObjectKind::Response: + return {.fields = FIELDS_RESPONSE, .maps = MAPS_RESPONSE}; + case OpenAPIObjectKind::Example: + return {.fields = FIELDS_EXAMPLE, .maps = {}}; + case OpenAPIObjectKind::Header: + return {.fields = FIELDS_HEADER, .maps = MAPS_HEADER}; + case OpenAPIObjectKind::Link: + return {.fields = FIELDS_LINK, .maps = MAPS_LINK}; + case OpenAPIObjectKind::SecurityScheme: + return {.fields = FIELDS_SECURITY_SCHEME, .maps = {}}; + case OpenAPIObjectKind::OAuthFlows: + return {.fields = FIELDS_OAUTH_FLOWS, .maps = {}}; + case OpenAPIObjectKind::OAuthFlow: + return {.fields = FIELDS_OAUTH_FLOW, .maps = MAPS_OAUTH_FLOW}; + case OpenAPIObjectKind::Tag: + return {.fields = FIELDS_TAG, .maps = {}}; + case OpenAPIObjectKind::ExternalDocumentation: + return {.fields = FIELDS_EXTERNAL_DOCS, .maps = {}}; + case OpenAPIObjectKind::Encoding: + return {.fields = FIELDS_ENCODING, .maps = MAPS_ENCODING}; + case OpenAPIObjectKind::MediaType: + return {.fields = FIELDS_MEDIA_TYPE, .maps = MAPS_MEDIA_TYPE}; + case OpenAPIObjectKind::Reference: + return {.fields = FIELDS_REFERENCE, .maps = {}}; + // The four Objects this specification defines as a map alone, whose keys + // are a path template, a status code, a runtime expression and the name of + // a security scheme. There is no field to rank, so these sort themselves + case OpenAPIObjectKind::Paths: + case OpenAPIObjectKind::Responses: + case OpenAPIObjectKind::Callbacks: + case OpenAPIObjectKind::SecurityRequirement: + return {.fields = {}, .maps = {}, .sorts_own_keys = true}; + // Handled before this is ever asked + case OpenAPIObjectKind::Schema: + return {}; + } + + std::unreachable(); +} + +// Byte order on the key, which is the whole of what a map whose keys the author +// chose needs. In a Responses Object it groups each class of status code with +// the range that generalises it, so `200` and `201` sort before `2XX` and that +// whole group sorts before `300`, while `default` lands after every code and +// every range because a letter follows every digit, and the extensions land +// last. In a Paths Object it puts every path before the extensions beside them, +// as a path begins with a slash +auto compare_keys(const JSON::String &left, const JSON::String &right) -> bool { + return left < right; +} + +// The rank of a field within a kind's table, where every Specification +// Extension takes the rank of the placeholder entry that stands for all of +// them. A field the table does not name sorts last, which only a Reference +// Object can reach +struct FieldComparison { + std::span fields; + + [[nodiscard]] auto rank(const JSON::String &field) const -> std::uint16_t { + constexpr auto UNRECOGNISED{std::numeric_limits::max()}; + const auto match{std::ranges::find( + this->fields, field.starts_with(OPENAPI_EXTENSION_PREFIX) + ? OPENAPI_EXTENSION_PREFIX + : JSON::StringView{field})}; + if (match == this->fields.end()) { + return UNRECOGNISED; + } + + return static_cast( + std::distance(this->fields.begin(), match)); + } + + auto operator()(const JSON::String &left, const JSON::String &right) const + -> bool { + const auto left_rank{this->rank(left)}; + const auto right_rank{this->rank(right)}; + if (left_rank == right_rank) { + return left < right; + } + + return left_rank < right_rank; + } +}; + +auto format_object(JSON &document, const OpenAPIFrame::Location &location) + -> void { + // What sits inside a Schema Object is JSON Schema's to order, and the pass + // over the schemas has already done it. Ranking one here against an empty + // table would put its keywords in alphabetical order and undo that + if (location.type == OpenAPIObjectKind::Schema) { + return; + } + + auto &object{get(document, location.pointer)}; + // A position the frame reports may hold something other than an object, as a + // Path Item Object holding nothing but a reference is read as the Object it + // stands in for + if (!object.is_object()) { + return; + } + + const auto row{row_of(location.type)}; + if (row.sorts_own_keys) { + object.reorder(compare_keys); + return; + } + + object.reorder(FieldComparison{.fields = row.fields}); + for (const auto &field : row.maps) { + auto *map{object.try_at(field)}; + if (map != nullptr && map->is_object()) { + map->reorder(compare_keys); + } + } +} + +} // namespace + +auto openapi_format(JSON &document, const OpenAPIFrame &frame) -> void { + assert(document.is_object()); + + // OpenAPI Specification 3.1.1, Section 4.3 leaves a Schema Object to JSON + // Schema, so ordering one is that implementation's to do. It goes first + // because the frame of those schemas holds pointers that borrow this + // document's own property names, and reordering an Object that holds a schema + // would leave them naming something else. Nothing this specification defines + // sits inside a Schema Object, so every schema position is deeper than the + // Object holding it and doing this first is enough to be safe + schema_format(document, frame.schemas()); + + // Unlike the schema frame above, these locations own the pointers they + // report, so reordering one Object leaves the rest addressable and no order + // among them is required + frame.for_each_object( + [&document](const auto &, const auto &location) -> void { + format_object(document, location); + }); +} + +} // namespace sourcemeta::core diff --git a/vendor/core/src/core/openapi/frame.cc b/vendor/core/src/core/openapi/frame.cc index d202a20a4..ba2a6303b 100644 --- a/vendor/core/src/core/openapi/frame.cc +++ b/vendor/core/src/core/openapi/frame.cc @@ -1,4 +1,5 @@ #include +#include #include "discriminator.h" #include "document.h" @@ -13,6 +14,7 @@ #include // std::optional #include // std::set #include // std::string_view +#include // std::tuple #include // std::move, std::pair, std::unreachable #include // std::vector @@ -22,8 +24,7 @@ using namespace std::string_view_literals; // A parent is given as one of the keys a location is held under rather than as // a bare pointer, so that following it is a lookup in the same map rather than // a key the reader has to rebuild -auto parent_of(const std::map &locations, +auto parent_of(const sourcemeta::core::OpenAPIFrame::Locations &locations, const sourcemeta::core::JSON::String &uri, const sourcemeta::core::OpenAPILocation &location) -> sourcemeta::core::JSON { @@ -44,103 +45,13 @@ auto parent_of(const std::map &locations, - const sourcemeta::core::JSON::String &location, - const char *message, - const sourcemeta::core::JSON::StringView field = {}) - -> sourcemeta::core::OpenAPIError { - const auto match{locations.find(location)}; - auto pointer{match == locations.cend() ? sourcemeta::core::EMPTY_POINTER - : match->second.pointer}; - if (!field.empty()) { - pointer = pointer.concat(sourcemeta::core::JSON::String{field}); - } - - return {sourcemeta::core::openapi_document_uri(location), std::move(pointer), - message}; -} - auto error_at(const sourcemeta::core::OpenAPIWalk &walk, const sourcemeta::core::JSON::String &location, const char *message, const sourcemeta::core::JSON::StringView field = {}) -> sourcemeta::core::OpenAPIError { - return error_at(walk.locations, location, message, field); -} - -// OpenAPI Specification 3.2.1, Section 4.22, of a Tag Object's `parent`: -// "The named tag MUST exist in the API description, and circular references -// between parent and child tags MUST NOT be used". A description spans every -// document it references, so a parent naming a tag this one does not declare -// is only missing where nothing is missing, and a cycle is a property of the -// tags held rather than of any one tag -auto check_tag_parents( - const sourcemeta::core::OpenAPIWalk &walk, - const std::map &locations, - const bool whole) -> void { - std::map - parents; - for (const auto &[location, edge] : walk.tag_parents) { - if (whole && !walk.tag_names.contains(edge.second)) { - throw error_at(locations, location, - "The Tag Object parent must name a tag the OpenAPI " - "Description declares", - "parent"); - } - - parents.insert_or_assign(edge.first, edge.second); - } - - // Walking upward from each tag terminates at a tag with no parent unless the - // chain comes back round, and a chain longer than the number of edges has - // come back round - for (const auto &[location, edge] : walk.tag_parents) { - auto name{edge.first}; - for (std::size_t step = 0; step <= parents.size(); step += 1) { - const auto next{parents.find(name)}; - if (next == parents.cend()) { - break; - } - - name = next->second; - if (step == parents.size()) { - throw error_at(locations, location, - "The Tag Object parents must not form a cycle", - "parent"); - } - } - } -} - -// OpenAPI Specification 3.1.1, Section 4.8.20: "The identified or reference -// operation MUST be unique, and in the case of an `operationId`, it MUST be -// resolved within the scope of the OpenAPI Description". Section 4.3.3 -// recommends resolving one "considering all Operation Objects from all parsed -// documents", and only one document is ever parsed, which is why nothing is -// decided here unless the frame stands alone -auto check_operation_id_links( - const sourcemeta::core::OpenAPIWalk &walk, - const std::map &locations) -> void { - for (const auto &[location, identifier] : walk.operation_id_links) { - // Section 4.8.20 goes on to say that an operation reached through a Path - // Item referenced more than once "cannot be resolved unambiguously", and - // that "in such ambiguous cases, the resulting behavior is - // implementation-defined and MAY result in an error". So naming nothing at - // all is the violation, and naming something twice over is not - if (!walk.operation_ids.contains(identifier)) { - throw error_at(locations, location, - "The Link Object operation identifier must name an " - "operation the OpenAPI Description declares", - "operationId"); - } - } + return sourcemeta::core::openapi_error_at(walk.locations, location, message, + field); } auto info_json(const sourcemeta::core::OpenAPIInfo &info) @@ -200,9 +111,12 @@ auto version_string(const sourcemeta::core::OpenAPIVersion version) // A Reference Object, and a Path Item Object that declares a `$ref`, stand in // for what they lead to. OpenAPI Specification 3.1.1, Section 4.8.9 has `$ref` -// "allow for a referenced definition of this path item" and leaves what a -// sibling field means undefined, so the definition is what the reference leads -// to rather than anything written alongside it +// "Allows for a referenced definition of this path item", and leaves undefined +// only what "appears both in the defined object and the referenced object", so +// what the reference leads to is where a place is looked for first. A field +// written beside it that the referenced Path Item Object does not declare is +// an ordinary field of the Object holding it, which the projection reads back +// rather than this auto follow_aliases(const sourcemeta::core::OpenAPIWalk &walk, const sourcemeta::core::JSON::String &position) -> sourcemeta::core::JSON::String { @@ -216,6 +130,25 @@ auto identity_of(const sourcemeta::core::OpenAPIWalk &walk, return sourcemeta::core::openapi_parameter_identity(walk, position); } +// Whether a Parameter Object is one the specification tells a reader to look +// past. Section 4.8.12 names three by their location and their name: "If `in` +// is `"header"` and the `name` field is `"Accept"`, `"Content-Type"` or +// `"Authorization"`, the parameter definition SHALL be ignored". Section +// 4.8.12.1 reads such a name under RFC 7230, which "states header names are +// case insensitive", and 3.2.1 Section 4.12.1 says as much under RFC 9110, so +// which letters it is written with settles nothing +auto ignored_by_the_specification( + const std::pair &identity) -> bool { + if (identity.second != "header") { + return false; + } + + auto name{identity.first}; + sourcemeta::core::to_lowercase(name); + return name == "accept" || name == "content-type" || name == "authorization"; +} + // The parameters in force where an operation sits. Section 4.8.9 has the ones // a Path Item Object declares "applicable for all the operations described // under this path. These parameters can be overridden at the operation level, @@ -241,12 +174,24 @@ auto parameters_of(const sourcemeta::core::OpenAPIWalk &walk, // A Reference Object that was never followed names no parameter, so // nothing can be said to override it const auto *identity{identity_of(walk, position)}; + if (identity != nullptr && ignored_by_the_specification(*identity)) { + continue; + } + if (identity == nullptr || !claimed.contains(*identity)) { result.push_back(position); } } - result.insert(result.cend(), operation.cbegin(), operation.cend()); + for (const auto &position : operation) { + const auto *identity{identity_of(walk, position)}; + if (identity != nullptr && ignored_by_the_specification(*identity)) { + continue; + } + + result.push_back(position); + } + return result; } @@ -271,8 +216,16 @@ auto tags_of(const sourcemeta::core::OpenAPIWalk &walk, // The servers in force where an operation sits. OpenAPI Specification 3.1.1, // Section 4.8.10 has an Operation Object's servers override those of "the Path -// Item Object or OpenAPI Object level", and Section 4.8.1 makes an empty array -// there stand for none being given at all, so an empty array carries on up +// Item Object or OpenAPI Object level", and Section 4.8.9 says as much of a +// Path Item Object's own. +// +// Neither says what an empty array means where it is written. Only Section +// 4.8.1 speaks of one, and only of the array the OpenAPI Object itself holds: +// "If the `servers` field is not provided, or is an empty array, the default +// value would be a Server Object with a url value of `/`". That sentence is no +// authority over the two levels below it, +// so reading an empty array there as nothing given is a choice this makes +// where the specification says nothing, rather than a rule it follows auto servers_of(const std::vector &operation, const std::vector &path_item, const std::vector &document) @@ -292,7 +245,8 @@ auto servers_of(const std::vector &operation, auto check_path_parameters( const sourcemeta::core::OpenAPIWalk &walk, const std::vector &templates, - const std::vector ¶meters) -> void { + const std::vector ¶meters, + const char *message) -> void { for (const auto &position : parameters) { const auto *identity{identity_of(walk, position)}; // A Reference Object that was never followed names no parameter, so @@ -302,13 +256,82 @@ auto check_path_parameters( } if (std::ranges::find(templates, identity->first) == templates.cend()) { - throw error_at(walk, position, - "A path Parameter Object must name a template expression " - "of the path it is under"); + throw error_at(walk, position, message); } } } +// Every Path Item Object a position leads through, from the one written down +// to the one the chain ends at. A `$ref` may lead to a Path Item Object that +// declares one of its own, and each of those is a place of the description in +// its own right, so reading only the two ends would pass over whatever the +// middle of a chain writes beside its own reference +auto aliased_path_items(const sourcemeta::core::OpenAPIWalk &walk, + const sourcemeta::core::JSON::String &position) + -> std::vector { + std::vector result; + // Every position walked through is one the caller or the walk already holds, + // so this runs once per path and keeps no string of its own + std::set seen; + const auto *current{&position}; + while (seen.insert(*current).second) { + const auto record{walk.path_items.find(*current)}; + if (record != walk.path_items.cend()) { + result.push_back(&record->second); + } + + const auto alias{walk.references.find(*current)}; + if (alias == walk.references.cend()) { + break; + } + + current = &alias->second.destination; + } + + return result; +} + +// The template expressions of every path that exposes a given Path Item +// Object, keyed by where that Path Item sits. The requirement above names the +// Paths Object rather than wherever the parameter happens to be written, so a +// Path Item reached as a webhook or through a callback expression is held to +// what the paths reaching that very Path Item declare, and one no path reaches +// leaves nothing for such a parameter to correspond to +auto exposing_expressions(const sourcemeta::core::OpenAPIWalk &walk) + -> std::pair>, + bool> { + std::map> + result; + bool whole{true}; + for (const auto &endpoint : walk.endpoints) { + if (endpoint.kind != sourcemeta::core::OpenAPIOperationKind::Path) { + continue; + } + + const auto position{follow_aliases(walk, endpoint.path_item)}; + // A path whose own Path Item Object the walk could not reach may be the + // very path that exposes another one, so what is gathered here is short of + // what the description holds and says nothing about a place it does not + // cover. OpenAPI Specification 3.2.1, Section 4.1.2.1 leaves no room to + // call a reference unresolvable while a document of the description has + // gone unread + if (!walk.path_items.contains(position)) { + whole = false; + continue; + } + + auto &expressions{result[position]}; + for (const auto &expression : + sourcemeta::core::openapi_brace_expressions(endpoint.path)) { + expressions.push_back(expression); + } + } + + return {std::move(result), whole}; +} + // OpenAPI Specification 3.2.1, Section 4.12, of a parameter whose location // is `querystring`: it "MUST NOT appear more than once, and MUST NOT appear in // the same operation (or in the operation's path-item) as any `in: "query"` @@ -352,25 +375,55 @@ auto check_querystring( // parameters in force for one operation rather than of either level alone, and // Section 3.5 excuses an empty Path Item from it, which is why nothing checks // it until there is an operation to check -auto check_path_templates( +auto collect_path_parameter_names( const sourcemeta::core::OpenAPIWalk &walk, - const sourcemeta::core::JSON::String &endpoint, - const std::vector &templates, - const std::vector ¶meters) -> void { - std::set named; + const std::vector ¶meters, + std::set &names) -> bool { for (const auto &position : parameters) { const auto *identity{identity_of(walk, position)}; // A reference the walk could not follow may be the very parameter a // template expression is looking for, and a description we do not hold in // full is one we cannot call incomplete. This is the same restraint - // Section 8.7.1 applies to a Link Object's operation identifier + // Section 4.3.3 applies to a Link Object's operation identifier, and it + // stops in the same place: one that ends inside this very document ends + // where no further reading can supply a parameter, so it is passed over + // rather than taken as a reason to say nothing if (identity == nullptr) { - return; + if (!sourcemeta::core::openapi_within_document( + follow_aliases(walk, position), walk.base)) { + return false; + } + + continue; } if (identity->second == "path") { - named.insert(identity->first); + names.insert(identity->first); + } + } + + return true; +} + +auto check_path_templates( + const sourcemeta::core::OpenAPIWalk &walk, + const sourcemeta::core::JSON::String &endpoint, + const std::vector &templates, + const std::vector &chain, + const std::vector ¶meters) -> void { + std::set named; + if (!collect_path_parameter_names(walk, parameters, named)) { + return; + } + + // Section 4.8.9 leaves undefined which of two lists a chain writes is in + // force, so a template expression is answered by a path parameter written at + // any place the chain leads through. Turning a description down on one + // reading alone would refuse what another equally licensed reading accepts + for (const auto *record : chain) { + if (!collect_path_parameter_names(walk, record->parameters, named)) { + return; } } @@ -383,59 +436,169 @@ auto check_path_templates( } } +} // namespace + +namespace sourcemeta::core { + // Every operation the description exposes, which Section 4.3.3 confines to what // the entry document reaches: "only the entry document's Paths Object // contributes URLs to the described API". A Callback Object holds Path Item // Objects of its own, so what an endpoint reaches may expose further endpoints -auto project(const sourcemeta::core::OpenAPIWalk &walk) - -> std::vector { - std::vector result; - std::vector pending{walk.endpoints}; - std::set seen; +auto openapi_project(const OpenAPIWalk &walk) -> std::vector { + std::vector result; + std::vector pending{walk.endpoints}; + const auto expressions{exposing_expressions(walk)}; + // What tells one exposure from another, held as the four things it is + // rather than as one string spelling them out. A position carries a `#` of + // its own, so any character picked to join them is a character one of them + // may hold, and the spelling would not tell every pair of exposures apart + std::set>> + seen; for (std::size_t index = 0; index < pending.size(); index += 1) { const auto kind{pending[index].kind}; const auto path{pending[index].path}; const auto endpoint{pending[index].path_item}; + const auto parent{pending[index].parent}; auto position{follow_aliases(walk, endpoint)}; // A Path Item that leads back to one already exposed the same way exposes - // nothing further, which is what stops a cycle of them - sourcemeta::core::JSON::String key{ - sourcemeta::core::openapi_operation_kind_name(kind)}; - key.append("#").append(path).append("#").append(position); - if (!seen.insert(std::move(key)).second) { + // nothing further, which is what stops a cycle of them. Section 4.8.10 + // hangs a Callback Object off "the parent operation", so one that two of + // them reach is reached twice rather than once and the Object it hangs off + // is part of what tells the two apart + if (!seen.emplace(kind, path, position, parent).second) { continue; } - const auto entry{walk.path_items.find(position)}; - if (entry == walk.path_items.cend()) { + // Section 4.8.9 leaves undefined only what "appears both in the defined + // object and the referenced object", so a field written beside a `$ref` + // that the referenced Path Item Object does not declare is an ordinary + // field of the Path Item Object holding it, and Section 3.5 counts a path + // parameter "included in the Path Item itself" wherever it is written. So + // what the reference leads to answers first, and what sits beside it + // answers for whatever that leaves unsaid + const auto chain{aliased_path_items(walk, endpoint)}; + if (chain.empty()) { continue; } + // Whether what the chain ends at is settled by the document in hand. It is + // settled when that place is a Path Item Object this holds, and equally + // when the chain ends on a pointer into this very document that leads + // nowhere, since no further reading can rescue one of those. 3.2.1 Section + // 4.1.2.1 asks for every document to be parsed before a reference is + // called unresolvable, which leaves only a chain ending in a document this + // has not read unsettled + const auto settled{walk.path_items.contains(position) || + openapi_within_document(position, walk.base)}; + + // The last place the chain leads through that this one holds answers + // first, which is what the chain ends at whenever that landed, and each + // place nearer to it answers before the one that names it. Section 4.8.9 + // leaves that order undefined between any two of them and so free to pick + const auto *parameters_of_path_item{&chain.back()->parameters}; + const auto *servers_of_path_item{&chain.back()->servers}; + auto methods{chain.back()->operations}; + for (const auto *record : std::ranges::reverse_view{chain}) { + if (parameters_of_path_item->empty()) { + parameters_of_path_item = &record->parameters; + } + + if (servers_of_path_item->empty()) { + servers_of_path_item = &record->servers; + } + + for (const auto &method : record->operations) { + if (std::ranges::none_of(methods, [&method](const auto &known) -> bool { + return known.first == method.first; + })) { + methods.push_back(method); + } + } + } + + const auto &path_item_parameters{*parameters_of_path_item}; + const auto &path_item_servers{*servers_of_path_item}; + // A webhook name and a callback expression are not templated paths, so - // only what the Paths Object exposes has any templating to correspond to - const auto templated{kind == sourcemeta::core::OpenAPIOperationKind::Path}; - const auto templates{ - templated ? sourcemeta::core::openapi_brace_expressions(path) - : std::vector{}}; - if (templated) { - check_path_parameters(walk, templates, entry->second.parameters); + // only what the Paths Object exposes has any templating of its own + const auto templated{kind == OpenAPIOperationKind::Path}; + const auto own_templates{templated ? openapi_brace_expressions(path) + : std::vector{}}; + // Section 4.8.12 asks a path Parameter Object to name "a template + // expression occurring within the path field in the Paths Object" rather + // than one of whichever path it happens to sit under, and one Path Item + // Object may be reached from several of them. So every path exposing this + // one answers, which is the reading a webhook and a callback already went + // by, and a Path Item Object no path reached falls back to the path being + // projected + const auto exposed{expressions.first.find(position)}; + const auto &templates{ + exposed == expressions.first.cend() ? own_templates : exposed->second}; + // A path answers for its own braces whatever else went unread, but the set + // gathered from every path exposing one Path Item Object is short by + // whatever a path this does not hold would have added to it, so only the + // first of those two is worth reading against on its own + const auto own_only{templated && exposed == expressions.first.cend()}; + const auto *const message{ + "A path Parameter Object must name a template expression of a path " + "that exposes it"}; + // What no path exposes is only known to be exposed by none once every path + // has been read, which a description held short of its documents leaves + // unsettled + // Section 4.8.12 binds every Parameter Object written down rather than + // whichever list the fold above carries forward, so each place the chain + // leads through answers for the ones it declares itself + // A path answers for its own braces whatever else is unread, but every + // other kind of endpoint is held to the expressions of the paths reaching + // the Path Item Object the chain ends at, and a chain that ends somewhere + // this does not hold is one whose end an unread document still decides + if ((own_only || expressions.second) && settled) { + for (const auto *record : chain) { + check_path_parameters(walk, templates, record->parameters, message); + // An Operation Object that a method of the same name nearer the end of + // the chain outranks is written down all the same, and Section 4.8.12 + // binds a Parameter Object wherever it is written + for (const auto &declared : record->operations) { + const auto operation{walk.operation_records.find(declared.second)}; + if (operation != walk.operation_records.cend()) { + check_path_parameters(walk, templates, operation->second.parameters, + message); + } + } + } } - for (const auto &[method, origin] : entry->second.operations) { + for (const auto &[method, origin] : methods) { const auto operation{walk.operation_records.find(origin)}; if (operation == walk.operation_records.cend()) { continue; } auto parameters{parameters_of(walk, operation->second.parameters, - entry->second.parameters)}; - if (templated) { - check_path_parameters(walk, templates, parameters); - check_path_templates(walk, endpoint, templates, parameters); + path_item_parameters)}; + // The two loops above already answer for every Parameter Object written + // at any place the chain leads through, and what is in force here is + // drawn from those same lists, so this holds nothing new. It abstains + // on the same terms all the same, as the expressions it would be read + // against are the ones an unread document settles + if ((own_only || expressions.second) && settled) { + check_path_parameters(walk, templates, parameters, message); + } + // This one turns on an expression having no parameter anywhere, and a + // chain that ends somewhere this does not hold may well end at the Path + // Item Object declaring it, so there is nothing to conclude yet + if (templated && settled) { + check_path_templates(walk, endpoint, own_templates, chain, + operation->second.parameters); } - check_querystring(walk, origin, parameters); + // Which of the chain's lists is in force decides whether these two ever + // meet, and that is not settled while the chain ends somewhere unread + if (settled) { + check_querystring(walk, origin, parameters); + } result.push_back( {.kind = kind, @@ -443,16 +606,17 @@ auto project(const sourcemeta::core::OpenAPIWalk &walk) .method = method, .origin = origin, .endpoint = endpoint, - .servers = servers_of(operation->second.servers, - entry->second.servers, walk.servers), + .parent = parent, + .servers = servers_of(operation->second.servers, path_item_servers, + walk.servers), // Section 4.8.10: "This definition overrides any declared top-level // security. To remove a top-level security declaration, an empty // array can be used", which is why declaring none and declaring an // empty array are not the same thing here - .security = operation->second.security.has_value() - ? operation->second.security.value() - : walk.security.value_or( - std::vector{}), + .security = + operation->second.security.has_value() + ? operation->second.security.value() + : walk.security.value_or(std::vector{}), .parameters = std::move(parameters), .tags = tags_of(walk, operation->second.tags)}); @@ -463,10 +627,10 @@ auto project(const sourcemeta::core::OpenAPIWalk &walk) } for (const auto &[expression, path_item] : entries->second) { - pending.push_back( - {.kind = sourcemeta::core::OpenAPIOperationKind::Callback, - .path = expression, - .path_item = path_item}); + pending.push_back({.kind = OpenAPIOperationKind::Callback, + .path = expression, + .path_item = path_item, + .parent = origin}); } } } @@ -475,22 +639,19 @@ auto project(const sourcemeta::core::OpenAPIWalk &walk) return result; } -} // namespace - -namespace sourcemeta::core { - struct OpenAPIFrame::Internal { OpenAPIVersion version; OpenAPIInfo info; // Canonicalising means this no longer borrows from what the caller passed JSON::String base; bool standalone; - std::map locations; - std::map references; + OpenAPIFrame::Locations locations; + OpenAPIFrame::References references; std::vector operations; // What a Discriminator Object names by URI, which is a reference the schemas // hold rather than one the shell around them does std::vector discriminators; + OpenAPIFrame::References security_references; // Reading inside a Schema Object is the business of whatever understands // JSON Schema, so this is that pass over every Schema Object position at // once. It is declared last so that it is destroyed first, as it holds @@ -510,10 +671,6 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, max_locations)}; this->internal_->version = walk.version; this->internal_->info = walk.info; - // What the caller passed in is where the entry document was retrieved from, - // and from 3.2 onwards the document may give itself a URI of its own, which - // the walk settles and everything it holds is keyed by - this->internal_->base = std::move(walk.base); // A frame stands alone when everything it references is inside it, which is // what a caller asks before deciding whether it has the whole description. // Which references leave it is what making it whole comes down to, so each @@ -527,11 +684,29 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, } } - // Projecting reads the whole walk, so nothing is taken out of it until after - this->internal_->operations = project(walk); + // And so does a Security Requirement Object that names a scheme by the URI + // of one, which 3.2 admits alongside the name of a component. Naming one + // that this document does not hold leaves the description no more whole than + // any other reference out of it would + for (auto &reference : walk.security_references) { + reference.second.dangling = + !walk.locations.contains(reference.second.destination); + if (reference.second.dangling) { + every_reference_lands = false; + } + } + + // Projecting reads the whole walk, the base included, so nothing is taken + // out of it until after + this->internal_->operations = openapi_project(walk); + // What the caller passed in is where the entry document was retrieved from, + // and from 3.2 onwards the document may give itself a URI of its own, which + // the walk settles and everything it holds is keyed by + this->internal_->base = std::move(walk.base); const auto walk_locations{walk.locations.size()}; this->internal_->locations = std::move(walk.locations); this->internal_->references = std::move(walk.references); + this->internal_->security_references = std::move(walk.security_references); // Every Schema Object position of the document at once, rather than one // pass each, so that a schema referring to another resolves against a frame @@ -552,25 +727,11 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, // // Locations are keyed by the base with the pointer hung off it, so a place // within another has that other one's key as a prefix and follows it here - std::vector enclosing; for (const auto &location : this->internal_->locations) { - if (location.second.type != OpenAPIObjectKind::Schema) { - continue; - } - - auto pointer{to_weak_pointer(location.second.pointer)}; - while (!enclosing.empty() && !pointer.starts_with(enclosing.back())) { - enclosing.pop_back(); - } - - if (!enclosing.empty()) { - throw OpenAPIError{ - this->internal_->base, location.second.pointer, - "A Schema Object must not sit within another Schema Object"}; + if (location.second.type == OpenAPIObjectKind::Schema) { + this->internal_->schema_paths.push_back( + to_weak_pointer(location.second.pointer)); } - - enclosing.push_back(pointer); - this->internal_->schema_paths.push_back(std::move(pointer)); } this->internal_->schema_resolver = resolver; @@ -620,13 +781,14 @@ OpenAPIFrame::OpenAPIFrame(const JSON &document, const SchemaWalker &walker, // that of, so these wait until the whole of it is settled, which counts what // the Schema Objects reach for as much as what the shell around them does if (this->internal_->standalone) { - check_operation_id_links(walk, this->internal_->locations); + sourcemeta::core::openapi_check_operation_id_links( + walk, this->internal_->locations); } // A tag the description declares elsewhere is one this cannot say is // missing, for the same reason as the identifiers above - check_tag_parents(walk, this->internal_->locations, - this->internal_->standalone); + sourcemeta::core::openapi_check_tag_parents(walk, this->internal_->locations, + this->internal_->standalone); } OpenAPIFrame::~OpenAPIFrame() = default; @@ -651,6 +813,50 @@ auto OpenAPIFrame::schemas() const noexcept -> const SchemaFrame & { return *(this->internal_->schemas); } +auto OpenAPIFrame::locations() const noexcept -> const Locations & { + return this->internal_->locations; +} + +auto OpenAPIFrame::references() const noexcept -> const References & { + return this->internal_->references; +} + +auto OpenAPIFrame::security_references() const noexcept -> const References & { + return this->internal_->security_references; +} + +auto OpenAPIFrame::operations() const noexcept + -> const std::vector & { + return this->internal_->operations; +} + +auto OpenAPIFrame::discriminators() const noexcept + -> const std::vector & { + return this->internal_->discriminators; +} + +auto OpenAPIFrame::traverse(const JSON::StringView uri) const + -> const Location * { + const auto match{this->internal_->locations.find(uri)}; + if (match == this->internal_->locations.cend()) { + return nullptr; + } + + return &match->second; +} + +auto OpenAPIFrame::uri(const Pointer &pointer) const -> JSON::String { + return openapi_location_uri(this->internal_->base, pointer); +} + +auto OpenAPIFrame::object_count() const noexcept -> std::size_t { + return this->internal_->locations.size(); +} + +auto OpenAPIFrame::reference_count() const noexcept -> std::size_t { + return this->internal_->references.size(); +} + auto OpenAPIFrame::to_json() const -> JSON { // Read through the accessors rather than the internal state, so that what // this reports and what a caller can observe cannot drift apart @@ -711,6 +917,12 @@ auto OpenAPIFrame::to_json() const -> JSON { entry.assign_assume_new("origin", JSON{operation.origin}); entry.assign_assume_new("endpoint", JSON{operation.endpoint}); + // Only an operation that a Callback Object exposes has one of these, which + // is the Operation Object that Callback Object hangs off + if (operation.parent.has_value()) { + entry.assign_assume_new("parent", JSON{operation.parent.value()}); + } + entry.assign_assume_new("tags", sourcemeta::core::to_json(operation.tags)); entry.assign_assume_new("servers", @@ -744,6 +956,22 @@ auto OpenAPIFrame::to_json() const -> JSON { result.assign_assume_new("discriminators", std::move(discriminators)); } + if (!this->internal_->security_references.empty()) { + auto references{JSON::make_array()}; + for (const auto &reference : this->internal_->security_references) { + auto entry{JSON::make_object()}; + entry.assign_assume_new("pointer", + JSON{to_string(reference.second.origin)}); + entry.assign_assume_new("original", JSON{reference.second.original}); + entry.assign_assume_new("destination", + JSON{reference.second.destination}); + entry.assign_assume_new("dangling", JSON{reference.second.dangling}); + references.push_back(std::move(entry)); + } + + result.assign_assume_new("securityReferences", std::move(references)); + } + return result; } diff --git a/vendor/core/src/core/openapi/helpers.h b/vendor/core/src/core/openapi/helpers.h index 73ce1d9a2..fde41a070 100644 --- a/vendor/core/src/core/openapi/helpers.h +++ b/vendor/core/src/core/openapi/helpers.h @@ -10,11 +10,13 @@ #include // std::array #include // std::size_t #include // std::uint8_t, std::uint64_t +#include // std::less #include // std::initializer_list #include // std::numeric_limits #include // std::map -#include // std::optional +#include // std::optional, std::nullopt #include // std::set +#include // std::span #include // std::string_view #include // std::pair, std::unreachable #include // std::vector @@ -23,68 +25,18 @@ namespace sourcemeta::core { using namespace std::string_view_literals; -// OpenAPI Specification 3.1.1, Section 4.9: "The field name MUST begin with -// `x-`, for example, `x-internal-id`" -constexpr auto OPENAPI_EXTENSION_PREFIX{"x-"sv}; -constexpr auto OPENAPI_HASH_DEPRECATED{JSON::Object::hash("deprecated"sv)}; - -constexpr auto OPENAPI_HASH_DESCRIPTION{JSON::Object::hash("description"sv)}; -constexpr auto OPENAPI_HASH_SUMMARY{JSON::Object::hash("summary"sv)}; -constexpr auto OPENAPI_HASH_URL{JSON::Object::hash("url"sv)}; -constexpr auto OPENAPI_HASH_NAME{JSON::Object::hash("name"sv)}; -constexpr auto OPENAPI_HASH_IN{JSON::Object::hash("in"sv)}; -constexpr auto OPENAPI_HASH_TAGS{JSON::Object::hash("tags"sv)}; -constexpr auto OPENAPI_HASH_SERVERS{JSON::Object::hash("servers"sv)}; - -/// A fixed field name paired with the hash of that name, so that looking one -/// up does not have to hash it again on every Object read -struct OpenAPIField { - JSON::StringView name; - JSON::Object::hash_type hash; -}; - -// What a reference expects to find at the far end of itself, which is fixed by -// where the reference sits rather than by anything the target says about -// itself. OpenAPI Specification 3.1.1, Section 4.3.1 calls this "the expected -// type of the reference", and it is what the place a reference lands on is -// held to -enum class OpenAPIObjectKind : std::uint8_t { - /// A whole OpenAPI Description, which is what the root of the document - /// framed is recorded as - Document, - // The eleven a reference may expect to find, the last of them only from 3.2 - // onwards, which is where a `content` map and the Components Object both - // learn to hold a Reference Object in place of a Media Type Object - PathItem, - Parameter, - RequestBody, - Response, - Example, - Header, - Link, - Callbacks, - SecurityScheme, - MediaType, - // The rest are never referenced, but every Object gets a location - Info, - Contact, - License, - Server, - ServerVariable, - Components, - Paths, - Operation, - ExternalDocumentation, - Encoding, - Responses, - Tag, - Reference, - Schema, - OAuthFlows, - OAuthFlow, - SecurityRequirement -}; - +// The walk builds what the frame will hand out, so each of these is one type +// with the frame's own. They are spelled unqualified here because the walk +// machinery is written against them throughout +using OpenAPILocation = OpenAPIFrame::Location; +using OpenAPIObjectKind = OpenAPIFrame::ObjectKind; +using OpenAPIOperationKind = OpenAPIFrame::OperationKind; +using OpenAPIReference = OpenAPIFrame::Reference; +using OpenAPIOperation = OpenAPIFrame::Operation; +using OpenAPIDiscriminator = OpenAPIFrame::Discriminator; + +// What a frame exports each of these as, which is the name a fixture +// records rather than the enumerator behind it inline auto openapi_kind_name(const OpenAPIObjectKind kind) noexcept -> JSON::StringView { switch (kind) { @@ -148,30 +100,6 @@ inline auto openapi_kind_name(const OpenAPIObjectKind kind) noexcept std::unreachable(); } - -/// Where an Object that stands in for another leads. OpenAPI Specification -/// 3.1.1 has a Reference Object and a Path Item Object each declare at most -/// one `$ref`, and a Schema Object's `$ref` never reaches here, so this is a -/// field of the Object that makes it rather than a table of its own -struct OpenAPIReference { - /// The value as the document wrote it - JSON::String original; - /// Where it points, resolved against the base and canonicalised. The - /// document it names and the fragment it carries are that string either side - /// of its `#`, so neither is repeated here - JSON::String destination; - /// Whether that destination is nowhere the frame holds, which is what makes - /// a description one that has to be made whole before it describes anything - bool dangling{false}; -}; - -/// How an Operation Object is reached from the entry document. OpenAPI -/// Specification 3.1.1, Section 4.3.3: "only the entry document's Paths Object -/// contributes URLs to the described API", so what an operation is reached -/// through is a property of the route to it rather than of where it is -/// defined -enum class OpenAPIOperationKind : std::uint8_t { Path, Webhook, Callback }; - inline auto openapi_operation_kind_name(const OpenAPIOperationKind kind) noexcept -> JSON::StringView { @@ -187,6 +115,65 @@ openapi_operation_kind_name(const OpenAPIOperationKind kind) noexcept std::unreachable(); } +// OpenAPI Specification 3.1.1, Section 4.9: "The field name MUST begin with +// `x-`, for example, `x-internal-id`" +constexpr auto OPENAPI_EXTENSION_PREFIX{"x-"sv}; +constexpr auto OPENAPI_HASH_DEPRECATED{JSON::Object::hash("deprecated"sv)}; + +constexpr auto OPENAPI_HASH_DESCRIPTION{JSON::Object::hash("description"sv)}; +constexpr auto OPENAPI_HASH_SUMMARY{JSON::Object::hash("summary"sv)}; +constexpr auto OPENAPI_HASH_URL{JSON::Object::hash("url"sv)}; +constexpr auto OPENAPI_HASH_NAME{JSON::Object::hash("name"sv)}; +constexpr auto OPENAPI_HASH_IN{JSON::Object::hash("in"sv)}; +constexpr auto OPENAPI_HASH_TAGS{JSON::Object::hash("tags"sv)}; +constexpr auto OPENAPI_HASH_SERVERS{JSON::Object::hash("servers"sv)}; + +/// A fixed field name paired with the hash of that name, so that looking one +/// up does not have to hash it again on every Object read +struct OpenAPIField { + JSON::StringView name; + JSON::Object::hash_type hash; +}; + +// OpenAPI Specification 3.1.1, Section 4.6: "Unless specified otherwise, all +// fields that are URIs MAY be relative references as defined by RFC3986", and +// one of those is resolved "using the referring document's base URI". So an +// Object that moves to another document carries fields that would otherwise +// go on resolving against a base that is no longer theirs. +// +// Which fields those are is what the row of each one says rather than what it +// is named. Section 4.6: "Note that some URI fields are named `url` for +// historical reasons, but the descriptive text for those fields uses the +// correct \"URI\" terminology". So a row reading URI belongs here and a row +// reading URL does not, as Section 4.7 resolves those "using the URLs defined +// in the Server Object as a Base URL" rather than against any document. The +// endpoints of a Security Scheme Object and of an OAuth Flow Object are that +// second kind, and moving one leaves what it names untouched because the +// Server Objects it reads against are not what moved. +// +// A `$ref` and an `operationRef` are left out, as the frame records those as +// references of their own. A Server Object `url` is left out too, as Section +// 4.8.5 makes it a URL template rather than a URI reference, and resolving one +// through a URI would mangle the variables it is written with. So is every +// field of the Objects that only ever sit at the root of a document, as those +// never move anywhere +constexpr std::array OPENAPI_URI_FIELDS_EXTERNAL_DOCS{ + {"url"sv}}; +constexpr std::array OPENAPI_URI_FIELDS_EXAMPLE{ + {"externalValue"sv}}; + +inline auto openapi_embedded_uri_fields(const OpenAPIObjectKind kind) noexcept + -> std::span { + switch (kind) { + case OpenAPIObjectKind::ExternalDocumentation: + return OPENAPI_URI_FIELDS_EXTERNAL_DOCS; + case OpenAPIObjectKind::Example: + return OPENAPI_URI_FIELDS_EXAMPLE; + default: + return {}; + } +} + /// What a Path Item Object declares that the endpoints reaching it need. Two /// endpoints may reach one Path Item, and following a reference reads its /// target once, so this is kept rather than read again @@ -226,51 +213,11 @@ struct OpenAPIEndpoint { OpenAPIOperationKind kind; JSON::String path; JSON::String path_item; -}; - -/// One operation of the described API, which is what an endpoint and the Path -/// Item it reaches come to between them -struct OpenAPIOperation { - OpenAPIOperationKind kind; - JSON::String path; - JSON::String method; - /// Where the Operation Object sits - JSON::String origin; - /// Where the Path Item Object that exposes it sits, which is the position - /// that gives it a URL rather than the one that defines it. The two differ - /// whenever a reference stands between them - JSON::String endpoint; - /// Where the Server Objects in force sit, empty when nothing declares any, - /// in which case Section 4.8.1 puts a single Server Object with a `url` of - /// `/` in their place - std::vector servers; - /// Where the Security Requirement Objects in force sit - std::vector security; - /// Where the Parameter Objects in force sit, which is what the Path Item - /// Object declares once anything the Operation Object overrides is taken - /// out, followed by what the Operation Object declares itself - std::vector parameters; - /// Where the Tag Object each of its tags names sits, in the order the - /// operation wrote them, with no value where the entry document declares no - /// tag by that name. Section 4.8.1 permits exactly that: "Not all tags that - /// are used by the Operation Object must be declared" - std::vector> tags; -}; - -struct OpenAPILocation { - OpenAPIObjectKind type; - Pointer pointer; - /// Set on the root of a document and on every Schema Object position it - /// holds: the default `$schema` in force there, resolved against the base. A - /// Schema Object that declares its own overrides it, which is a matter for - /// whatever reads inside one - JSON::String dialect; - /// Set on a Schema Object position alone: the base its document keys every - /// location by, which is what a relative reference inside that schema - /// resolves against until an `$id` says otherwise. RFC 3986 Section 5.1.1 - /// makes an `$id` the higher precedence source, so this is a default in the - /// same way the dialect above is - JSON::String base; + /// Where the Operation Object that a Callback Object hangs off sits, which + /// only an expression of such an Object is exposed by. Section 4.8.10 makes + /// that Object the one a callback is "related to", so two of them reaching + /// one Callback Object expose it twice rather than once + std::optional parent{std::nullopt}; }; // What every check needs to reach beyond the Object in front of it: the @@ -281,6 +228,16 @@ struct OpenAPILocation { // clash is only ever caught within it struct OpenAPIWalk { JSON::String base; + // Where the document was retrieved from, which `$self` may take the place of + // as the base every URI it holds resolves against. OpenAPI Specification + // 3.2.1, Section 4.5.2 keeps the addresses of the API itself out of that: + // "Because the API is a distinct entity from the OpenAPI document, RFC3986's + // base URI rules for the OpenAPI document do not apply", and 3.2.1 + // Section 4.5.2.1 says which base does apply instead: "For API URLs the + // `$self` field, which identifies the OpenAPI document, is ignored and the + // retrieval URI is used instead". So this is kept apart from the base above + // rather than replaced by it + JSON::String retrieval; // The document the checks are reading, which a reference that stays inside // it resolves its fragment against const JSON *document{nullptr}; @@ -302,12 +259,12 @@ struct OpenAPIWalk { /// as a fragment, or by the pointer alone when no base was established. The /// document an Object sits in is that key up to its fragment, so nothing /// records it a second time - std::map locations; + OpenAPIFrame::Locations locations; /// Every reference the description makes, whether or not it was followed, /// keyed by the location of the Object that makes it. Kept apart from the /// locations themselves only because reading one Object twice records it /// twice, and what it stands in for must survive that - std::map references; + OpenAPIFrame::References references; /// Every Path Item Object and Operation Object read, along with the Callback /// Objects that hold more of them, all keyed by where they sit. The /// projection that turns these into operations runs once the walk is over, @@ -333,19 +290,33 @@ struct OpenAPIWalk { std::optional> security; /// The names the entry document declares as security schemes, which is what /// a Security Requirement Object anywhere in the description may name - std::set security_schemes; + JSONPropertySet security_schemes; + /// Where a Security Requirement Object names a Security Scheme Object by the + /// URI of one rather than by the name of a component, and what each of those + /// names leads to. OpenAPI Specification 3.2.1 admits both spellings. This + /// is kept apart from the references above because a single one of those + /// Objects may name several schemes, which is more than one entry keyed by + /// the Object that makes it, so each is keyed by the member that spells it + /// instead. Reading one Object twice, which following a reference into the + /// document being read does, must still record it once + OpenAPIFrame::References security_references; + /// Whether an entry document is what the names above came from, which is + /// what makes this a document the description reaches rather than the one + /// that describes the API + bool referenced{false}; /// Where the entry document declares each Tag Object, by the name it gave /// it, which is the name an Operation Object's tags resolve against std::map tags; /// What each Tag Object that declares a parent is called and which tag it - /// names, keyed by where that Tag Object sits. Section 4.22 has the named - /// tag exist and forbids a cycle, neither of which can be settled until + /// names, keyed by where that Tag Object sits. 3.2.1 Section 4.22 has the + /// named tag exist and forbids a cycle, neither of which can be settled until /// every tag has been read std::map> tag_parents; - /// Every name any document declares a Tag Object under. Section 4.22 has a - /// parent name "a tag that MUST exist in the API description", which is the - /// whole of it rather than the entry document alone, so this is a wider set - /// than the one above it + /// Every name any document declares a Tag Object under. 3.2.1 Section 4.22 + /// has a parent name "The `name` of a tag that this tag is nested under", + /// and of what it names, "The named tag MUST exist in the API description". + /// A description is the whole of what it spans rather than the entry + /// document alone, so this is a wider set than the one above it std::set tag_names; /// The operation each Link Object names, keyed by where that Link Object /// sits. Section 4.3.3 has resolving one of these require "parsing all @@ -436,7 +407,9 @@ inline auto openapi_child(const Pointer &base, const std::size_t index) // fields: fixed fields, which have a declared name, and patterned fields, // which have a declared pattern for the field name". These objects declare // `^x-` as their only pattern, so a member that is neither is not a field that -// this specification defines +// this specification defines. That sentence only tells the two apart, and what +// turns the leftover into a refusal is Section 4.9, which holds an extension +// to "MUST begin with `x-`, for example, `x-internal-id`" template auto openapi_reject_unknown_fields( const JSON &object, const std::array &fields, @@ -451,6 +424,30 @@ auto openapi_reject_unknown_fields( } } +// A field table may hold a field whose Description cell then restricts where +// it applies, which is a field the Object defines rather than one it does not. +// The two are turned down alike, so which of them a member is has to be said +// here for the reason given to be a true one +template +auto openapi_reject_unknown_fields( + const JSON &object, const std::array &fields, + const Pointer &base, const char *message, + const std::array &restricted, + const char *restricted_message) -> void { + for (const auto &entry : object.as_object()) { + if (entry.first.starts_with(OPENAPI_EXTENSION_PREFIX) || + std::ranges::find(fields, entry.first) != fields.cend()) { + continue; + } + + throw OpenAPIError{openapi_child(base, entry.first), + std::ranges::find(restricted, entry.first) != + restricted.cend() + ? restricted_message + : message}; + } +} + // A field table is a property of the revision a document declares, so an // Object whose table grew between revisions has one array per revision and // which of them applies is asked here rather than at every call site @@ -466,12 +463,29 @@ auto openapi_reject_unknown_fields( } } +template +auto openapi_reject_unknown_fields( + const JSON &object, const std::array &fields, + const std::array &later, const Pointer &base, + const char *message, const OpenAPIWalk &walk, + const std::array &restricted, + const char *restricted_message) -> void { + if (walk.version == OpenAPIVersion::OPENAPI_3_2) { + openapi_reject_unknown_fields(object, later, base, message, restricted, + restricted_message); + } else { + openapi_reject_unknown_fields(object, fields, base, message, restricted, + restricted_message); + } +} + // A location is keyed the way a schema frame keys its own: the document base // with the pointer hung off it as a fragment, or the pointer alone when no // base was established, and the base by itself for the root of a document. // // RFC 6901 Section 6: "A JSON Pointer can be represented in a URI fragment -// identifier by encoding it into octets using UTF-8, while percent-encoding +// identifier by encoding it into octets using UTF-8 [RFC3629], while +// percent-encoding // those characters not allowed by the fragment rule in [RFC3986]". A path // template holds braces and a callback expression holds a `#`, neither of // which a fragment admits, so writing the pointer out as it stands would give @@ -521,13 +535,41 @@ inline auto openapi_document_uri(const JSON::String &uri) -> JSON::String { return JSON::String{take_until(uri, '#')}; } +// Whether a location key names a place in a given document. A pointer into a +// document already in hand that leads nowhere leads nowhere for good, while +// one naming another document says only that the document has gone unread, so +// telling the two apart is what several checks abstain on. They abstain on one +// answer rather than each asking in its own words +inline auto openapi_within_document(const JSON::String &uri, + const JSON::String &base) -> bool { + return take_until(uri, '#') == base; +} + +// Where a problem found once the walk is over belongs. A location says which +// base it is keyed by and where under it the Object sits, and a field hangs +// off that when the problem is with one rather than with the Object holding it +inline auto openapi_error_at(const OpenAPIFrame::Locations &locations, + const JSON::String &location, const char *message, + const JSON::StringView field = {}) + -> OpenAPIError { + const auto match{locations.find(location)}; + auto pointer{match == locations.cend() ? EMPTY_POINTER + : match->second.pointer}; + if (!field.empty()) { + pointer = pointer.concat(JSON::String{field}); + } + + return {openapi_document_uri(location), std::move(pointer), message}; +} + // OpenAPI Specification 3.1.1, Section 4.6 determines a document's base URI // "in accordance with RFC3986 Section 5.1.2 - 5.1.4", a range that starts at // 5.1.2 and so leaves out 5.1.1, "Base URI Embedded in Content". A 3.1 // document therefore has no way of declaring its own base, and what remains is // 5.1.3, "Base URI from the Retrieval URI", which only the caller can supply. -// Section 4.6 says as much: implementations "SHOULD allow users to provide -// documents with their intended retrieval URIs" +// 3.2.1 Section 4.1.2.2.1 says as much, and 3.1 leaves it unsaid rather than +// otherwise: implementations "SHOULD allow users to provide documents with +// their intended retrieval URIs" inline auto openapi_canonical_base(const std::string_view input) -> JSON::String { if (input.empty()) { diff --git a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi.h b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi.h index ec953893d..98008f132 100644 --- a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi.h +++ b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi.h @@ -6,20 +6,26 @@ #endif #include +#include #include // NOLINTBEGIN(misc-include-cleaner) #include // NOLINTEND(misc-include-cleaner) +#include // std::invocable, std::predicate +#include // std::size_t #include // std::uint8_t, std::uint64_t +#include // std::function, std::less #include // std::numeric_limits +#include // std::map #include // std::unique_ptr #include // std::optional, std::nullopt #include // std::string_view +#include // std::vector /// @defgroup openapi OpenAPI -/// @brief A growing implementation of the OpenAPI Specification 3.1. +/// @brief A growing implementation of the OpenAPI Specification. /// /// This module reports where an OpenAPI Description declares its JSON Schemas /// and leaves what is inside them to a JSON Schema implementation. @@ -43,7 +49,8 @@ enum class OpenAPIVersion : std::uint8_t { /// @ingroup openapi /// Determine the version of an OpenAPI Description from its `openapi` field -/// without framing it, returning no value for a version we do not recognise. +/// without framing it, returning no value for a version we do not recognise +/// and for anything that declares no such field to read. /// The patch component of the field carries no meaning, so every `3.1.x` /// release maps to the same result. For example: /// @@ -137,9 +144,212 @@ struct OpenAPIInfo { /// std::cout << std::endl; /// ``` /// -/// A frame is analysed once, on construction, and is immutable afterwards. +/// A frame is analysed once, on construction, and what it reports never +/// changes afterwards. Reading one is not thread safe even so, as it answers +/// out of caches it fills as it goes. A frame cannot be copied or moved, so it +/// is built where it is read. class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { public: + /// What kind of Object an OpenAPI Description holds at a given position. For + /// a position a reference names, this is what it expects to find at the far + /// end of itself, which is fixed by where the reference sits rather than by + /// anything the target says about itself. OpenAPI Specification 3.1.1, + /// Section 4.3.1 calls this "the expected type of the reference", and it is + /// what the place a reference lands on is held to + enum class ObjectKind : std::uint8_t { + /// A whole OpenAPI Description, which is what the root of the document + /// framed is recorded as + Document, + // The eleven a reference may expect to find, the last of them only from 3.2 + // onwards, which is where a `content` map and the Components Object both + // learn to hold a Reference Object in place of a Media Type Object + /// A Path Item Object, which describes the operations available on one path + PathItem, + /// A Parameter Object, which describes one parameter of an operation + Parameter, + /// A Request Body Object, which describes the body an operation takes + RequestBody, + /// A Response Object, which describes one response of an operation + Response, + /// An Example Object, which pairs an example value with its metadata + Example, + /// A Header Object, which describes one header of a response or an encoding + Header, + /// A Link Object, which names a design time relationship to an operation + Link, + /// A Callback Object, which maps runtime expressions to out of band + /// requests + Callbacks, + /// A Security Scheme Object, which describes one way of authenticating + SecurityScheme, + /// A Media Type Object, which describes one representation of a body + MediaType, + // The rest are never referenced, but every Object gets a location + /// An Info Object, which carries the metadata about the API + Info, + /// A Contact Object, which names who to reach about the API + Contact, + /// A License Object, which names the license the API is offered under + License, + /// A Server Object, which names a host the API is served from + Server, + /// A Server Variable Object, which describes one substitution a server URL + /// template takes + ServerVariable, + /// A Components Object, which holds the Objects a description reuses + Components, + /// A Paths Object, which maps path templates to what describes them + Paths, + /// An Operation Object, which describes one operation on a path + Operation, + /// An External Documentation Object, which points at documentation held + /// elsewhere + ExternalDocumentation, + /// An Encoding Object, which describes how one property of a body is + /// serialised + Encoding, + /// A Responses Object, which maps status codes to what describes them + Responses, + /// A Tag Object, which adds metadata to a tag that the operations name + Tag, + /// A Reference Object, which stands in for another Object + Reference, + /// A Schema Object, whose contents are JSON Schema's to make sense of + /// rather than this specification's + Schema, + /// An OAuth Flows Object, which holds the flows an OAuth scheme supports + OAuthFlows, + /// An OAuth Flow Object, which describes one flow a scheme supports + OAuthFlow, + /// A Security Requirement Object, which names the schemes that apply + SecurityRequirement + }; + + /// How an Operation Object is reached from the entry document. OpenAPI + /// Specification 3.1.1, Section 4.3.3: "only the entry document's Paths + /// Object contributes URLs to the described API", so what an operation is + /// reached through is a property of the route to it rather than of where it + /// is defined + enum class OperationKind : std::uint8_t { + /// Reached through the Paths Object of the entry document + Path, + /// Reached through the webhooks that the entry document declares + Webhook, + /// Reached through a Callback Object that an operation declares + Callback + }; + + /// Where an Object that stands in for another leads. OpenAPI Specification + /// 3.1.1 has a Reference Object and a Path Item Object each declare at most + /// one `$ref`, and a Schema Object's `$ref` never reaches here, so this is a + /// field of the Object that makes it rather than a table of its own + struct Reference { + /// The value as the document wrote it + JSON::String original; + /// Where it points, resolved against the base and canonicalised. The + /// document it names and the fragment it carries are that string either + /// side of its `#`, so neither is repeated here + JSON::String destination; + /// Whether that destination is nowhere the frame holds, which is what makes + /// a description one that has to be made whole before it describes anything + bool dangling{false}; + /// What the position that spells it expects to find at the far end, which + /// OpenAPI Specification 3.1.1, Section 4.3.1 fixes by where the reference + /// sits rather than by anything the target says about itself + ObjectKind expected{ObjectKind::Document}; + /// Where the member that spells it sits, which is the one place a rewrite + /// of this reference has to write to + Pointer origin; + }; + + /// One operation of the described API, which is what an endpoint and the Path + /// Item it reaches come to between them + struct Operation { + /// How the entry document exposes it + OperationKind kind; + /// The path template, the webhook name, or the runtime expression that + /// exposes it, according to how it is reached + JSON::String path; + /// The method it answers to, as the field that declares it is spelled + JSON::String method; + /// Where the Operation Object sits + JSON::String origin; + /// Where the Path Item Object that exposes it sits, which is the position + /// that gives it a URL rather than the one that defines it. The two differ + /// whenever a reference stands between them + JSON::String endpoint; + /// Where the Operation Object that a Callback Object hangs off sits, with + /// no value for an operation the Paths Object or the webhooks exposes. + /// Section 4.8.10 has a Callback Object be "a map of possible out-of band + /// callbacks related to the parent operation", and what the expression it + /// is keyed by evaluates against is that Object's request, so one Callback + /// Object two Operation Objects reach describes one callback for each of + /// them + std::optional parent{std::nullopt}; + /// Where the Server Objects in force sit, empty when nothing declares any, + /// in which case Section 4.8.1 puts a single Server Object with a `url` of + /// `/` in their place + std::vector servers; + /// Where the Security Requirement Objects in force sit + std::vector security; + /// Where the Parameter Objects in force sit, which is what the Path Item + /// Object declares once anything the Operation Object overrides is taken + /// out, followed by what the Operation Object declares itself + std::vector parameters; + /// Where the Tag Object each of its tags names sits, in the order the + /// operation wrote them, with no value where the entry document declares no + /// tag by that name. Section 4.8.1 permits exactly that: "Not all tags that + /// are used by the Operation Object must be declared" + std::vector> tags; + }; + + /// Where a Discriminator Object names a schema, by the name of a component or + /// by URI. OpenAPI Specification 3.1.1, Section 4.3 lists the URI form of a + /// `mapping` among the fields that connect the documents of a description, + /// and Section 4.3.3 lists the name form among the connections it makes by + /// name, so either way one of these is a place the description reaches for + struct Discriminator { + /// Where the mapping value sits, as a pointer from the root of the document + Pointer origin; + /// Where it points, resolved and canonicalised + JSON::String destination; + /// What it resolved against, which is the nearest identifier an enclosing + /// schema declares for the URI form, and the description itself for the + /// name form + JSON::String scope; + }; + + /// Where an Object sits in a description, and what the frame knows + /// about it + struct Location { + /// What kind of Object sits here + ObjectKind type; + /// Where in the document it sits, as a pointer from the root + Pointer pointer; + /// Set on the root of a document and on every Schema Object position it + /// holds, empty elsewhere: the default `$schema` in force there, resolved + /// against the base. A + /// Schema Object that declares its own overrides it, which is a matter for + /// whatever reads inside one + JSON::String dialect; + /// Set on a Schema Object position alone, empty elsewhere: the base its + /// document keys every + /// location by, which is what a relative reference inside that schema + /// resolves against until an `$id` says otherwise. RFC 3986 Section 5.1.1 + /// makes an `$id` the higher precedence source, so this is a default in the + /// same way the dialect above is + JSON::String base; + }; + + /// The Objects a frame holds, keyed by the URI addressing each one. + /// Comparison is transparent so that a lookup may take a view of a URI + /// without building a string of it + using Locations = std::map>; + + /// The references a frame holds, keyed by the URI of the Object that makes + /// each one + using References = std::map>; + /// Frame an OpenAPI Description from a given document. That document must /// outlive the frame, as the metadata it reports borrows from it. The given /// base need not, as the frame canonicalises it into a string of its own @@ -147,8 +357,9 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { /// The base is the retrieval URI of the document. OpenAPI 3.1 offers a /// document no way of declaring an identity of its own, so under that /// revision this is the only way to give the description one. From 3.2 - /// onwards a document may declare `$self`, which takes precedence and is - /// resolved against this when relative + /// onwards a document may declare `$self`, which takes precedence once it is + /// absolute, resolving against this when relative and standing aside when + /// neither gives it a scheme /// /// Only the given document is read. A reference that leaves it is recorded /// and left there, and a frame holding one of those does not stand alone @@ -159,10 +370,31 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { /// description may be written against is the caller's to state: pass /// sourcemeta::core::schema_walker and /// sourcemeta::core::schema_resolver for the dialects that are published, - /// and a resolver of your own for one that is not + /// and a resolver of your own for one that is not. The frame keeps the + /// resolver and asks it again when exporting, so whatever it reaches for has + /// to be there for as long as the frame is + /// + /// The places the description holds and the places its Schema Objects hold + /// are places of the one description, so they spend from the one allowance. + /// Bound it to throw sourcemeta::core::OpenAPIFrameLimitError rather than + /// register past it, which reports the allowance the caller set rather than + /// whatever was left of it + /// + /// The base must carry a scheme. One that does not is refused before the + /// document is read, which is why such a refusal names no place within it /// /// A document that does not conform to the specification is rejected here - /// rather than reported back + /// rather than reported back, by throwing sourcemeta::core::OpenAPIError. + /// What sits inside a Schema Object is held to JSON Schema instead, so one + /// naming a dialect nothing resolves throws + /// sourcemeta::core::SchemaResolutionError, one declaring an identifier or a + /// reference that is no URI throws sourcemeta::core::SchemaKeywordError, and + /// two colliding on an identifier or on an anchor throw + /// sourcemeta::core::SchemaFrameError and + /// sourcemeta::core::SchemaAnchorCollisionError respectively. One whose + /// dialect or base dialect cannot be settled at all throws + /// sourcemeta::core::SchemaUnknownDialectError or + /// sourcemeta::core::SchemaUnknownBaseDialectError OpenAPIFrame( const JSON &document, const SchemaWalker &walker, const SchemaResolver &resolver, std::string_view default_base = "", @@ -226,7 +458,8 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { /// against, canonicalised, or the empty URI reference when nothing /// established one, which leaves those references relative. It is the /// `$self` the entry document declares, and the retrieval URI the caller - /// supplied when it declares none. For example: + /// supplied when it declares none or when what it declares cannot be made + /// absolute, in either case stripped of any fragment. For example: /// /// ```cpp /// #include @@ -249,7 +482,9 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { /// Check whether everything this description references is inside what was /// framed, which counts what its Schema Objects reference as much as what - /// the shell around them does. For example: + /// the shell around them does. The dialect a Schema Object names is not one + /// of those, as a schema is under no obligation to carry the meta-schema it + /// is written against. For example: /// /// ```cpp /// #include @@ -296,8 +531,293 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { /// ``` [[nodiscard]] auto schemas() const noexcept -> const SchemaFrame &; - /// Export the frame as JSON. This is the complete state of the frame, and - /// for now its only window + /// Every Object the description holds, keyed by the URI addressing each one + [[nodiscard]] auto locations() const noexcept -> const Locations &; + + /// Every reference the description makes, keyed by the URI of the Object that + /// makes it rather than by its `$ref` member, so that where a reference comes + /// from is a location like any other + [[nodiscard]] auto references() const noexcept -> const References &; + + /// Every scheme a Security Requirement Object names by the URI of one, which + /// OpenAPI Specification 3.2.1 admits alongside the name of a component. + /// These are kept apart from the references above because a single such + /// Object may name several, while every other way of naming an Object is + /// written down where the Object that makes it sits + [[nodiscard]] auto security_references() const noexcept -> const References &; + + /// Every operation the described API exposes + [[nodiscard]] auto operations() const noexcept + -> const std::vector &; + + /// Every URI a Discriminator Object names, which is a reference the schemas + /// hold rather than one the shell around them does + [[nodiscard]] auto discriminators() const noexcept + -> const std::vector &; + + /// Iterate over every Object the description holds, along with the URI the + /// frame keys it by. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// frame.for_each_object([](const auto &uri, const auto &location) { + /// if (location.type == + /// sourcemeta::core::OpenAPIFrame::ObjectKind::Schema) { + /// std::cout << "a schema at " << uri << "\n"; + /// } + /// }); + /// ``` + template F> + auto for_each_object(const F &callback) const -> void { + for (const auto &entry : this->locations()) { + callback(entry.first, entry.second); + } + } + + /// Check whether any Object the description holds satisfies the predicate. + /// For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// assert(frame.any_object([](const auto &, const auto &location) { + /// return location.type == + /// sourcemeta::core::OpenAPIFrame::ObjectKind::Paths; + /// })); + /// ``` + template F> + [[nodiscard]] auto any_object(const F &predicate) const -> bool { + for (const auto &entry : this->locations()) { + if (predicate(entry.first, entry.second)) { + return true; + } + } + + return false; + } + + /// Iterate over every reference the description makes, along with the URI of + /// the Object that makes it. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": { "/users": { "$ref": "#/components/pathItems/Users" } }, + /// "components": { "pathItems": { "Users": {} } } + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// frame.for_each_reference([](const auto &origin, const auto &reference) { + /// std::cout << origin << " -> " << reference.destination << "\n"; + /// }); + /// ``` + template F> + auto for_each_reference(const F &callback) const -> void { + for (const auto &entry : this->references()) { + callback(entry.first, entry.second); + } + } + + /// Iterate over every scheme a Security Requirement Object names by the URI + /// of one, along with the URI of the Object that names it. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// frame.for_each_security_reference( + /// [](const auto &origin, const auto &reference) { + /// std::cout << origin << " -> " << reference.destination << "\n"; + /// }); + /// ``` + template F> + auto for_each_security_reference(const F &callback) const -> void { + for (const auto &entry : this->security_references()) { + callback(entry.first, entry.second); + } + } + + /// Iterate over every operation the described API exposes. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": { "/users": { "get": { "responses": { + /// "200": { "description": "Some users" } } } } } + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// frame.for_each_operation([](const auto &operation) { + /// std::cout << operation.method << " " << operation.path << "\n"; + /// }); + /// ``` + template F> + auto for_each_operation(const F &callback) const -> void { + for (const auto &operation : this->operations()) { + callback(operation); + } + } + + /// Iterate over every URI a Discriminator Object names. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// frame.for_each_discriminator([](const auto &discriminator) { + /// std::cout << discriminator.destination << "\n"; + /// }); + /// ``` + template F> + auto for_each_discriminator(const F &callback) const -> void { + for (const auto &discriminator : this->discriminators()) { + callback(discriminator); + } + } + + /// The Object a URI names, or nothing when the frame holds none. This is what + /// turns the destination of a reference into the Object it lands on, as every + /// destination is one of the URIs this frame addresses its Objects by. For + /// example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver, + /// "https://example.com/openapi.json"}; + /// + /// assert(frame.traverse("https://example.com/openapi.json#/info")->type == + /// sourcemeta::core::OpenAPIFrame::ObjectKind::Info); + /// ``` + [[nodiscard]] auto traverse(const JSON::StringView uri) const + -> const Location *; + + /// The URI this frame addresses the given position by, which is the inverse + /// of traversal. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver, + /// "https://example.com/openapi.json"}; + /// + /// assert(frame.uri(sourcemeta::core::Pointer{"info"}) == + /// "https://example.com/openapi.json#/info"); + /// ``` + [[nodiscard]] auto uri(const Pointer &pointer) const -> JSON::String; + + /// How many Objects the description holds + [[nodiscard]] auto object_count() const noexcept -> std::size_t; + + /// How many references the description makes, not counting what a Security + /// Requirement Object names by the URI of a scheme + [[nodiscard]] auto reference_count() const noexcept -> std::size_t; + + /// Export the frame as JSON. This is the complete state of the frame. It asks + /// the resolver the frame kept, so a meta-schema that has gone out of reach + /// since throws sourcemeta::core::SchemaResolutionError here rather than at + /// construction. For example: + /// + /// ```cpp + /// #include + /// #include + /// #include + /// + /// const auto document{sourcemeta::core::parse_json(R"({ + /// "openapi": "3.1.1", + /// "info": { "title": "Example", "version": "1.0.0" }, + /// "paths": {} + /// })")}; + /// + /// const sourcemeta::core::OpenAPIFrame frame{ + /// document, sourcemeta::core::schema_walker, + /// sourcemeta::core::schema_resolver}; + /// + /// assert(frame.to_json().at("version").to_string() == "3.1"); + /// ``` [[nodiscard]] auto to_json() const -> JSON; private: @@ -315,6 +835,220 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrame { #endif }; +/// @ingroup openapi +/// What a sourcemeta::core::OpenAPIResolver hands back: either a document it +/// owns, or a reference to one that outlives the call +using OpenAPIResolverResult = OwnedOrReference; + +/// @ingroup openapi +/// How bundling reaches the other documents that an OpenAPI Description is +/// split across. Every document handed back must itself be an OpenAPI +/// Description, and a URI that names nothing is reported by handing back no +/// value. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// static auto resolver(const std::string_view identifier) +/// -> sourcemeta::core::OpenAPIResolverResult { +/// if (identifier == "https://example.com/shared.json") { +/// return sourcemeta::core::parse_json(R"JSON({ +/// "openapi": "3.1.1", +/// "info": { "title": "Shared", "version": "1.0.0" }, +/// "components": {} +/// })JSON"); +/// } +/// +/// return std::nullopt; +/// } +/// ``` +using OpenAPIResolver = std::function; + +/// @ingroup openapi +/// +/// This function reorders an OpenAPI Description in place, following an +/// opinionated OpenAPI aware order, and hands every Schema Object it holds to +/// the JSON Schema formatter. Note that doing so invalidates the given frame, +/// as the locations it holds point into the document. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// #include +/// +/// auto document = sourcemeta::core::parse_json(R"JSON({ +/// "info": { "version": "1.0.0", "title": "Example" }, +/// "paths": {}, +/// "openapi": "3.1.1" +/// })JSON"); +/// +/// const sourcemeta::core::OpenAPIFrame frame{ +/// document, sourcemeta::core::schema_walker, +/// sourcemeta::core::schema_resolver}; +/// +/// sourcemeta::core::openapi_format(document, frame); +/// sourcemeta::core::prettify(document, std::cout); +/// ``` +SOURCEMETA_CORE_OPENAPI_EXPORT +auto openapi_format(JSON &document, const OpenAPIFrame &frame) -> void; + +/// @ingroup openapi +/// Everything bundling takes beyond the document and how to reach the rest +struct OpenAPIBundleOptions { + /// A callback to report what bundling embedded, as the URI of the place it + /// came from and a pointer from the root of the document it landed at. + /// Bundling renames what it embeds to hold up as a component name, so this + /// is the only way to know which place became which component + using Callback = + std::function; + + /// A callback to name what bundling embeds, given the URI of the place it + /// came from and the Components Object member it goes under. Whatever it + /// hands back is held to the keys that the specification admits and to + /// being one the description does not already give a meaning to, so it is + /// what bundling starts from rather than the last word + using Namer = std::function; + + /// The URI the document was retrieved from, which every relative reference + /// it makes resolves against. A document that names itself takes that name + /// as its base instead, leaving this as the one a relative such name + /// resolves against + std::string_view default_base{}; + /// The maximum number of locations that analysis may register. How many + /// documents bundling ends up reading follows from what the resolvers hand + /// back rather than from the document the caller passed in, and every walk + /// and every frame that bundling constructs spends from this one allowance, + /// throwing sourcemeta::core::OpenAPIBundleLimitError once it runs out. + /// + /// Bundling settles by reading what it has produced so far over and over + /// until a pass brings nothing new in, so this bounds the reading rather + /// than the result. One place counts once per pass that goes by it and once + /// more for each document brought in alongside it, which puts the allowance + /// a whole description needs well above the number of places it holds. Note + /// too that a document is read in full before anything charges for it, so + /// this bounds how many oversized documents are read rather than whether + /// one is + std::uint64_t max_locations{std::numeric_limits::max()}; + /// A callback to report each place that bundling embedded + Callback callback{}; + /// A callback to name each place that bundling embeds + Namer namer{}; +}; + +/// @ingroup openapi +/// Bundle an OpenAPI Description by embedding everything it references from +/// another document into its own Components Object. The walker and the +/// resolver are what reading inside a Schema Object takes, and the OpenAPI +/// resolver is how the rest of the description is reached. No document the +/// description spans may declare a revision of the OpenAPI Specification other +/// than the one the entry document declares. The specification does not ask +/// for that. It is a choice this makes, as what this produces is one document +/// that declares one revision, and there is none to pick that can express both +/// what one revision holds and what another does. This overload mutates the +/// input document. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// #include +/// +/// static auto resolver(const std::string_view identifier) +/// -> sourcemeta::core::OpenAPIResolverResult { +/// assert(identifier == "https://example.com/shared.json"); +/// return sourcemeta::core::parse_json(R"JSON({ +/// "openapi": "3.1.1", +/// "info": { "title": "Shared", "version": "1.0.0" }, +/// "components": { +/// "responses": { "NotFound": { "description": "Not found" } } +/// } +/// })JSON"); +/// } +/// +/// auto document{sourcemeta::core::parse_json(R"JSON({ +/// "openapi": "3.1.1", +/// "info": { "title": "Example", "version": "1.0.0" }, +/// "paths": { +/// "/pets": { +/// "get": { +/// "responses": { +/// "404": { "$ref": "shared.json#/components/responses/NotFound" } +/// } +/// } +/// } +/// } +/// })JSON")}; +/// +/// sourcemeta::core::openapi_bundle( +/// document, sourcemeta::core::schema_walker, +/// sourcemeta::core::schema_resolver, resolver, +/// {.default_base = "https://example.com/openapi.json"}); +/// +/// assert(document.at("components").at("responses").defines("NotFound")); +/// ``` +SOURCEMETA_CORE_OPENAPI_EXPORT +auto openapi_bundle(JSON &document, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIResolver &resolver, + const OpenAPIBundleOptions &options = {}) -> void; + +/// @ingroup openapi +/// Bundle an OpenAPI Description by embedding everything it references from +/// another document into its own Components Object. No document the description +/// spans may declare a revision of the OpenAPI Specification other than the one +/// the entry document declares, which is a choice this makes rather than one +/// the specification asks for. This overload returns a new document, without +/// mutating the input. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// #include +/// +/// static auto resolver(const std::string_view identifier) +/// -> sourcemeta::core::OpenAPIResolverResult { +/// assert(identifier == "https://example.com/shared.json"); +/// return sourcemeta::core::parse_json(R"JSON({ +/// "openapi": "3.1.1", +/// "info": { "title": "Shared", "version": "1.0.0" }, +/// "components": { +/// "responses": { "NotFound": { "description": "Not found" } } +/// } +/// })JSON"); +/// } +/// +/// const auto document{sourcemeta::core::parse_json(R"JSON({ +/// "openapi": "3.1.1", +/// "info": { "title": "Example", "version": "1.0.0" }, +/// "paths": { +/// "/pets": { +/// "get": { +/// "responses": { +/// "404": { "$ref": "shared.json#/components/responses/NotFound" } +/// } +/// } +/// } +/// } +/// })JSON")}; +/// +/// const auto result{sourcemeta::core::openapi_bundle( +/// document, sourcemeta::core::schema_walker, +/// sourcemeta::core::schema_resolver, resolver, +/// {.default_base = "https://example.com/openapi.json"})}; +/// +/// assert(result.at("components").at("responses").defines("NotFound")); +/// ``` +SOURCEMETA_CORE_OPENAPI_EXPORT +auto openapi_bundle(const JSON &document, const SchemaWalker &walker, + const SchemaResolver &schema_resolver, + const OpenAPIResolver &resolver, + const OpenAPIBundleOptions &options = {}) -> JSON; + } // namespace sourcemeta::core #endif diff --git a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h index 9f4cbc516..bac2bb518 100644 --- a/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h +++ b/vendor/core/src/core/openapi/include/sourcemeta/core/openapi_error.h @@ -80,6 +80,124 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIError : public std::exception { const char *message_; }; +/// @ingroup openapi +/// An error that represents a document of an OpenAPI Description that nothing +/// could produce. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// const sourcemeta::core::OpenAPIResolutionError error{ +/// "https://example.com/openapi.json", sourcemeta::core::Pointer{"$ref"}, +/// "https://example.com/shared.json", "Could not resolve"}; +/// assert(error.identifier() == "https://example.com/shared.json"); +/// ``` +class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIResolutionError + : public std::exception { +public: + /// Create a resolution error from the document the reference was made in, + /// where in it the reference sits, what it names, and a message + OpenAPIResolutionError(JSON::String base, Pointer location, + JSON::String identifier, const char *message) + : base_{std::move(base)}, location_{std::move(location)}, + identifier_{std::move(identifier)}, message_{message} {} + OpenAPIResolutionError(JSON::String base, Pointer location, + JSON::String identifier, std::string message) = delete; + OpenAPIResolutionError(JSON::String base, Pointer location, + JSON::String identifier, + std::string &&message) = delete; + OpenAPIResolutionError(JSON::String base, Pointer location, + JSON::String identifier, + std::string_view message) = delete; + + [[nodiscard]] auto what() const noexcept -> const char * override { + return this->message_; + } + + /// The URI that nothing could produce a document for + [[nodiscard]] auto identifier() const noexcept -> JSON::StringView { + return this->identifier_; + } + + /// Get where the reference is, as a pointer from the root of the document + /// that makes it + [[nodiscard]] auto location() const noexcept -> const Pointer & { + return this->location_; + } + + /// Get the base URI of the document that makes the reference + [[nodiscard]] auto base() const noexcept -> JSON::StringView { + return this->base_; + } + +private: + JSON::String base_; + Pointer location_; + JSON::String identifier_; + const char *message_; +}; + +/// @ingroup openapi +/// An error that represents a reference of an OpenAPI Description that names +/// something it may not. For example: +/// +/// ```cpp +/// #include +/// #include +/// #include +/// +/// const sourcemeta::core::OpenAPIReferenceError error{ +/// "https://example.com/openapi.json", sourcemeta::core::Pointer{"$ref"}, +/// "https://example.com/shared.json#/info", "Wrong kind"}; +/// assert(error.identifier() == "https://example.com/shared.json#/info"); +/// ``` +class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIReferenceError + : public std::exception { +public: + /// Create a reference error from the document the reference was made in, + /// where in it the reference sits, what it names, and a message + OpenAPIReferenceError(JSON::String base, Pointer location, + JSON::String identifier, const char *message) + : base_{std::move(base)}, location_{std::move(location)}, + identifier_{std::move(identifier)}, message_{message} {} + OpenAPIReferenceError(JSON::String base, Pointer location, + JSON::String identifier, std::string message) = delete; + OpenAPIReferenceError(JSON::String base, Pointer location, + JSON::String identifier, + std::string &&message) = delete; + OpenAPIReferenceError(JSON::String base, Pointer location, + JSON::String identifier, + std::string_view message) = delete; + + [[nodiscard]] auto what() const noexcept -> const char * override { + return this->message_; + } + + /// The URI that the reference names, resolved and canonicalised + [[nodiscard]] auto identifier() const noexcept -> JSON::StringView { + return this->identifier_; + } + + /// Get where the reference is, as a pointer from the root of the document + /// that makes it + [[nodiscard]] auto location() const noexcept -> const Pointer & { + return this->location_; + } + + /// Get the base URI of the document that makes the reference + [[nodiscard]] auto base() const noexcept -> JSON::StringView { + return this->base_; + } + +private: + JSON::String base_; + Pointer location_; + JSON::String identifier_; + const char *message_; +}; + /// @ingroup openapi /// An error that represents framing that ran past what the caller allowed it /// to register. For example: @@ -111,6 +229,37 @@ class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIFrameLimitError std::uint64_t limit_; }; +/// @ingroup openapi +/// An error that represents bundling that ran past what the caller allowed it +/// to analyse. For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// const sourcemeta::core::OpenAPIBundleLimitError error{100}; +/// assert(error.limit() == 100); +/// ``` +class SOURCEMETA_CORE_OPENAPI_EXPORT OpenAPIBundleLimitError + : public std::exception { +public: + /// Create a bundling limit error + OpenAPIBundleLimitError(const std::uint64_t limit) : limit_{limit} {} + + [[nodiscard]] auto what() const noexcept -> const char * override { + return "The OpenAPI Description exceeds the maximum number of locations " + "that bundling may analyse"; + } + + /// The maximum number of locations that bundling was allowed to register + [[nodiscard]] auto limit() const noexcept -> std::uint64_t { + return this->limit_; + } + +private: + std::uint64_t limit_; +}; + #if defined(_MSC_VER) #pragma warning(pop) #endif diff --git a/vendor/core/src/core/openapi/parameter.h b/vendor/core/src/core/openapi/parameter.h index 7b8c0dd34..db60f6dad 100644 --- a/vendor/core/src/core/openapi/parameter.h +++ b/vendor/core/src/core/openapi/parameter.h @@ -43,7 +43,7 @@ constexpr std::array OPENAPI_PARAMETER_SCHEMA_FIELDS{ {"name"sv, "in"sv, "description"sv, "required"sv, "deprecated"sv, "schema"sv, "style"sv, "explode"sv, "example"sv, "examples"sv}}; -// Section 4.12, of `allowReserved`: "This field only applies to `in` and +// 3.2.1 Section 4.12, of `allowReserved`: "This field only applies to `in` and // `style` values that automatically percent-encode (that is: `in: path`, // `in: query`, and `in: cookie` with `style: form`)". 3.1 held the same field // to `query` alone @@ -69,10 +69,31 @@ constexpr std::array {"name"sv, "in"sv, "description"sv, "required"sv, "deprecated"sv, "content"sv, "allowEmptyValue"sv, "example"sv, "examples"sv}}; -// Section 4.12 scopes `allowReserved` to the locations that percent-encode of -// their own accord. A query parameter has a field table of its own, so what is -// asked here is whether one of the other two locations is such a place, and a -// cookie parameter takes the `form` style when it declares none +// Section 4.8.12.2.1 holds `allowEmptyValue` among the fields that "MAY be used +// with either `content` or `schema`" and restricts it in its own Description +// cell alone: "This field is valid only for `query` parameters". So a +// parameter elsewhere writing it writes a field this Object defines rather +// than one it does not, which is a different thing to be turned down for +constexpr std::array + OPENAPI_PARAMETER_INAPPLICABLE_COMMON_FIELDS{{"allowEmptyValue"sv}}; + +// 3.2.1 Section 4.12.2.2 holds `allowReserved` among the fields for use with +// `schema` and restricts it the same way: "This field only applies to `in` +// and `style` values that automatically percent-encode". 3.1 restricts the +// same field to `query` alone. A `content` form reaches neither field table, +// so this stands for the `schema` form where both fields are written down +constexpr std::array + OPENAPI_PARAMETER_INAPPLICABLE_SCHEMA_FIELDS{ + {"allowEmptyValue"sv, "allowReserved"sv}}; + +constexpr auto OPENAPI_PARAMETER_INAPPLICABLE_MESSAGE{ + "The Parameter Object does not admit this field as declared"}; + +// 3.2.1 Section 4.12 scopes `allowReserved` to the locations that +// percent-encode of their own accord. A query parameter has a field table of +// its own, so what is asked here is whether one of the other two locations is +// such a place, and a cookie parameter takes the `form` style when it declares +// none inline auto openapi_parameter_admits_reserved(const JSON::StringView location, const JSON &value) -> bool { if (location == "path"sv) { @@ -144,8 +165,8 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, base, "The Parameter Object must declare a schema or a content"}; } - // Section 4.12, of `querystring`: its value "MUST be specified using the - // `content` field" + // 3.2.1 Section 4.12, of `querystring`: its value "MUST be specified using + // the `content` field" if (parameter_location == "querystring"sv && content == nullptr) { throw OpenAPIError{base, "A querystring Parameter Object must declare a content"}; @@ -162,7 +183,9 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, openapi_reject_unknown_fields( value, OPENAPI_PARAMETER_CONTENT_FIELDS_3_1, OPENAPI_PARAMETER_CONTENT_FIELDS_3_2, base, - "The Parameter Object does not define this field", walk); + "The Parameter Object does not define this field", walk, + OPENAPI_PARAMETER_INAPPLICABLE_COMMON_FIELDS, + OPENAPI_PARAMETER_INAPPLICABLE_MESSAGE); } } else if (in_query) { openapi_reject_unknown_fields( @@ -172,11 +195,15 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, openapi_parameter_admits_reserved(parameter_location, value)) { openapi_reject_unknown_fields( value, OPENAPI_PARAMETER_RESERVED_SCHEMA_FIELDS_3_2, base, - "The Parameter Object does not define this field"); + "The Parameter Object does not define this field", + OPENAPI_PARAMETER_INAPPLICABLE_COMMON_FIELDS, + OPENAPI_PARAMETER_INAPPLICABLE_MESSAGE); } else { openapi_reject_unknown_fields( value, OPENAPI_PARAMETER_SCHEMA_FIELDS, base, - "The Parameter Object does not define this field"); + "The Parameter Object does not define this field", + OPENAPI_PARAMETER_INAPPLICABLE_SCHEMA_FIELDS, + OPENAPI_PARAMETER_INAPPLICABLE_MESSAGE); } openapi_check_optional_string( @@ -208,12 +235,20 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, "A path Parameter Object must be required"}; } - // OpenAPI Specification 3.1.1, Section 4.8.12: "If `in` is `"path"`, the - // `name` field MUST correspond to a template expression occurring within - // the path field in the Paths Object", and a template expression is - // delimited by braces, so its name can hold neither - if (parameter_name.find('{') != JSON::StringView::npos || - parameter_name.find('}') != JSON::StringView::npos) { + // OpenAPI Specification 3.2.1, Section 4.12.2.1 has such a name "correspond + // to a single template expression occurring within the path field in the + // Paths Object", and 3.2.1 Section 4.8.2 writes out what one may hold: + // "template-expression-param-name = 1*( %x00-7A / %x7C / %x7E-10FFFF ) ; + // every Unicode character except { and }". No revision of 3.1 carries that + // grammar, and all 3.1.1 Section 3.5 says of the shape is "template + // expressions, delimited by curly braces", which leaves a brace within a + // name to whatever reads the path. Holding a 3.1 document to this would + // also leave it nothing to declare, as the expression of `/a/{x{y}` reads + // as `x{y` there and the correspondence below would then ask for the very + // name this turns down + if (walk.version == OpenAPIVersion::OPENAPI_3_2 && + (parameter_name.find('{') != JSON::StringView::npos || + parameter_name.find('}') != JSON::StringView::npos)) { throw OpenAPIError{openapi_child(base, "name"sv), "A path Parameter Object name must not hold braces"}; } @@ -257,8 +292,16 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, openapi_expect_schema(*schema, openapi_child(base, "schema"sv), "A Schema Object must be an object or a boolean", walk); - // OpenAPI Specification 3.1.1, Section 4.8.12 lists the styles each location - // admits, and the meta-schema enumerates them per location + // Section 4.8.12.3 gives the styles an `in` column, and both revisions fill + // it with the same four locations. 3.2.1 Section 4.12.3 goes on to close the + // table, "Combinations not represented in this table are not permitted", + // which 3.1 does not say. It does not need to: the column is the table, and + // Section 4.8.21 reads it as binding where it derives a Header Object's own + // restriction from it, "All traits that are affected by the location MUST be + // applicable to a location of `header` (for example, `style`) ... and + // `style`, if used, MUST be limited to `"simple"`". So each location admits + // what its column says in both revisions, and only the `cookie` style 3.2 + // adds is held back const auto *style{value.try_at("style", OPENAPI_HASH_STYLE)}; if (style != nullptr) { if (parameter_location == "path"sv) { @@ -278,10 +321,9 @@ inline auto openapi_check_parameter(const JSON &value, const Pointer &base, "The Parameter Object style must be a string", "The Parameter Object style is not one a query parameter admits"); } else if (walk.version == OpenAPIVersion::OPENAPI_3_2) { - // OpenAPI Specification 3.2.1, Section 4.12.3 adds a `cookie` style, - // "analogous to `form`, but following RFC6265 `Cookie` syntax rules", - // and the same table now states that "combinations not represented in - // this table are not permitted" + // 3.2.1 Section 4.12.3 adds a `cookie` style, "Analogous to `form`, but + // following [RFC6265] `Cookie` syntax rules", which no revision of 3.1 + // carries openapi_expect_enumeration( *style, base, "style"sv, {"form"sv, "cookie"sv}, "The Parameter Object style must be a string", @@ -344,9 +386,9 @@ inline auto openapi_check_parameters(const JSON &value, const Pointer &base, // These own their strings, as an identity read back through a reference // borrows from a map the walk keeps writing to std::set> seen; - // Section 4.12, of `querystring`: it "MUST NOT appear more than once, and - // MUST NOT appear in the same operation (or in the operation's path-item) as - // any `in: "query"` parameters". Both halves hold of a single list on its + // 3.2.1 Section 4.12, of `querystring`: it "MUST NOT appear more than once, + // and MUST NOT appear in the same operation (or in the operation's path-item) + // as any `in: "query"` parameters". Both halves hold of a single list on its // own, and what the two levels come to between them is settled where they // meet std::size_t querystrings{0}; diff --git a/vendor/core/src/core/openapi/path_item.h b/vendor/core/src/core/openapi/path_item.h index e3fdf76cf..af7732f9a 100644 --- a/vendor/core/src/core/openapi/path_item.h +++ b/vendor/core/src/core/openapi/path_item.h @@ -187,11 +187,13 @@ inline auto openapi_check_operation(const JSON &value, const Pointer &base, const auto location{openapi_child(base, "callbacks"sv)}; openapi_expect_object(*callbacks, location, "The Operation Object callbacks must be an object"); + // Section 4.8.10: "callbacks | Map[string, Callback Object | Reference + // Object] | A map of possible out-of band callbacks related to the parent + // operation [...] The key is a unique identifier for the Callback Object". + // Its keys are identifiers rather than field names, and like the webhooks + // map and unlike the Paths Object this one carries no extension carve-out, + // so a member named `x-` is a callback for (const auto &entry : callbacks->as_object()) { - if (entry.first.starts_with(OPENAPI_EXTENSION_PREFIX)) { - continue; - } - const auto callback{openapi_child(location, entry.first)}; openapi_check_callbacks_or_reference(entry.second, callback, walk); record.callbacks.push_back(openapi_location_uri(walk.base, callback)); diff --git a/vendor/core/src/core/openapi/paths.h b/vendor/core/src/core/openapi/paths.h index 3623598da..bb9af1c90 100644 --- a/vendor/core/src/core/openapi/paths.h +++ b/vendor/core/src/core/openapi/paths.h @@ -138,7 +138,15 @@ inline auto openapi_check_paths(const JSON &document, OpenAPIWalk &walk) // writes out a grammar and forbids repeating an expression: "Each // template expression MUST NOT appear more than once in a single path // template". Both are new in 3.2, so a path 3.1 accepts is still accepted - // when a document declares 3.1 + // when a document declares 3.1. + // + // The grammar is introduced as a definition, and neither of the two + // requirements beside it speaks of the key's shape, so what licenses + // turning a key down for its shape is the same clause the Server Object + // goes by: 3.2.1 Section 4.12.4, "All API URLs MUST successfully parse and + // percent-decode using [RFC3986] rules", a path being appended to a server + // URL to make one. The grammar is the whole of what the specification says + // a path template is, and it is read for nothing but the shape if (walk.version == OpenAPIVersion::OPENAPI_3_2) { if (!openapi_is_path_template(entry.first)) { throw OpenAPIError{location, diff --git a/vendor/core/src/core/openapi/response.h b/vendor/core/src/core/openapi/response.h index 32c892ac0..a6c704008 100644 --- a/vendor/core/src/core/openapi/response.h +++ b/vendor/core/src/core/openapi/response.h @@ -106,13 +106,19 @@ inline auto openapi_check_responses(const JSON &value, const Pointer &base, openapi_record(walk, base, OpenAPIObjectKind::Responses); openapi_expect_object(value, base, "The Responses Object must be an object"); - // The meta-schema bounds this map from below, and requires a `default` when - // no status code is named + // Section 4.8.16: "The Responses Object MUST contain at least one response + // code". Neither revision says which keys count as one, and the OpenAPI + // Initiative settled that twice over: the published meta-schema carries + // "either default, or at least one response code property must exist", and + // their own conformance corpus files a Responses Object holding nothing but + // a `default` under the passing cases of both revisions. So a `default` + // answers for a response here, and only an Object answering for none at all + // is turned down if (value.empty()) { throw OpenAPIError{base, "The Responses Object must not be empty"}; } - bool names_a_status_code{false}; + bool answers_for_a_response{false}; for (const auto &entry : value.as_object()) { if (entry.first.starts_with(OPENAPI_EXTENSION_PREFIX)) { continue; @@ -120,6 +126,7 @@ inline auto openapi_check_responses(const JSON &value, const Pointer &base, const auto location{openapi_child(base, entry.first)}; if (entry.first == "default"sv) { + answers_for_a_response = true; openapi_check_response_or_reference(entry.second, location, walk); continue; } @@ -130,14 +137,13 @@ inline auto openapi_check_responses(const JSON &value, const Pointer &base, "code or a status code range"}; } - names_a_status_code = true; + answers_for_a_response = true; openapi_check_response_or_reference(entry.second, location, walk); } - if (!names_a_status_code && - value.try_at("default", OPENAPI_HASH_DEFAULT) == nullptr) { + if (!answers_for_a_response) { throw OpenAPIError{ - base, "The Responses Object must declare a default or a status code"}; + base, "The Responses Object must declare a status code or a default"}; } } diff --git a/vendor/core/src/core/openapi/security.h b/vendor/core/src/core/openapi/security.h index 3df4a23af..4503beb52 100644 --- a/vendor/core/src/core/openapi/security.h +++ b/vendor/core/src/core/openapi/security.h @@ -191,7 +191,7 @@ inline auto openapi_check_oauth_flow(const JSON &value, const Pointer &base, } // OpenAPI Specification 3.1.1, Section 4.8.29: "scopes | Map[string, string] - // | REQUIRED. The available scopes for the OAuth2 security scheme" + // | oauth2 | REQUIRED. The available scopes for the OAuth2 security scheme." const auto &scopes{ openapi_require(value, "scopes"sv, OPENAPI_HASH_SCOPES, base, "The OAuth Flow Object must declare its scopes")}; @@ -233,8 +233,8 @@ inline auto openapi_check_oauth_flows(const JSON &value, const Pointer &base, true, false, walk); } - // Section 4.28 adds the device authorization flow, whose required URLs are - // its own and the token URL + // 3.2.1 Section 4.28 adds the device authorization flow, whose required URLs + // are its own and the token URL const auto *device{ value.try_at("deviceAuthorization"sv, OPENAPI_HASH_DEVICE_AUTHORIZATION)}; if (device != nullptr) { @@ -355,7 +355,7 @@ inline auto openapi_check_security_scheme(const JSON &value, OPENAPI_SECURITY_SCHEME_OAUTH2_FIELDS_3_2, base, "The Security Scheme Object does not define this field", walk); - // Section 4.27: "oauth2MetadataUrl | string | oauth2" + // 3.2.1 Section 4.27: "oauth2MetadataUrl | string | oauth2" const auto *metadata{ value.try_at("oauth2MetadataUrl", OPENAPI_HASH_OAUTH2_METADATA_URL)}; if (metadata != nullptr) { @@ -432,6 +432,19 @@ inline auto openapi_check_security_scheme_name(const JSON::StringView name, "scheme or the URI of one"}; } + // A name that leads out of the document it was written in is one this does + // not hold, and saying so is what keeps a description that spans more than + // one document from reading as whole. Every other way of naming another + // Object is written down where the Object that makes it sits, which one of + // these cannot be, as a single Security Requirement Object may name several + walk.security_references.insert_or_assign( + openapi_location_uri(walk.base, origin), + OpenAPIReference{.original = JSON::String{name}, + .destination = target.value().recompose(), + .dangling = false, + .expected = OpenAPIObjectKind::SecurityScheme, + .origin = origin}); + // Naming a whole OpenAPI Description is naming something that is not a // Security Scheme Object openapi_follow_target(target.value(), origin, @@ -449,7 +462,7 @@ inline auto openapi_check_security_requirement(const JSON &value, // declared in the Security Schemes under the Components Object". This // Object declares no pattern but its names, so a member called `x-` is a // scheme name and is held to the same requirement - if (!walk.security_schemes.contains(entry.first)) { + if (!walk.security_schemes.contains(entry.first, entry.hash)) { openapi_check_security_scheme_name( entry.first, openapi_child(base, entry.first), walk); } diff --git a/vendor/core/src/core/openapi/server.h b/vendor/core/src/core/openapi/server.h index b4ab2d144..1cd19fe59 100644 --- a/vendor/core/src/core/openapi/server.h +++ b/vendor/core/src/core/openapi/server.h @@ -7,11 +7,13 @@ #include #include +#include #include #include // std::ranges::any_of #include // std::array #include // std::size_t +#include // std::optional, std::nullopt #include // std::set #include // std::string_view #include // std::move @@ -115,7 +117,184 @@ inline auto openapi_check_server_variable(const JSON &value, // // A variable name admits "every Unicode character except { and }", which of // the bytes of one holds only of a brace, so it is read byte by byte while a -// literal is read a character at a time +// literal is read a character at a time. +// +// That grammar is this specification's own rather than RFC 6570's, and the two +// disagree. RFC 6570 Section 2.3 admits only ALPHA, DIGIT and an underscore, +// plus a percent-encoded triplet into a variable name, so a name holding a +// hyphen is a server variable here and no expression at all there. RFC 6570 +// Section 2.2 further reads a leading `+` as an operator, where this grammar +// reads it as the first character of the name, so one template means two things +// depending on which of the two is doing the reading. +// +// RFC 6570 does govern this specification, but elsewhere. Appendix C of 3.1.1 +// scopes it to serialising a value: "Serialization is defined in terms of +// RFC6570 URI Templates in three scenarios", being a Parameter Object and a +// Header Object that declare a schema, and an Encoding Object for a form. What +// a Server Object URL or a templated path is made of is not one of those, so +// reading either through an RFC 6570 implementation would reject names this +// grammar admits. Hence a reader of its own, kept here rather than beside one +// that answers for a specification this does not follow + +// What a server URL template means is the URL that substituting its variables +// produces, resolved against the base of the document that declares it. So a +// template that moves to another document has to be resolved against the one +// it came from first. +// +// RFC 3986 Section 5.2.2 branches on one question alone, which is whether the +// reference declares a scheme, a network path, an absolute path or a relative +// path. Everything past that only prepends a fixed part of the base, and for a +// relative path merges the base's directory. So resolving and substituting +// commute exactly when the literals of a template settle which of the four it +// is, and a template that leaves that to a variable means no one thing that +// resolving could preserve. That case is turned down rather than guessed at, +// and is reported by handing back no value +// What a Server Object URL template names once every variable stands for what +// it is declared to. OpenAPI Specification 3.1.1, Section 4.8.6 makes a Server +// Variable Object's `default` "REQUIRED. The default value to use for +// substitution, which SHALL be sent if an alternate value is not supplied", so +// a template always names at least one concrete URL +inline auto openapi_substitute_server_variables(const JSON::StringView address, + const JSON &variables, + const JSON::StringView varied, + const JSON::StringView value) + -> std::optional { + JSON::String result; + JSON::StringView::size_type cursor{0}; + while (cursor < address.size()) { + const auto opening{address.find('{', cursor)}; + if (opening == JSON::StringView::npos) { + result.append(address.substr(cursor)); + break; + } + + const auto closing{address.find('}', opening)}; + if (closing == JSON::StringView::npos) { + return std::nullopt; + } + + result.append(address.substr(cursor, opening - cursor)); + const auto *variable{ + variables.try_at(address.substr(opening + 1, closing - opening - 1))}; + if (variable == nullptr || !variable->is_object()) { + return std::nullopt; + } + + const auto name{address.substr(opening + 1, closing - opening - 1)}; + if (!varied.empty() && name == varied) { + result.append(value); + cursor = closing + 1; + continue; + } + + const auto *fallback{variable->try_at("default")}; + if (fallback == nullptr || !fallback->is_string()) { + return std::nullopt; + } + + result.append(fallback->to_string()); + cursor = closing + 1; + } + + return result; +} + +// Whether a server URL template names an absolute URI whatever its variables +// stand for. Section 4.8.6 bounds that: a `default` is "REQUIRED. The default +// value to use for substitution, which SHALL be sent if an alternate value is +// not supplied", and where an `enum` is present "the value MUST exist in the +// enum's values", so between them they are the whole of what one may stand +// for. A single value that leaves the template relative leaves what it names +// for the document holding it to settle, which is the one thing moving it +// changes, so the default answering alone says too little +inline auto +openapi_is_absolute_server_url_template(const JSON::StringView address, + const JSON &variables) -> bool { + const auto baseline{ + openapi_substitute_server_variables(address, variables, {}, {})}; + if (!baseline.has_value() || !URI::is_uri(baseline.value())) { + return false; + } + + for (const auto &variable : variables.as_object()) { + const auto *choices{variable.second.try_at("enum")}; + if (choices == nullptr || !choices->is_array()) { + continue; + } + + for (const auto &choice : choices->as_array()) { + if (!choice.is_string()) { + return false; + } + + const auto candidate{openapi_substitute_server_variables( + address, variables, variable.first, choice.to_string())}; + if (!candidate.has_value() || !URI::is_uri(candidate.value())) { + return false; + } + } + } + + return true; +} + +inline auto openapi_resolve_server_url(const JSON::StringView address, + const JSON::String &base, + const JSON *const variables) + -> std::optional { + const auto variable{address.find('{')}; + // A template that declares no variable is a URI reference already + if (variable == JSON::StringView::npos) { + const auto resolved{openapi_resolve_uri(address, base)}; + return resolved.has_value() + ? std::optional{resolved.value().recompose()} + : std::nullopt; + } + + const auto literals{address.substr(0, variable)}; + const auto scheme{literals.find(':')}; + const auto separator{literals.find('/')}; + + // RFC 3986 Section 5.2.2 resolves a reference that declares a scheme to + // itself, so whatever a variable stands for past that point cannot change the + // result + if (scheme != JSON::StringView::npos && + (separator == JSON::StringView::npos || scheme < separator)) { + return JSON::String{address}; + } + + // Until a slash settles it, a variable may still stand for a colon and make + // what comes before it a scheme, which leaves the kind of reference this is + // for the values of the variables to decide rather than the template. They + // do decide it, as every one of them declares what it stands for by default, + // and a template that names an absolute URI that way names the same place + // wherever the Object holding it comes to sit + if (separator == JSON::StringView::npos) { + if (variables == nullptr || !variables->is_object()) { + return std::nullopt; + } + + if (!openapi_is_absolute_server_url_template(address, *variables)) { + return std::nullopt; + } + + return JSON::String{address}; + } + + // RFC 3986 Section 5.2.4 removes dot segments from the path that merging + // produces, which is an operation on whole segments. Splitting at a segment + // boundary is what keeps it from reaching into what a variable stands for + const auto boundary{literals.find_last_of('/') + 1}; + const auto resolved{openapi_resolve_uri(address.substr(0, boundary), base)}; + if (!resolved.has_value()) { + return std::nullopt; + } + + auto result{resolved.value().recompose()}; + result.append(address.substr(boundary)); + return result; +} + inline auto openapi_is_server_url_template(const JSON::StringView address) -> bool { if (address.empty()) { @@ -176,10 +355,17 @@ inline auto openapi_check_server(const JSON &value, const Pointer &base, url, base, "url"sv, "The Server Object URL must be a string")}; // OpenAPI Specification 3.1.2, Section 4.8.5 adds to that row: "Query and - // fragment MUST NOT be part of this URL". A query begins at the first `?` - // and a fragment at the first `#`, so either character in the template - // starts one, whether or not it sits inside a variable expression. A - // percent-encoded one is neither and is left alone + // fragment MUST NOT be part of this URL". That revision is held to every + // 3.1 document rather than to the ones that declare it, as Section 4.1 has + // a patch release "address errors in, or provide clarifications to, this + // document, not the feature set" and goes on: "The patch version SHOULD NOT + // be considered by tooling, making no distinction between `3.1.0` and + // `3.1.1`". So reading this one as a clarification of what 3.1 always meant + // is what that rule asks for. + // + // A query begins at the first `?` and a fragment at the first `#`, so either + // character in the template starts one, whether or not it sits inside a + // variable expression. A percent-encoded one is neither and is left alone if (address.find('?') != JSON::StringView::npos || address.find('#') != JSON::StringView::npos) { throw OpenAPIError{ @@ -191,7 +377,21 @@ inline auto openapi_check_server(const JSON &value, const Pointer &base, // of 3.2 writes out a grammar for it and forbids repeating a variable: // "Each server variable MUST NOT appear more than once in the URL template". // Both are new in 3.2, so a URL 3.1 accepts is still accepted when a - // document declares 3.1 + // document declares 3.1. + // + // The grammar is introduced as a definition rather than as a requirement, so + // what licenses turning a URL down for its shape is 3.2.1 Section 4.12.4, + // "All API URLs MUST successfully parse and percent-decode using [RFC3986] + // rules", together with 3.2.1 Section 4 making the text "the only normative + // description of the format". The grammar is the whole of what either + // revision says a server URL template is, and nothing beyond its shape is + // read from it. + // + // That MUST is also 3.1.2 Section 4.8.12.4, which by Section 4.1 reaches + // every 3.1 document, so holding this to 3.2 alone leaves a 3.1 URL that no + // amount of parsing can rescue unreported. That is under-reporting rather + // than a wrong refusal, and closing it would turn down documents accepted + // until now, so it waits for a release that can carry it if (walk.version == OpenAPIVersion::OPENAPI_3_2) { if (!openapi_is_server_url_template(address)) { throw OpenAPIError{openapi_child(base, "url"sv), diff --git a/vendor/core/src/core/openapi/tag.h b/vendor/core/src/core/openapi/tag.h index 24895165e..b5240eb6e 100644 --- a/vendor/core/src/core/openapi/tag.h +++ b/vendor/core/src/core/openapi/tag.h @@ -64,7 +64,7 @@ inline auto openapi_check_tag(const JSON &value, const Pointer &base, openapi_check_optional_string(value, base, "kind"sv, OPENAPI_HASH_KIND, "The Tag Object kind must be a string"); - // Section 4.22: "parent | string | The `name` of a tag that this tag is + // 3.2.1 Section 4.22: "parent | string | The `name` of a tag that this tag is // nested under. The named tag MUST exist in the API description, and // circular references between parent and child tags MUST NOT be used". // Neither of those can be settled until every tag has been read diff --git a/vendor/core/src/core/punycode/punycode.cc b/vendor/core/src/core/punycode/punycode.cc index 006f3240e..0e2749573 100644 --- a/vendor/core/src/core/punycode/punycode.cc +++ b/vendor/core/src/core/punycode/punycode.cc @@ -237,11 +237,16 @@ static auto punycode_decode(const std::string_view encoded, break; } + // The digit check at the top of this loop already throws on the inputs + // that would overflow here. It lets a digit through only while the + // running weight leaves room for it, and a digit is never below the + // threshold at this point, so both could hold at once only while the + // threshold stays under half the base. Bias adaptation cannot reach far + // enough for that to last beyond the sixth digit of a code point, and the + // running weight is still an order of magnitude under this bound there const std::uint32_t base_minus_threshold = BASE - threshold; - if (weight_factor > - std::numeric_limits::max() / base_minus_threshold) { - throw PunycodeError("Decode overflow"); - } + assert(weight_factor <= + std::numeric_limits::max() / base_minus_threshold); weight_factor *= base_minus_threshold; } @@ -256,13 +261,13 @@ static auto punycode_decode(const std::string_view encoded, throw PunycodeError("Decode overflow"); } + // RFC 3492 Section 6.2 fails here when the code point decoded is a basic + // one. That cannot happen: the running code point starts at the first + // non-basic one and only ever grows, and the check above rules out the wrap + // that is the one way it could come back under current_code_point += increment; insertion_index %= output_length; - if (current_code_point < INITIAL_N) { - throw PunycodeError("Decoded basic code point"); - } - if (current_code_point > 0x10FFFF || (current_code_point >= 0xD800 && current_code_point <= 0xDFFF)) { throw PunycodeError("Invalid code point"); diff --git a/vendor/core/src/core/regex/CMakeLists.txt b/vendor/core/src/core/regex/CMakeLists.txt index f10eec028..877984360 100644 --- a/vendor/core/src/core/regex/CMakeLists.txt +++ b/vendor/core/src/core/regex/CMakeLists.txt @@ -5,6 +5,11 @@ if(SOURCEMETA_CORE_INSTALL) sourcemeta_library_install(NAMESPACE sourcemeta PROJECT core NAME regex) endif() +# An object library hands its objects to the targets that name it, and a +# second object library in between only passes on usage requirements, so the +# code generator has to be named here rather than left to the library that +# actually calls into it +target_link_libraries(sourcemeta_core_regex PRIVATE SLJIT::sljit) target_link_libraries(sourcemeta_core_regex PRIVATE PCRE2::pcre2) target_link_libraries(sourcemeta_core_regex PRIVATE sourcemeta::core::text) target_link_libraries(sourcemeta_core_regex PRIVATE sourcemeta::core::unicode) diff --git a/vendor/core/src/core/uri/filesystem.cc b/vendor/core/src/core/uri/filesystem.cc index 853fc551e..e7d783907 100644 --- a/vendor/core/src/core/uri/filesystem.cc +++ b/vendor/core/src/core/uri/filesystem.cc @@ -22,9 +22,6 @@ auto is_localhost_host(const std::string_view host) -> bool { auto append_raw_segment(std::optional &path, const std::string_view segment) -> void { - if (segment.empty()) { - return; - } if (!path.has_value()) { path = std::string{segment}; return; @@ -114,12 +111,20 @@ auto URI::from_path(const std::filesystem::path &path) -> URI { URI result{"file://"}; auto iterator = final_path.begin(); - // For UNC paths, the first segment is the hostname - if (is_unc) { + // For UNC paths, the first segment is the hostname, which a root made of + // nothing but separators does not have + if (is_unc && iterator != final_path.end()) { result.host_ = iterator->string(); std::advance(iterator, 1); } + // A path made of nothing but separators is the root itself, and the strip + // above leaves no segment for the loop to walk. RFC 8089 spells the root as + // an empty authority followed by "/", so it is set here rather than lost + if (normalized.empty()) { + result.path_ = "/"; + } + // Process remaining path segments for (; iterator != final_path.end(); ++iterator) { if (iterator->empty()) { diff --git a/vendor/core/src/core/yaml/include/sourcemeta/core/yaml.h b/vendor/core/src/core/yaml/include/sourcemeta/core/yaml.h index e9b614bb6..8274b590d 100644 --- a/vendor/core/src/core/yaml/include/sourcemeta/core/yaml.h +++ b/vendor/core/src/core/yaml/include/sourcemeta/core/yaml.h @@ -12,8 +12,10 @@ #include // NOLINTEND(misc-include-cleaner) +#include // std::size_t #include // std::filesystem #include // std::basic_istream +#include // std::optional, std::nullopt #include // std::basic_ostream /// @defgroup yaml YAML @@ -170,6 +172,31 @@ auto read_yaml_or_json(const std::filesystem::path &path, JSON &output, SOURCEMETA_CORE_YAML_EXPORT auto parse_yaml(const JSON::String &input, YAMLRoundTrip &roundtrip) -> JSON; +/// @ingroup yaml +/// +/// Create a JSON document from a C++ standard input stream that represents a +/// YAML document, collecting round-trip metadata to reproduce the original +/// formatting. The stream is left just after the document that was read, so +/// that a stream holding several documents can be read one document at a +/// time. For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// #include +/// +/// std::istringstream stream{"hello: world\n---\nsecond: document\n"}; +/// while (stream.peek() != std::char_traits::eof()) { +/// sourcemeta::core::YAMLRoundTrip roundtrip; +/// const sourcemeta::core::JSON document = +/// sourcemeta::core::parse_yaml(stream, roundtrip); +/// } +/// ``` +SOURCEMETA_CORE_YAML_EXPORT +auto parse_yaml(std::basic_istream &stream, + YAMLRoundTrip &roundtrip) -> JSON; + /// @ingroup yaml /// /// Parse a YAML string with round-trip metadata into an existing JSON value, @@ -181,6 +208,57 @@ SOURCEMETA_CORE_YAML_EXPORT auto parse_yaml(const JSON::String &input, YAMLRoundTrip &roundtrip, JSON &output, const JSON::ParseCallback &callback) -> void; +/// @ingroup yaml +/// +/// Parse a YAML document from a C++ standard input stream with round-trip +/// metadata into an existing JSON value, invoking the given callback during +/// parsing. The stream is left just after the document that was read, so that +/// a stream holding several documents can be read one document at a time. The +/// result is constructed directly into the given reference rather than +/// returned by value to ensure that references passed through the parse +/// callback remain valid after parsing completes. +SOURCEMETA_CORE_YAML_EXPORT +auto parse_yaml(std::basic_istream &stream, + YAMLRoundTrip &roundtrip, JSON &output, + const JSON::ParseCallback &callback) -> void; + +/// @ingroup yaml +/// +/// Read a JSON document from a file location that represents a YAML file, +/// collecting round-trip metadata to reproduce the original formatting. Unlike +/// the stream overload, the file must hold a single document, as a file that +/// carries more cannot be written back from one set of metadata. For example: +/// +/// ```cpp +/// #include +/// #include +/// +/// #include +/// +/// sourcemeta::core::YAMLRoundTrip roundtrip; +/// const sourcemeta::core::JSON document = +/// sourcemeta::core::read_yaml("test.yaml", roundtrip); +/// sourcemeta::core::stringify_yaml(document, std::cout, roundtrip); +/// ``` +/// +/// If parsing fails, sourcemeta::core::YAMLFileParseError will be thrown. +SOURCEMETA_CORE_YAML_EXPORT +auto read_yaml(const std::filesystem::path &path, YAMLRoundTrip &roundtrip) + -> JSON; + +/// @ingroup yaml +/// +/// Read a YAML file with round-trip metadata into an existing JSON value, +/// invoking the given callback during parsing. The file must hold a single +/// document. The result is constructed directly into the given reference +/// rather than returned by value to ensure that references passed through the +/// parse callback remain valid after parsing completes. +/// +/// If parsing fails, sourcemeta::core::YAMLFileParseError will be thrown. +SOURCEMETA_CORE_YAML_EXPORT +auto read_yaml(const std::filesystem::path &path, YAMLRoundTrip &roundtrip, + JSON &output, const JSON::ParseCallback &callback) -> void; + /// @ingroup yaml /// /// Stringify a JSON document as YAML, using round-trip metadata collected @@ -201,14 +279,23 @@ auto parse_yaml(const JSON::String &input, YAMLRoundTrip &roundtrip, /// sourcemeta::core::parse_yaml(input, roundtrip); /// sourcemeta::core::stringify_yaml(document, std::cout, roundtrip); /// ``` +/// +/// Each level of nesting is laid out with the width the document was written +/// with, unless an indentation is given, which overrides it. A width of zero +/// would run a nested collection into the one that holds it, so it is treated +/// as one. SOURCEMETA_CORE_YAML_EXPORT auto stringify_yaml(const JSON &document, std::basic_ostream &stream, - const YAMLRoundTrip &roundtrip) -> void; + const YAMLRoundTrip &roundtrip, + const std::optional indentation = std::nullopt) + -> void; /// @ingroup yaml /// -/// Stringify a JSON document as YAML. For example: +/// Stringify a JSON document as YAML, laying out each level of nesting with +/// the given number of spaces. A width of zero would run a nested collection +/// into the one that holds it, so it is treated as one. For example: /// /// ```cpp /// #include @@ -222,8 +309,8 @@ auto stringify_yaml(const JSON &document, /// ``` SOURCEMETA_CORE_YAML_EXPORT auto stringify_yaml(const JSON &document, - std::basic_ostream &stream) - -> void; + std::basic_ostream &stream, + const std::size_t indentation = 2) -> void; } // namespace sourcemeta::core diff --git a/vendor/core/src/core/yaml/include/sourcemeta/core/yaml_roundtrip.h b/vendor/core/src/core/yaml/include/sourcemeta/core/yaml_roundtrip.h index 9a655dfa1..874288dd8 100644 --- a/vendor/core/src/core/yaml/include/sourcemeta/core/yaml_roundtrip.h +++ b/vendor/core/src/core/yaml/include/sourcemeta/core/yaml_roundtrip.h @@ -84,6 +84,13 @@ class SOURCEMETA_CORE_YAML_EXPORT YAMLRoundTrip { std::optional content_value; /// The anchor name attached to the node std::optional anchor; + /// The tag attached to the node, as it was written + std::optional tag; + /// The type the tagged node held when the tag was recorded, so that a tag + /// is only reproduced while the document still holds that kind of value + std::optional tag_type; + /// Whether the tag precedes the anchor + bool tag_before_anchor{false}; /// The comments preceding the node std::vector comments_before; /// The comment on the same line as the node @@ -92,6 +99,16 @@ class SOURCEMETA_CORE_YAML_EXPORT YAMLRoundTrip { std::optional comment_on_indicator; /// Whether the flow collection uses compact formatting bool compact_flow{false}; + /// Whether the flow collection is padded with a space inside its delimiters + bool padded_flow{false}; + /// Whether a block sequence that is the value of a mapping key sits at the + /// same indentation as that key rather than one level in. + /// See https://yaml.org/spec/1.2.2/#821-block-sequences + bool unindented_sequence{false}; + /// How many items the sequence held when it was read, so that comments and + /// node properties recorded against a position are only reproduced while + /// that position still holds the item they were read from + std::optional sequence_size; }; /// The recorded formatting for each node by pointer @@ -102,6 +119,11 @@ class SOURCEMETA_CORE_YAML_EXPORT YAMLRoundTrip { std::unordered_map key_styles; /// The original quoted content for each mapping key std::unordered_map key_quoted_contents; + /// The directive and comment lines that precede the document start marker, + /// in the order they were written. A document that carries a directive + /// always begins with an explicit start marker. + /// See https://yaml.org/spec/1.2.2/#912-document-markers + std::vector document_prefix; /// Whether the document begins with an explicit start marker bool explicit_document_start{false}; /// Whether the document ends with an explicit end marker @@ -120,6 +142,11 @@ class SOURCEMETA_CORE_YAML_EXPORT YAMLRoundTrip { std::vector trailing_comments; /// The indentation width used when emitting the document std::size_t indent_width{2}; + /// Whether the document begins with a byte order mark + bool byte_order_mark{false}; + /// Whether the document separates its lines with a carriage return and a + /// line feed rather than a line feed alone + bool carriage_returns{false}; }; #if defined(_MSC_VER) diff --git a/vendor/core/src/core/yaml/lexer.h b/vendor/core/src/core/yaml/lexer.h index 049440ff7..f42fbafe6 100644 --- a/vendor/core/src/core/yaml/lexer.h +++ b/vendor/core/src/core/yaml/lexer.h @@ -74,6 +74,14 @@ class Lexer { this->validate_characters(); } + // A carriage return is a line break rather than comment content, so it never + // belongs to the text of a comment that a carriage return ends. + // See https://yaml.org/spec/1.2.2/#66-comments + static auto comment_text(const std::string_view raw) -> std::string { + return std::string{raw.ends_with('\r') ? raw.substr(0, raw.size() - 1) + : raw}; + } + // The number of leading bytes consumed by a stripped byte order mark, so a // caller reading from a stream can map a consumed count back to the original // input offset @@ -344,6 +352,26 @@ class Lexer { return this->position_; } + // The line break that closes a document suffix belongs to the document that + // is ending rather than to whatever follows it. + // See https://yaml.org/spec/1.2.2/#912-document-markers + auto skip_line_break() -> void { + if (this->position_ >= this->input_.size()) { + return; + } + + const auto current{this->input_[this->position_]}; + if (current != '\n' && current != '\r') { + return; + } + + this->advance(1); + if (current == '\r' && this->position_ < this->input_.size() && + this->input_[this->position_] == '\n') { + this->advance(1); + } + } + auto take_inline_comment() -> std::optional { auto result{std::move(this->inline_comment_buffer_)}; this->inline_comment_buffer_.reset(); @@ -443,7 +471,13 @@ class Lexer { if (this->position_ >= this->input_.size()) { break; } - if (this->input_[this->position_] == '\n') { + // A lone carriage return is a line break of its own, while the one that + // opens a carriage return and line feed pair leaves the counting to the + // line feed. See https://yaml.org/spec/1.2.2/#54-line-break-characters + const auto character{this->input_[this->position_]}; + if (character == '\n' || + (character == '\r' && (this->position_ + 1 >= this->input_.size() || + this->input_[this->position_ + 1] != '\n'))) { this->line_++; this->column_ = 1; } else { @@ -537,8 +571,8 @@ class Lexer { this->advance(1); } if (this->roundtrip_) { - std::string text{this->input_.substr( - comment_start, this->position_ - comment_start)}; + std::string text{comment_text(this->input_.substr( + comment_start, this->position_ - comment_start))}; if (comment_line == this->comment_reference_line_ && this->comment_reference_line_ > 0 && !this->inline_comment_buffer_.has_value()) { @@ -1090,59 +1124,20 @@ class Lexer { return codepoint_to_utf8(character); } - [[nodiscard]] auto calculate_parent_indentation( - const std::size_t indicator_position) const noexcept -> std::size_t { - std::size_t line_start{indicator_position}; - while (line_start > 0 && this->input_[line_start - 1] != '\n' && - this->input_[line_start - 1] != '\r') { - line_start--; - } - - std::size_t leading_spaces{0}; - std::size_t scan_position{line_start}; - while (scan_position < this->input_.size() && - this->input_[scan_position] == ' ') { - leading_spaces++; - scan_position++; - } - - bool in_sequence_entry{false}; - if (scan_position < this->input_.size() - 1 && - this->input_[scan_position] == '-' && - this->input_[scan_position + 1] == ' ') { - in_sequence_entry = true; - } - - bool is_mapping_value_same_line{false}; - for (std::size_t index = line_start; index < indicator_position; ++index) { - if (this->input_[index] == ':') { - is_mapping_value_same_line = true; - break; - } - } - - if (in_sequence_entry && is_mapping_value_same_line) { - return leading_spaces + 2; - } - - if (is_mapping_value_same_line) { - return leading_spaces; - } - - return 0; - } - auto detect_block_scalar_indent(const std::size_t explicit_indent, - const std::size_t indicator_position, const std::uint64_t start_line, const std::uint64_t start_column) -> std::size_t { std::size_t content_indent{0}; if (explicit_indent > 0) { - const auto parent_indent{ - this->calculate_parent_indentation(indicator_position)}; - content_indent = parent_indent + explicit_indent; + // The content indentation level of a block scalar is the indentation + // level of the node itself plus the indicator, and the node at the + // document root sits one level further out than the leftmost column. + // See https://yaml.org/spec/1.2.2/#8111-block-indentation-indicator + content_indent = this->block_indent_ == SIZE_MAX + ? explicit_indent - 1 + : this->block_indent_ + explicit_indent; } else { const auto saved_position{this->position_}; const auto saved_line{this->line_}; @@ -1196,7 +1191,6 @@ class Lexer { auto scan_block_scalar(const ScalarStyle style) -> Token { const auto start_line{this->line_}; const auto start_column{this->column_}; - const auto indicator_position{this->position_}; this->advance(1); @@ -1228,8 +1222,8 @@ class Lexer { this->advance(1); } if (this->roundtrip_) { - this->block_scalar_comment_ = std::string{this->input_.substr( - comment_start, this->position_ - comment_start)}; + this->block_scalar_comment_ = comment_text(this->input_.substr( + comment_start, this->position_ - comment_start)); } } else if (current == '\n' || current == '\r') { break; @@ -1260,7 +1254,7 @@ class Lexer { } const auto content_indent{this->detect_block_scalar_indent( - explicit_indent, indicator_position, start_line, start_column)}; + explicit_indent, start_line, start_column)}; std::size_t blank_line_count{0}; bool previous_was_more_indented{false}; diff --git a/vendor/core/src/core/yaml/parser.h b/vendor/core/src/core/yaml/parser.h index ffef25ffe..a26c7088f 100644 --- a/vendor/core/src/core/yaml/parser.h +++ b/vendor/core/src/core/yaml/parser.h @@ -9,7 +9,7 @@ #include #include -#include // std::max +#include // std::max, std::find_if #include // assert #include // std::uint64_t, std::int64_t #include // std::optional @@ -46,6 +46,7 @@ class Parser { : lexer_{lexer}, callback_{callback}, roundtrip_{roundtrip} {} auto parse() -> JSON { + bool empty_document{false}; std::optional token; if (!this->pending_tokens_.empty()) { @@ -72,7 +73,7 @@ class Parser { if (token->type == TokenType::DirectiveYAML || token->type == TokenType::DirectiveTag || token->type == TokenType::DirectiveReserved) { - this->process_directives(token.value()); + this->process_directives(token.value(), true); } if (token->type == TokenType::DocumentStart) { @@ -82,6 +83,7 @@ class Parser { this->roundtrip_->explicit_document_start = true; } this->document_start_line_ = token->line; + this->lexer_->skip_line_break(); const auto pos_before_next{this->lexer_->position()}; token = this->lexer_->next(); if (this->roundtrip_ != nullptr) { @@ -95,8 +97,17 @@ class Parser { if (token.has_value() && token->type == TokenType::DocumentStart) { this->pending_tokens_.push_back(token.value()); this->pending_token_position_ = pos_before_next; + return JSON{nullptr}; } - return JSON{nullptr}; + + if (this->roundtrip_ != nullptr) { + this->roundtrip_->post_start_comments = + this->lexer_->take_preceding_comments(); + } + + // A document with no node of its own still ends the way any other + // does, so what follows it is read the same way + empty_document = true; } } else if (!token.has_value() || token->type == TokenType::StreamEnd) [[unlikely]] { @@ -114,26 +125,33 @@ class Parser { return JSON{nullptr}; } - if (this->roundtrip_ != nullptr) { - auto comments{this->lexer_->take_preceding_comments()}; - this->lexer_->take_inline_comment(); - if (this->roundtrip_->explicit_document_start) { - this->roundtrip_->post_start_comments = std::move(comments); - } else { - this->roundtrip_->leading_comments = std::move(comments); + JSON result{nullptr}; + auto pos_before_token{this->lexer_->position()}; + + if (!empty_document) { + if (this->roundtrip_ != nullptr) { + auto comments{this->lexer_->take_preceding_comments()}; + this->lexer_->take_inline_comment(); + if (this->roundtrip_->explicit_document_start) { + this->roundtrip_->post_start_comments = std::move(comments); + } else { + this->roundtrip_->leading_comments = std::move(comments); + } } - } - auto result{this->parse_value(token.value(), JSON::ParseContext::Root, 0, - EMPTY_PROPERTY)}; + result = this->parse_value(token.value(), JSON::ParseContext::Root, 0, + EMPTY_PROPERTY); - auto pos_before_token{this->lexer_->position()}; - token = this->next_token(); - if (this->roundtrip_ != nullptr) { - auto root_inline{this->lexer_->take_inline_comment()}; - if (root_inline.has_value()) { - this->roundtrip_->styles[this->pointer_stack_].comment_inline = - std::move(root_inline); + this->attach_leading_comments_to_first_key(result); + + pos_before_token = this->lexer_->position(); + token = this->next_token(); + if (this->roundtrip_ != nullptr) { + auto root_inline{this->lexer_->take_inline_comment()}; + if (root_inline.has_value()) { + this->roundtrip_->styles[this->pointer_stack_].comment_inline = + std::move(root_inline); + } } } while (token.has_value() && token->type == TokenType::DocumentEnd) { @@ -143,6 +161,7 @@ class Parser { this->lexer_->take_preceding_comments(); this->roundtrip_->explicit_document_end = true; } + this->lexer_->skip_line_break(); pos_before_token = this->lexer_->position(); token = this->next_token(); if (this->roundtrip_ != nullptr) { @@ -153,7 +172,9 @@ class Parser { if (this->roundtrip_ != nullptr) { auto trailing{this->lexer_->take_preceding_comments()}; - if (!trailing.empty()) { + const bool ends_the_stream{!token.has_value() || + token->type == TokenType::StreamEnd}; + if (!trailing.empty() && ends_the_stream) { this->roundtrip_->trailing_comments = std::move(trailing); } } @@ -177,6 +198,20 @@ class Parser { return this->lexer_->position(); } + // Metadata collected for a round-trip describes the one document it was read + // from, so a stream that carries more than that cannot be written back + auto validate_single_document() -> void { + auto token{this->next_token()}; + while (token.has_value() && token->type == TokenType::DocumentEnd) { + token = this->next_token(); + } + + if (token.has_value() && token->type != TokenType::StreamEnd) [[unlikely]] { + throw YAMLParseError{token->line, token->column, + "Unexpected content after document"}; + } + } + auto validate_end_of_stream() -> void { auto token{this->next_token()}; // The preceding parse already consumed a document, so its end marker, if @@ -296,11 +331,59 @@ class Parser { return total; } - auto process_directives(Token &token) -> void { + // A comment block that runs straight into the first key of the document reads + // as belonging to that key, so it is recorded there and travels with the key + // if the document is later rearranged. A blank line in between instead marks + // the block as a header for the document as a whole + auto attach_leading_comments_to_first_key(const JSON &result) -> void { + if ((this->roundtrip_ == nullptr) || !result.is_object() || + result.empty()) { + return; + } + + auto &comments{this->roundtrip_->explicit_document_start + ? this->roundtrip_->post_start_comments + : this->roundtrip_->leading_comments}; + const auto blank{ + std::find_if(comments.crbegin(), comments.crend(), + [](const auto &comment) { return comment.empty(); })}; + if (blank == comments.crbegin()) { + return; + } + + const auto first{blank.base()}; + Pointer pointer{result.as_object().cbegin()->first}; + auto &attached{this->roundtrip_->styles[pointer].comments_before}; + attached.insert(attached.cbegin(), first, comments.cend()); + comments.erase(first, comments.cend()); + } + + // A flow collection may be written with a space just inside its delimiters, + // which the first token after the opening one gives away + auto record_flow_padding(const Token &start_token, + const std::optional &first) -> void { + if ((this->roundtrip_ == nullptr) || !first.has_value() || + first->line != start_token.line || + first->column <= start_token.column + 1) { + return; + } + + this->roundtrip_->styles[this->pointer_stack_].padded_flow = true; + } + + auto process_directives(Token &token, const bool record = false) -> void { bool seen_yaml_directive{false}; while (token.type == TokenType::DirectiveYAML || token.type == TokenType::DirectiveTag || token.type == TokenType::DirectiveReserved) { + if (record && this->roundtrip_ != nullptr) { + auto &prefix{this->roundtrip_->document_prefix}; + for (auto &comment : this->lexer_->take_preceding_comments()) { + prefix.push_back(std::move(comment)); + } + + prefix.emplace_back(token.value); + } if (token.type == TokenType::DirectiveYAML) { if (seen_yaml_directive) [[unlikely]] { throw YAMLParseError{token.line, token.column, @@ -535,6 +618,8 @@ class Parser { std::optional anchor_name; std::uint64_t anchor_line{0}; std::optional tag; + std::optional raw_tag; + bool tag_before_anchor{false}; std::size_t anchor_count{0}; std::optional anchor_inline_comment; Token current_token{token}; @@ -563,11 +648,24 @@ class Parser { anchor_count++; } else { tag = this->resolve_tag(current_token.value); + if (this->roundtrip_ != nullptr) { + raw_tag = std::string{current_token.value}; + tag_before_anchor = anchor_count == 0; + } } auto next{this->lexer_->next()}; if ((this->roundtrip_ != nullptr) && anchor_name.has_value()) { anchor_inline_comment = this->lexer_->take_inline_comment(); + } else if ((this->roundtrip_ != nullptr) && + context == JSON::ParseContext::Root && + this->document_start_line_ > 0 && + current_token.line == this->document_start_line_ && + !this->roundtrip_->document_start_comment.has_value()) { + // A comment that trails the node properties of the root node still + // sits on the document start marker line, which is where it is written + this->roundtrip_->document_start_comment = + this->lexer_->take_inline_comment(); } if (!next.has_value() || next->type == TokenType::StreamEnd || next->type == TokenType::DocumentEnd || @@ -588,6 +686,7 @@ class Parser { style.comment_inline = std::move(anchor_inline_comment); } } + this->record_tag(raw_tag, tag_before_anchor, empty_value); if ((this->roundtrip_ != nullptr) && context != JSON::ParseContext::Root) { this->pointer_stack_.pop_back(); @@ -602,23 +701,29 @@ class Parser { if (after.has_value() && after->type == TokenType::BlockMappingValue) { this->pending_tokens_.push_back(current_token); this->pending_tokens_.push_back(after.value()); + JSON empty_value{nullptr}; + if (tag.has_value() && tag.value() == "tag:yaml.org,2002:str") { + empty_value = JSON{std::string{}}; + } if (anchor_name.has_value()) { this->register_anchored_null(anchor_name.value(), token, context, index, property, anchor_inline_comment); } + this->record_tag(raw_tag, tag_before_anchor, empty_value); if ((this->roundtrip_ != nullptr) && context != JSON::ParseContext::Root) { this->pointer_stack_.pop_back(); } - return JSON{nullptr}; + return empty_value; } if (after.has_value()) { this->pending_tokens_.push_back(after.value()); } } - if (anchor_name.has_value() && context == JSON::ParseContext::Index && + if ((anchor_name.has_value() || tag.has_value()) && + context == JSON::ParseContext::Index && current_token.type == TokenType::BlockSequenceEntry) { const auto block_indent{this->lexer_->block_indent()}; const auto entry_indent{ @@ -627,13 +732,21 @@ class Parser { : 0UZ}; if (block_indent != SIZE_MAX && entry_indent <= block_indent) { this->pending_tokens_.push_back(current_token); - this->register_anchored_null(anchor_name.value(), token, context, - index, property, anchor_inline_comment); + JSON empty_value{nullptr}; + if (tag.has_value() && tag.value() == "tag:yaml.org,2002:str") { + empty_value = JSON{std::string{}}; + } + if (anchor_name.has_value()) { + this->register_anchored_null(anchor_name.value(), token, context, + index, property, + anchor_inline_comment); + } + this->record_tag(raw_tag, tag_before_anchor, empty_value); if ((this->roundtrip_ != nullptr) && context != JSON::ParseContext::Root) { this->pointer_stack_.pop_back(); } - return JSON{nullptr}; + return empty_value; } } } @@ -646,6 +759,7 @@ class Parser { empty_value = JSON{std::string{}}; } this->pending_tokens_.push_back(current_token); + this->record_tag(raw_tag, tag_before_anchor, empty_value); if ((this->roundtrip_ != nullptr) && context != JSON::ParseContext::Root) { this->pointer_stack_.pop_back(); @@ -675,7 +789,17 @@ class Parser { switch (current_token.type) { case TokenType::Scalar: { auto next{this->next_token()}; - if (next.has_value() && next->type == TokenType::BlockMappingValue) { + // YAML 1.2.2 Section 7.4.2: the value of an entry of a flow collection + // is a single node, and a pair carrying no brackets of its own is a + // node only where a flow sequence takes its entries. Leaving the + // indicator unread in that position hands the scalar back on its own, + // which is what lets the caller report the separator the entry is + // really missing + const auto pair_without_brackets_allowed{ + this->lexer_->flow_level() == 0 || + context != JSON::ParseContext::Property}; + if (next.has_value() && next->type == TokenType::BlockMappingValue && + pair_without_brackets_allowed) { if (current_token.multiline) [[unlikely]] { throw YAMLParseError{current_token.line, current_token.column, "Multi-line implicit mapping key"}; @@ -812,6 +936,13 @@ class Parser { } } + this->record_tag(raw_tag, tag_before_anchor, result); + + if ((this->roundtrip_ != nullptr) && result.is_array()) { + this->roundtrip_->styles[this->pointer_stack_].sequence_size = + result.size(); + } + if ((this->roundtrip_ != nullptr) && context != JSON::ParseContext::Root) { this->pointer_stack_.pop_back(); } @@ -1090,6 +1221,7 @@ class Parser { bool found_compact_separator{false}; auto token{this->next_token()}; + this->record_flow_padding(start_token, token); while (token.has_value() && token->type != TokenType::MappingEnd) { if (token->type == TokenType::FlowEntry) { @@ -1224,6 +1356,7 @@ class Parser { bool found_compact_separator{false}; auto token{this->next_token()}; + this->record_flow_padding(start_token, token); std::size_t element_index{0}; while (token.has_value() && token->type != TokenType::SequenceEnd) { @@ -1345,6 +1478,11 @@ class Parser { const auto sequence_indent{ base_column > 0 ? static_cast(base_column - 1) : 0UZ}; this->detect_indent_width(key_column, base_column); + if ((this->roundtrip_ != nullptr) && + context == JSON::ParseContext::Property && key_column > 0 && + base_column == key_column) { + this->roundtrip_->styles[this->pointer_stack_].unindented_sequence = true; + } this->lexer_->set_block_indent(sequence_indent); this->record_preceding_comments_for_index(0); @@ -1699,6 +1837,37 @@ class Parser { return anchored.value; } + // A node that stands on a later line than the key it belongs to has to be + // indented past the mapping for that mapping to own it. A block sequence is + // the one exception, as it may sit at the very indentation of its key. + // See https://yaml.org/spec/1.2.2/#821-block-sequences + [[nodiscard]] auto starts_mapping_value(const Token &token, + const std::uint64_t key_line, + const std::uint64_t base_column) const + -> bool { + if (token.line == key_line) { + return true; + } + + return token.type == TokenType::BlockSequenceEntry + ? token.column >= base_column + : token.column > base_column; + } + + // A node property that decorates a block node rather than a key of the + // mapping it sits in has to be indented past that mapping, so one that opens + // a block sequence from the mapping's own indentation has nowhere to belong. + // See https://yaml.org/spec/1.2.2/#822-block-mappings + auto reject_misplaced_property(const Token &property, + const std::optional &node) const + -> void { + if (node.has_value() && node->type == TokenType::BlockSequenceEntry) + [[unlikely]] { + throw YAMLParseError{property.line, property.column, + "Node property at wrong indentation level"}; + } + } + auto next_token() -> std::optional { std::optional result; if (!this->pending_tokens_.empty()) { @@ -1754,7 +1923,7 @@ class Parser { next->type == TokenType::StreamEnd || next->type == TokenType::DocumentEnd) { if (next.has_value() && next->type == TokenType::Scalar && - (next->line == key_line || next->column != base_column)) { + this->starts_mapping_value(next.value(), key_line, base_column)) { this->record_inline_comment_for_key(key, next->line != key_line); auto value{this->parse_value(next.value(), JSON::ParseContext::Property, 0, key, key_line, key_column)}; @@ -1779,18 +1948,22 @@ class Parser { next->type == TokenType::BlockSequenceEntry || next->type == TokenType::Anchor || next->type == TokenType::Tag || next->type == TokenType::Alias) { - if (next->type == TokenType::BlockSequenceEntry && next->line == key_line) - [[unlikely]] { - throw YAMLParseError{ - next->line, next->column, - "Block sequence entry on same line as mapping key"}; - } - this->record_inline_comment_for_key(key, next->line != key_line); - auto value{this->parse_value(next.value(), JSON::ParseContext::Property, - 0, key, key_line, key_column)}; - result.assign(key, std::move(value)); - next = this->next_token(); - this->record_inline_comment_for_key(key); + if (!this->starts_mapping_value(next.value(), key_line, base_column)) { + result.assign(key, JSON{nullptr}); + } else { + if (next->type == TokenType::BlockSequenceEntry && + next->line == key_line) [[unlikely]] { + throw YAMLParseError{ + next->line, next->column, + "Block sequence entry on same line as mapping key"}; + } + this->record_inline_comment_for_key(key, next->line != key_line); + auto value{this->parse_value(next.value(), JSON::ParseContext::Property, + 0, key, key_line, key_column)}; + result.assign(key, std::move(value)); + next = this->next_token(); + this->record_inline_comment_for_key(key); + } } else { result.assign(key, JSON{nullptr}); } @@ -1901,16 +2074,27 @@ class Parser { auto effective_column{next->column}; std::optional subsequent_key_tag; + // A node property introduces the key it decorates, so a mapping that + // does not reach that key must leave the property alone as well + if ((next->type == TokenType::Anchor || next->type == TokenType::Tag) && + effective_column != base_column) { + break; + } + if (next->type == TokenType::Anchor) { + const auto property_token{next.value()}; next = this->next_token(); + this->reject_misplaced_property(property_token, next); if (!next.has_value() || next->type != TokenType::Scalar) { continue; } } if (next->type == TokenType::Tag) { + const auto property_token{next.value()}; subsequent_key_tag = this->resolve_tag(next->value); next = this->next_token(); + this->reject_misplaced_property(property_token, next); if (!next.has_value() || next->type != TokenType::Scalar) { continue; } @@ -1948,7 +2132,8 @@ class Parser { next = this->next_token(); if (!next.has_value() || next->type == TokenType::Scalar) { - if (next.has_value()) { + if (next.has_value() && + this->starts_mapping_value(next.value(), key_line, base_column)) { auto value{this->parse_value(next.value(), JSON::ParseContext::Property, 0, key, key_line, key_column)}; @@ -1962,12 +2147,15 @@ class Parser { next->type == TokenType::DocumentStart) { result.assign(key, JSON{nullptr}); break; - } else { + } else if (this->starts_mapping_value(next.value(), key_line, + base_column)) { auto value{this->parse_value(next.value(), JSON::ParseContext::Property, 0, key, key_line, key_column)}; result.assign(key, std::move(value)); next = this->next_token(); + } else { + result.assign(key, JSON{nullptr}); } continue; } @@ -2006,7 +2194,7 @@ class Parser { if (!next.has_value() || next->type == TokenType::Scalar) { if (next.has_value() && - (next->line == key_line || next->column != base_column)) { + this->starts_mapping_value(next.value(), key_line, base_column)) { this->record_inline_comment_for_key(key, next->line != key_line); auto after{this->next_token()}; if (after.has_value()) { @@ -2028,12 +2216,15 @@ class Parser { next->type == TokenType::DocumentStart) { result.assign(key, JSON{nullptr}); break; - } else { + } else if (this->starts_mapping_value(next.value(), key_line, + base_column)) { this->record_inline_comment_for_key(key, next->line != key_line); auto value{this->parse_value(next.value(), JSON::ParseContext::Property, 0, key, key_line, key_column)}; result.assign(key, std::move(value)); next = this->next_token(); + } else { + result.assign(key, JSON{nullptr}); } } @@ -2164,6 +2355,18 @@ class Parser { this->roundtrip_->styles[this->pointer_stack_].collection = style; } + auto record_tag(const std::optional &raw_tag, + const bool tag_before_anchor, const JSON &value) -> void { + if ((this->roundtrip_ == nullptr) || !raw_tag.has_value()) { + return; + } + + auto &node_style{this->roundtrip_->styles[this->pointer_stack_]}; + node_style.tag = raw_tag.value(); + node_style.tag_type = value.type(); + node_style.tag_before_anchor = tag_before_anchor; + } + auto record_scalar_style(const Token &token, const JSON &value) -> void { if (this->roundtrip_ == nullptr) { return; diff --git a/vendor/core/src/core/yaml/stringify.h b/vendor/core/src/core/yaml/stringify.h index 2f70421ef..d866aa075 100644 --- a/vendor/core/src/core/yaml/stringify.h +++ b/vendor/core/src/core/yaml/stringify.h @@ -6,33 +6,51 @@ #include #include -#include // std::array -#include // assert -#include // std::to_chars -#include // std::modf -#include // std::size_t -#include // std::basic_ostream -#include // std::string -#include // std::string_view -#include // std::pair -#include // std::vector +#include // std::max +#include // std::array +#include // assert +#include // std::to_chars +#include // std::modf +#include // std::size_t +#include // std::optional +#include // std::basic_ostream +#include // std::string +#include // std::string_view +#include // std::unordered_map +#include // std::pair +#include // std::vector namespace sourcemeta::core::yaml { using OutputStream = std::basic_ostream; static constexpr std::size_t INDENT_WIDTH{2}; +static constexpr std::size_t ONE_COLUMN{1}; static constexpr std::array HEX_DIGITS{{'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f'}}; -inline auto write_indent(OutputStream &stream, const std::size_t indent, - const std::size_t width = INDENT_WIDTH) -> void { - for (std::size_t index{0}; index < indent * width; ++index) { +// A processor is free to write line breaks with whatever convention suits the +// document. See https://yaml.org/spec/1.2.2/#54-line-break-characters +inline auto write_break(OutputStream &stream, const YAMLRoundTrip *roundtrip) + -> void { + if ((roundtrip != nullptr) && roundtrip->carriage_returns) { + stream.put('\r'); + } + + stream.put('\n'); +} + +inline auto write_indent(OutputStream &stream, const std::size_t columns) + -> void { + for (std::size_t index{0}; index < columns; ++index) { stream.put(' '); } } +// The width of the "- " indicator that opens a block sequence entry +static constexpr std::size_t SEQUENCE_INDICATOR_WIDTH{2}; + inline auto looks_like_number(const std::string &value) -> bool { std::size_t start{0}; if (value[0] == '-' || value[0] == '+') { @@ -214,49 +232,48 @@ inline auto write_string(OutputStream &stream, const std::string &value) } inline auto write_block_scalar( - OutputStream &stream, const std::string &value, const std::size_t indent, - const YAMLRoundTrip::ScalarStyle style, + OutputStream &stream, const std::string &value, + const std::size_t content_columns, const YAMLRoundTrip::ScalarStyle style, const YAMLRoundTrip::Chomping chomping, const std::optional &header_comment = std::nullopt, - const std::size_t indent_width = INDENT_WIDTH, - const std::size_t explicit_indent = 0, - const bool indent_before_chomping = false) -> void { + const std::size_t indicator = 0, const bool indent_before_chomping = false, + const YAMLRoundTrip *roundtrip = nullptr) -> void { stream.put(style == YAMLRoundTrip::ScalarStyle::Literal ? '|' : '>'); - if (indent_before_chomping && explicit_indent > 0) { - stream.put(static_cast('0' + explicit_indent)); + if (indent_before_chomping && indicator > 0) { + stream.put(static_cast('0' + indicator)); } if (chomping == YAMLRoundTrip::Chomping::Strip) { stream.put('-'); } else if (chomping == YAMLRoundTrip::Chomping::Keep) { stream.put('+'); } - if (!indent_before_chomping && explicit_indent > 0) { - stream.put(static_cast('0' + explicit_indent)); + if (!indent_before_chomping && indicator > 0) { + stream.put(static_cast('0' + indicator)); } if (header_comment.has_value()) { stream.put(' '); const auto &comment{header_comment.value()}; stream.write(comment.data(), static_cast(comment.size())); } - stream.put('\n'); + write_break(stream, roundtrip); std::size_t position{0}; while (position < value.size()) { auto line_end{value.find('\n', position)}; if (line_end == std::string::npos) { - write_indent(stream, indent, indent_width); + write_indent(stream, content_columns); stream.write(value.data() + position, static_cast(value.size() - position)); - stream.put('\n'); + write_break(stream, roundtrip); break; } if (line_end > position) { - write_indent(stream, indent, indent_width); + write_indent(stream, content_columns); } stream.write(value.data() + position, static_cast(line_end - position)); - stream.put('\n'); + write_break(stream, roundtrip); position = line_end + 1; } } @@ -277,6 +294,85 @@ inline auto matches_recorded_value(const YAMLRoundTrip::NodeStyle &style, same_value(style.content_value.value(), value); } +// The content indentation level of a block scalar that carries no indicator is +// read off its first non-empty line, so detection fails when that line begins +// with a space, and when the content holds no non-empty line at all. +// See https://yaml.org/spec/1.2.2/#8111-block-indentation-indicator +inline auto block_detection_fails(const std::string &value) -> bool { + std::size_t position{0}; + while (position < value.size()) { + const auto line_end{value.find('\n', position)}; + const auto length{line_end == std::string::npos ? value.size() - position + : line_end - position}; + if (length > 0) { + return value[position] == ' '; + } + + if (line_end == std::string::npos) { + break; + } + + position = line_end + 1; + } + + return !value.empty(); +} + +// Every line a block scalar writes is closed with a line break, so the chomping +// indicator is what decides which of the trailing breaks of the value survive +// being read back. See +// https://yaml.org/spec/1.2.2/#8112-block-chomping-indicator +inline auto required_chomping(const std::string &value) + -> YAMLRoundTrip::Chomping { + const auto body{value.find_last_not_of('\n')}; + if (body == std::string::npos) { + return value.empty() ? YAMLRoundTrip::Chomping::Clip + : YAMLRoundTrip::Chomping::Keep; + } + + const auto breaks{value.size() - body - 1}; + if (breaks == 0) { + return YAMLRoundTrip::Chomping::Strip; + } + + return breaks == 1 ? YAMLRoundTrip::Chomping::Clip + : YAMLRoundTrip::Chomping::Keep; +} + +// Clipping and keeping both leave a single trailing line break in place, so a +// recorded indicator that only differs in that way still says the same thing +// and is worth writing back as it was +inline auto block_chomping(const YAMLRoundTrip::NodeStyle &style, + const std::string &value) + -> YAMLRoundTrip::Chomping { + const auto required{required_chomping(value)}; + if (!style.chomping.has_value() || style.chomping.value() == required) { + return required; + } + + return (required == YAMLRoundTrip::Chomping::Clip && !value.empty() && + style.chomping.value() == YAMLRoundTrip::Chomping::Keep) + ? YAMLRoundTrip::Chomping::Keep + : required; +} + +// The folded style joins the lines of its content, so it can only stand for a +// value that has no line break of its own to lose +inline auto folding_preserves(const std::string &value) -> bool { + const auto body{value.find_last_not_of('\n')}; + return body == std::string::npos || value.find('\n') == std::string::npos || + value.find('\n') > body; +} + +// The text a block scalar writes, which is the original one for as long as the +// document still holds the value it was read from +inline auto block_scalar_content(const YAMLRoundTrip::NodeStyle &style, + const JSON &value) -> const std::string & { + return style.block_content.has_value() && matches_recorded_value(style, value) + ? style.block_content.value() + : value.to_string(); +} + inline auto find_anchor(const AnchorValues &anchors, const std::string_view name) -> const JSON * { for (auto iterator{anchors.crbegin()}; iterator != anchors.crend(); @@ -320,6 +416,66 @@ inline auto write_anchor(OutputStream &stream, const std::string &name, anchors.emplace_back(name, &value); } +// A tag names the kind of value its node holds, so it may only be emitted +// while the document still holds that kind of value +inline auto matching_tag(const YAMLRoundTrip::NodeStyle *style, + const JSON &value) -> const std::string * { + if ((style == nullptr) || !style->tag.has_value() || + !style->tag_type.has_value() || style->tag_type.value() != value.type()) { + return nullptr; + } + + return &style->tag.value(); +} + +// Comments and node properties are read off a position in a sequence, so they +// only stand for what is written there while the sequence still holds the same +// items it was read from +inline auto keeps_item_annotations(const YAMLRoundTrip::NodeStyle *style, + const JSON &value) -> bool { + return (style == nullptr) || !style->sequence_size.has_value() || + style->sequence_size.value() == value.size(); +} + +inline auto has_node_properties(const YAMLRoundTrip::NodeStyle *style, + const JSON &value) -> bool { + return (matching_tag(style, value) != nullptr) || + ((style != nullptr) && style->anchor.has_value()); +} + +// Writes the tag and anchor that decorate a node, in the order they were +// written, and reports whether anything was written at all, as a node property +// has to be separated from the node it decorates +inline auto write_node_properties(OutputStream &stream, const JSON &value, + const YAMLRoundTrip::NodeStyle *style, + AnchorValues &anchors) -> bool { + const auto *tag{matching_tag(style, value)}; + const bool anchor{(style != nullptr) && style->anchor.has_value()}; + if ((tag == nullptr) && !anchor) { + return false; + } + + if ((tag != nullptr) && style->tag_before_anchor) { + stream.write(tag->data(), static_cast(tag->size())); + if (anchor) { + stream.put(' '); + } + } + + if (anchor) { + write_anchor(stream, style->anchor.value(), value, anchors); + } + + if ((tag != nullptr) && !style->tag_before_anchor) { + if (anchor) { + stream.put(' '); + } + stream.write(tag->data(), static_cast(tag->size())); + } + + return true; +} + inline auto write_string_with_style(OutputStream &stream, const JSON &value, const YAMLRoundTrip *roundtrip, const Pointer &pointer) -> void { @@ -508,32 +664,53 @@ inline auto is_implicit_null(const JSON &value, const YAMLRoundTrip *roundtrip, return !match->second.scalar.has_value(); } -inline auto write_flow_anchor(OutputStream &stream, const JSON &value, - const YAMLRoundTrip *roundtrip, - AnchorValues &anchors, const Pointer &pointer) +inline auto write_flow_properties(OutputStream &stream, const JSON &value, + const YAMLRoundTrip *roundtrip, + AnchorValues &anchors, const Pointer &pointer) -> void { if (roundtrip == nullptr) { return; } const auto match{roundtrip->styles.find(pointer)}; - if (match != roundtrip->styles.end() && match->second.anchor.has_value()) { - write_anchor(stream, match->second.anchor.value(), value, anchors); + if (match != roundtrip->styles.end() && + write_node_properties(stream, value, &match->second, anchors)) { stream.put(' '); } } +// The same as writing the properties of a flow node, but for a node that is +// written with no value at all, so nothing follows to be separated from +inline auto write_flow_node_properties(OutputStream &stream, const JSON &value, + const YAMLRoundTrip *roundtrip, + AnchorValues &anchors, + const Pointer &pointer) -> void { + if (roundtrip == nullptr) { + return; + } + + const auto match{roundtrip->styles.find(pointer)}; + if (match != roundtrip->styles.end()) { + write_node_properties(stream, value, &match->second, anchors); + } +} + inline auto write_flow_mapping(OutputStream &stream, const JSON &value, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, Pointer &pointer) -> void { bool compact{false}; + bool padded{false}; if (roundtrip != nullptr) { const auto match{roundtrip->styles.find(pointer)}; if (match != roundtrip->styles.end()) { compact = match->second.compact_flow; + padded = match->second.padded_flow; } } stream.put('{'); + if (padded) { + stream.put(' '); + } bool first{true}; for (const auto &entry : value.as_object()) { if (!first) { @@ -547,12 +724,18 @@ inline auto write_flow_mapping(OutputStream &stream, const JSON &value, pointer.push_back(entry.first); write_key_string(stream, entry.first, roundtrip, pointer); stream.write(": ", 2); - if (!is_implicit_null(entry.second, roundtrip, pointer)) { - write_flow_anchor(stream, entry.second, roundtrip, anchors, pointer); + if (is_implicit_null(entry.second, roundtrip, pointer)) { + write_flow_node_properties(stream, entry.second, roundtrip, anchors, + pointer); + } else { + write_flow_properties(stream, entry.second, roundtrip, anchors, pointer); write_inline_value(stream, entry.second, roundtrip, anchors, pointer); } pointer.pop_back(); } + if (padded) { + stream.put(' '); + } stream.put('}'); } @@ -561,13 +744,20 @@ inline auto write_flow_sequence(OutputStream &stream, const JSON &value, AnchorValues &anchors, Pointer &pointer) -> void { bool compact{false}; + bool padded{false}; + bool annotations{true}; if (roundtrip != nullptr) { const auto match{roundtrip->styles.find(pointer)}; if (match != roundtrip->styles.end()) { compact = match->second.compact_flow; + padded = match->second.padded_flow; + annotations = keeps_item_annotations(&match->second, value); } } stream.put('['); + if (padded) { + stream.put(' '); + } bool first{true}; std::size_t item_index{0}; for (const auto &item : value.as_array()) { @@ -580,21 +770,28 @@ inline auto write_flow_sequence(OutputStream &stream, const JSON &value, } first = false; pointer.push_back(item_index); - write_flow_anchor(stream, item, roundtrip, anchors, pointer); + if (annotations) { + write_flow_properties(stream, item, roundtrip, anchors, pointer); + } write_inline_value(stream, item, roundtrip, anchors, pointer); pointer.pop_back(); item_index++; } + if (padded) { + stream.put(' '); + } stream.put(']'); } inline auto write_block_mapping(OutputStream &stream, const JSON &value, - std::size_t indent, bool skip_first_indent, + std::size_t columns, std::size_t width, + bool skip_first_indent, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, Pointer &pointer) -> void; inline auto write_block_sequence(OutputStream &stream, const JSON &value, - std::size_t indent, bool skip_first_indent, + std::size_t columns, std::size_t width, + bool skip_first_indent, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, Pointer &pointer) -> void; @@ -609,9 +806,16 @@ inline auto emit_inline_comment(OutputStream &stream, } inline auto write_node(OutputStream &stream, const JSON &value, - const std::size_t indent, const bool skip_first_indent, + const std::size_t columns, const std::size_t width, + const std::size_t block_indicator, + const bool skip_first_indent, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, - Pointer &pointer) -> void { + Pointer &pointer, const bool skip_properties = false, + const bool annotations = true) -> void { + // Only the document root sits at the leftmost column, and a block scalar + // there has no column of its own, so its content is pushed one nesting level + // in to leave room for whatever follows it + const auto block_columns{columns == 0 ? width : columns}; const YAMLRoundTrip::NodeStyle *node_style{nullptr}; if (roundtrip != nullptr) { const auto style_match{roundtrip->styles.find(pointer)}; @@ -620,96 +824,118 @@ inline auto write_node(OutputStream &stream, const JSON &value, } } + const YAMLRoundTrip::NodeStyle *annotation_style{annotations ? node_style + : nullptr}; + const auto *alias{matching_alias(value, roundtrip, anchors, pointer)}; if (alias != nullptr) { write_alias(stream, *alias); - emit_inline_comment(stream, node_style); - stream.put('\n'); + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); return; } - bool has_anchor{false}; - if ((node_style != nullptr) && node_style->anchor.has_value()) { - write_anchor(stream, node_style->anchor.value(), value, anchors); - has_anchor = true; - } + const bool has_properties{ + skip_properties + ? false + : write_node_properties(stream, value, annotation_style, anchors)}; const bool flow{ (node_style != nullptr) && node_style->collection.has_value() && node_style->collection.value() == YAMLRoundTrip::CollectionStyle::Flow}; + // An indicator is a single digit, so content that needs one but sits too far + // in has to give up on block style and be quoted instead + const bool block_style{ + (node_style != nullptr) && value.is_string() && + node_style->scalar.has_value() && + (node_style->scalar.value() == YAMLRoundTrip::ScalarStyle::Literal || + node_style->scalar.value() == YAMLRoundTrip::ScalarStyle::Folded) && + ((block_indicator >= 1 && block_indicator <= 9) || + !block_detection_fails(block_scalar_content(*node_style, value)))}; + if (value.is_object() && !value.empty()) { if (flow) { - if (has_anchor) { + if (has_properties) { stream.put(' '); } write_flow_mapping(stream, value, roundtrip, anchors, pointer); - emit_inline_comment(stream, node_style); - stream.put('\n'); + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); } else { - if (has_anchor) { - emit_inline_comment(stream, node_style); - stream.put('\n'); + if (has_properties) { + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); } - write_block_mapping(stream, value, indent, - has_anchor ? false : skip_first_indent, roundtrip, + write_block_mapping(stream, value, columns, width, + has_properties ? false : skip_first_indent, roundtrip, anchors, pointer); } } else if (value.is_array() && !value.empty()) { if (flow) { - if (has_anchor) { + if (has_properties) { stream.put(' '); } write_flow_sequence(stream, value, roundtrip, anchors, pointer); - emit_inline_comment(stream, node_style); - stream.put('\n'); + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); } else { - if (has_anchor) { - emit_inline_comment(stream, node_style); - stream.put('\n'); + if (has_properties) { + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); } - write_block_sequence(stream, value, indent, - has_anchor ? false : skip_first_indent, roundtrip, - anchors, pointer); - } - } else if ((node_style != nullptr) && value.is_string() && - node_style->scalar.has_value() && - (node_style->scalar.value() == - YAMLRoundTrip::ScalarStyle::Literal || - node_style->scalar.value() == - YAMLRoundTrip::ScalarStyle::Folded)) { - if (has_anchor) { + // A block sequence may sit at the indentation of the mapping key it + // belongs to rather than one level further in + const auto sequence_columns{(node_style != nullptr) && + node_style->unindented_sequence && + columns >= block_indicator + ? columns - block_indicator + : columns}; + write_block_sequence(stream, value, sequence_columns, width, + has_properties ? false : skip_first_indent, + roundtrip, anchors, pointer); + } + } else if (block_style) { + if (has_properties) { stream.put(' '); } - const auto chomping{ - node_style->chomping.value_or(YAMLRoundTrip::Chomping::Clip)}; - const auto &content{node_style->block_content.has_value() && - matches_recorded_value(*node_style, value) - ? node_style->block_content.value() - : value.to_string()}; - write_block_scalar(stream, content, indent, node_style->scalar.value(), - chomping, node_style->comment_inline, - roundtrip->indent_width, node_style->explicit_indent, - node_style->indent_before_chomping); + const auto &content{block_scalar_content(*node_style, value)}; + const auto &text{value.to_string()}; + // The recorded style only reproduces the text it was read from, so once the + // document holds something else, a style that would lose a line break in + // the process gives way to one that keeps every line as it is + const bool original{node_style->block_content.has_value() && + matches_recorded_value(*node_style, value)}; + const auto style{original || folding_preserves(text) + ? node_style->scalar.value() + : YAMLRoundTrip::ScalarStyle::Literal}; + const std::optional header_comment{ + (annotation_style != nullptr) ? annotation_style->comment_inline + : std::nullopt}; + const bool indicated{block_detection_fails(content) || + node_style->explicit_indent > 0}; + write_block_scalar(stream, content, block_columns, style, + block_chomping(*node_style, text), header_comment, + indicated ? block_indicator : 0, + node_style->indent_before_chomping, roundtrip); } else { - if (has_anchor) { + if (has_properties) { stream.put(' '); } write_inline_value(stream, value, roundtrip, anchors, pointer); - emit_inline_comment(stream, node_style); - stream.put('\n'); + emit_inline_comment(stream, annotation_style); + write_break(stream, roundtrip); } } inline auto write_block_mapping(OutputStream &stream, const JSON &value, - const std::size_t indent, + const std::size_t columns, + const std::size_t width, const bool skip_first_indent, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, Pointer &pointer) -> void { assert(value.is_object() && !value.empty()); - const auto width{(roundtrip != nullptr) ? roundtrip->indent_width - : INDENT_WIDTH}; bool first{true}; for (const auto &entry : value.as_object()) { pointer.push_back(entry.first); @@ -729,16 +955,16 @@ inline auto write_block_mapping(OutputStream &stream, const JSON &value, if ((entry_style != nullptr) && !entry_style->comments_before.empty()) { for (const auto &comment : entry_style->comments_before) { if (comment.empty()) { - stream.put('\n'); + write_break(stream, roundtrip); } else { - write_indent(stream, indent, width); + write_indent(stream, columns); stream.write(comment.data(), static_cast(comment.size())); - stream.put('\n'); + write_break(stream, roundtrip); } } } - write_indent(stream, indent, width); + write_indent(stream, columns); } first = false; @@ -749,13 +975,12 @@ inline auto write_block_mapping(OutputStream &stream, const JSON &value, (roundtrip != nullptr) && entry.second.is_null() && !entry_is_alias && ((entry_style == nullptr) || !entry_style->scalar.has_value())}; if (implicit_null) { - if ((entry_style != nullptr) && entry_style->anchor.has_value()) { + if (has_node_properties(entry_style, entry.second)) { stream.put(' '); - write_anchor(stream, entry_style->anchor.value(), entry.second, - anchors); + write_node_properties(stream, entry.second, entry_style, anchors); } emit_inline_comment(stream, entry_style); - stream.put('\n'); + write_break(stream, roundtrip); } else { bool has_indicator_comment{false}; if ((entry_style != nullptr) && @@ -765,13 +990,12 @@ inline auto write_block_mapping(OutputStream &stream, const JSON &value, const auto &comment{entry_style->comment_on_indicator.value()}; stream.write(comment.data(), static_cast(comment.size())); - stream.put('\n'); - write_indent(stream, indent + 1, width); + write_break(stream, roundtrip); + write_indent(stream, columns + width); } if (!has_indicator_comment) { - const bool has_prefix{ - entry_is_alias || - ((entry_style != nullptr) && entry_style->anchor.has_value())}; + const bool has_prefix{entry_is_alias || + has_node_properties(entry_style, entry.second)}; const bool entry_flow{(entry_style != nullptr) && entry_style->collection.has_value() && entry_style->collection.value() == @@ -781,12 +1005,12 @@ inline auto write_block_mapping(OutputStream &stream, const JSON &value, !entry.second.empty() && !entry_flow && !has_prefix}; if (nested) { emit_inline_comment(stream, entry_style); - stream.put('\n'); + write_break(stream, roundtrip); } else { stream.put(' '); } } - write_node(stream, entry.second, indent + 1, + write_node(stream, entry.second, columns + width, width, width, has_indicator_comment ? true : false, roundtrip, anchors, pointer); } @@ -796,21 +1020,29 @@ inline auto write_block_mapping(OutputStream &stream, const JSON &value, } inline auto write_block_sequence(OutputStream &stream, const JSON &value, - const std::size_t indent, + const std::size_t columns, + const std::size_t width, const bool skip_first_indent, const YAMLRoundTrip *roundtrip, AnchorValues &anchors, Pointer &pointer) -> void { assert(value.is_array() && !value.empty()); - const auto width{(roundtrip != nullptr) ? roundtrip->indent_width - : INDENT_WIDTH}; + const YAMLRoundTrip::NodeStyle *sequence_style{nullptr}; + if (roundtrip != nullptr) { + const auto style_match{roundtrip->styles.find(pointer)}; + if (style_match != roundtrip->styles.end()) { + sequence_style = &style_match->second; + } + } + + const bool annotations{keeps_item_annotations(sequence_style, value)}; bool first{true}; std::size_t item_index{0}; for (const auto &item : value.as_array()) { pointer.push_back(item_index); const YAMLRoundTrip::NodeStyle *item_style{nullptr}; - if (roundtrip != nullptr) { + if (annotations && (roundtrip != nullptr)) { const auto style_match{roundtrip->styles.find(pointer)}; if (style_match != roundtrip->styles.end()) { item_style = &style_match->second; @@ -824,16 +1056,16 @@ inline auto write_block_sequence(OutputStream &stream, const JSON &value, if ((item_style != nullptr) && !item_style->comments_before.empty()) { for (const auto &comment : item_style->comments_before) { if (comment.empty()) { - stream.put('\n'); + write_break(stream, roundtrip); } else { - write_indent(stream, indent, width); + write_indent(stream, columns); stream.write(comment.data(), static_cast(comment.size())); - stream.put('\n'); + write_break(stream, roundtrip); } } } - write_indent(stream, indent, width); + write_indent(stream, columns); } first = false; @@ -843,9 +1075,9 @@ inline auto write_block_sequence(OutputStream &stream, const JSON &value, if (implicit_null) { stream.put('-'); if (item_style != nullptr) { - if (item_style->anchor.has_value()) { + if (has_node_properties(item_style, item)) { stream.put(' '); - write_anchor(stream, item_style->anchor.value(), item, anchors); + write_node_properties(stream, item, item_style, anchors); } if (item_style->comment_on_indicator.has_value() && !item_style->comment_on_indicator.value().empty()) { @@ -856,7 +1088,7 @@ inline auto write_block_sequence(OutputStream &stream, const JSON &value, } } emit_inline_comment(stream, item_style); - stream.put('\n'); + write_break(stream, roundtrip); } else { bool has_indicator{false}; if ((item_style != nullptr) && @@ -870,13 +1102,15 @@ inline auto write_block_sequence(OutputStream &stream, const JSON &value, stream.write(comment.data(), static_cast(comment.size())); } - stream.put('\n'); - write_indent(stream, indent + 1, width); + write_break(stream, roundtrip); + write_indent(stream, columns + SEQUENCE_INDICATOR_WIDTH); } if (!has_indicator) { stream.write("- ", 2); } - write_node(stream, item, indent + 1, true, roundtrip, anchors, pointer); + write_node(stream, item, columns + SEQUENCE_INDICATOR_WIDTH, width, + SEQUENCE_INDICATOR_WIDTH, true, roundtrip, anchors, pointer, + false, annotations); } pointer.pop_back(); @@ -884,44 +1118,165 @@ inline auto write_block_sequence(OutputStream &stream, const JSON &value, } } +// An anchor only names something when an alias that refers to it is written +// later, as an alias may not stand before the anchor it names. A caller that +// rearranges the document can move an anchor past its aliases, which expands +// them and leaves the anchor naming nothing. +// See https://yaml.org/spec/1.2.2/#71-alias-nodes +struct AnchorDefinition { + Pointer pointer; + std::string name; + std::size_t position; +}; + +inline auto scan_anchor_uses( + const JSON &value, const YAMLRoundTrip &roundtrip, Pointer &pointer, + std::size_t &position, std::vector &definitions, + std::unordered_map &aliases) -> void { + const auto alias{roundtrip.aliases.find(pointer)}; + if (alias != roundtrip.aliases.cend()) { + aliases[alias->second] = position; + } else { + const auto style{roundtrip.styles.find(pointer)}; + if (style != roundtrip.styles.cend() && style->second.anchor.has_value()) { + definitions.emplace_back(pointer, style->second.anchor.value(), position); + } + } + + position += 1; + + if (value.is_object()) { + for (const auto &entry : value.as_object()) { + pointer.push_back(entry.first); + scan_anchor_uses(entry.second, roundtrip, pointer, position, definitions, + aliases); + pointer.pop_back(); + } + } else if (value.is_array()) { + std::size_t index{0}; + for (const auto &item : value.as_array()) { + pointer.push_back(index); + scan_anchor_uses(item, roundtrip, pointer, position, definitions, + aliases); + pointer.pop_back(); + index += 1; + } + } +} + +// An anchor that never had an alias is markup the document was written with, so +// only one whose aliases have all moved ahead of it is dropped +inline auto collect_dead_anchors(const JSON &document, + const YAMLRoundTrip &roundtrip) + -> std::vector { + std::vector dead; + if (roundtrip.aliases.empty()) { + return dead; + } + + Pointer pointer; + std::size_t position{0}; + std::vector definitions; + std::unordered_map aliases; + scan_anchor_uses(document, roundtrip, pointer, position, definitions, + aliases); + + for (const auto &definition : definitions) { + const auto match{aliases.find(definition.name)}; + if (match != aliases.cend() && match->second < definition.position) { + dead.push_back(definition.pointer); + } + } + + return dead; +} + template