From af97e87c92c8050d3ad68dc47498993aa9aac3e5 Mon Sep 17 00:00:00 2001 From: PiX <69745008+pixincreate@users.noreply.github.com> Date: Tue, 6 Oct 2026 22:46:05 +0530 Subject: [PATCH 1/3] scanner: Harden detection and scan coverage Match complete credentials and make incomplete coverage explicit. Bound scan resources, protect output writes, and validate configuration. Add paired accuracy cases, regression tests, and workload measurements. Baseline reviewed synthetic fixtures and document remaining CI and large-repository scan limits. Assisted-by: GPT-6.1 Sol Signed-off-by: PiX <69745008+pixincreate@users.noreply.github.com> --- .keywatch-baseline.json | 1211 +++++++++++------ CHANGELOG.md | 35 + README.md | 67 +- action.yml | 8 +- detectors.toml | 112 +- docs/plans/2026-10-06-hardening.md | 165 +++ docs/security-and-validation.md | 148 ++ .../keywatch_action_scenarios.py | 7 +- scripts/action_validation/validate.py | 2 +- scripts/benchmark.py | 69 + src/baseline.rs | 8 +- src/cli.rs | 10 +- src/config.rs | 28 +- src/config/tests/application.rs | 22 +- src/detector.rs | 47 +- src/lib.rs | 23 +- src/report.rs | 29 +- src/report/sarif.rs | 63 +- src/run_error.rs | 4 + src/scanner.rs | 212 ++- src/scanner/error.rs | 2 + src/scanner/files.rs | 78 +- src/scanner/limits.rs | 73 + src/scanner/lines.rs | 226 +-- src/scanner/staged.rs | 315 +++-- src/utils.rs | 77 +- templates/pre-push.sh | 4 +- tests/accuracy_corpus.toml | 123 ++ tests/accuracy_corpus_tests.rs | 69 + tests/baseline_tests.rs | 115 ++ tests/detector_tests.rs | 171 ++- tests/hardening_tests.rs | 620 +++++++++ tests/hooks_tests.rs | 50 +- tests/report_tests.rs | 51 +- tests/scanner_tests.rs | 187 +++ 35 files changed, 3558 insertions(+), 873 deletions(-) create mode 100644 docs/plans/2026-10-06-hardening.md create mode 100644 docs/security-and-validation.md create mode 100644 scripts/benchmark.py create mode 100644 src/scanner/limits.rs create mode 100644 tests/accuracy_corpus.toml create mode 100644 tests/accuracy_corpus_tests.rs create mode 100644 tests/hardening_tests.rs diff --git a/.keywatch-baseline.json b/.keywatch-baseline.json index 06b4f93..06d3539 100644 --- a/.keywatch-baseline.json +++ b/.keywatch-baseline.json @@ -31,14 +31,21 @@ }, { "file_path": "detectors.toml", - "line_number": 216, - "finding_type": "Certificate", - "matched_content_hash": "648ee3671e3bc408c3bce7d0ce237d9716da2c360a5c57c235a21067ac11eea9", - "plugin_name": "CertificateDetector" + "line_number": 139, + "finding_type": "Password", + "matched_content_hash": "c4b6db313d1e0ff8fa67c051492f789b8a72be53a36ae9a809bedf3c9da567b0", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "detectors.toml", + "line_number": 145, + "finding_type": "Password", + "matched_content_hash": "45621b4aab2b2d2dfbb9ecca8abd5e2cd338cfe8c7e993f3c2d9abb2d98124ce", + "plugin_name": "PasswordDetector" }, { "file_path": "detectors.toml", - "line_number": 829, + "line_number": 883, "finding_type": "Base64 Encoded String", "matched_content_hash": "e7e83ac014759b0de9f4e42e44daa4ae4820947de0c4436363b752b701dd147f", "plugin_name": "Base64Detector" @@ -117,7 +124,7 @@ "file_path": "docs/architecture/detector-config-trust.svg", "line_number": 7, "finding_type": "Base64 Encoded String", - "matched_content_hash": "c77955587c1ef977a8ef77ed7c143faf1c8922f858da143af3e146a3f95fe841", + "matched_content_hash": "b7141e1d4f3ed2f03840807e34d0514f9142d1617d99e6e284495a2391ee5a79", "plugin_name": "Base64Detector" }, { @@ -173,14 +180,14 @@ "file_path": "docs/architecture/scan-pipeline.svg", "line_number": 7, "finding_type": "Base64 Encoded String", - "matched_content_hash": "f3753f090549b3529cacad629e2bfcbd8864d45851010ffedc5159fc3bb74e91", + "matched_content_hash": "8dc72c32e291de9c0401c3550fa2c8df4698939bb6edc5a0fcdc60f0c768513c", "plugin_name": "Base64Detector" }, { "file_path": "docs/architecture/scan-pipeline.svg", "line_number": 14, "finding_type": "Base64 Encoded String", - "matched_content_hash": "ba34e96d0f5312b12c26d734c3cd0c5f468e318de78c5fb54c13f5b6df6515a8", + "matched_content_hash": "9655d730afa93f28500efce8fd3efadc67e1d32ae4d6768e3ac7da8c1fa4dd11", "plugin_name": "Base64Detector" }, { @@ -253,61 +260,47 @@ "matched_content_hash": "b5458e408e59b20b5f8bb66bb996cf3229bba0f4260ab705070fe7bf7029afa4", "plugin_name": "Base64Detector" }, - { - "file_path": "src/baseline.rs", - "line_number": 313, - "finding_type": "Random String", - "matched_content_hash": "c40c9141cd77d55f1ba9d21fb045f8e7c5e900c16f2e563779e7ed75cee6b15b", - "plugin_name": "RandomString" - }, - { - "file_path": "src/config/tests/application.rs", - "line_number": 317, - "finding_type": "Credit Card Number", - "matched_content_hash": "4541206d542811878a9374508fe296fa321a8b56c2902736a02f388836f6e108", - "plugin_name": "CreditCardDetector" - }, { "file_path": "src/detector.rs", - "line_number": 387, + "line_number": 188, "finding_type": "Random String", - "matched_content_hash": "6bb761274b9fe9cb8eaad9e1c7a0a519c6aa21f619a88bf7c10b6ddc41476822", + "matched_content_hash": "b072b74dff00d183f00c1f247fd961d781e865fdf8409f66ed8b6ee324f7fa16", "plugin_name": "RandomString" }, { "file_path": "src/detector.rs", - "line_number": 387, + "line_number": 188, "finding_type": "Base64 Encoded String", - "matched_content_hash": "55aed33555bb9537c54581d477cb02670b2dc79458c39e02fd8796ccb3aa9cc2", + "matched_content_hash": "629c198cd173e43cfaf3fc9b505d06104466a1e966d5491b8c82d5c251b13f67", "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 763, - "finding_type": "Credit Card Number", - "matched_content_hash": "4541206d542811878a9374508fe296fa321a8b56c2902736a02f388836f6e108", - "plugin_name": "CreditCardDetector" + "line_number": 232, + "finding_type": "Base64 Encoded String", + "matched_content_hash": "21860d0774458a5952aeaabd9c216ac03212211ed5191516ba2e6bcfd9f73d05", + "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 764, - "finding_type": "Credit Card Number", - "matched_content_hash": "13ae894eedbfba2dbd06400ba5b215ffd661885646ab86e050fb1a0d192c1c5b", - "plugin_name": "CreditCardDetector" + "line_number": 242, + "finding_type": "Base64 Encoded String", + "matched_content_hash": "69abc601294453463474ddc77dcbcf003202a1dfa2e1f886d880151676c42e5f", + "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 765, - "finding_type": "Credit Card Number", - "matched_content_hash": "0d30829f4cbd240de78f8dc72d0a5ed0a77887656aa572ee8fb1392cf9ae34a1", - "plugin_name": "CreditCardDetector" + "line_number": 362, + "finding_type": "Random String", + "matched_content_hash": "6bb761274b9fe9cb8eaad9e1c7a0a519c6aa21f619a88bf7c10b6ddc41476822", + "plugin_name": "RandomString" }, { "file_path": "src/detector.rs", - "line_number": 780, - "finding_type": "Random String", - "matched_content_hash": "e51298df0e431de2bfdf6180e3a7b9f3f092c3e9a912facd350e8c7179936e75", - "plugin_name": "RandomString" + "line_number": 362, + "finding_type": "Base64 Encoded String", + "matched_content_hash": "55aed33555bb9537c54581d477cb02670b2dc79458c39e02fd8796ccb3aa9cc2", + "plugin_name": "Base64Detector" }, { "file_path": "src/scanner/lines.rs", @@ -353,18 +346,11 @@ }, { "file_path": "tests/baseline_tests.rs", - "line_number": 53, + "line_number": 168, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 254, - "finding_type": "Generic Key/Secret", - "matched_content_hash": "78eb69b065478e9702683ad7731074c1b46c2ab9c59ef29e9a38754005970888", - "plugin_name": "GenericKeyValueDetector" - }, { "file_path": "tests/detector_tests.rs", "line_number": 271, @@ -416,756 +402,686 @@ }, { "file_path": "tests/detector_tests.rs", - "line_number": 422, - "finding_type": "Credit Card Number", - "matched_content_hash": "4541206d542811878a9374508fe296fa321a8b56c2902736a02f388836f6e108", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 423, - "finding_type": "Credit Card Number", - "matched_content_hash": "e2e5b50c7d336fb0ed238a9c1dd7520b847fe55b990c8f90ecb810a398854d52", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 424, - "finding_type": "Credit Card Number", - "matched_content_hash": "13ae894eedbfba2dbd06400ba5b215ffd661885646ab86e050fb1a0d192c1c5b", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 425, - "finding_type": "Credit Card Number", - "matched_content_hash": "7eafa99f1c4d8c35d2a84d390f9e1a9806fb518aaad3bd0c6dde7c9669e1ab97", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 433, - "finding_type": "Credit Card Number", - "matched_content_hash": "b542aa1c9d50b5b2f457050ecab5fa170d57ae38ebaeb3ef7d4cfcf97b9d4389", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 433, - "finding_type": "Credit Card Number", - "matched_content_hash": "52d19e14a2590e721948f58bf1e9dacf8a07049ee1b3238f193a8f15c0f997bf", - "plugin_name": "CreditCardDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 456, + "line_number": 421, "finding_type": "Phone Number", "matched_content_hash": "9a90be7e1667c88c036f8f271e9746c37d99740f5762fb056090339a75ef24e6", "plugin_name": "PhoneNumberDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 456, + "line_number": 421, "finding_type": "Phone Number", "matched_content_hash": "950cf54889ab5374a01729d84c2e3cf6a086f403b24e4109f522d22d03f1fcf3", "plugin_name": "PhoneNumberDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 456, + "line_number": 421, "finding_type": "Phone Number", "matched_content_hash": "af0d73c957c0706f71d0421fd5dd40748b5663537855374b60d71c0624962486", "plugin_name": "PhoneNumberDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 467, + "line_number": 432, "finding_type": "Phone Number", "matched_content_hash": "33360227b18134e6e594a3df05104457bac9713999bb6374dc9bd74fd54c49ee", "plugin_name": "PhoneNumberDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 477, + "line_number": 442, "finding_type": "SSH Private Key", "matched_content_hash": "1b01887f477e98dd56ec542c12433dee8c323176bd532c05163823079263ba31", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 478, + "line_number": 443, "finding_type": "SSH Private Key", "matched_content_hash": "32006da0b4e4851aa7946ffa4f040f364cac7c317f312b5d742ca5833dd8760f", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 479, + "line_number": 444, "finding_type": "SSH Private Key", "matched_content_hash": "678a65e8968aabff441076ae306e13d4fc85b1d36d8036a10d2100c1dc40d251", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 490, - "finding_type": "Random String", - "matched_content_hash": "e51298df0e431de2bfdf6180e3a7b9f3f092c3e9a912facd350e8c7179936e75", - "plugin_name": "RandomString" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 515, + "line_number": 480, "finding_type": "AWS Access Key", "matched_content_hash": "0cbae582394e61dc81d9946ed87ec44f4c83cf1167181f62cd1454eb5e2e5469", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 542, + "line_number": 507, "finding_type": "Generic Key/Secret", "matched_content_hash": "f6bd37622d846ac435a7b7dcbde2347d59494dfd3c3a787604e8b91491a39c95", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 543, + "line_number": 508, "finding_type": "Generic Key/Secret", "matched_content_hash": "c98bc65cd589366ec37fedb07e874646469d675fc56078e5b8a3e18d2a70e51e", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 544, + "line_number": 509, "finding_type": "Password", "matched_content_hash": "af29a321067a1bbd4a7d3f56ba583751c3f0496bd204f642e1e5764766fe9a8c", "plugin_name": "PasswordDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 570, + "line_number": 537, "finding_type": "Email Address", "matched_content_hash": "ebe162e5d3cc06b42201b0bfe39fde3379a3fe97c9836631f1750f21db142808", "plugin_name": "EmailDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 574, + "line_number": 541, "finding_type": "Email Address", "matched_content_hash": "5900c8c1bad1f56492d653fe4ef68106fd227f791667e7825e183060067aedca", "plugin_name": "EmailDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 591, + "line_number": 558, "finding_type": "Phone Number", "matched_content_hash": "8f44b5588c26ec7be961e53be0c11ced694d220020312a942da3a6fa0a1297d2", "plugin_name": "PhoneNumberDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 605, + "line_number": 572, "finding_type": "Base64 Encoded String", "matched_content_hash": "097eda6585aa25d794c938beab55a00e7f2e7471a39375a3356535aa2e76efa6", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 611, + "line_number": 578, "finding_type": "Random String", "matched_content_hash": "f56884431ad6ea3a4fb741ca53cae314b4d3202d6f14bf69aafe86440f799403", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 611, + "line_number": 578, "finding_type": "Base64 Encoded String", "matched_content_hash": "a011d6c6ecfe7ed080c7c4bfcc08efc5a3744a7be7a117c466d2536c2ee30a9b", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 611, + "line_number": 578, "finding_type": "Generic Key/Secret", "matched_content_hash": "c9bff1b4953e132d00a1d95477c81e7e73f15847b0e7d538c5a3a9c022858899", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 623, - "finding_type": "Random String", - "matched_content_hash": "4b98f33784e3a2e93ade7cd0d84c9bebe903fc6cae40c951d055e13e241440f4", - "plugin_name": "RandomString" + "line_number": 599, + "finding_type": "Password", + "matched_content_hash": "6f8d7f24b4f8974bcbb118d7030beb032445fb0e798cd478923680ee1ca8457d", + "plugin_name": "PasswordDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 623, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "e64a42fa536515c68483f9b16cae91558748e69c9e7119bc8f32964322a67fe2", - "plugin_name": "Base64Detector" + "line_number": 608, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "064402036625a9e1d42472b70bd4906899c127023416b4977afcaca7e36b52ea", + "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 624, - "finding_type": "Random String", - "matched_content_hash": "74b06d8603b542ac6581dcac272c221820e7dddb4710cd68ad2542892a6c8f2a", - "plugin_name": "RandomString" + "line_number": 609, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "7e30fc480fb50248115dba9574bfc592039b5b4007770a0a456cc3f32aa36824", + "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 624, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "b5a2901b320d7344b1ed5676cf19aa3bc3d14ed2e513b5b086e4f01f9d85a0da", - "plugin_name": "Base64Detector" + "line_number": 610, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "c70bdcf84ffafe6934969523fc4584da57bedb67ac9c6bc06df2e504170d306d", + "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 626, - "finding_type": "Random String", - "matched_content_hash": "8f492c68c7d44324c4babb1ab7c115ba365aa9e702b85cb96b2432ed6a9967f2", - "plugin_name": "RandomString" + "line_number": 612, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "a94a7aba4833d3b1dd72bc0be4cd5a891558c9135e30cf0d4c382f39b110ada8", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 614, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "cab6c8062cc0037e4dd528446ba4d2292c523217459bef6964a42ef147bd891e", + "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 626, + "line_number": 649, "finding_type": "Base64 Encoded String", - "matched_content_hash": "764c955c280af7eb75cdd5ced467826bbec01bb7474379d1bbae74b91f107c97", + "matched_content_hash": "00ec0f07296eaa8344f902a2897f3d48514dfeb72701bf26f9ceb78fa0c8ac15", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 627, + "line_number": 673, "finding_type": "Random String", - "matched_content_hash": "be963ea18a45cce7e6304bbbb555137f550f13d598e71b748057ce45b0dfe762", + "matched_content_hash": "80867cc8c03f35c4dc16374edb49d71bb2bb293e5ccff8fda2818b78a64d58e7", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 627, + "line_number": 673, "finding_type": "Base64 Encoded String", - "matched_content_hash": "9793e393aabc5b9fe76299897081c6f14e6393998549defea81667d04bf22de0", + "matched_content_hash": "0d204a2c7bf6cbe32065cc1e0f830481e15711186ddf26803cdaa28dfa7a30eb", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 628, + "line_number": 673, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "2db775a823d4902177d098997dcff95b72df6dd55a5c781011553870aea65539", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 701, + "finding_type": "JWT Token", + "matched_content_hash": "0d6273826316eeee8f374526763d751ca73f134c33a6f7d97c2e133cce5c3ca3", + "plugin_name": "JWTokenDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 719, "finding_type": "Random String", "matched_content_hash": "b28d3a52df9a1c7a7d59cbfb6741fdd3d35216675d170a9bc7dbef2744f2b3ec", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 628, + "line_number": 719, "finding_type": "Base64 Encoded String", "matched_content_hash": "41e2a39a10ab1a5d623590de9962259020ba4218cb57ebbcd7e90b7f87aa1546", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 653, + "line_number": 744, "finding_type": "Random String", "matched_content_hash": "30223dbf4c6f1329a106b8f7583cb64cf8d3f583f3eefda7041c662fc2fa87d9", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 658, - "finding_type": "Generic Key/Secret", - "matched_content_hash": "7e5cddbc0123c9ac48d6ec86899945d296c02974ff3fd24c89cbdb5a8e7cd74e", - "plugin_name": "GenericKeyValueDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 672, - "finding_type": "Generic Key/Secret", - "matched_content_hash": "31ab04ad69305e88b42ca887b277c80c84694d95c2867d19cb48cf3ea745e0fb", - "plugin_name": "GenericKeyValueDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 705, + "line_number": 796, "finding_type": "Base64 Encoded String", "matched_content_hash": "e10e88c7f8ade4bef7e1129d88ee488f3ca18b10225fe63173133eda5a7e4ecf", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 734, + "line_number": 825, "finding_type": "Generic Key/Secret", "matched_content_hash": "3c8321f2331972cb4b8c59d16de3fae99fe333508dfd8741b332817f0c0045c4", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 762, + "line_number": 853, "finding_type": "Random String", "matched_content_hash": "c8930ac0699b2ac783fbeeaa531643cf1f97933a23bf50c1e83ca93df8a2c86a", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 762, + "line_number": 853, "finding_type": "Netlify Token", "matched_content_hash": "a689cfd83519a1d7baacb966f5703133ec6282972f9a80fd6f51295295ffcc3c", "plugin_name": "NetlifyTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 762, + "line_number": 853, "finding_type": "Generic Key/Secret", "matched_content_hash": "73e2d76ad4b9f5d13b11b59426ec03ea630309c848008d952e026a173e7b2ab4", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 771, + "line_number": 862, "finding_type": "Generic Key/Secret", "matched_content_hash": "98c20b40b5197777f3a7dd7251776a09823b3fe2e37c83ca43d36ce74ab49412", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 779, - "finding_type": "Random String", - "matched_content_hash": "0d9d69f46579dfe3ed087d18bb0225e0e8412ff02ab21bfa6e1e3b4b50b865f7", - "plugin_name": "RandomString" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 786, + "line_number": 877, "finding_type": "Generic Key/Secret", "matched_content_hash": "d48fd8fe619073c8e7853ae91002e34b1687914ed270ebf78ffaa4e73c573bd7", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 801, + "line_number": 892, "finding_type": "Generic Key/Secret", "matched_content_hash": "8d4694948a126f262ba0e7839b70fe61f9392018de5334500bd2656f6baf11e5", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 801, + "line_number": 892, "finding_type": "Email Address", "matched_content_hash": "6c6b72437ed83fbd7189e7b99110468ad4356bb9b3231e1178edd2bfc48750fd", "plugin_name": "EmailDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 806, + "line_number": 897, "finding_type": "Generic Key/Secret", "matched_content_hash": "062952b9995efce6d1c968542689fdd6d3c57e8e362c428f91acaf6450a79f21", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 810, + "line_number": 901, "finding_type": "Generic Key/Secret", "matched_content_hash": "cc716c48492ba96fbcd95232be522664f95b6d6b112af2919f1f9caae91825cf", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 810, + "line_number": 901, "finding_type": "Razorpay API Key", "matched_content_hash": "4357d9b5d7262f331e5bf679bca5850fd5c3d0cc7a3f3f59f32b34ce22ca83a0", "plugin_name": "RazorpayAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 817, + "line_number": 908, "finding_type": "IP Address", - "matched_content_hash": "8847dd86386a47986b28e9cc2a6975f510292e2f5a028634b6542164ff4bb6f8", + "matched_content_hash": "1b7185fb80bd4ae042a8de37da216e5f3c979ae606434c0b35ee7b474f07587c", "plugin_name": "IPAddressDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 817, + "line_number": 908, "finding_type": "IP Address", - "matched_content_hash": "1b7185fb80bd4ae042a8de37da216e5f3c979ae606434c0b35ee7b474f07587c", + "matched_content_hash": "cfe93fba9bff2613ae8e0471d8b04396114d1e72983bc64332837adbd65c23ce", "plugin_name": "IPAddressDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 817, + "line_number": 908, "finding_type": "IP Address", "matched_content_hash": "2e8d922b332b47bc5e30c20c2956e1cb947ef66523fa17aa3ff616c5719d9b61", "plugin_name": "IPAddressDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 817, + "line_number": 908, "finding_type": "IP Address", "matched_content_hash": "60b9673abd1aecb769a499d1c2a75df1f5976a0e5951621663c052d10cb70dc7", "plugin_name": "IPAddressDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 880, + "line_number": 977, "finding_type": "GCP Service Account Key", "matched_content_hash": "8b77dbdb569eb8b6189e0493fad34e25ced6a8ccbd8331ec36d5a80d2d84eb5f", "plugin_name": "GCPServiceAccountKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 919, + "line_number": 1016, "finding_type": "AWS Access Key", "matched_content_hash": "865d995932d48bd4ccdc53238645ff6d5d43e060bbf692d66e3370f5ce1cf746", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 922, + "line_number": 1019, "finding_type": "Generic Key/Secret", "matched_content_hash": "8fdc6b643e52c153a747ab70a7f80c53e79f120c7b47904413d9e5b77ca972e1", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 922, + "line_number": 1019, "finding_type": "AWS Secret Access Key", "matched_content_hash": "6b085daba2016507175f442843f08fe87148c879fb05e9c53cd6dc346ae9d75f", "plugin_name": "AWSSecretKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 922, + "line_number": 1019, "finding_type": "Base64 Encoded String", "matched_content_hash": "6c197abf076cfea6ed1ec1d1e35ff8f3930724213b0841bed6c2509b29784ff6", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 917, + "line_number": 1023, "finding_type": "GitHub Token", - "matched_content_hash": "3e4f8cb3b94e0a1b7c86b4a50946e22263d19a40548afffc5354777ae3be0d2a", + "matched_content_hash": "b16bd9f07a6786888caf195fb2e6713ab305d373c68c05c878cfbbea71855cb2", "plugin_name": "GitHubTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 917, + "line_number": 1023, "finding_type": "Generic Key/Secret", - "matched_content_hash": "3b11fd40f888000b57ad9c8722bc266e1806df159a61b3f6c4d29d4d20a67e59", + "matched_content_hash": "45c883873c36ab6f5a63ecbcd3c44b7e8aa8c12f75c12b746083ebe6ebc493b4", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 921, + "line_number": 1027, "finding_type": "GitHub Token", - "matched_content_hash": "de2aa2561e7217217afb6937556a045ff3a6ee826f6d4aea799927194eabd5db", + "matched_content_hash": "275e1a2a3f41699adf8ac2b219f5e689a381a12f0e9f195e322c978bffb56ba6", "plugin_name": "GitHubTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 934, + "line_number": 1031, "finding_type": "Random String", "matched_content_hash": "8423d71d9a6fd5e09a6741c6bb3d01bc1d773a95513fa5447211f96da8a6d395", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 934, + "line_number": 1031, "finding_type": "GitHub Fine-Grained PAT", "matched_content_hash": "4c282b9c573df8daea4bfc25b4466c3106e1765396b102899ee7b48d4775dac5", "plugin_name": "GitHubFineGrainedPATDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 938, + "line_number": 1035, "finding_type": "Random String", "matched_content_hash": "337a2f33f5738ea49abef3c1c3253f8909f924af9bc7337e0869b8795e79695d", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 938, + "line_number": 1035, "finding_type": "Generic Key/Secret", "matched_content_hash": "1512aba8af9492ea212748038b4868c715167a18d740e14ac82fb4ef910fb6d2", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 938, + "line_number": 1035, "finding_type": "Slack Token", "matched_content_hash": "ec24b1ae025c523258b45905e028da5592eb7378c0a756d00e6e95cae7c98c63", "plugin_name": "SlackTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 954, - "finding_type": "Random String", - "matched_content_hash": "2466e925ab7084a466f8cee95f8844a2212f11515d48a8053f8cb29dad5e205a", - "plugin_name": "RandomString" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 954, + "line_number": 1051, "finding_type": "Slack App Token", "matched_content_hash": "a1fdb7135bb38d021cba98feaac017fd093e4a11b1c03d00420f016b56db26bb", "plugin_name": "SlackAppTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 958, + "line_number": 1055, "finding_type": "Slack Webhook URL", "matched_content_hash": "f9dedf87592799bbc78d9a5a41856c71fb077483182f7ac6f1cdeb254188582b", "plugin_name": "SlackWebhookDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 967, + "line_number": 1064, "finding_type": "Random String", "matched_content_hash": "dfed8fce3f43297f4774db0a5451c1f46a34b8085039e627e188b2267264c415", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 967, + "line_number": 1064, "finding_type": "Generic Key/Secret", "matched_content_hash": "546029563145946161a1a36146c7ae5fd9b1a7c985b794dea7e2e681bbd6a388", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 967, + "line_number": 1064, "finding_type": "New Relic API Key", "matched_content_hash": "6728d07e118225e1ed0379b523c096195f632bb98cc784eb9b82e3d01bc9ab32", "plugin_name": "NewRelicAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 975, + "line_number": 1072, "finding_type": "Base64 Encoded String", "matched_content_hash": "a08c024ab29d4928fc16f9bb9234d63a1cca7c0db43575e5decbe3fca10a9e75", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 975, + "line_number": 1072, "finding_type": "OpenAI API Key", "matched_content_hash": "7a83e11ca64ed2441c10eacfc2a9ece3683813e7cee863d22ce22a537b7439e8", "plugin_name": "OpenAIProjectKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 979, + "line_number": 1076, "finding_type": "Base64 Encoded String", "matched_content_hash": "a8e8235879316e24e151682d10003df635a18fb3bb706b07ce3589d8a8173ff0", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 979, + "line_number": 1076, "finding_type": "Kimi/Moonshot API Key", "matched_content_hash": "f11819d7089a7701b2db1ae015c5c0a05762720502531dda40c008aeb391626d", "plugin_name": "KimiMoonshotAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 981, + "line_number": 1078, "finding_type": "Stripe API Key", "matched_content_hash": "3f312e5c13b12595f8f3b6f74a8ae6128c1b5a8fb99ca52ea1b4455aa9477331", "plugin_name": "StripeAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 982, + "line_number": 1079, "finding_type": "Stripe API Key", "matched_content_hash": "c6be1124d243824046a8dfeea346c27527a971b5c6e35e88e95bb5ceed26e83e", "plugin_name": "StripeAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 987, + "line_number": 1084, "finding_type": "Aadhaar Card Number", "matched_content_hash": "0476e6cf99db1c000b7e0a433ff83591ffe0472f3156c9ebac7b11d0862f3429", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 988, + "line_number": 1085, "finding_type": "PAN Card Number", "matched_content_hash": "75d7a58ed9996980fee805fb623d81193cdb5e9efddf2fa12b595171ca16530e", "plugin_name": "PANCardDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 989, + "line_number": 1086, "finding_type": "Voter ID (EPIC)", "matched_content_hash": "e530110a549dd903837a0e2af06b406b01367b4572d07bffee9387334c027f20", "plugin_name": "VoterIDDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 990, + "line_number": 1087, "finding_type": "Social Security Number", "matched_content_hash": "f9b230216fb066be06e470cb3c0128af42871ef7c232e1359abbb8f28283a79a", "plugin_name": "SSNDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 991, + "line_number": 1088, "finding_type": "ABHA Health ID", "matched_content_hash": "f18a39a3c0ab355e4e17257dc1e75985da5a8da0ff0992267750e6c005a32cdc", "plugin_name": "ABHADetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 998, - "finding_type": "JWT Token", - "matched_content_hash": "63fab2d473c9588a8eb97b746b1c9471a007965a7bcc984b50282214f6dfbca3", - "plugin_name": "JWTokenDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 1001, + "line_number": 1098, "finding_type": "Database URL", "matched_content_hash": "92a764286dc65f3b07669e8a6cfaa88538b7a9ddb279e506df47abc31e6fc67f", "plugin_name": "DatabaseURLDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1004, + "line_number": 1101, "finding_type": "Database URL", "matched_content_hash": "aa01165935d3f8891e4beb2392f0c5cdd5963b5fbe041e4ec9a3fb65d01e098c", "plugin_name": "DatabaseURLDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1004, + "line_number": 1101, "finding_type": "MongoDB Connection String", "matched_content_hash": "86e05b03473347c72e1cbe549b27bea5396b50802162a4ecfc7ac2fcb8fb12f8", "plugin_name": "MongoDBConnectionStringDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1004, + "line_number": 1101, "finding_type": "Email Address", "matched_content_hash": "6b49e15c059cfbe0fe9f4f9262a1ada9fd0e803098f284fd8b2f7df7a9ea7789", "plugin_name": "EmailDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1008, + "line_number": 1105, "finding_type": "SendGrid API Key", "matched_content_hash": "9211accd29c5d5e26cbc040c59724ed4f59b2ab9a6fd5ef16ec518eb103f91ed", "plugin_name": "SendGridAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1008, + "line_number": 1105, "finding_type": "Base64 Encoded String", "matched_content_hash": "83c33a81e18a5ec4b40f7d9314cb8d63d94c0f324938630cd44ca64230bf0e49", "plugin_name": "Base64Detector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1012, + "line_number": 1109, "finding_type": "Random String", "matched_content_hash": "a907b01629778ccd357a125e843211d716ec4d15c8a5d4a29643aa257beaea77", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1012, + "line_number": 1109, "finding_type": "DigitalOcean API Token", "matched_content_hash": "eac5db004534e626e7b61daf10ca90fb6fe258c812cd7845f4f75c03d496c2b2", "plugin_name": "DigitalOceanTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1016, + "line_number": 1113, "finding_type": "Random String", "matched_content_hash": "c99d7da93fde37dd019d0cf73b366d4142c7e5af291ae011f627da6eef9283a2", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1016, + "line_number": 1113, "finding_type": "NPM Token", "matched_content_hash": "93fa4d52f8628507260c420eda9d4e5b8b53083eaa9c1650697f15ef6c1d4cda", "plugin_name": "NPMTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1020, + "line_number": 1117, "finding_type": "Heroku API Key", "matched_content_hash": "e8cd71caa187d4d908480c35dabd5ac3eaad88e33174fac1229926b78b715a97", "plugin_name": "HerokuAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1020, - "finding_type": "Generic Key/Secret", - "matched_content_hash": "ad7c54874c2910b3ff65345d4bfba97c7aae5e95307bfa745ecf2c2c6c406b29", - "plugin_name": "GenericKeyValueDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 1024, + "line_number": 1121, "finding_type": "Random String", "matched_content_hash": "f48ffeb7e4c083b563be0cff8e4158476658cfc15dc493c14d127fecf31259ae", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1024, + "line_number": 1121, "finding_type": "Groq API Key", "matched_content_hash": "659ba6bae01aab6d09327e7f2d2f808140fd6b990db40a35b1a9cb967ad4b6f4", "plugin_name": "GroqAPIKeyDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1028, + "line_number": 1125, "finding_type": "Random String", "matched_content_hash": "15ce329320873469fe3be8c5266a0d158b2b49a3903447e1d5e0cae1cd4b54b3", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1028, + "line_number": 1125, "finding_type": "Hugging Face Token", "matched_content_hash": "7e620527f0ee998f548f41283f1fd607497f176e8b801154231db33265b5046d", "plugin_name": "HuggingFaceTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1032, + "line_number": 1129, "finding_type": "GitLab Personal Access Token", "matched_content_hash": "1c920d68e02f0cf8f21a1be01c6e91256e4e7e60461406a4bb103af5a4e99539", "plugin_name": "GitLabPersonalAccessTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1036, + "line_number": 1133, "finding_type": "HashiCorp Vault Token", "matched_content_hash": "0e290623d9b5c162f39a3b894bc2a500c542cb879e2b295d321677485eb3315b", "plugin_name": "HashicorpVaultTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1040, + "line_number": 1137, "finding_type": "Google OAuth Token", "matched_content_hash": "14bcf91874e68f421e946b8414c43fbe04f1d5ee1b97fd58d4d2bb10a08cb82b", "plugin_name": "GoogleOAuthTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1044, + "line_number": 1141, "finding_type": "Random String", "matched_content_hash": "0b71a8751aa19a3057b98d957c12b1d0f713b6fed41fb2c1eb244ff45c0942ae", "plugin_name": "RandomString" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1044, + "line_number": 1141, "finding_type": "Shopify Access Token", "matched_content_hash": "a954f6f8f04f0be943d4b794a5d1881ccab4704c42cc7f1032988e969c75e1c8", "plugin_name": "ShopifyAccessTokenDetector" }, { "file_path": "tests/detector_tests.rs", - "line_number": 1048, + "line_number": 1145, "finding_type": "Base64 Encoded String", "matched_content_hash": "554736805d6e994bf786b21a30653266d29aa65775bbc17eb9f910d2556da91b", "plugin_name": "Base64Detector" @@ -1184,6 +1100,13 @@ "matched_content_hash": "d6d60d3e50a63a67cc37ebfac022834f34a7ec0c74d9b5952b8b8ccd50b5f1ff", "plugin_name": "GenericKeyValueDetector" }, + { + "file_path": "tests/exit_tests.rs", + "line_number": 706, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", + "plugin_name": "GenericKeyValueDetector" + }, { "file_path": "tests/fixtures/fp_corpus/README.md", "line_number": 3, @@ -1263,126 +1186,119 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 42, - "finding_type": "Email Address", - "matched_content_hash": "07976d47040b0974eaade10d0da9a9253abeaa82912a2c18e744f0590c90f637", - "plugin_name": "EmailDetector" - }, - { - "file_path": "tests/scanner_tests.rs", - "line_number": 109, + "line_number": 296, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 110, + "line_number": 297, "finding_type": "Generic Key/Secret", "matched_content_hash": "f6bd37622d846ac435a7b7dcbde2347d59494dfd3c3a787604e8b91491a39c95", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 113, + "line_number": 300, "finding_type": "SendGrid API Key", "matched_content_hash": "324af8a2810ec90cd74766d9fa88ddb6957d311ffa77a7a4f1e533f230e6745c", "plugin_name": "SendGridAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 113, + "line_number": 300, "finding_type": "Base64 Encoded String", "matched_content_hash": "c281ba6691c2e2772aebdbd6a7a05c869f7627d7c69eab07edfa130cee7bda70", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 114, + "line_number": 301, "finding_type": "Base64 Encoded String", "matched_content_hash": "703296baa6f0ae75d7b4c6e41b9908603d1273c9db9a28d6fefd4e23d8f3c8b5", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 114, + "line_number": 301, "finding_type": "OpenAI API Key", "matched_content_hash": "7a0d70456feea263871762c537e5e6141866b456eadc938eb9840e59eb08d1a9", "plugin_name": "OpenAIAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 157, + "line_number": 344, "finding_type": "Stripe API Key", "matched_content_hash": "c371dd8bd98e42a547fdc2378597152357423ba782bbb8ecbbafd0508ec8b825", "plugin_name": "StripeAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 187, + "line_number": 374, "finding_type": "Generic Key/Secret", "matched_content_hash": "4c8b5530b17fef887a9b327f69a38f1958e4b656f2b1c54b1d426320ead690ff", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 187, + "line_number": 374, "finding_type": "AWS Secret Access Key", "matched_content_hash": "2c770dc5ac4de63cd2cb89e604080a34ce55dd946586ba7e0d961b31e321403f", "plugin_name": "AWSSecretKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 187, + "line_number": 374, "finding_type": "Base64 Encoded String", "matched_content_hash": "332a14e95304a48a4ef671073cc98f0fc4046c61042ac20287bfa22b37c27d01", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 188, + "line_number": 375, "finding_type": "Generic Key/Secret", "matched_content_hash": "6c6d557cc63eda114745424f991b2019fdfde2844412a1f58fb5527f0e0bef9f", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 227, + "line_number": 414, "finding_type": "SSH Private Key", "matched_content_hash": "678a65e8968aabff441076ae306e13d4fc85b1d36d8036a10d2100c1dc40d251", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 227, + "line_number": 414, "finding_type": "Private Key Content", "matched_content_hash": "b94c156724c086556087773aad998c0791dc79172caa4b3184a7330cfa7c1d76", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 227, + "line_number": 414, "finding_type": "Base64 Encoded String", "matched_content_hash": "4587ecda3342a38feff73eef527c2582add9f98a892b185b54b470bb91bc631e", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 229, + "line_number": 416, "finding_type": "SSH Private Key", "matched_content_hash": "ca54604a9ed82ad96e5f5d68450002ef41112a7058469bbbdde94f83b267dca5", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 229, + "line_number": 416, "finding_type": "Private Key Content", "matched_content_hash": "43531827dd6142886cfeb10207f046021a4eb6c575828583ad2cb20d9430c72f", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 230, + "line_number": 417, "finding_type": "Base64 Encoded String", "matched_content_hash": "131407c23861d5a739223f303ce0b62808fc0dabe0301622d9836dc3c4999fcb", "plugin_name": "Base64Detector" @@ -1396,7 +1312,7 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 268, + "line_number": 455, "finding_type": "Email Address", "matched_content_hash": "48f63a76aa3c93efd693e0d5a960f5fe3ddd697e6a280f812bf678de80106a25", "plugin_name": "EmailDetector" @@ -1439,234 +1355,171 @@ { "file_path": "tests/scanner_tests.rs", "line_number": 497, - "finding_type": "Generic Key/Secret", + "finding_type": "Password", "matched_content_hash": "795c0808e6128ceee90a9dada1ad2ad243d07534def7532d5adf7e82a7237374", - "plugin_name": "GenericKeyValueDetector" + "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", "line_number": 536, - "finding_type": "Generic Key/Secret", + "finding_type": "Password", "matched_content_hash": "ead01d09cbf0176f24f604a4a4f955b32c34b155fe614e81ecc52f2c6fe89a3f", - "plugin_name": "GenericKeyValueDetector" + "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 667, + "line_number": 854, "finding_type": "Aadhaar Card Number", "matched_content_hash": "73e8a0879ef4b4acf94b2c188c0b620e4dd198bef454f6270118d63fd87ff4a3", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 667, + "line_number": 854, "finding_type": "Aadhaar Card Number", "matched_content_hash": "4714ac4f70659987acae4a4760aedecd3924449f855e3be8115a51eb1ad35f93", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 667, + "line_number": 854, "finding_type": "Aadhaar Card Number", "matched_content_hash": "c941e88b8d0be7288dcc792a8ecfae22482b8a83ea4494f1ca0e36293009730b", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 695, + "line_number": 882, "finding_type": "Voter ID (EPIC)", "matched_content_hash": "e530110a549dd903837a0e2af06b406b01367b4572d07bffee9387334c027f20", "plugin_name": "VoterIDDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 695, + "line_number": 882, "finding_type": "Voter ID (EPIC)", "matched_content_hash": "8d6bf65189e5bdf52955b4a1592eb9c2fb0560709152317a1dabc36971c38f30", "plugin_name": "VoterIDDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 719, + "line_number": 906, "finding_type": "PAN Card Number", "matched_content_hash": "75d7a58ed9996980fee805fb623d81193cdb5e9efddf2fa12b595171ca16530e", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 719, + "line_number": 906, "finding_type": "PAN Card Number", "matched_content_hash": "4dd22fad4bf96036ceea29ca253be8d72851cc289d7e694c9b208b7370812ee6", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 743, + "line_number": 930, "finding_type": "ABHA Health ID", "matched_content_hash": "25b618101c3bfdb4438894aa540e108cd150903bd24da57ac24f0d4174c0fe7e", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 743, + "line_number": 930, "finding_type": "ABHA Health ID", "matched_content_hash": "397d0bc36d77ccfc6f74c2eaa6615bb75c3fe3bdc3df26e6f53f413cfcdacdb3", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 768, + "line_number": 955, "finding_type": "Aadhaar Card Number", "matched_content_hash": "8d7ab03162973f8bc2baf7c8556af862841d23c4d66f7f18f8cf5c5777f8b263", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 768, + "line_number": 955, "finding_type": "ABHA Health ID", "matched_content_hash": "89deb2b58721d6303daa4a62f2dee30b9a388afb905fc4d3595ec88ca875fb15", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 768, + "line_number": 955, "finding_type": "PAN Card Number", "matched_content_hash": "18fe123b5e00bbed7228c4ededc8a4e6e0bfd59169af97ec2e62b595eac11dde", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 873, + "line_number": 1060, "finding_type": "Random String", "matched_content_hash": "cbe1ce18b874bc08437699d863cd422c49e104462e7d5ac6bd390acc0d7c973a", "plugin_name": "RandomString" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 873, + "line_number": 1060, "finding_type": "Google API Key", "matched_content_hash": "aec6660855470791a5b4f5a6c1ac4b61e61d2b942b1cdec400fb7811743798d0", "plugin_name": "GoogleAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 920, + "line_number": 1107, "finding_type": "Password", "matched_content_hash": "67ef748345ad7f183084a7449ec05906e648c882400b9d940f2fadec23a7b197", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1015, + "line_number": 1202, "finding_type": "AWS Access Key", "matched_content_hash": "05c0aace2b76ca255ed3a7a953016d981477226dccc3b0e709d00174c8bc48b5", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1658, + "line_number": 1845, "finding_type": "Generic Key/Secret", "matched_content_hash": "97894929682fc00219686473cbcfa3731d73b23e88ffa7c19a511c1bbfa18aa5", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1658, + "line_number": 1845, "finding_type": "AWS Secret Access Key", "matched_content_hash": "58db90a4f2acf80492ed15e73ad77de6c84b9aace839d821931aa0474c898386", "plugin_name": "AWSSecretKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1862, + "line_number": 2049, "finding_type": "AWS Access Key", "matched_content_hash": "357b7fb7890985d4c94a43012d1f7aefe25757f36d8810388357993bb38bd8e7", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2114, + "line_number": 2301, "finding_type": "AWS Access Key", "matched_content_hash": "0cbae582394e61dc81d9946ed87ec44f4c83cf1167181f62cd1454eb5e2e5469", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2415, + "line_number": 2602, "finding_type": "Password", "matched_content_hash": "6121378fbc25183476235d474ee367f7dbd0bbe3d96642daaeb483b5e9108cdb", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2529, + "line_number": 2716, "finding_type": "Base64 Encoded String", "matched_content_hash": "37f93597692f90e64e009b94ce81237f0f5a6d03ffce8085e91814efe71e4431", "plugin_name": "Base64Detector" }, - { - "file_path": "src/detector.rs", - "line_number": 193, - "finding_type": "Random String", - "matched_content_hash": "b072b74dff00d183f00c1f247fd961d781e865fdf8409f66ed8b6ee324f7fa16", - "plugin_name": "RandomString" - }, - { - "file_path": "src/detector.rs", - "line_number": 193, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "629c198cd173e43cfaf3fc9b505d06104466a1e966d5491b8c82d5c251b13f67", - "plugin_name": "Base64Detector" - }, - { - "file_path": "src/detector.rs", - "line_number": 234, - "finding_type": "Random String", - "matched_content_hash": "005cf828a6da2ba5f43f9bc052eb0db15b6aae1cc493d7f0c92792d06bb037c9", - "plugin_name": "RandomString" - }, - { - "file_path": "src/detector.rs", - "line_number": 234, - "finding_type": "GitHub Token", - "matched_content_hash": "b16bd9f07a6786888caf195fb2e6713ab305d373c68c05c878cfbbea71855cb2", - "plugin_name": "GitHubTokenDetector" - }, - { - "file_path": "src/detector.rs", - "line_number": 237, - "finding_type": "Random String", - "matched_content_hash": "f3aca0a62b8d61959b005ce780f01a952a647059583578014588434d4226e33e", - "plugin_name": "RandomString" - }, - { - "file_path": "src/detector.rs", - "line_number": 240, - "finding_type": "Random String", - "matched_content_hash": "cf3b0601cd155fba56ea1cf383f57a9fd45a85bf1cdb677838c6676d62bf222f", - "plugin_name": "RandomString" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 926, - "finding_type": "GitHub Token", - "matched_content_hash": "b16bd9f07a6786888caf195fb2e6713ab305d373c68c05c878cfbbea71855cb2", - "plugin_name": "GitHubTokenDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 926, - "finding_type": "Generic Key/Secret", - "matched_content_hash": "45c883873c36ab6f5a63ecbcd3c44b7e8aa8c12f75c12b746083ebe6ebc493b4", - "plugin_name": "GenericKeyValueDetector" - }, - { - "file_path": "tests/detector_tests.rs", - "line_number": 930, - "finding_type": "GitHub Token", - "matched_content_hash": "275e1a2a3f41699adf8ac2b219f5e689a381a12f0e9f195e322c978bffb56ba6", - "plugin_name": "GitHubTokenDetector" - }, { "file_path": "tests/scanner_tests.rs", "line_number": 2623, @@ -1683,87 +1536,633 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2670, + "line_number": 2857, "finding_type": "Base64 Encoded String", "matched_content_hash": "f6c41d46bc806e05582dd7504058eb98e514473b4637fcbd3f1f46a12d6a1399", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2670, + "line_number": 2810, "finding_type": "Generic Key/Secret", "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2677, + "line_number": 2864, "finding_type": "Base64 Encoded String", "matched_content_hash": "de00639949784d409b3d3e7b866432f64ca1fe35e047b312a012bbc0a19501b9", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2719, + "line_number": 2906, "finding_type": "SSH Private Key", "matched_content_hash": "1b01887f477e98dd56ec542c12433dee8c323176bd532c05163823079263ba31", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2719, + "line_number": 2906, "finding_type": "Private Key Content", "matched_content_hash": "d3022d6b263b30e0ab3e0d77220f81e0bebfcc0b723f84f77d2d82d306d22e50", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2719, + "line_number": 2906, "finding_type": "Base64 Encoded String", "matched_content_hash": "81623b8aeb6fe8c68c0a53b2d7a12ecf1d044b3ac15886fdaced0eb177f385dc", "plugin_name": "Base64Detector" }, { - "file_path": "tests/exit_tests.rs", - "line_number": 706, + "file_path": ".github/workflows/ci.yml", + "line_number": 77, "finding_type": "Generic Key/Secret", "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", "plugin_name": "GenericKeyValueDetector" }, { - "file_path": "src/detector.rs", - "line_number": 237, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "21860d0774458a5952aeaabd9c216ac03212211ed5191516ba2e6bcfd9f73d05", - "plugin_name": "Base64Detector" + "file_path": "scripts/benchmark.py", + "line_number": 57, + "finding_type": "AWS Access Key", + "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", + "plugin_name": "AWSKeyDetector" }, { - "file_path": "src/detector.rs", - "line_number": 247, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "69abc601294453463474ddc77dcbcf003202a1dfa2e1f886d880151676c42e5f", - "plugin_name": "Base64Detector" + "file_path": "tests/accuracy_corpus.toml", + "line_number": 4, + "finding_type": "Password", + "matched_content_hash": "e8eb0049db9c4975d70e3abda509cd398bf52fdb873f2cec2d65ca3d869149b5", + "plugin_name": "PasswordDetector" }, { - "file_path": "docs/architecture/detector-config-trust.svg", - "line_number": 7, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "b7141e1d4f3ed2f03840807e34d0514f9142d1617d99e6e284495a2391ee5a79", - "plugin_name": "Base64Detector" + "file_path": "tests/accuracy_corpus.toml", + "line_number": 8, + "finding_type": "Password", + "matched_content_hash": "a5e35f41efcb2f1feace9f5de93895899afe5cb9fa8ac783698e6f3e953fb986", + "plugin_name": "PasswordDetector" }, { - "file_path": "docs/architecture/scan-pipeline.svg", - "line_number": 7, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "8dc72c32e291de9c0401c3550fa2c8df4698939bb6edc5a0fcdc60f0c768513c", - "plugin_name": "Base64Detector" + "file_path": "tests/accuracy_corpus.toml", + "line_number": 12, + "finding_type": "Password", + "matched_content_hash": "da59f2312034f2f4d9b3e07888c42f842ab8319bff412a137fbcc7143a53a4a4", + "plugin_name": "PasswordDetector" }, { - "file_path": "docs/architecture/scan-pipeline.svg", - "line_number": 14, - "finding_type": "Base64 Encoded String", - "matched_content_hash": "9655d730afa93f28500efce8fd3efadc67e1d32ae4d6768e3ac7da8c1fa4dd11", - "plugin_name": "Base64Detector" + "file_path": "tests/accuracy_corpus.toml", + "line_number": 16, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "882038bbf112978faba52d26bc7b3c94baf342a49829780a8e60c6fc151d5306", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 20, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "1abd585a115aba9f26337d05f03f6f15a2c52f9210652b2cf77a9f314d48d2d4", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 24, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "355085b30c615c9b0bab3120b15d62be0eefb58eb6b0c97f3434d9289d36c944", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 28, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "2b4f47d7b0e620bb678ccec70c5c47c17332ceadf650f53ce5dd81ab20263e00", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 32, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "18f5030adb92484a0f5d6d603fb2cee4c322951a24de4ed855f71e6e0ca49fa8", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 36, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "d2e344619a7cf9e6286920a10d54d3e49cf1ef4482697e4ff67473426b5c2e80", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 40, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "44f155dc61976abb0aa3f0fad1b8b0c4f585d07e1184d9957be6c7650e5d3bfa", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 44, + "finding_type": "Password", + "matched_content_hash": "d8880112f59bc42ba1980daec1c8ea00037a09cd98e8e9d36ccd8e3c635d5fa0", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 48, + "finding_type": "Password", + "matched_content_hash": "1e5e3c3f3c82f9c1819e26230e521021d0faef82800732b02a90ffc079ef49f3", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 52, + "finding_type": "Password", + "matched_content_hash": "b852778a6601bce63299473637d31fc855da0175928516bea4c59ac3df9700b4", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 56, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "9a0da032992f2dcc88ac52febfb9eab1388f4d44bedd723bea26035760b24f9b", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 60, + "finding_type": "Password", + "matched_content_hash": "4273229cb3a44618a8db7e8a7a0dba51a429ffd163ff9a1a31b6cf1f4ce39eda", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 64, + "finding_type": "AWS Access Key", + "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", + "plugin_name": "AWSKeyDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 68, + "finding_type": "Generic Payment Gateway Key", + "matched_content_hash": "6b960145fdcc53a7bcbd0bbab24b32e51e3c4025572aa20900bdc83e865fbe27", + "plugin_name": "PaymentGatewayKeyDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 72, + "finding_type": "Generic Payment Gateway Key", + "matched_content_hash": "8a639559f1fbf8d34771ae27bec0ed243ad52417018123a284abf0aeb5175bb3", + "plugin_name": "PaymentGatewayKeyDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 76, + "finding_type": "Service Account Key", + "matched_content_hash": "e0b04164f54f55b8cf99ad57c71965cb715c00583e4607985ded3a01bc427ea6", + "plugin_name": "ServiceAccountKeyDetector" + }, + { + "file_path": "tests/accuracy_corpus.toml", + "line_number": 80, + "finding_type": "Service Account Key", + "matched_content_hash": "878f21920cfd94fc48d0f11f773524d220294cbf7714bb91ce2ac619d099810a", + "plugin_name": "ServiceAccountKeyDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 93, + "finding_type": "Password", + "matched_content_hash": "e8eb0049db9c4975d70e3abda509cd398bf52fdb873f2cec2d65ca3d869149b5", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 94, + "finding_type": "Password", + "matched_content_hash": "0277e63a3110fdb712c7d51a71dbd221f813df135a2a55175c5710196a608ae9", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 99, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "b45cc2d7a8e0aaa40496f39d55356fc05dee4e8bdb55c2903632c8485d14c757", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 100, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "62e816fbbfa820a8570f6c5415eb7b8d227df44a74cbd2357a9d88075e36e0bf", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 272, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "07f27c3cf9a9135a82973a847dd2a2009ee31cce8852e834c75d3ae8ed094e0e", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 273, + "finding_type": "Password", + "matched_content_hash": "75c459789c1be82f54084983aab18a5cf5ae6166e79e1192227557606394e34d", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 274, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "20a38545364d35a169c0167d08a9889eeeaee5757cb19d01634b49afc9aa714a", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 275, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "4c2a667d907169e6fe6bb9f6c589581f8f7f0e9ed2e65af673087263878f41d9", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 391, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "bdc71c84eb3243ace39f5c4b9a03dedc91d6f07e7fe678be32861ea5b93230fc", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 392, + "finding_type": "Password", + "matched_content_hash": "c6b9567a921c6601e5cc219f1f63918bfc550e103fdf08b86f22244d13f1bb16", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 508, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "acffc9647fcecadef45aa6e9cf8763e9093cfb331651cc886d7a3a0db8fcc11a", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 509, + "finding_type": "Password", + "matched_content_hash": "3ba8c78e4ea6517fd0d22026ca20956ab75adceeea6c2a56b5bf475cbbcf7f44", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 599, + "finding_type": "Password", + "matched_content_hash": "41b7bcf5abc47673eea6c106d900d4327dc090a41ec1e2c185fa6b8d1d90ae61", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 825, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "25a05ff1d0ba9fc3484b0d1c7d54263030337c0003ae3a611ef335fa8810b8ee", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 862, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "d1d824be95eb78fdb9d11a3ac68b50ac7d1fc17a780f3c102818c641a7b69459", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 877, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "f99ea06430382525b294943073d9ac58e1095d192d3eddae0ebb4b751496d224", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 897, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "0339ed14e99e51f3fbb56887f46dd6edcf5ed097c72a1c0e731487b28669d76a", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 901, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "df2094339d0c083541b06b6ee2aa6d449470f0443a51f6599f2cd37530c02bb3", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1023, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "ab5b521f493c97a47bcb4e8ffce12c137b1da758b7f5fa27575f3ec4eb2c199b", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1035, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "1a36c8a92e306647a30628eabfa4f41409eca8a55c08440562f94da65ecbdcfb", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1064, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "463d94f21daf853f009394a4c7caf5a6700e3c29cc4e59019ef1d07128e74929", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1101, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "37fa9fd681522303706db1400cf029676475d33a88a0434eaf64c9c8b502a2b5", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1117, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "329ac35590009cf405b8bfc2df4a03fd133d9e7d1b854fb9825a354b6787f141", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/hardening_tests.rs", + "line_number": 58, + "finding_type": "AWS Access Key", + "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", + "plugin_name": "AWSKeyDetector" + }, + { + "file_path": "tests/run_cli_error_tests.rs", + "line_number": 9, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 93, + "finding_type": "Password", + "matched_content_hash": "e8eb0049db9c4975d70e3abda509cd398bf52fdb873f2cec2d65ca3d869149b5", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 94, + "finding_type": "Password", + "matched_content_hash": "431e6d62bea13942f291d35cde3a76a858835a050ac27f881909c18a3ae77f4f", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 95, + "finding_type": "Password", + "matched_content_hash": "d3bad73e57534c0b8e2be356be0e85b0a7fc8dda8460129a6f46a4751398025b", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 105, + "finding_type": "Password", + "matched_content_hash": "e0bcedac4b8f91c6d19b8f3490206f289394d980e40d663e002797cd1edd9b6c", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 125, + "finding_type": "Password", + "matched_content_hash": "cc115cb1c6cabde0bca4f9eda1f19a1e03ce8d4bcbbd39fbcc5494adaf6d513f", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 126, + "finding_type": "Password", + "matched_content_hash": "2601374643a4111d9a57b343dd3fc21a9d3120be180cafb72bb1c50d2d0d12e8", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 127, + "finding_type": "Password", + "matched_content_hash": "55c75e5fa5cdf17af1924e269d3e5a50482e5f176e8678b8b8aa80cefdde7327", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 128, + "finding_type": "Password", + "matched_content_hash": "b8fe043c2960a207b6e4e08c756b8599c659b5d611bb0125129d62ea294e4061", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 129, + "finding_type": "Password", + "matched_content_hash": "a5e35f41efcb2f1feace9f5de93895899afe5cb9fa8ac783698e6f3e953fb986", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 130, + "finding_type": "Password", + "matched_content_hash": "efc3112c94494384c049ee16980a5036654a064ebb52427e44ba36fd9f16336a", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 170, + "finding_type": "Password", + "matched_content_hash": "1d0d219f9ad298656f91a44acefd09073c64ae21b222418470d74ff9f879875a", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 180, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "b45cc2d7a8e0aaa40496f39d55356fc05dee4e8bdb55c2903632c8485d14c757", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 181, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "4c72db1edb621dfd726fb04d1aed8c690c18a54e4308a6139f96f041766227bc", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 182, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "8c4338da401e17fb3fd8e54984aad04556c75ee4dfb6257d5b56ef5746e7cfb6", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 183, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "2f93ff759c4ed619c6226d6a2469a849f8b27639f142d56cd1146791795e9697", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 184, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "39c7920b0af4c5950547533611929a853899c7e8e9e2db6402e54bd6b7e4ae24", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 185, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "bb05efe9a21754d719f32dcb799814a54b8b05deb52d9e380d0a7b74858b6858", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 186, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "0576d8963c5377bd32001b76e85400f83fa8d063abd62cbe1aab91335ab556d7", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 187, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "ccb05cd9847eccd058e416fb03c4a3d3680b9ea1b86b1388acd01f50646e2b2d", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 188, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "dfdf7e4c468aa63e7a470befd56a4c8a8b2ecf00b6668161f8f535cf58cf9ef8", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 216, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "882038bbf112978faba52d26bc7b3c94baf342a49829780a8e60c6fc151d5306", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 216, + "finding_type": "Password", + "matched_content_hash": "da59f2312034f2f4d9b3e07888c42f842ab8319bff412a137fbcc7143a53a4a4", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 243, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "bedf750ffbf75b7493c11baad571804a3edb884f6536722a5d99e8fea2e648ab", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 244, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "62673b45fa4d97d4e3914584c611fc886c814691240b08da38c11b6155a50d1b", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 245, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "f8abdc480c8cf97c232ae37084a8dd211daa71b7c8d15fbdcdf7456d3cd4bdde", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 247, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "5698e9b921268822471d96185883282d0aee44c9f40e39863abfd5334f004a11", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 250, + "finding_type": "Password", + "matched_content_hash": "d8880112f59bc42ba1980daec1c8ea00037a09cd98e8e9d36ccd8e3c635d5fa0", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 251, + "finding_type": "Password", + "matched_content_hash": "1e5e3c3f3c82f9c1819e26230e521021d0faef82800732b02a90ffc079ef49f3", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 252, + "finding_type": "Password", + "matched_content_hash": "b852778a6601bce63299473637d31fc855da0175928516bea4c59ac3df9700b4", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 254, + "finding_type": "Password", + "matched_content_hash": "b2c167e2202c4fe897b118e21d5758cf6147186b276660620d9c85e462f5a307", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 455, + "finding_type": "Password", + "matched_content_hash": "3021414aa2f536cb3b4b850cc0573e71e97596e81014e0e62d1abf7785a2f998", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 525, + "finding_type": "Password", + "matched_content_hash": "406dd360d9af78b059bb954a52e93435b61065bb2fe4d742acef80ae4223c549", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 526, + "finding_type": "Password", + "matched_content_hash": "5492a757450efbf6b362c4e37a438efcf02ffeebf1992acb4d364f33adfaf04e", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 647, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "450344b31602c3e15912c9ca9f0504aa45cd6d2816934c07f8bdf07434266f1d", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 648, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "84a0a3964fed0da6095ff2ebc1ec44efb3c973fdf5c8728e4838be7e232ff47a", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 684, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "158e13610e827eb0e4e836966cad520a90a938d98b85b95b38c9294103dfe20f", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 723, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "4580776c2c5ad71d3dd4cfade7fb51fa6b5a06315aaa5704d925d64c0293b7b0", + "plugin_name": "GenericKeyValueDetector" } ] } diff --git a/CHANGELOG.md b/CHANGELOG.md index c80e2ad..39dfa64 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,41 @@ All notable changes to this project will be documented in this file. ## [Unreleased] +### Breaking changes + +- `CreditCardDetector` and the `luhn` validator are removed; payment-card numbers are no longer a finding type, and custom rules that use `validate = "luhn"` must switch to a supported validator +- Incomplete reports use `INCOMPLETE` instead of `PASS` or `FAIL`. + Finding status and coverage are separate fields. +- `--fail-on-unscannable` applies in every exit mode, and incomplete scans cannot update baselines. +- Inputs have a 16 MiB ceiling, lines have a 1 MiB ceiling, and scans enforce path and finding budgets. + `--max-file-size` can lower the input ceiling but cannot raise it. + +### Changed + +- Password, Generic Key/Secret, Base64, Random String, IP Address, Email, Certificate, and JWT detectors use narrower shape rules and allowlists. + These rules suppress selected declarations, unquoted references, language descriptors, checksum prefixes, loopback addresses, prose mentions, and placeholder signatures. +- Password matches include complete quoted values and typed Rust `&str` or `&'static str` literals. + Generic Key/Secret matches include `/` and `+` instead of stopping at those characters. +- Password and Generic Key/Secret detectors recognize quoted JSON keys. + Generic Key/Secret detection also covers `apikey`, `accesskey`, and `securitykey`. +- Quoted snake-case and kebab-case values no longer receive a general exemption from Generic Key/Secret detection. + Placeholder exemptions match complete, selected values rather than prefixes or arbitrary suffixes. +- Corrected matches can produce findings that an existing baseline does not suppress. + The baseline file format remains unchanged. + Review these findings before updating your baseline. +- Git scans pin path prefixes and hunk context, disclose shallow history, and scan historical text rendered as binary. +- Custom multiline patterns no longer depend on guessing flags from regex source or fixed overlap windows. + Base64-decoded multiline findings use the encoded source line. +- Report and baseline writes use same-directory atomic replacement and reject destination symlinks or unsafe immediate parent directories. +- Unknown override names and duplicate detector names fail before configuration changes apply. + Recursive scans report skipped symlinks and unreadable entries as incomplete coverage. +- `--scan-lockfiles` includes lockfiles without removing the default checksum-noise policy. +- Pre-push hooks and the GitHub Action fail on incomplete coverage. + Reinstall hooks to use the changed template. +- Reports identify the scanner version and effective detector definitions. + SARIF locations percent-encode filename characters. +- A labeled synthetic credential corpus and a repeatable workload benchmark protect detection and coverage contracts. + ## [3.0.0] - 2026-09-14 ### Breaking changes diff --git a/README.md b/README.md index 33d595c..2ba5776 100644 --- a/README.md +++ b/README.md @@ -3,6 +3,9 @@ KeyWatch scans files, directories, and git repositories for secrets such as API keys, tokens, passwords, and private keys. It runs as a command-line tool, as a git hook, as a GitHub Action, and as a container image. +See `CHANGELOG.md` for differences between this source tree and published releases. +Do not assume an installed binary or hook contains unreleased changes. + ## Install Install with cargo: @@ -64,8 +67,8 @@ Reports never contain the full matched text unless you pass `--show-secrets`. | Option | Purpose | | ------------------------- | ----------------------------------------------------------------------------------------------------------------- | | `--exclude ` | Skip paths that match these comma-separated glob patterns | -| `--exit-mode ` | `strict` fails on any finding (default), `critical` fails only on HIGH or CRITICAL findings, `always` never fails | -| `--fail-on-unscannable` | Fail when a file or directory could not be read; applies in `strict` exit mode and not with `--update-baseline` | +| `--exit-mode ` | Set the finding policy: `strict` fails on any finding, `critical` fails on HIGH or CRITICAL findings, `always` ignores findings | +| `--fail-on-unscannable` | Fail on incomplete coverage in every exit mode; incomplete scans cannot update baselines | | `--baseline ` | Use a specific baseline file | | `--no-baseline-discovery` | Do not look for a baseline file automatically | | `--update-baseline` | Record the current findings in the baseline instead of reporting them | @@ -75,31 +78,48 @@ Reports never contain the full matched text unless you pass `--show-secrets`. | `--no-repo-config` | Do not look for `.keywatch.toml` in the scanned tree; an explicit `--config` still loads | | `--no-config-discovery` | Shorthand for `--trusted-detectors` plus `--no-repo-config`; the installed hooks pass it | | `--show-secrets` | Include the full matched text in reports | -| `--max-file-size ` | Skip files larger than this size and report them as unscannable | +| `--max-file-size ` | Lower the 16 MiB input ceiling; larger inputs are unscannable | +| `--scan-lockfiles` | Include lockfiles excluded by the default noise policy | Notes: -- Lock files such as `Cargo.lock`, `package-lock.json`, `pnpm-lock.yaml`, and `yarn.lock` are always skipped. - They contain checksums, not credentials. +- Lockfiles such as `Cargo.lock`, `package-lock.json`, `pnpm-lock.yaml`, and `yarn.lock` are skipped by default to reduce checksum findings. + Lockfiles can contain credentials; use `--scan-lockfiles` when your policy requires them. - `--staged` reads the content you staged with `git add`, not the files on disk. A secret that is staged but already removed from the working copy is still found. A secret whose lines were staged in separate commits can span change hunks the diff never shows together; run `key-watch scan .` on the tree to catch that case. - `--git-history` scans every branch and tag. Use `--rev-range` to scan only a range of commits. + Shallow history produces incomplete coverage; fetch complete history before using a history gate. + KeyWatch does not fetch history automatically. + Text that Git renders as binary is scanned from its committed blobs, including deleted versions. - A scan path that does not exist, is a symbolic link, or cannot be read is an error. The scan never reports a clean result for input it could not read. + Recursive scans do not follow symbolic links; skipped links and unreadable entries produce incomplete coverage unless you exclude their paths. - Files that start with a UTF-16 byte-order mark are decoded and scanned. Other files that contain NUL bytes are treated as binary and reported as unscannable. - Base64 runs of 24 or more characters are decoded, and the decoded text is scanned as well. An encoded credential is reported at the line that contains it. - GitHub tokens are checked against their built-in checksum, so lookalike strings do not appear in results. +JSON reports separate `finding_status` from `coverage`. +The combined `status` is `INCOMPLETE` when requested inputs cannot be scanned completely, even when no finding exists. +Reports include the scanner version and a fingerprint of the effective detector definitions. +This fingerprint identifies rules; it does not authenticate the binary or describe every exclusion and suppression. +Report and baseline files use atomic replacement and reject destination symlinks or unsafe immediate parent directories. +Use directories whose parents and ancestors you control. + +Each logical input has a 16 MiB byte ceiling and a 1 MiB line ceiling. +Scans also limit path records, finding counts, and retained finding text. +Exceeding a limit produces incomplete coverage or a runtime error, not a clean report. +See [Security and validation](docs/security-and-validation.md) for limits, measured workloads, and remaining risks. + ### Exit codes | Code | Meaning | | ---- | ----------------------------------------------------------------- | -| 0 | No secrets found, or `--exit-mode always` | -| 1 | Secrets found, or an unreadable file with `--fail-on-unscannable` | +| 0 | Finding policy passes and no explicit coverage failure applies | +| 1 | Finding policy fails, or coverage is incomplete with `--fail-on-unscannable` | | 2 | Invalid input, configuration error, or runtime error | ## Git hooks @@ -111,6 +131,7 @@ KeyWatch installs two git hooks: Findings in lines you did not change never block a commit. - The **pre-push** hook scans the commits you are about to push. It runs in `critical` exit mode, so HIGH and CRITICAL findings block the push; MEDIUM and LOW findings are reported but do not block. + Incomplete coverage also blocks the push. Uncommitted files never block a push. Install and remove hooks inside a repository: @@ -144,7 +165,7 @@ key-watch hook uninstall pre-commit --global - Hooks respect a committed baseline file and `keywatch:ignore` markers. - KeyWatch refuses to overwrite or remove a hook file it did not install. - The first push of a branch scans the full history of that branch, because every commit on it is new to the remote. - If that push reports old findings, record them in the baseline first. + Review old findings before accepting any in a baseline. - A global install sets `core.hooksPath` in your git configuration. Git then ignores each repository's own `.git/hooks` scripts. To keep a repository's own hooks instead, run `git config core.hooksPath .git/hooks` inside that repository. @@ -153,7 +174,10 @@ key-watch hook uninstall pre-commit --global ## Baselines A baseline records findings you have reviewed and accepted, so later scans report only new findings. -The baseline file stores fingerprints of the findings, never the secrets themselves, and is safe to commit. +The baseline file stores unsalted hashes of matched text, not the plaintext secrets. +An attacker can guess low-entropy values offline from these hashes. +Review this disclosure risk before committing a baseline. +Protect baseline changes and inline ignore markers through code review. ```sh # Record the current findings @@ -166,6 +190,12 @@ key-watch scan . KeyWatch finds a committed `.keywatch-baseline.json` automatically. You do not need to pass `--baseline` on every scan. +Detector corrections can change the matched text used for a fingerprint without changing the baseline file format. +An entry created from a truncated match does not suppress a corrected, complete match. +Review findings that reappear after an upgrade before you run `--update-baseline`. +Do not accept them only to restore a clean scan. +An incomplete scan cannot create, merge, or prune a baseline. + ## Ignore a single line Add `keywatch:ignore` to a line to suppress findings on that line: @@ -192,6 +222,7 @@ enabled = false ``` Unknown keys in the configuration file are rejected, so a misspelled key cannot silently weaken a scan. +Unknown detector override names and duplicate detector names are also rejected before configuration changes apply. ## GitHub Action @@ -218,8 +249,10 @@ jobs: ``` The Action installs a released KeyWatch binary, verifies its checksum, and writes a JSON report. +It fails on incomplete coverage, including reports from binaries that only expose unscannable file counts. It supports Linux x64 and macOS runners. Pin an exact release tag or commit SHA when you need a fixed version. +Checksums from the same release detect corruption; they do not independently authenticate a compromised release publisher. | Input | Default | Purpose | | ----------- | ---------------------- | -------------------------------------------------------------------- | @@ -277,11 +310,12 @@ Yellow boxes are external adapters such as git and the installed hook scripts, w ### Scan pipeline -![KeyWatch scan pipeline](docs/architecture/scan-pipeline.svg) - -Path scans collect files and scan them in parallel. -Stdin and git-based scans stream their input in overlapping chunks. -`--update-baseline` writes the baseline instead of producing a report. +Path scans collect files and process batches of four files in parallel. +Stdin and Git blobs use bounded complete inputs. +Git text scans inspect added lines and bounded addition hunks. +Multiline matching scans each complete bounded input or hunk. +Reports separate findings from coverage status. +`--update-baseline` writes the baseline instead of producing a report, but rejects incomplete coverage. ### Detector and configuration trust @@ -289,7 +323,8 @@ Stdin and git-based scans stream their input in overlapping chunks. Detector rules and repository configuration are separate systems. External detector sources take precedence, and the compiled-in rules are the fallback. -Trusted scans ignore files supplied by the scanned repository but still honor explicit configuration and operator-supplied detector sources. +Trusted scans use embedded detector rules and still honor configuration explicitly selected with `--config`. +Repository baselines and inline suppressions remain policy inputs that require review. ### Core data types @@ -298,7 +333,7 @@ Trusted scans ignore files supplied by the scanned repository but still honor ex - **Severity** — `Critical`, `High`, `Medium`, `Low`. - **KeywatchConfig** — parsed `.keywatch.toml`: custom rules, per-detector overrides, and exclude patterns. - **Baseline** — versioned fingerprint entries that filter out known findings. -- **ScanMetadata** — files scanned, total lines, and skipped files, reported alongside findings. +- **ScanMetadata** — files scanned, total lines, skipped files, coverage warnings, and the effective detector fingerprint. The diagram sources are in `docs/architecture/*.d2`. After editing them, run `scripts/render-diagrams.sh render` with D2 v0.7.1, or `scripts/render-diagrams.sh check` to detect stale images. diff --git a/action.yml b/action.yml index 4fefc8e..3c0eee5 100644 --- a/action.yml +++ b/action.yml @@ -236,7 +236,7 @@ runs: if ((${#extra_args[@]})); then for arg in "${extra_args[@]}"; do case "$arg" in - --verbose|--verbose=*|-v*|--format|--format=*|-f|-f*|--output|--output=*|-o|-o*|--exit-mode|--exit-mode=*|--config|--config=*|--no-config-discovery|--no-config-discovery=*) + --verbose|--verbose=*|-v*|--format|--format=*|-f|-f*|--output|--output=*|-o|-o*|--exit-mode|--exit-mode=*|--config|--config=*|--no-config-discovery|--no-config-discovery=*|--fail-on-unscannable|--fail-on-unscannable=*) echo "ERROR: '$arg' is managed by action inputs and cannot be passed through args" >&2 exit 1 ;; @@ -255,7 +255,7 @@ runs: mkdir -p "$(dirname -- "$report_path")" fi - keywatch_args=(scan --no-config-discovery) + keywatch_args=(scan --no-config-discovery --fail-on-unscannable) keywatch_args+=(--exit-mode "$INPUT_EXIT_MODE") if [ -n "$INPUT_CONFIG" ]; then keywatch_args+=(--config "$INPUT_CONFIG") @@ -288,6 +288,10 @@ runs: fi action_status=$scan_status + if [ "$report_status" = "ok" ] && [ "$scan_status" -eq 0 ] && jq -e '.coverage == "INCOMPLETE" or ((.unscannable.count // 0) > 0)' "$report_path" >/dev/null; then + echo "ERROR: Scan coverage is incomplete" >&2 + action_status=1 + fi if [ "$report_status" != "ok" ] && [ "$scan_status" -eq 0 ]; then echo "ERROR: KeyWatch exited successfully but the JSON report is $report_status" >&2 action_status=2 diff --git a/detectors.toml b/detectors.toml index 94630a8..067a8f7 100644 --- a/detectors.toml +++ b/detectors.toml @@ -86,6 +86,11 @@ pattern = "\\beyJ[A-Za-z0-9-_=]+\\.eyJ[A-Za-z0-9-_=]+\\.?[A-Za-z0-9-_.+/=]*\\b" finding_type = "JWT Token" severity = "MEDIUM" keywords = ["eyJ"] +# Test fixtures replace the signature with a short placeholder +# (ZHVtbXlfc2lnbmF0dXJl is "dummy_signature"). A real HS256 signature is 43 +# base64url chars and an RS256 signature is longer, so a final segment of 32 +# characters or fewer is padding, not a signature. +allowlist = ["\\.[A-Za-z0-9_-]{1,32}$"] [[detectors]] name = "SSHPrivateKeyDetector" @@ -102,29 +107,35 @@ severity = "HIGH" [[detectors]] name = "PasswordDetector" -pattern = "(?i)(password|passwd|pwd)\\s*[:=]\\s*(?-i)[\"']?([^\"'\\n]+)[\"']?" +# Each value branch captures the complete value. Typed Rust assignments must +# match the literal after `=`, not the type annotation before it. +pattern = '''(?i)(?:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*(?-i)(?:"((?:\\[^\r\n]|[^"\\\r\n])*)"|'((?:\\[^\r\n]|[^'\\\r\n])*)'|([^\s"'`;,\r\n]+)(?:[ \t]*;)?)''' finding_type = "Password" severity = "HIGH" keywords = ["password", "passwd", "pwd"] # $PWD is the shell working-directory variable (docker --volume "$PWD:...", # Makefiles), never a password literal. allowlist = [ - "^\\$?PWD:", + "^\\$PWD:", # Rust expressions are plumbing, not literals: type paths (`Secret::new`), # generics (`Secret`), constructor calls (`Some(...)`) and field or # method access (`config.password.clone()`). Quoted literals and bare # alphanumeric values stay reported. - "[:=]\\s*[A-Za-z_][A-Za-z0-9_]*(?:::[A-Za-z_][A-Za-z0-9_]*)+", - "[:=]\\s*[A-Za-z_][A-Za-z0-9_]*<[^\"'\\n]*>", - "[:=]\\s*[A-Za-z_][A-Za-z0-9_]*\\([^\"'\\n]*\\)", - "[:=]\\s*[A-Za-z_][A-Za-z0-9_]*\\.[A-Za-z_][A-Za-z0-9_]*", + "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*(?:::[A-Za-z_][A-Za-z0-9_]*)+$", + "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*<[^\"'\\s]*>$", + "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*\\([^\"'\\s]*\\)$", + "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*(?:\\.[A-Za-z_][A-Za-z0-9_]*(?:\\([^\"'\\s]*\\))?)+$", # A bare CamelCase value is a Rust type (`password: String,`), matching the # GenericKeyValueDetector rule. Digits or symbols keep it reported. - "[:=]\\s*[A-Z][a-z]+(?:[A-Z][a-z]*)*\\b", - # Placeholder values in templates and .env.example files. Case-sensitive - # (placeholders are lowercase by convention) and shape-checked: a mixed - # value like "ReplaceThisRealSecret123" does not match and stays reported. - "[=:\"]?\\s*(changeme|changeit|your(-[a-z0-9]+)+|replace(-[a-z0-9]+)+|placeholder|dummy|xxxxx+)", + "[:=][ \\t]*[A-Z][a-z]+(?:[A-Z][a-z]*)*$", + # Suppress only complete, selected documentation placeholders. + # Extended values such as `dummy7!` must still report. + '''[:=][ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', + # Match empty literals completely before suppressing them. + '''[:=][ \t]*(?:""|'')$''', + # Proto field numbers and Rust parameter types do not hold literals. + "[:=][ \\t]*\\d+[ \\t]*;$", + "[:=][ \\t]*&str\\)?$", ] [[detectors]] @@ -141,6 +152,10 @@ allowlist = [ "^noreply@", "^no-reply@", "@users\\.noreply\\.github\\.com$", + # Placeholder local parts and the test.com domain, seen throughout checked-in + # fixtures. Real people at gmail.com or a company domain stay reported. + "^example@", + "@test\\.com$", ] [[detectors]] @@ -148,6 +163,10 @@ name = "IPAddressDetector" pattern = "\\b(?:25[0-5]|2[0-4]\\d|1\\d{2}|[1-9]?\\d)\\.(?:25[0-5]|2[0-4]\\d|1\\d{2}|[1-9]?\\d)\\.(?:25[0-5]|2[0-4]\\d|1\\d{2}|[1-9]?\\d)\\.(?:25[0-5]|2[0-4]\\d|1\\d{2}|[1-9]?\\d)\\b" finding_type = "IP Address" severity = "LOW" +# Loopback and the wildcard bind address are configuration, not credentials. +# Other private and public addresses (10.x, 192.168.x, arbitrary postman +# payloads) stay reported. +allowlist = ["^127\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}$", "^0\\.0\\.0\\.0$"] [[detectors]] name = "PhoneNumberDetector" @@ -160,19 +179,6 @@ severity = "LOW" # allowlist matches the whole number rather than just the exchange. allowlist = ["^(?:\\+1[-.\\s]?)?\\(?\\d{3}\\)?[-.\\s]*555-01\\d{2}$"] -[[detectors]] -name = "CreditCardDetector" -# Issuer prefixes (Visa/Mastercard/Amex/Discover/UnionPay) in 4-digit groups, -# including Discover's 644-649 and 65 ranges. The old "any 13-16 digits with -# arbitrary spaces" shape matched commit hashes, timestamps and ids, and even -# spanned the gap between two unrelated numbers ("index aabbcc0..1111111 -# 100644"). Luhn on its own does not fix that: roughly one in ten random -# digit runs satisfies it. -pattern = "\\b(?:4\\d{3}|5[1-5]\\d{2}|2[2-7]\\d{2}|6011|64[4-9]\\d{2}|65\\d{2}|3[47]\\d{2})[ -]?\\d{4}[ -]?\\d{4}[ -]?\\d{1,4}\\b" -validate = "luhn" -finding_type = "Credit Card Number" -severity = "HIGH" - [[detectors]] name = "SSNDetector" pattern = "\\b\\d{3}-\\d{2}-\\d{4}\\b" @@ -185,13 +191,15 @@ keywords = ["ssn", "social security", "social_security", "social-security"] [[detectors]] name = "GenericKeyValueDetector" -pattern = "(?i)(api_?key|secret|password|token|access_?key|security_?key|_key|credential|auth)\\s*[:=]\\s*(?-i)[\"']?([a-zA-Z0-9\\-_=]{10,})[\"']?" +# Quoted branches require a closing quote. Include the full token alphabet +# so `/` and `+` do not truncate the value used for baseline identity. +pattern = '''(?i)(?:api_?key|secret|password|token|access_?key|security_?key|_key|credential|auth)["']?[ \t]*[:=][ \t]*(?-i)(?:"([-A-Za-z0-9_/+=]{10,})"|'([-A-Za-z0-9_/+=]{10,})'|([-A-Za-z0-9_/+=]{10,}))''' finding_type = "Generic Key/Secret" severity = "HIGH" # Every pattern alternative needs a covering keyword: without "auth" and # "_key" here, `auth = ...` and `encryption_key = ...` lines never reached # the regex at all. -keywords = ["api_key", "secret", "password", "token", "access_key", "credential", "auth", "_key"] +keywords = ["api_key", "apikey", "secret", "password", "token", "access_key", "accesskey", "securitykey", "credential", "auth", "_key"] entropy = 2.5 # An unquoted, all-lowercase snake_case value is a variable reference, not a # credential: `let payment_method_token = card_token;` reads identically to @@ -203,17 +211,24 @@ entropy = 2.5 allowlist = [ "[:=]\\s*[a-z]+(?:_[a-z]+)+$", "[:=]\\s*[A-Z][a-z]+(?:[A-Z][a-z]*)*$", - # Placeholder values in templates and .env.example files (case-sensitive: - # mixed-case shapes are plausibly real weak credentials). - "[=:]\\s*[\"']?(changeme|changeit|your(-[a-z0-9]+)+|replace(-[a-z0-9]+)+|placeholder|dummy|xxxxx+)", - # The AWS documentation example secret; its slash stops this pattern's - # value class, so only the leading fragment is visible in the match. - "wJalrXUtnFEMI", + # Suppress only complete, selected documentation placeholders. + '''[:=][ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', + # Match the complete AWS documentation example, not a shared prefix. + '''[:=][ \t]*(?:"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"|'wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY'|wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY)$''', + # Unquoted code references remain heuristics. Quoted snake-case and + # kebab-case values can hold credentials and must not borrow these rules. + "[:=]\\s*_*[a-z][a-z0-9]*(?:_[a-z][a-z0-9]*)+$", + "[:=]\\s*[a-z]{10,}$", + "[:=]\\s*[a-z][a-z]*(?:[A-Z][a-z]*)+$", + "[:=]\\s*[A-Z][A-Z0-9]*(?:_+[A-Z0-9]+)+$", ] [[detectors]] name = "CertificateDetector" -pattern = "-----BEGIN CERTIFICATE-----" +# A PEM block in source starts its line or directly follows a string quote; +# a mention inside prose (a doc comment that describes the header) does not +# start a certificate. +pattern = "(?:^\\s*|[\"'\x60])-----BEGIN CERTIFICATE-----" finding_type = "Certificate" severity = "LOW" keywords = ["BEGIN CERTIFICATE"] @@ -239,7 +254,18 @@ finding_type = "Base64 Encoded String" severity = "LOW" entropy = 4.2 # The AWS documentation example secret pairs with the example access key. -allowlist = ["^wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY$"] +# Beyond that: CamelCase prose identifiers (PaymentMethodServiceEligibility- +# Request) are word runs, not random data; eyJ/ZXlK prefixes mark base64 +# encoded JSON, a signature or fixture shape rather than a credential; a +# dotted or slashed alphanumeric path is a language/type descriptor +# (com/michaelkeevildown/..., Ljava/security/MessageDigest). Random base64 +# with a plus sign, padding or mixed word-less shape stays reported. +allowlist = [ + "^wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY$", + "^[A-Z][a-z]+(?:[A-Z][a-z]*)+$", + "^(?:eyJ|ZXlK)[A-Za-z0-9+/]*={0,2}$", + "^L?[A-Za-z][A-Za-z0-9]*(?:/[A-Za-z0-9]+){2,};?$", +] [[detectors]] name = "HighEntropyDetector" @@ -277,7 +303,7 @@ name = "PaymentGatewayKeyDetector" pattern = "\\b(?:api_?key|secret)_(?:test|live)_[0-9a-zA-Z]{10,}\\b" finding_type = "Generic Payment Gateway Key" severity = "HIGH" -keywords = ["api_key_test", "api_key_live", "secret_test", "secret_live"] +keywords = ["api_key_test", "api_key_live", "apikey_test", "apikey_live", "secret_test", "secret_live"] [[detectors]] name = "RandomString" @@ -292,7 +318,19 @@ allowlist = [ "^\"[a-z]+(?:_[a-z]+)+\"$", # npm/shield-style checksum strings carry their algorithm prefix inside # the quotes, so the prefix is visible in the match. - "^\"sha(256|384|512)-", + "(?i)^\"sha-?(?:256|384|512)(?:-|=)", + # Quoted identifiers, test names and enum values: kebab-case topics, + # camelCase and PascalCase names, SCREAMING_SNAKE codes, single lowercase + # words and UUIDs. Generated secrets mix digits, symbols or case in shapes + # these rules do not cover. + "^\"[A-Za-z]+(?:-[A-Za-z0-9]+)+\"$", + "^\"[A-Za-z][a-z]*(?:[A-Z][a-z]*)+\"$", + "^\"[A-Z][A-Z0-9]*(?:_+[A-Z0-9]+)+\"$", + "^\"[a-z]+\"$", + "^\"[0-9a-z]{8}-(?:[0-9a-z]{4}-){3}[0-9a-z]{12}\"$", + # Quoted hex digests and base64-encoded JSON payloads from fixtures. + "^\"[a-f0-9]{40,}\"$", + "^\"(?:eyJ|ZXlK)[A-Za-z0-9+/]*={0,2}\"$", ] @@ -902,7 +940,7 @@ name = "ServiceAccountKeyDetector" pattern = "(?i)service[_\\-]?account[_\\-]?key[\\s'\":=]+[\"']?([a-zA-Z0-9\\-_]{20,})[\"']?" finding_type = "Service Account Key" severity = "HIGH" -keywords = ["service_account"] +keywords = ["service_account", "service-account", "serviceaccount"] [[detectors]] name = "MasterAPIKeyDetector" diff --git a/docs/plans/2026-10-06-hardening.md b/docs/plans/2026-10-06-hardening.md new file mode 100644 index 0000000..c10b51a --- /dev/null +++ b/docs/plans/2026-10-06-hardening.md @@ -0,0 +1,165 @@ +# KeyWatch hardening plan + +## Goal and constraints + +Close the verified detection and scan-coverage gaps before making production-readiness claims. +Keep the CLI and module architecture. +Preserve the existing false-positive changes, except for exemptions that hide credentials. +Do not add project-specific fixture exemptions or dependencies. +Do not commit, publish, or replace the installed binary without permission. + +## Stage 1: Complete credential matches + +Status: Implemented and tested. +Repository CI remains blocked by baseline drift. + +1. Add failing scanner tests for complete typed password values and token suffixes. +2. Fix `PasswordDetector` and `GenericKeyValueDetector` patterns and keyword coverage. +3. Support quoted JSON keys and multiple credential fields on one line. +4. Remove quoted snake-case and kebab-case exemptions. +5. Restrict placeholder exemptions to complete, selected documentation values. +6. Test baseline suppression through the executable in filesystem and staged modes. +7. Update conflicting test expectations and baseline compatibility documentation. +8. Run formatting, lint, debug tests, release tests, and the Rust 1.85 build check. +9. Inspect baseline drift without automatically accepting findings. + +### Files + +- `detectors.toml` +- `tests/scanner_tests.rs` +- `tests/baseline_tests.rs` +- `tests/detector_tests.rs` +- `README.md` +- `CHANGELOG.md` + +### Approved acceptance tests + +1. Typed Rust passwords and tokens containing `/` or `+` include the complete value. +2. Changing a password or token suffix produces a finding despite an existing baseline. +3. Moving an unchanged credential to another line remains suppressed by its baseline. +4. JSON credentials and bare `apikey`, `accesskey`, and `securitykey` names report. +5. Extended placeholders, passphrases, and quoted snake-case or kebab-case credentials report. +6. Exact selected placeholders and unquoted code references remain suppressed. + +Keep detector names, severity levels, entropy thresholds, custom-rule behavior, and baseline format unchanged. +Corrected match text can change fingerprints. +Do not use legacy prefix fingerprints to suppress complete matches. +Review affected findings before updating a baseline. +This stage does not provide a complete Rust or JSON parser. + +### Validation results + +- `cargo test --locked --all-features --all-targets` exits with status 0. + All 285 debug tests pass. +- `cargo test --locked --release --all-features --all-targets` exits with status 0. + All 285 release tests pass. +- `cargo clippy --locked --all-features --all-targets -- -D warnings` exits with status 0. +- `cargo fmt --all --check` and `rustup run nightly cargo fmt --all --check` exit with status 0. +- `rustup run 1.85.0 cargo check --locked --offline --all-features --all-targets` exits with status 0. +- `git diff --check` exits with status 0. + +Baseline inspection uses temporary copies and the repository's configured exclusions. +The scan reports 71 unbaselined findings with 62 distinct baseline identities. +These findings occur in test and workflow files. +The temporary merge also refreshes 49 existing locations. +The temporary prune removes 38 stale identities. +The repository baseline remains unchanged by this stage. +Review the findings before accepting any baseline changes. +The repository self-scan and baseline drift checks do not pass until this work is resolved. + +## Stage 2: Git coverage and scan outcomes + +Status: Implemented and tested through executable Git and report regressions. + +Define predictable Git prefixes and hunk context. +Disclose shallow history without fetching automatically. +Address historical binary blobs and incomplete-scan exit behavior. +Separate detection results from coverage status. +Test exclusion paths, line attribution, shallow clones, and binary attributes through executable scans. + +## Stage 3: Resource limits + +Status: Implemented and tested. +Synthetic workload measurements are recorded in `docs/security-and-validation.md`. +Production-environment measurements remain a deployment requirement. +A whole-workspace measurement reaches the finding budget and exits with status 2 before producing a report. +Do not claim that this source tree can complete every large-repository scan. + +Use checked size conversion and consistent byte limits across scan modes. +Bound long lines, chunks, decoded content, and retained findings. +Measure memory and execution time on representative repositories. +Do not claim large-project readiness without those measurements. + +## Stage 4: Storage and configuration safety + +Status: Implemented and tested. + +Protect report and baseline writes against symlinks and partial writes. +Reject unknown override names. +Disclose skipped paths and define lockfile policy. +Document offline guessing risks for low-entropy baseline values. + +## Stage 5: Accuracy and deployment evidence + +Status: Implemented and tested within the documented scope. +The labeled corpus, benchmark script, report provenance, hook checks, and Action checks are included. +Published-binary authentication and untested platforms remain operational requirements. + +Expand the labeled positive and negative corpus with paired examples. +Measure precision and recall without claiming universal detection. +Review suppression ownership, release provenance, and supported platforms. +Document unresolved coverage limits and the need for additional security controls. + +You approved implementation of stages 2 through 5, with review after completion. +Changes to PII defaults, provider verification, release automation, or installation are not approved by this plan. + +## Final review points + +Review `docs/security-and-validation.md` for coverage, limits, trust boundaries, and measured accuracy. +The installed binary and existing installed hooks remain unchanged. +No commits, publication, dependency changes, or automatic baseline acceptance occur during this work. +The source changes preserve baseline format `1.0` but change some finding identities. +The JSON status can be `INCOMPLETE`, and explicit coverage failure overrides every severity exit mode. +Report and baseline writes require safe, operator-controlled directories. +Configuration rejects unknown overrides, duplicate detector names, and non-finite entropy thresholds. + +### Final validation evidence + +The following checks exit with status 0: + +- `cargo test --locked --all-features --all-targets --quiet`: All 307 debug tests pass. +- `cargo test --locked --release --all-features --all-targets --quiet`: All 307 release tests pass. +- `cargo clippy --locked --all-features --all-targets -- -D warnings`. +- `rustup run nightly cargo fmt --all --check`. +- `rustup run 1.85.0 cargo check --locked --offline --all-features --all-targets`. +- `git diff --check`. +- `uv run --no-project --no-config python -B scripts/action_validation/validate.py`. +- `cargo audit --no-fetch --json`: The cached database reports zero vulnerabilities and no warnings. + +The staged and history regressions include deleted binary paths with control characters. +Configuration tests reject non-finite thresholds before they can affect detection or report fingerprints. +The two documented false-positive corpus residuals remain unchanged. +The 34-case labeled corpus passes, but does not establish real-world precision or recall. + +### Baseline review before draft PR preparation + +The final baseline inspection uses temporary copies, trusted rules, and the repository's configured exclusions. +The self-scan exits with status 1, reports complete coverage, and finds 103 unbaselined occurrences. +These occurrences have 87 distinct baseline identities in tests, detector definitions, workflows, and the benchmark script. +The temporary merge changes the entry count from 225 to 312 and refreshes 63 existing locations. +The temporary prune removes 38 stale identities and leaves 274 entries. +The repository baseline remains unchanged by the hardening stages. +Review these findings before accepting them. +The repository self-scan and baseline drift checks remain blocked until that review is complete. + +### Draft PR preparation + +You approve review and baselining of confirmed synthetic fixtures before creating the draft PR. +The reviewed fixture update changes the baseline from 225 to 309 entries. +It includes test credentials, benchmark credentials, and the CI test token. +The required installed-scanner check of staged files exits with status 0. +The source-built repository self-scan still exits with status 1 and reports complete coverage. +Three findings remain: one workflow expression and two detector-definition matches. +Those findings are not accepted by the fixture baseline update. +CI self-scan and baseline drift checks remain blocked for your review. +The whole-workspace resource-budget limitation also remains unresolved. diff --git a/docs/security-and-validation.md b/docs/security-and-validation.md new file mode 100644 index 0000000..7f255fc --- /dev/null +++ b/docs/security-and-validation.md @@ -0,0 +1,148 @@ +# Security and validation + +## Intended use + +Use KeyWatch as an additional secret-detection control, not as proof that a repository contains no credentials. +Regex rules, entropy checks, and identifier exemptions can produce false positives and false negatives. +Provider verification is not implemented, and the scanner does not determine whether a credential is active. +The installed binary, published Action binary, container image, and hook templates can differ from this source tree. +Check their versions and release notes before you deploy them. + +## Trust boundaries + +Repository content, Git configuration, filenames, diffs, and object data are untrusted inputs. +Operator-selected configuration controls rules, exclusions, and severity. +Baseline entries and `keywatch:ignore` markers suppress findings. +Require review for these suppressions and for changes to scan policy. +Hooks ignore repository detector configuration but still trust repository baselines and inline ignore markers. +This does not create a tamper-proof gate against a malicious contributor. + +For CI gates, use trusted detector rules, an operator-owned configuration, and reviewed suppressions. +Use `--fail-on-unscannable` to reject incomplete coverage. +Supply a trusted baseline explicitly when the repository baseline is not an approved policy source. +Protect configuration and baseline files through repository access controls and code review. +Do not expose privileged CI credentials to code from untrusted pull requests. + +## Findings, coverage, and limits + +Reports distinguish detected findings from scan coverage. +`INCOMPLETE` means some requested content or history could not be scanned. +It does not mean that the unavailable content contains a secret. +`COMPLETE` means the scanner completes its selected scope within its implemented capabilities. +It does not establish universal credential detection or include deliberately excluded files. +Review exclusions and coverage warnings with the findings. + +The scanner enforces these limits: + +| Resource | Limit | +| --- | --- | +| Raw logical input or Git addition hunk | 16 MiB | +| Line | 1 MiB | +| Path records | 100,000 | +| Findings per input and aggregate checks | 10,000 | +| Retained finding text and metadata per input and aggregate checks | 32 MiB | +| Concurrent filesystem batch | Four files | +| Retained Git standard error | 64 KiB | + +`--max-file-size` lowers the raw input ceiling in filesystem, stdin, staged, and history modes. +Larger values do not raise the hard ceiling. +Limits produce incomplete coverage or a runtime error. +Partial input is not accepted as a clean scan or used to update a baseline. +Batch buffers, decoded content, detector definitions, paths, and reports also consume memory. +The retained-text limit is not a process memory limit. +Execution has no built-in wall-clock deadline; enforce a CI job timeout. + +Multiline matching uses bounded complete inputs rather than fixed overlap windows. +Git text scans still inspect additions, not complete snapshots of every historical text file. +Credentials split across separate hunks or commits can therefore be missed. +Run a filesystem scan of the checked-out tree as well as the required Git scan. +Text rendered as binary uses actual Git blobs, including historical and deleted versions. +True binary content remains unscannable. +Shallow history remains incomplete until you fetch full history outside the scanner. +Archives, arbitrary obfuscation, recursive encoding, and complete language parsing are not supported. + +## Secret storage and output + +Reports redact matched content unless you explicitly use `--show-secrets`. +Do not upload unredacted reports to shared CI artifacts or logs. +The process still holds matched content in memory while scanning. +Restrict access to the scanner process and its environment. + +Baseline hashes use domain-separated SHA-256 without a salt or keyed secret. +An attacker who obtains a baseline can test guesses for weak passwords offline. +Review this disclosure risk before publishing a baseline. +Rotate leaked credentials; adding a baseline entry does not remove a leak. + +Report and baseline writes use private same-directory temporary files and atomic replacement. +They reject destination symlinks and unsafe immediate parent directories. +Control the parent directory and all ancestors. +This is not a defense against an attacker who can replace those ancestors. +The implementation does not promise directory-sync durability after power loss. + +## Accuracy evidence + +`tests/accuracy_corpus.toml` contains 34 synthetic cases: 20 credentials and 14 noncredentials. +The scanner reports the expected detector for every positive case and no findings for every negative case. +The measured case-level results are 20 true positives, 14 true negatives, zero false positives, and zero false negatives. +Precision and recall are both 1.000 on this corpus only. +These results do not measure real-world accuracy or prove that other provider formats are covered. +The separate false-positive corpus retains its two documented residual findings. +Expand the corpus when real reports reveal another failure class. +Add both the unwanted finding and a similar credential that must remain detectable. + +## Synthetic workload measurements + +Run the benchmark with an optimized binary: + +```sh +cargo build --locked --release +uv run --no-project --no-config python -B scripts/benchmark.py target/release/key-watch +``` + +The script removes its generated files after execution. +The measured run uses macOS and the release build of this source tree. +The ordinary workload contains 2,000 files, 400,000 lines, and 9,600,000 bytes. + +| Workload | Seconds | Peak resident bytes | Exit | Coverage | +| --- | ---: | ---: | ---: | --- | +| Many small files | 0.533 | 49,119,232 | 0 | Complete | +| Oversized line | 0.244 | 34,996,224 | 1 | Incomplete | +| Finding budget exceeded | 0.136 | 32,948,224 | 1 | Incomplete | +| 100,000 rejected multiline matches | 0.210 | 31,784,960 | 0 | Complete | + +These are single-run measurements, not throughput guarantees or comparisons with other scanners. +No production CI environment, Windows runtime, or container runtime benchmark is measured here. +Measure your repository and CI environment before you rely on its scan duration or memory use. + +### Existing repository measurement + +A local Rust workspace scan excludes `target/**` and `**/node_modules/**` and disables configuration and baseline discovery. +It reaches the finding or retained-text budget after 20.328 seconds, with 126,812,160 peak resident bytes. +The command exits with status 2 and does not produce a report. +`--exit-mode always` does not suppress this runtime failure. +This measurement does not establish complete coverage of that workspace. +A source partition completes 340 files and 355,501 lines in 1.992 seconds, with 81,838,080 peak resident bytes. +Its report has complete coverage and 102 findings; the finding status is `FAIL`. +The command exits with status 0 because it uses `--exit-mode always`. +These findings are not labeled, so this measurement does not establish accuracy. +Baseline filtering occurs after scanning and does not bypass the raw-finding budget. +Use reviewed scan partitions and measure them before deployment. +Do not treat a budget failure as a clean result or increase limits without measuring the effect. + +## Deployment and release evidence + +JSON and SARIF reports contain a fingerprint of the effective detector definitions. +JSON also identifies the scanner version; SARIF identifies it in the tool driver. +The fingerprint covers patterns, keywords, allowlists, validators, entropy thresholds, and severity. +It is not a signature, binary attestation, or complete record of exclusions and baseline policy. +Record the scanned commit, command, configuration, baseline, and binary digest with your CI evidence. + +Download checksums from the same release detect corruption, not compromise of the release publisher. +Pin Action references and container digests when you require reproducible inputs. +The Action supports Linux x64 and macOS, not Windows or Linux ARM64. +Reinstall hooks after changing the binary and hook templates. +Local tests do not validate an unpublished binary through the public release download path. +Release provenance, independently authenticated binaries, and platform deployment tests remain separate operational requirements. + +Do not claim that this project is an independently audited or complete production security gate. +Use additional secret-management controls, credential rotation, and review processes. diff --git a/scripts/action_validation/keywatch_action_scenarios.py b/scripts/action_validation/keywatch_action_scenarios.py index c02fe4c..9f2caae 100644 --- a/scripts/action_validation/keywatch_action_scenarios.py +++ b/scripts/action_validation/keywatch_action_scenarios.py @@ -140,6 +140,8 @@ def write_keywatch_stub(bin_dir: Path) -> None: " if [ \"$1\" = \"--output\" ]; then shift; out=\"$1\"; fi\n" " shift || true\ndone\ncase \"$KEYWATCH_REPORT_MODE\" in\n" " valid) printf '%s\\n' '{\"findings\":[{},{}]}' > \"$out\" ;;\n" + " incomplete) printf '%s\\n' '{\"findings\":[],\"coverage\":\"INCOMPLETE\"}' > \"$out\" ;;\n" + " legacy-incomplete) printf '%s\\n' '{\"findings\":[],\"unscannable\":{\"count\":1}}' > \"$out\" ;;\n" " malformed) printf '%s\\n' 'not-json' > \"$out\" ;;\n missing) ;;\n *) exit 99 ;;\nesac\n" "exit \"$KEYWATCH_STUB_EXIT\"\n", ) @@ -155,7 +157,7 @@ def run_scan_scenarios(scan_block: str) -> None: "valid", 0, ("exit_code=0", "findings_count=2"), - ("--no-config-discovery", "scan/match.txt"), + ("--no-config-discovery", "--fail-on-unscannable", "scan/match.txt"), ("scan/*.txt", "--verbose"), ), ScanScenario( @@ -176,6 +178,9 @@ def run_scan_scenarios(scan_block: str) -> None: ScanScenario("verbose-compact-short-rejected", ".", "-vv", 0, "valid", 1, expected_stderr=("managed by action inputs",)), ScanScenario("verbose-mode-long-allowed", ".", "--verbose-mode", 0, "valid", 0, expected_capture=("--verbose-mode",)), ScanScenario("scanner-nonzero-propagates", ".", "", 1, "valid", 1, ("exit_code=1", "findings_count=2")), + ScanScenario("incomplete-coverage-blocks", ".", "", 0, "incomplete", 1, ("exit_code=1", "findings_count=0")), + ScanScenario("legacy-incomplete-blocks", ".", "", 0, "legacy-incomplete", 1, ("exit_code=1", "findings_count=0")), + ScanScenario("coverage-override-rejected", ".", "--fail-on-unscannable=false", 0, "valid", 1, expected_stderr=("managed by action inputs",)), ScanScenario( "missing-report-zero-fails", ".", diff --git a/scripts/action_validation/validate.py b/scripts/action_validation/validate.py index 5465287..12bf5bb 100644 --- a/scripts/action_validation/validate.py +++ b/scripts/action_validation/validate.py @@ -121,7 +121,7 @@ def main() -> int: require(fragment not in shell, f"forbidden shell fragment remains: {fragment}") require("${{ inputs." not in shell, "inputs must be routed through step env, not run blocks") - require("keywatch_args=(scan --no-config-discovery)" in shell, "Action scans must disable untrusted config discovery") + require("keywatch_args=(scan --no-config-discovery --fail-on-unscannable)" in shell, "Action scans must disable untrusted config discovery and fail on incomplete coverage") require("keywatch_args+=(--config \"$INPUT_CONFIG\")" in shell, "explicit trusted config input must be supported") require("read -r -a paths" in shell, "paths input must be parsed without shell evaluation") require("compgen -G \"$path_token\"" in shell, "path globs must expand without shell evaluation") diff --git a/scripts/benchmark.py b/scripts/benchmark.py new file mode 100644 index 0000000..8d05522 --- /dev/null +++ b/scripts/benchmark.py @@ -0,0 +1,69 @@ +"""Measure synthetic scan workloads without keeping generated fixtures.""" + +import argparse +import json +import re +import subprocess +import sys +import tempfile +import time +from pathlib import Path + + +def measure(binary: Path, directory: Path, name: str, args: list[str]) -> dict: + command = [str(binary), "scan", "--no-config-discovery", "--no-baseline-discovery", + "--verbose", "--fail-on-unscannable", *args] + if sys.platform == "darwin": + command = ["/usr/bin/time", "-l", *command] + elif sys.platform.startswith("linux"): + command = ["/usr/bin/time", "-v", *command] + start = time.perf_counter() + result = subprocess.run(command, cwd=directory, capture_output=True, text=True, check=False) + elapsed = time.perf_counter() - start + report = json.loads(result.stdout) + peak = re.search(r"(\d+)\s+maximum resident set size", result.stderr) + linux_peak = re.search(r"Maximum resident set size \(kbytes\): (\d+)", result.stderr) + return { + "scenario": name, + "seconds": round(elapsed, 3), + "peak_rss_bytes": int(peak[1]) if peak else int(linux_peak[1]) * 1024 if linux_peak else None, + "exit_code": result.returncode, + "files_scanned": report["files_scanned"], + "coverage": report["coverage"], + "findings": len(report["findings"]), + } + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("binary", type=Path) + parser.add_argument("--files", type=int, default=2000) + parser.add_argument("--lines", type=int, default=200) + args = parser.parse_args() + if args.files < 1 or args.lines < 1: + parser.error("File and line counts must be positive") + binary = args.binary.resolve(strict=True) + results = [] + with tempfile.TemporaryDirectory(prefix="keywatch-benchmark-") as temporary: + directory = Path(temporary) + tree = directory / "tree" + tree.mkdir() + content = "// ordinary fixture line\n" * args.lines + for index in range(args.files): + (tree / f"fixture-{index}.rs").write_text(content, encoding="utf-8") + results.append(measure(binary, directory, "many-small-files", ["tree"])) + (directory / "long-line.txt").write_text("x" * (1024 * 1024 + 1), encoding="utf-8") + results.append(measure(binary, directory, "oversized-line", ["long-line.txt"])) + (directory / "many-findings.txt").write_text("AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n" * 10001, encoding="utf-8") + results.append(measure(binary, directory, "finding-budget", ["many-findings.txt"])) + (directory / "rejected.txt").write_text("ordinary\n" * 100000, encoding="utf-8") + (directory / "rule.toml").write_text("[[rules]]\nname = 'RejectedFixture'\nfinding_type = 'Fixture'\npattern = '\\n'\nallowlist = ['\\n']\n", encoding="utf-8") + results.append(measure(binary, directory, "rejected-multiline", ["rejected.txt", "--config", "rule.toml"])) + print(json.dumps({"platform": sys.platform, "files": args.files, "lines_per_file": args.lines, "results": results}, indent=2)) + expected = [(0, "COMPLETE"), (1, "INCOMPLETE"), (1, "INCOMPLETE"), (0, "COMPLETE")] + if any((row["exit_code"], row["coverage"]) != expectation for row, expectation in zip(results, expected)): + raise SystemExit("A workload did not meet its coverage contract") + + +if __name__ == "__main__": + main() diff --git a/src/baseline.rs b/src/baseline.rs index 66fec3d..db84594 100644 --- a/src/baseline.rs +++ b/src/baseline.rs @@ -216,9 +216,11 @@ impl Baseline { .map_err(|source| BaselineError::Serialize { source })?; json.push('\n'); - fs::write(path, json).map_err(|source| BaselineError::Write { - path: path.to_path_buf(), - source, + crate::utils::atomic_write(path, json.as_bytes()).map_err(|source| { + BaselineError::Write { + path: path.to_path_buf(), + source, + } })?; Ok(()) diff --git a/src/cli.rs b/src/cli.rs index 957f3ef..0a9beb3 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -142,16 +142,18 @@ pub struct ScanArgs { /// --config still loads) #[arg(long, default_value_t = false)] pub no_repo_config: bool, - /// Exit 1 when any scanned file could not be read (Strict exit mode only; - /// not applied by --update-baseline) + /// Exit 1 when scan coverage is incomplete, in every exit mode #[arg(long, default_value_t = false)] pub fail_on_unscannable: bool, /// Output format for the report (json or sarif) #[arg(long, value_enum, default_value_t = OutputFormat::Json)] pub format: OutputFormat, - /// Skip files larger than this many megabytes and report them as - /// unscannable (default: no limit) + /// Include lockfiles that the default noise policy excludes + #[arg(long, default_value_t = false)] + pub scan_lockfiles: bool, + + /// Lower the 16 MiB input ceiling and report larger inputs as unscannable #[arg(long, value_name = "MB", value_parser = clap::value_parser!(u64).range(1..))] pub max_file_size: Option, } diff --git a/src/config.rs b/src/config.rs index 157d5f5..ad25978 100644 --- a/src/config.rs +++ b/src/config.rs @@ -41,6 +41,10 @@ pub enum ConfigError { Invalid { source: toml::de::Error }, #[error("custom rule '{name}': {source}")] CustomRule { name: String, source: DetectorError }, + #[error("Unknown detector override '{name}'")] + UnknownOverride { name: String }, + #[error("Duplicate detector name '{name}'")] + DuplicateDetector { name: String }, } impl KeywatchConfig { @@ -83,6 +87,27 @@ impl KeywatchConfig { } } + let mut names: std::collections::HashSet<&str> = detectors + .iter() + .map(|detector| detector.name.as_str()) + .collect(); + for detector in &staged_detectors { + if !names.insert(&detector.name) { + return Err(ConfigError::DuplicateDetector { + name: detector.name.clone(), + }); + } + } + if let Some(overrides) = &self.overrides { + let mut override_names: Vec<_> = overrides.keys().collect(); + override_names.sort(); + for name in override_names { + if !names.contains(name.as_str()) { + return Err(ConfigError::UnknownOverride { name: name.clone() }); + } + } + } + // Phase 2 — commit: all validation passed, mutate. detectors.extend(staged_detectors); @@ -183,7 +208,8 @@ pub struct CustomRule { pub keywords: Option>, /// Minimum Shannon entropy a match must reach to be reported. pub entropy: Option, - /// Extra structural check applied to each match, e.g. `validate = "luhn"`. + /// Extra structural check applied to each match, e.g. + /// `validate = "verhoeff"`. pub validate: Option, /// Accepted for configuration compatibility and otherwise ignored; earlier /// releases parsed the key but never surfaced it. diff --git a/src/config/tests/application.rs b/src/config/tests/application.rs index 3e93768..96ae897 100644 --- a/src/config/tests/application.rs +++ b/src/config/tests/application.rs @@ -273,11 +273,11 @@ allowlist = ["ABCD-EFGH-JKLM"] entropy = 0.0 [[rules]] -name = "Luhnish" -pattern = '\b[0-9]{13,16}\b' -finding_type = "Cardish" +name = "Verhoeffish" +pattern = '\b[0-9]{12}\b' +finding_type = "Idish" severity = "HIGH" -validate = "luhn" +validate = "verhoeff" "#; let config: KeywatchConfig = toml::from_str(toml).expect("parse config"); let mut detectors = Vec::new(); @@ -305,16 +305,16 @@ validate = "luhn" "non-allowlisted code must be reported" ); - let luhn = detectors + let verhoeff = detectors .iter() - .find(|d| d.name == "Luhnish") - .expect("Luhnish should exist"); + .find(|d| d.name == "Verhoeffish") + .expect("Verhoeffish should exist"); assert!( - !reported("card = 4111111111111112", luhn), - "Luhn-invalid match must be rejected" + !reported("id = 123456789012", verhoeff), + "Verhoeff-invalid match must be rejected" ); assert!( - reported("card = 4111111111111111", luhn), - "Luhn-valid match must be reported" + reported("id = 100000000004", verhoeff), + "Verhoeff-valid match must be reported" ); } diff --git a/src/detector.rs b/src/detector.rs index 151fac9..7656dac 100644 --- a/src/detector.rs +++ b/src/detector.rs @@ -37,16 +37,14 @@ pub enum DetectorError { detector: String, source: ParseValidatorError, }, + #[error("entropy threshold in detector '{detector}' must be finite")] + InvalidEntropy { detector: String }, } /// Extra structural check a detector can require of its matches, for /// patterns whose shape alone is too permissive. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ContentValidator { - /// Payment card numbers carry a Luhn check digit. Without it, a 13-16 - /// digit pattern matches every commit hash fragment, timestamp and - /// numeric id in a codebase. - Luhn, /// Aadhaar numbers carry a Verhoeff check digit. Without it, every /// 12-digit run (the tail of a UUID, a numeric id) reports HIGH. Verhoeff, @@ -65,7 +63,6 @@ impl FromStr for ContentValidator { fn from_str(value: &str) -> Result { match value.trim().to_lowercase().as_str() { - "luhn" => Ok(Self::Luhn), "verhoeff" => Ok(Self::Verhoeff), "supabase-service-role" => Ok(Self::SupabaseServiceRole), "github-token-checksum" => Ok(Self::GithubTokenChecksum), @@ -261,25 +258,6 @@ mod github_checksum_tests { } } -/// Luhn checksum, ignoring embedded separators. -fn passes_luhn(matched: &str) -> bool { - let digits: Vec = matched.chars().filter_map(|c| c.to_digit(10)).collect(); - if !(13..=19).contains(&digits.len()) { - return false; - } - let sum: u32 = digits - .iter() - .rev() - .enumerate() - .map(|(index, digit)| match index % 2 { - 1 if *digit > 4 => digit * 2 - 9, - 1 => digit * 2, - _ => *digit, - }) - .sum(); - sum % 10 == 0 -} - pub struct Detector { pub name: String, pub regex: Regex, @@ -312,6 +290,12 @@ impl Detector { source, })?; + if entropy_threshold.is_some_and(|threshold| !threshold.is_finite()) { + return Err(DetectorError::InvalidEntropy { + detector: name.to_string(), + }); + } + let mut compiled_allowlist = Vec::new(); for pattern in allowlist { let compiled = @@ -347,7 +331,6 @@ impl Detector { /// Whether a match satisfies the detector's structural validator. pub fn passes_validation(&self, matched: &str) -> bool { match self.validator { - Some(ContentValidator::Luhn) => passes_luhn(matched), Some(ContentValidator::Verhoeff) => passes_verhoeff(matched), Some(ContentValidator::SupabaseServiceRole) => { Self::passes_supabase_service_role(matched) @@ -751,25 +734,13 @@ fn initialize_detectors_from_config( #[cfg(test)] mod accept_unit_tests { - use super::{Detector, passes_luhn, shannon_entropy}; + use super::{Detector, shannon_entropy}; fn detector(pattern: &str, keywords: &[&str]) -> Detector { let keywords: Vec = keywords.iter().map(|k| k.to_string()).collect(); Detector::new("T", pattern, "T", "LOW", &[], &keywords, None).expect("valid detector") } - #[test] - fn luhn_accepts_known_cards_and_separators() { - assert!(passes_luhn("4111111111111111")); - assert!(passes_luhn("4111-1111-1111-1111")); - assert!(passes_luhn("4111 1111 1111 1111")); - assert!(!passes_luhn("4111111111111112")); - assert!(!passes_luhn("")); - assert!(!passes_luhn("no digits")); - // Below the 13-digit floor. - assert!(!passes_luhn("41111111111")); - } - #[test] fn shannon_entropy_matches_information_theory() { assert_eq!(shannon_entropy(""), 0.0); diff --git a/src/lib.rs b/src/lib.rs index 8c2fd62..129d84f 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -84,6 +84,9 @@ fn run_scan_command(args: &ScanArgs) -> Result { } if args.update_baseline { + if !scan_metadata.is_complete() { + return Err(RunCliError::IncompleteBaselineScan); + } update_baseline(&args, &findings, &mut loaded_baseline, prune, &anchor)?; return Ok(0); } @@ -222,13 +225,13 @@ fn emit_scan_result( let suppressed = scan_metadata.suppressed_by_baseline; let severity_counts = report::get_severity_counts(&findings); let mut exit_code = calculate_exit_code(&findings, &args.exit_mode); - let unscannable_failure = args.fail_on_unscannable - && matches!(args.exit_mode, ExitMode::Strict) - && !scan_metadata.unscannable_files.is_empty(); + let incomplete = !scan_metadata.is_complete(); + let unscannable_failure = args.fail_on_unscannable && incomplete; if unscannable_failure { exit_code = 1; } let unscannable_count = scan_metadata.unscannable_files.len(); + let coverage_warnings = scan_metadata.coverage_warnings.clone(); let findings_count = findings.len(); // Non-verbose runs still need to say WHERE each finding is; a bare count // forces a second scan with --verbose to act on anything. Matched text @@ -273,12 +276,22 @@ fn emit_scan_result( for line in &finding_lines { emit(line)?; } + if !args.verbose { + for warning in coverage_warnings { + emit(&format!("WARNING: {warning}"))?; + } + } let summary = match findings_count { _ if args.verbose => report_out.clone(), // "No secrets found." next to exit code 1 is contradictory; name the // actual failure instead. - 0 if unscannable_failure => format!( - "WARNING: {unscannable_count} file(s) could not be scanned (--fail-on-unscannable)" + 0 if incomplete => format!( + "WARNING: Scan incomplete; {unscannable_count} file(s) could not be scanned{}", + if unscannable_failure { + " (--fail-on-unscannable)" + } else { + "" + } ), 0 => "No secrets found.".to_string(), count => format!( diff --git a/src/report.rs b/src/report.rs index 866e975..77a761d 100644 --- a/src/report.rs +++ b/src/report.rs @@ -109,6 +109,7 @@ impl fmt::Display for Severity { pub enum ScanStatus { Pass, Fail, + Incomplete, } #[derive(Serialize, Deserialize, Clone, Debug)] @@ -136,6 +137,14 @@ pub struct ScanMetadata { /// separately instead of masquerading as operator-requested skips. pub unscannable_files: Vec, pub suppressed_by_baseline: usize, + pub coverage_warnings: Vec, + pub detector_fingerprint: String, +} + +impl ScanMetadata { + pub fn is_complete(&self) -> bool { + self.unscannable_files.is_empty() && self.coverage_warnings.is_empty() + } } /// How many paths an exclusion removed, and a bounded sample of them. @@ -158,7 +167,12 @@ impl ExcludedSummary { #[derive(Serialize)] pub struct Report { + pub scanner_version: &'static str, + pub detector_fingerprint: String, pub status: ScanStatus, + pub finding_status: ScanStatus, + pub coverage: &'static str, + pub coverage_warnings: Vec, pub findings: Vec, pub files_scanned: usize, pub total_lines: usize, @@ -178,7 +192,15 @@ pub fn create_report( scan_time: String, show_secrets: bool, ) -> Result { - let status = if findings.is_empty() { + let finding_status = if findings.is_empty() { + ScanStatus::Pass + } else { + ScanStatus::Fail + }; + let complete = metadata.is_complete(); + let status = if !complete { + ScanStatus::Incomplete + } else if findings.is_empty() { ScanStatus::Pass } else { ScanStatus::Fail @@ -195,7 +217,12 @@ pub fn create_report( .collect() }; let report = Report { + scanner_version: env!("CARGO_PKG_VERSION"), + detector_fingerprint: metadata.detector_fingerprint, status, + finding_status, + coverage: if complete { "COMPLETE" } else { "INCOMPLETE" }, + coverage_warnings: metadata.coverage_warnings, findings, files_scanned: metadata.files_scanned, total_lines: metadata.total_lines, diff --git a/src/report/sarif.rs b/src/report/sarif.rs index 9a27c53..f4f3a15 100644 --- a/src/report/sarif.rs +++ b/src/report/sarif.rs @@ -2,6 +2,40 @@ use super::{Finding, ScanMetadata, Severity}; use serde::Serialize; use std::collections::BTreeMap; +fn artifact_uri(path: &str) -> String { + let normalized = if cfg!(windows) { + path.replace('\\', "/") + } else { + path.to_string() + }; + let mut uri = if std::path::Path::new(path).is_absolute() { + if cfg!(windows) && normalized.starts_with("//") { + "file:".to_string() + } else if cfg!(windows) { + "file:///".to_string() + } else { + "file://".to_string() + } + } else { + String::new() + }; + for (index, byte) in normalized.bytes().enumerate() { + if byte.is_ascii_alphanumeric() + || matches!(byte, b'-' | b'_' | b'.' | b'~' | b'/') + || (cfg!(windows) + && index == 1 + && byte == b':' + && normalized.as_bytes()[0].is_ascii_alphabetic()) + { + uri.push(char::from(byte)); + } else { + use std::fmt::Write; + write!(uri, "%{byte:02X}").expect("Writing to a String cannot fail"); + } + } + uri +} + /// Generate a SARIF 2.1.0 report from findings. pub fn create_sarif_report( findings: Vec, @@ -98,7 +132,7 @@ pub fn create_sarif_report( let level = severity_to_sarif_level(finding.severity); let rule_id_clone = rule_id.clone(); let severity_str = finding.severity.as_str(); - let uri = finding.file_path; + let uri = artifact_uri(&finding.file_path); let start_line = finding.line_number; // No per-rule confidence model exists, so no `precision` claim is @@ -127,11 +161,36 @@ pub fn create_sarif_report( }) .collect(); - let status = if results.is_empty() { "pass" } else { "fail" }; + let finding_status = if results.is_empty() { "pass" } else { "fail" }; + let status = if metadata.is_complete() { + finding_status + } else { + "incomplete" + }; // Scan counts only: the BTreeMap serializes in key order, so the payload // is deterministic, and no matched content enters the run metadata. let mut properties = BTreeMap::new(); + properties.insert( + "detectorFingerprint".to_string(), + serde_json::json!(metadata.detector_fingerprint), + ); + properties.insert( + "findingStatus".to_string(), + serde_json::json!(finding_status), + ); + properties.insert( + "coverage".to_string(), + serde_json::json!(if metadata.is_complete() { + "complete" + } else { + "incomplete" + }), + ); + properties.insert( + "coverageWarnings".to_string(), + serde_json::json!(metadata.coverage_warnings), + ); properties.insert( "status".to_string(), serde_json::Value::String(status.to_string()), diff --git a/src/run_error.rs b/src/run_error.rs index 1821b83..fc5208d 100644 --- a/src/run_error.rs +++ b/src/run_error.rs @@ -38,6 +38,10 @@ pub enum RunCliError { "--update-baseline could not resolve a baseline file: pass --baseline , or run where .keywatch-baseline.json can be discovered or created" )] MissingBaselineForUpdate, + #[error( + "Cannot update a baseline from an incomplete scan; resolve coverage warnings and unscannable files first" + )] + IncompleteBaselineScan, #[error("Baseline file not found: '{path}' (pass --update-baseline to create it)")] BaselineNotFound { path: String }, #[error( diff --git a/src/scanner.rs b/src/scanner.rs index 284a6cc..ed93ad0 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -18,6 +18,7 @@ use std::path::{Path, PathBuf}; mod error; mod files; +mod limits; mod lines; mod staged; @@ -28,7 +29,7 @@ use files::{ is_default_excluded_file, matches_exclude_patterns, path_has_git_dir, }; use lines::{LineScanContext, scan_file_stream, scan_stream}; -use staged::{StagedScan, scan_git_output, scan_index_blobs, scan_staged_diff}; +use staged::{StagedScan, scan_git_output, scan_index_blobs, scan_staged_diff_with_limit}; /// Git config that must be overridden for every diff-based scan: the parser /// depends on undecorated `diff --git`/`@@`/`+` framing and literal `a/`/`b/` /// path prefixes, so user git config that colors, re-prefixes, quotes, or @@ -49,6 +50,12 @@ const GIT_DIFF_FRAMING_ARGS: &[&str] = &[ "core.quotePath=false", "-c", "diff.relative=false", + "-c", + "diff.srcPrefix=a/", + "-c", + "diff.dstPrefix=b/", + "-c", + "diff.interHunkContext=0", ]; pub fn run_scan( @@ -56,22 +63,21 @@ pub fn run_scan( config: Option<&KeywatchConfig>, ) -> Result<(Vec, ScanMetadata), ScannerError> { let detectors = resolve_detectors(args, config)?; - // A pattern carrying the dot-matches-newline flag anywhere — `(?s)` or - // the grouped `(?s:...)` form — spans lines and must run per-chunk, not - // per-line, or multiline secrets slip past it. - let (multiline_detectors, line_detectors): (Vec<_>, Vec<_>) = detectors - .iter() - .partition(|detector| detector.is_multiline()); + limits::input_limit(args.max_file_size)?; + // Line-local anchors retain their behavior. The complete-input pass + // reports only matches that actually span lines, without flag guessing. + let line_detectors: Vec<_> = detectors.iter().collect(); + let multiline_detectors = &line_detectors; // Resolved once: every scan mode must skip the baseline file itself. let excluded_baseline = baseline_exclusion(args); - let (findings, metadata) = if args.git_history { + let (findings, mut metadata) = if args.git_history { scan_git_history( args, config, excluded_baseline.as_ref(), - &multiline_detectors, + multiline_detectors, &line_detectors, )? } else if args.staged { @@ -79,28 +85,47 @@ pub fn run_scan( args, config, excluded_baseline.as_ref(), - &multiline_detectors, + multiline_detectors, &line_detectors, )? } else if args.stdin { - scan_stdin(&multiline_detectors, &line_detectors)? + scan_stdin(args, multiline_detectors, &line_detectors)? } else { scan_filesystem( args, config, excluded_baseline.as_ref(), - &multiline_detectors, + multiline_detectors, &line_detectors, )? }; // Every mode funnels through here, so deduplication and the canonical // order hold for reports, baselines and exit codes alike. + limits::check_findings(&findings)?; let mut findings = dedupe_findings(findings); sort_findings(&mut findings); + metadata.detector_fingerprint = detector_fingerprint(&detectors); Ok((findings, metadata)) } +fn detector_fingerprint(detectors: &[Detector]) -> String { + use sha2::{Digest, Sha256}; + let rules: Vec<_> = detectors.iter().map(|detector| serde_json::json!({ + "name": detector.name, + "pattern": detector.regex.as_str(), + "finding_type": detector.finding_type, + "severity": detector.severity.as_str(), + "allowlist": detector.allowlist.iter().map(|regex| regex.as_str()).collect::>(), + "keywords": detector.keywords, + "entropy": detector.entropy_threshold, + "validator": detector.validator.map(|validator| format!("{validator:?}")), + })).collect(); + hex::encode(Sha256::digest( + serde_json::Value::Array(rules).to_string().as_bytes(), + )) +} + /// Builds the detector set for this scan and applies user configuration on /// top. Trusted mode ignores repository-supplied detector files. fn resolve_detectors( @@ -155,6 +180,9 @@ fn scan_git_history( "--no-ext-diff", "--no-textconv", "--no-color", + "--full-index", + "--format=", + "--no-renames", ]); // Without a range, walk every ref: a secret committed on a side branch // is exactly as leaked as one on the checked-out branch. An explicit @@ -174,24 +202,45 @@ fn scan_git_history( command, |stderr| ScannerError::GitLogNonZero { stderr }, |reader| { - scan_staged_diff( + scan_staged_diff_with_limit( reader, &exclude_patterns, excluded_baseline, &repo_root, multiline_detectors, line_detectors, + staged::DiffScanPolicy { + max_bytes: limits::input_limit(args.max_file_size)?, + scan_lockfiles: args.scan_lockfiles, + }, ) }, |source| ScannerError::RunGitLog { source }, )?; let mut metadata = history.metadata; - // Blobs are only re-readable from the index, not from history, so a - // git-rendered binary in history is unscannable rather than excluded. - metadata.unscannable_files = history.unscannable_from_diff; - let mut findings = history.findings; + let (blob_findings, blob_lines, skipped) = staged::scan_history_blobs( + &repo_root, + &history.history_blobs, + Some(limits::input_limit(args.max_file_size)?), + multiline_detectors, + line_detectors, + )?; + metadata.files_scanned += history.history_blobs.len() - skipped.len(); + metadata.total_lines += blob_lines; + metadata.unscannable_files.extend(skipped); + findings.extend(blob_findings); + let shallow = std::process::Command::new("git") + .current_dir(&repo_root) + .args(["rev-parse", "--is-shallow-repository"]) + .output() + .map_err(|source| ScannerError::RunGitLog { source })?; + if !shallow.status.success() || shallow.stdout != b"false\n" { + metadata.coverage_warnings.push( + "Git history is shallow or its completeness could not be determined; fetch complete history before using a history gate.".to_string(), + ); + } sort_findings(&mut findings); Ok((findings, metadata)) } @@ -216,6 +265,8 @@ fn scan_staged( "--no-ext-diff", "--no-textconv", "--no-color", + "--no-renames", + "--full-index", "--", ]); command.args(&args.paths); @@ -224,13 +275,17 @@ fn scan_staged( command, |stderr| ScannerError::GitDiffNonZero { stderr }, |reader| { - scan_staged_diff( + scan_staged_diff_with_limit( reader, &exclude_patterns, excluded_baseline, &repo_root, multiline_detectors, line_detectors, + staged::DiffScanPolicy { + max_bytes: limits::input_limit(args.max_file_size)?, + scan_lockfiles: args.scan_lockfiles, + }, ) }, |source| ScannerError::RunGitDiff { source }, @@ -240,11 +295,12 @@ fn scan_staged( mut findings, mut metadata, unscannable_from_diff, + .. } = staged; let (blob_findings, blob_lines, skipped) = scan_index_blobs( &unscannable_from_diff, - args.max_file_size.map(|megabytes| megabytes * 1024 * 1024), + Some(limits::input_limit(args.max_file_size)?), multiline_detectors, line_detectors, )?; @@ -259,13 +315,19 @@ fn scan_staged( } fn scan_stdin( + args: &ScanArgs, multiline_detectors: &[&Detector], line_detectors: &[&Detector], ) -> Result<(Vec, ScanMetadata), ScannerError> { let stdin = std::io::stdin(); let reader = BufReader::new(stdin); - let (findings, total_lines) = - scan_stream(reader, "", multiline_detectors, line_detectors)?; + let bytes = limits::read_bounded(reader, "", limits::input_limit(args.max_file_size)?)?; + let (findings, total_lines) = scan_stream( + std::io::Cursor::new(bytes), + "", + multiline_detectors, + line_detectors, + )?; let metadata = ScanMetadata { files_scanned: 1, @@ -273,6 +335,7 @@ fn scan_stdin( excluded_files: Vec::new(), unscannable_files: Vec::new(), suppressed_by_baseline: 0, + ..Default::default() }; Ok((findings, metadata)) @@ -298,16 +361,6 @@ impl FileOutcome { } } - fn ignored() -> Self { - Self { - findings: Vec::new(), - lines_seen: 0, - scanned: false, - excluded: None, - unscannable: None, - } - } - fn unreadable(path: String) -> Self { Self { findings: Vec::new(), @@ -326,6 +379,7 @@ fn scan_filesystem( multiline_detectors: &[&Detector], line_detectors: &[&Detector], ) -> Result<(Vec, ScanMetadata), ScannerError> { + let exclude_patterns = compile_exclude_patterns(args, config)?; let mut target_paths: Vec = Vec::new(); let mut unlistable_dirs: Vec = Vec::new(); @@ -333,6 +387,11 @@ fn scan_filesystem( // the scanner will not read (symlink, device, FIFO) must not produce a // silent "No secrets found" pass. for path_str in &args.paths { + if target_paths.len() + unlistable_dirs.len() >= limits::MAX_PATHS { + return Err(ScannerError::ResourceLimit { + reason: "Input count exceeds the scan budget".to_string(), + }); + } let path = Path::new(path_str); let metadata = match fs::symlink_metadata(path) { Ok(metadata) => metadata, @@ -369,7 +428,13 @@ fn scan_filesystem( source, }); } - collect_files(path_str, &mut target_paths, path_str, &mut unlistable_dirs); + collect_files( + path_str, + &mut target_paths, + path_str, + &mut unlistable_dirs, + &exclude_patterns, + )?; } else { return Err(ScannerError::ScanPathUnsupported { path: path_str.clone(), @@ -394,31 +459,45 @@ fn scan_filesystem( } let unique_paths: Vec<_> = unique_paths.into_values().collect(); - let exclude_patterns = compile_exclude_patterns(args, config)?; let line_scan_context = LineScanContext::new(line_detectors); let scan_base_dir = std::env::current_dir().unwrap_or_else(|_| PathBuf::from(".")); - let max_bytes = args.max_file_size.map(|megabytes| megabytes * 1024 * 1024); + let max_bytes = Some(limits::input_limit(args.max_file_size)?); let settings = FileScanSettings { exclude_patterns: &exclude_patterns, excluded_baseline, scan_base_dir: &scan_base_dir, max_bytes, + scan_lockfiles: args.scan_lockfiles, }; - let results: Vec = unique_paths - .into_par_iter() - .map(|(path, roots)| { - scan_one_path( - &path, - &roots, - &settings, - multiline_detectors, - &line_scan_context, - ) - }) - .collect(); - - let (findings, mut metadata) = aggregate_file_outcomes(results); + let mut findings = Vec::new(); + let mut metadata = ScanMetadata::default(); + for batch in unique_paths.chunks(4) { + let results: Vec = batch + .par_iter() + .map(|(path, roots)| { + scan_one_path( + path, + roots, + &settings, + multiline_detectors, + &line_scan_context, + ) + }) + .collect(); + + let (batch_findings, batch_metadata) = aggregate_file_outcomes(results); + findings.extend(batch_findings); + limits::check_findings(&findings)?; + metadata.files_scanned += batch_metadata.files_scanned; + metadata.total_lines += batch_metadata.total_lines; + metadata + .excluded_files + .extend(batch_metadata.excluded_files); + metadata + .unscannable_files + .extend(batch_metadata.unscannable_files); + } metadata.unscannable_files.extend(unlistable_dirs); Ok((findings, metadata)) } @@ -456,6 +535,7 @@ struct FileScanSettings<'scan> { excluded_baseline: Option<&'scan PathBuf>, scan_base_dir: &'scan Path, max_bytes: Option, + scan_lockfiles: bool, } fn scan_one_path( @@ -471,7 +551,7 @@ fn scan_one_path( if matches_exclude_patterns(path, roots, settings.exclude_patterns) || is_baseline_file(path, settings.scan_base_dir, settings.excluded_baseline) - || is_default_excluded_file(path) + || (!settings.scan_lockfiles && is_default_excluded_file(path)) { return FileOutcome::skipped_but_reported(path.to_string()); } @@ -479,13 +559,13 @@ fn scan_one_path( let metadata = match fs::symlink_metadata(path) { Ok(metadata) => metadata, Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - return FileOutcome::ignored(); + return FileOutcome::unreadable(path.to_string()); } Err(_) => return FileOutcome::unreadable(path.to_string()), }; let file_type = metadata.file_type(); if file_type.is_symlink() || !file_type.is_file() { - return FileOutcome::ignored(); + return FileOutcome::unreadable(path.to_string()); } // Over the size cap: unscannable, never silently clean. The cap bounds // scan time on huge single files (throughput is per-file). @@ -496,7 +576,7 @@ fn scan_one_path( let mut reader = match fs::File::open(path) { Ok(file) => BufReader::new(file), Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - return FileOutcome::ignored(); + return FileOutcome::unreadable(path.to_string()); } Err(_) => return FileOutcome::unreadable(path.to_string()), }; @@ -505,15 +585,22 @@ fn scan_one_path( // identifies it reliably; decode and scan the text. match std::io::BufRead::fill_buf(&mut reader) { Ok(head) if head.starts_with(&[0xFF, 0xFE]) || head.starts_with(&[0xFE, 0xFF]) => { - let mut bytes = Vec::new(); - if std::io::Read::read_to_end(&mut reader, &mut bytes).is_err() { - return FileOutcome::unreadable(path.to_string()); - } + let bytes = match limits::read_bounded( + &mut reader, + path, + settings.max_bytes.unwrap_or(limits::MAX_INPUT_BYTES), + ) { + Ok(bytes) => bytes, + Err(_) => return FileOutcome::unreadable(path.to_string()), + }; let Some(text) = lines::decode_utf16_bom(&bytes) else { return FileOutcome::unreadable(path.to_string()); }; let (findings, total_lines) = - lines::scan_content(&text, path, multiline_detectors, line_scan_context); + match lines::scan_content(&text, path, multiline_detectors, line_scan_context) { + Ok(scanned) => scanned, + Err(_) => return FileOutcome::unreadable(path.to_string()), + }; return FileOutcome { findings, lines_seen: total_lines, @@ -525,8 +612,13 @@ fn scan_one_path( Ok(_) => {} Err(_) => return FileOutcome::unreadable(path.to_string()), } - let scanned = match scan_file_stream(&mut reader, path, multiline_detectors, line_scan_context) - { + let scanned = match scan_file_stream( + &mut reader, + path, + multiline_detectors, + line_scan_context, + settings.max_bytes.unwrap_or(limits::MAX_INPUT_BYTES), + ) { Ok(scanned) => scanned, Err(_) => return FileOutcome::unreadable(path.to_string()), }; @@ -579,6 +671,8 @@ fn aggregate_file_outcomes(results: Vec) -> (Vec, ScanMeta excluded_files, unscannable_files, suppressed_by_baseline: 0, + coverage_warnings: Vec::new(), + ..Default::default() }; (findings, metadata) diff --git a/src/scanner/error.rs b/src/scanner/error.rs index 643ac47..b5c46f7 100644 --- a/src/scanner/error.rs +++ b/src/scanner/error.rs @@ -6,6 +6,8 @@ use thiserror::Error; #[derive(Debug, Error)] #[non_exhaustive] pub enum ScannerError { + #[error("Scan resource limit: {reason}")] + ResourceLimit { reason: String }, #[error("{source}")] DetectorInit { source: DetectorInitError }, #[error("{source}")] diff --git a/src/scanner/files.rs b/src/scanner/files.rs index 47c7927..e7adf89 100644 --- a/src/scanner/files.rs +++ b/src/scanner/files.rs @@ -54,10 +54,8 @@ pub(super) fn baseline_exclusion(args: &ScanArgs) -> Option { fs::canonicalize(baseline_path).ok() } -/// Lockfiles hold checksums and resolved URLs, never credentials, and their -/// generated hashes otherwise flood reports and baselines with "Random -/// String" findings. Excluded by basename at any depth in every -/// filesystem-backed mode, matching gitleaks; `--stdin` is unaffected. +/// Lockfiles are excluded by default to reduce checksum noise. +/// They can contain credentials; users can include them explicitly. const DEFAULT_EXCLUDED_FILES: [&str; 13] = [ "bun.lock", "bun.lockb", @@ -111,38 +109,70 @@ pub(super) fn collect_files( targets: &mut Vec, root: &str, unlistable_dirs: &mut Vec, -) { + exclude_patterns: &[Pattern], +) -> Result<(), ScannerError> { // A directory that cannot be listed hides everything beneath it; record // it as unscannable instead of silently reporting a clean scan, so // --fail-on-unscannable catches it. - let entries = match fs::read_dir(dir_path) { - Ok(entries) => entries, - Err(_) => { - unlistable_dirs.push(dir_path.to_string()); - return; - } - }; - for entry in entries.flatten() { - let Ok(file_type) = entry.file_type() else { - continue; + let mut directories = vec![PathBuf::from(dir_path)]; + let roots = [Some(root.to_string())]; + while let Some(directory) = directories.pop() { + let entries = match fs::read_dir(&directory) { + Ok(entries) => entries, + Err(_) => { + unlistable_dirs.push(directory.display().to_string()); + continue; + } }; - if file_type.is_symlink() { - continue; - } - let path = entry.path(); - if file_type.is_file() { - if let Some(path_str) = path.to_str() { + for entry in entries { + if targets.len() + unlistable_dirs.len() + directories.len() >= super::limits::MAX_PATHS + { + return Err(ScannerError::ResourceLimit { + reason: "Filesystem input count exceeds the scan budget".to_string(), + }); + } + let entry = match entry { + Ok(entry) => entry, + Err(_) => { + unlistable_dirs.push(directory.display().to_string()); + continue; + } + }; + let path = entry.path(); + if path.file_name().is_some_and(|name| name == ".git") { + continue; + } + let Some(path_str) = path.to_str() else { + unlistable_dirs.push(path.display().to_string()); + continue; + }; + if matches_exclude_patterns(path_str, &roots, exclude_patterns) { targets.push(ScanTarget { path: path_str.to_string(), root: Some(root.to_string()), }); + continue; } - } else if file_type.is_dir() && path.file_name().is_none_or(|name| name != ".git") { - if let Some(path_str) = path.to_str() { - collect_files(path_str, targets, root, unlistable_dirs); + let file_type = match entry.file_type() { + Ok(file_type) => file_type, + Err(_) => { + unlistable_dirs.push(path_str.to_string()); + continue; + } + }; + if file_type.is_file() { + targets.push(ScanTarget { + path: path_str.to_string(), + root: Some(root.to_string()), + }); + } else if file_type.is_dir() { + directories.push(path); + } else { + unlistable_dirs.push(path_str.to_string()); } } } + Ok(()) } pub(super) fn path_has_git_dir(path: &Path) -> bool { diff --git a/src/scanner/limits.rs b/src/scanner/limits.rs new file mode 100644 index 0000000..0ded683 --- /dev/null +++ b/src/scanner/limits.rs @@ -0,0 +1,73 @@ +use super::ScannerError; +use crate::report::Finding; +use std::io::Read; + +pub(super) const MAX_INPUT_BYTES: u64 = 16 * 1024 * 1024; +pub(super) const MAX_LINE_BYTES: usize = 1024 * 1024; +pub(super) const MAX_PATHS: usize = 100_000; +const MAX_FINDINGS: usize = 10_000; +const MAX_FINDING_BYTES: usize = 32 * 1024 * 1024; + +pub(super) fn input_limit(megabytes: Option) -> Result { + let requested = megabytes + .unwrap_or(16) + .checked_mul(1024 * 1024) + .filter(|bytes| *bytes > 0) + .ok_or_else(|| ScannerError::ResourceLimit { + reason: "Invalid maximum file size".to_string(), + })?; + Ok(requested.min(MAX_INPUT_BYTES)) +} + +pub(super) fn read_bounded( + reader: impl Read, + path: &str, + limit: u64, +) -> Result, ScannerError> { + let mut bytes = Vec::new(); + reader + .take(limit + 1) + .read_to_end(&mut bytes) + .map_err(|source| ScannerError::ReadStream { + path: path.to_string(), + source, + })?; + if bytes.len() as u64 > limit { + return Err(ScannerError::ResourceLimit { + reason: format!("Input exceeds {limit} bytes: {path}"), + }); + } + Ok(bytes) +} + +#[derive(Default)] +pub(super) struct FindingBudget { + count: usize, + bytes: usize, +} + +impl FindingBudget { + pub(super) fn reserve(&mut self, bytes: usize) -> Result<(), ScannerError> { + if self.count >= MAX_FINDINGS || bytes > MAX_FINDING_BYTES.saturating_sub(self.bytes) { + return Err(ScannerError::ResourceLimit { + reason: "Finding count or retained text exceeds the scan budget".to_string(), + }); + } + self.count += 1; + self.bytes += bytes; + Ok(()) + } +} + +pub(super) fn check_findings(findings: &[Finding]) -> Result<(), ScannerError> { + let mut budget = FindingBudget::default(); + for finding in findings { + budget.reserve( + finding.file_path.len() + + finding.matched_content.len() + + finding.finding_type.len() + + finding.detector_name.len(), + )?; + } + Ok(()) +} diff --git a/src/scanner/lines.rs b/src/scanner/lines.rs index a3ed984..12eb5bd 100644 --- a/src/scanner/lines.rs +++ b/src/scanner/lines.rs @@ -1,12 +1,12 @@ //! Per-line and chunk scanning: the keyword prefilter, the detector accept //! chain, and the stream/chunk drivers shared by every scan mode. +use super::limits::{FindingBudget, MAX_INPUT_BYTES, MAX_LINE_BYTES, read_bounded}; use crate::detector::Detector; use crate::report::Finding; use crate::scanner::ScannerError; use aho_corasick::AhoCorasick; use regex::Regex; -use std::collections::HashSet; use std::io::BufRead; const INLINE_SUPPRESS: &str = "keywatch:ignore"; @@ -28,13 +28,34 @@ pub(super) fn read_raw_line( raw_line: &mut Vec, ) -> Result { raw_line.clear(); - let bytes_read = - reader - .read_until(b'\n', raw_line) + let mut bytes_read = 0; + loop { + let available = reader + .fill_buf() .map_err(|source| ScannerError::ReadStream { path: path.to_string(), source, })?; + if available.is_empty() { + break; + } + let take = available + .iter() + .position(|byte| *byte == b'\n') + .map_or(available.len(), |position| position + 1); + if take > MAX_LINE_BYTES.saturating_sub(raw_line.len()) { + return Err(ScannerError::ResourceLimit { + reason: format!("Line exceeds {MAX_LINE_BYTES} bytes: {path}"), + }); + } + let finished = available[take - 1] == b'\n'; + raw_line.extend_from_slice(&available[..take]); + reader.consume(take); + bytes_read += take; + if finished { + break; + } + } if bytes_read == 0 { return Ok(false); } @@ -201,6 +222,7 @@ impl<'detectors> LineScanContext<'detectors> { pub(super) struct LineScratch { lowered_line: String, candidates: Vec, + pub(super) budget: FindingBudget, } pub(super) fn scan_line_detectors( line: &str, @@ -209,14 +231,14 @@ pub(super) fn scan_line_detectors( context: &LineScanContext<'_>, scratch: &mut LineScratch, findings: &mut Vec, -) { +) -> Result<(), ScannerError> { to_lowercase_into(line, &mut scratch.lowered_line); if is_inline_suppressed(&scratch.lowered_line) { - return; + return Ok(()); } - run_line_detectors(line, line_number, path, context, scratch, findings); - scan_decoded_base64(line, line_number, path, context, scratch, findings); + run_line_detectors(line, line_number, path, context, scratch, findings)?; + scan_decoded_base64(line, line_number, path, context, scratch, findings) } /// Decodes base64 runs on the line and scans the decoded text once (no @@ -231,7 +253,7 @@ fn scan_decoded_base64( context: &LineScanContext<'_>, scratch: &mut LineScratch, findings: &mut Vec, -) { +) -> Result<(), ScannerError> { /// Shorter decoded payloads cannot hold a credential worth reporting. const MIN_DECODED_LENGTH: usize = 16; @@ -255,9 +277,23 @@ fn scan_decoded_base64( continue; }; for decoded_line in text.lines() { - run_line_detectors(decoded_line, line_number, path, context, scratch, findings); + run_line_detectors(decoded_line, line_number, path, context, scratch, findings)?; + } + let first_multiline = findings.len(); + scan_multiline_chunk( + &text, + line_number - 1, + path, + context.line_detectors, + findings, + &mut scratch.budget, + false, + )?; + for finding in &mut findings[first_multiline..] { + finding.line_number = line_number; } } + Ok(()) } /// The detector matching core, shared by the raw line and its decoded @@ -270,7 +306,7 @@ fn run_line_detectors( context: &LineScanContext<'_>, scratch: &mut LineScratch, findings: &mut Vec, -) { +) -> Result<(), ScannerError> { to_lowercase_into(line, &mut scratch.lowered_line); context .prefilter @@ -294,6 +330,9 @@ fn run_line_detectors( continue; }; if detector.accepts_captures(&captures) { + scratch.budget.reserve( + path.len() + matched.len() + detector.finding_type.len() + detector.name.len(), + )?; findings.push(Finding { file_path: path.to_string(), line_number, @@ -305,6 +344,7 @@ fn run_line_detectors( } } } + Ok(()) } pub(super) fn scan_multiline_chunk( @@ -313,39 +353,57 @@ pub(super) fn scan_multiline_chunk( path: &str, multiline_detectors: &[&Detector], findings: &mut Vec, - reported: &mut HashSet<(usize, usize, String)>, -) { + budget: &mut FindingBudget, + respect_suppression: bool, +) -> Result<(), ScannerError> { if multiline_detectors.is_empty() { - return; + return Ok(()); } let lowered_chunk = chunk.to_lowercase(); for detector in multiline_detectors { + let mut previous_start = 0; + let mut line_in_chunk = 1; + let mut line_start = 0; if detector.has_keywords(&lowered_chunk) { for captures in detector.regex.captures_iter(chunk) { let Some(matched) = captures.get(0) else { continue; }; - let line_in_chunk = chunk[..matched.start()].matches('\n').count() + 1; - let line_start = chunk[..matched.start()] - .rfind('\n') - .map(|i| i + 1) - .unwrap_or(0); - // Start position identifies a match exactly; sliding windows - // re-scan their carry, so the same match can be seen twice. - if !reported.insert(( - line_offset + line_in_chunk, - matched.start() - line_start, - detector.name.clone(), - )) { + if !matched.as_str().contains('\n') { continue; } - let line_content = chunk - .lines() - .nth(line_in_chunk.saturating_sub(1)) - .unwrap_or_default(); - let line_is_suppressed = is_inline_suppressed(&line_content.to_lowercase()); - - if !line_is_suppressed && detector.accepts_captures(&captures) { + let trimmed = matched.as_str().trim_end_matches(['\r', '\n']); + if !trimmed.contains('\n') + && detector + .regex + .find(trimmed) + .is_some_and(|found| found.as_str() == trimmed) + { + // A trailing line delimiter does not create another + // credential when the line-local pass covers the match. + continue; + } + if !detector.accepts_captures(&captures) { + continue; + } + // Matches occur in source order. Advance once through each + // intervening prefix instead of recounting the whole input. + for (offset, _) in chunk[previous_start..matched.start()].match_indices('\n') { + line_in_chunk += 1; + line_start = previous_start + offset + 1; + } + previous_start = matched.start(); + let line_content = chunk[line_start..].split('\n').next().unwrap_or_default(); + let line_is_suppressed = + respect_suppression && is_inline_suppressed(&line_content.to_lowercase()); + + if !line_is_suppressed { + budget.reserve( + path.len() + + matched.len() + + detector.finding_type.len() + + detector.name.len(), + )?; findings.push(Finding { file_path: path.to_string(), line_number: line_offset + line_in_chunk, @@ -358,16 +416,17 @@ pub(super) fn scan_multiline_chunk( } } } + Ok(()) } pub(super) fn scan_content( content: &str, path: &str, multiline_detectors: &[&Detector], context: &LineScanContext<'_>, -) -> (Vec, usize) { +) -> Result<(Vec, usize), ScannerError> { let mut findings = Vec::new(); let mut total_lines = 0; - let mut reported = HashSet::new(); + let mut scratch = LineScratch::default(); scan_multiline_chunk( content, @@ -375,11 +434,16 @@ pub(super) fn scan_content( path, multiline_detectors, &mut findings, - &mut reported, - ); + &mut scratch.budget, + true, + )?; - let mut scratch = LineScratch::default(); for (line_idx, line) in content.lines().enumerate() { + if line.len() > MAX_LINE_BYTES { + return Err(ScannerError::ResourceLimit { + reason: format!("Line exceeds {MAX_LINE_BYTES} bytes: {path}"), + }); + } total_lines += 1; scan_line_detectors( line, @@ -388,13 +452,11 @@ pub(super) fn scan_content( context, &mut scratch, &mut findings, - ); + )?; } - (findings, total_lines) + Ok((findings, total_lines)) } -const CHUNK_SIZE: usize = 1000; -const OVERLAP_LINES: usize = 50; /// How a streaming scan treats NUL bytes: stdin is scanned through, while /// a filesystem file containing one is binary and stops the scan. @@ -412,80 +474,27 @@ pub(super) struct StreamScan { pub(super) binary: bool, } -/// Shared streaming core for stdin and file mode. Reads line by line so -/// memory stays bounded regardless of input size, decodes lossily so one -/// invalid byte cannot abort the scan, and runs multiline detectors over -/// sliding windows whose carry keeps boundary-crossing secrets whole. -/// -/// Multiline matches are deduplicated by start position: a secret entirely -/// inside the carry would otherwise be reported by both adjacent windows. -/// Matches longer than OVERLAP_LINES that cross a window boundary are still -/// missed - inherent to a fixed window. +/// Reads a bounded logical input and checks complete cross-line matches. fn scan_lines( reader: &mut R, path: &str, multiline_detectors: &[&Detector], context: &LineScanContext, binary_handling: BinaryHandling, + max_bytes: u64, ) -> Result { - let mut findings = Vec::new(); - let mut reported: HashSet<(usize, usize, String)> = HashSet::new(); - let mut total_lines = 0; - let mut buffer: Vec = Vec::with_capacity(CHUNK_SIZE + OVERLAP_LINES); - // Lines preceding buffer[0]: scan_multiline_chunk adds its 1-based - // in-window line index to this offset. - let mut lines_before_window = 0; - let mut scratch = LineScratch::default(); - let mut binary = false; - let mut raw_line: Vec = Vec::new(); - - loop { - if !read_raw_line(reader, path, &mut raw_line)? { - break; - } - if matches!(binary_handling, BinaryHandling::StopAtNul) && raw_line.contains(&0) { - binary = true; - break; - } - total_lines += 1; - let line = String::from_utf8_lossy(&raw_line).into_owned(); - scan_line_detectors( - &line, - total_lines, - path, - context, - &mut scratch, - &mut findings, - ); - buffer.push(line); - - if buffer.len() >= CHUNK_SIZE + OVERLAP_LINES { - let chunk = buffer.join("\n"); - scan_multiline_chunk( - &chunk, - lines_before_window, - path, - multiline_detectors, - &mut findings, - &mut reported, - ); - lines_before_window += buffer.len() - OVERLAP_LINES; - buffer.drain(..buffer.len() - OVERLAP_LINES); - } - } - - if !binary && !buffer.is_empty() { - let chunk = buffer.join("\n"); - scan_multiline_chunk( - &chunk, - lines_before_window, + let bytes = read_bounded(reader, path, max_bytes)?; + let binary = matches!(binary_handling, BinaryHandling::StopAtNul) && bytes.contains(&0); + let (findings, total_lines) = if binary { + (Vec::new(), 0) + } else { + scan_content( + &String::from_utf8_lossy(&bytes), path, multiline_detectors, - &mut findings, - &mut reported, - ); - } - + context, + )? + }; Ok(StreamScan { findings, total_lines, @@ -506,6 +515,7 @@ pub(super) fn scan_stream( multiline_detectors, &context, BinaryHandling::ScanThrough, + MAX_INPUT_BYTES, )?; Ok((scanned.findings, scanned.total_lines)) } @@ -518,6 +528,7 @@ pub(super) fn scan_file_stream( path: &str, multiline_detectors: &[&Detector], context: &LineScanContext, + max_bytes: u64, ) -> Result { scan_lines( reader, @@ -525,6 +536,7 @@ pub(super) fn scan_file_stream( multiline_detectors, context, BinaryHandling::StopAtNul, + max_bytes, ) } diff --git a/src/scanner/staged.rs b/src/scanner/staged.rs index e3fd3fa..e2b60d3 100644 --- a/src/scanner/staged.rs +++ b/src/scanner/staged.rs @@ -1,6 +1,7 @@ //! The unified git-diff parser behind `--staged` and `--git-history`, plus //! the blob reader for files git renders as binary. +use super::limits::{self, FindingBudget, MAX_INPUT_BYTES, MAX_PATHS}; use crate::detector::Detector; use crate::report::{Finding, ScanMetadata}; use crate::scanner::ScannerError; @@ -10,7 +11,6 @@ use crate::scanner::lines::{ scan_multiline_chunk, }; use glob::Pattern; -use std::collections::HashSet; use std::io::{BufRead, BufReader}; use std::path::{Path, PathBuf}; @@ -38,11 +38,18 @@ pub(super) fn scan_git_output( let stderr = child.stderr.take(); let stderr_reader = std::thread::spawn(move || { use std::io::Read; - let mut buffer = String::new(); + let mut buffer = Vec::new(); if let Some(mut stderr) = stderr { - let _ = stderr.read_to_string(&mut buffer); + let mut chunk = [0u8; 8192]; + while let Ok(count) = stderr.read(&mut chunk) { + if count == 0 { + break; + } + let retained = count.min((64 * 1024usize).saturating_sub(buffer.len())); + buffer.extend_from_slice(&chunk[..retained]); + } } - buffer + String::from_utf8_lossy(&buffer).into_owned() }); let scanned = scan(BufReader::new(stdout)); if scanned.is_err() { @@ -116,6 +123,10 @@ fn c_unquote(quoted: &str) -> String { match bytes.next() { Some(b'"') => unescaped.push(b'"'), Some(b'\\') => unescaped.push(b'\\'), + Some(b'a') => unescaped.push(0x07), + Some(b'b') => unescaped.push(0x08), + Some(b'f') => unescaped.push(0x0c), + Some(b'v') => unescaped.push(0x0b), Some(b't') => unescaped.push(b'\t'), Some(b'n') => unescaped.push(b'\n'), Some(b'r') => unescaped.push(b'\r'), @@ -161,32 +172,43 @@ struct StagedDiffState { next_line_number: usize, hunk_start: usize, hunk_added: Vec, + hunk_bytes: usize, total_lines: usize, scanned_files: std::collections::BTreeSet, excluded_files: Vec, unscannable_from_diff: Vec, + unscannable_files: Vec, + current_oids: Vec, + history_blobs: std::collections::BTreeSet<(String, String)>, } impl StagedDiffState { /// Scans the buffered added lines of the hunk that just ended, or the /// whole diff when the stream ends. - fn flush_hunk(&mut self, multiline_detectors: &[&Detector], findings: &mut Vec) { + fn flush_hunk( + &mut self, + multiline_detectors: &[&Detector], + findings: &mut Vec, + budget: &mut FindingBudget, + ) -> Result<(), ScannerError> { if self.hunk_added.is_empty() { - return; + return Ok(()); } if let Some(path) = self.current_path.as_deref() { let chunk = self.hunk_added.join("\n"); - let mut reported = HashSet::new(); scan_multiline_chunk( &chunk, self.hunk_start.saturating_sub(1), path, multiline_detectors, findings, - &mut reported, - ); + budget, + true, + )?; } self.hunk_added.clear(); + self.hunk_bytes = 0; + Ok(()) } fn handle_added_line( @@ -195,16 +217,23 @@ impl StagedDiffState { context: &LineScanContext<'_>, scratch: &mut LineScratch, findings: &mut Vec, - ) { + ) -> Result<(), ScannerError> { let line_number = self.next_line_number; self.next_line_number += 1; let Some(path) = self.current_path.as_deref() else { - return; + return Ok(()); }; self.total_lines += 1; self.scanned_files.insert(path.to_string()); - scan_line_detectors(content, line_number, path, context, scratch, findings); + scan_line_detectors(content, line_number, path, context, scratch, findings)?; + self.hunk_bytes += content.len() + 1; + if self.hunk_bytes as u64 > MAX_INPUT_BYTES { + return Err(ScannerError::ResourceLimit { + reason: "Diff hunk exceeds the input budget".to_string(), + }); + } self.hunk_added.push(content.to_string()); + Ok(()) } /// A `+++ b/path` header selects the post-image path unless an exclusion @@ -215,12 +244,13 @@ impl StagedDiffState { exclude_patterns: &[Pattern], excluded_baseline: Option<&PathBuf>, base_dir: &Path, + scan_lockfiles: bool, ) { self.current_path = match parse_diff_target_path(target) { Some(path) if matches_exclude_patterns(&path, &[], exclude_patterns) || is_baseline_file(&path, base_dir, excluded_baseline) - || is_default_excluded_file(&path) => + || (!scan_lockfiles && is_default_excluded_file(&path)) => { self.excluded_files.push(path); None @@ -239,59 +269,86 @@ impl StagedDiffState { exclude_patterns: &[Pattern], excluded_baseline: Option<&PathBuf>, base_dir: &Path, + scan_lockfiles: bool, ) { - let path = parse_binary_marker_path(marker); - // A deleted file has no staged content left to read. - if path == "/dev/null" { + let Some((path, deleted)) = parse_binary_marker_path(marker) else { + self.unscannable_files + .push(format!("Unrecognized binary path: {marker}")); return; - } + }; if matches_exclude_patterns(&path, &[], exclude_patterns) || is_baseline_file(&path, base_dir, excluded_baseline) - || is_default_excluded_file(&path) + || (!scan_lockfiles && is_default_excluded_file(&path)) { self.excluded_files.push(path); } else { - self.unscannable_from_diff.push(path); + for oid in &self.current_oids { + self.history_blobs.insert((path.clone(), oid.clone())); + } + if !deleted { + self.unscannable_from_diff.push(path); + } } } } -/// Best-effort path from a `Binary files a/x and b/x differ` marker: the -/// post-image side, for surfacing skipped files. Sides that git quoted are -/// separated by a quoted-space-quoted delimiter, which also keeps `" and "` -/// inside a quoted name from splitting the pair. -fn parse_binary_marker_path(marker: &str) -> String { - let Some(paths) = marker.strip_suffix(" differ") else { - return marker.to_string(); - }; - let target = if paths.contains("\" and \"") { - paths - .rsplit("\" and \"") - .next() - .map(|tail| format!("\"{tail}")) - } else { - paths.rsplit(" and ").next().map(str::to_string) - }; - match target { - Some(target) => { - let unquoted = c_unquote(&target); - unquoted.strip_prefix("b/").unwrap_or(&unquoted).to_string() +/// Validate both sides before applying exclusions. Rename detection is off, +/// so two non-null sides must refer to the same path. +fn parse_binary_marker_path(marker: &str) -> Option<(String, bool)> { + let paths = marker.strip_suffix(" differ")?; + for (offset, _) in paths.match_indices(" and ") { + let old = c_unquote(&paths[..offset]); + let new = c_unquote(&paths[offset + 5..]); + match (old.strip_prefix("a/"), new.strip_prefix("b/")) { + (Some(old), Some(new)) if old == new => return Some((new.to_string(), false)), + (None, Some(new)) if old == "/dev/null" => return Some((new.to_string(), false)), + (Some(old), None) if new == "/dev/null" => return Some((old.to_string(), true)), + _ => {} } - None => marker.to_string(), } + None } /// Scans only the added lines of the staged diff, attributing findings to the /// real file path and post-image line number so `--baseline` entries match. /// Hunk state is tracked because added content may itself start with "+". /// Lines are decoded lossily so one non-UTF-8 file cannot abort the scan. +#[cfg(test)] pub(super) fn scan_staged_diff( + reader: ReaderType, + exclude_patterns: &[Pattern], + excluded_baseline: Option<&PathBuf>, + base_dir: &Path, + multiline_detectors: &[&Detector], + line_detectors: &[&Detector], +) -> Result { + scan_staged_diff_with_limit( + reader, + exclude_patterns, + excluded_baseline, + base_dir, + multiline_detectors, + line_detectors, + DiffScanPolicy { + max_bytes: MAX_INPUT_BYTES, + scan_lockfiles: false, + }, + ) +} + +pub(super) struct DiffScanPolicy { + pub max_bytes: u64, + pub scan_lockfiles: bool, +} + +pub(super) fn scan_staged_diff_with_limit( mut reader: ReaderType, exclude_patterns: &[Pattern], excluded_baseline: Option<&PathBuf>, base_dir: &Path, multiline_detectors: &[&Detector], line_detectors: &[&Detector], + policy: DiffScanPolicy, ) -> Result { let context = LineScanContext::new(line_detectors); let mut findings = Vec::new(); @@ -300,11 +357,22 @@ pub(super) fn scan_staged_diff( let mut raw_line: Vec = Vec::new(); while read_raw_line(&mut reader, "", &mut raw_line)? { + if state.scanned_files.len() + + state.excluded_files.len() + + state.unscannable_from_diff.len() + + state.unscannable_files.len() + + state.history_blobs.len() + > MAX_PATHS + { + return Err(ScannerError::ResourceLimit { + reason: "Git input count exceeds the scan budget".to_string(), + }); + } let line = String::from_utf8_lossy(&raw_line); if state.in_hunk { if let Some(content) = line.strip_prefix('+') { - state.handle_added_line(content, &context, &mut scratch, &mut findings); + state.handle_added_line(content, &context, &mut scratch, &mut findings)?; continue; } if line.starts_with('-') || line.starts_with('\\') { @@ -313,7 +381,7 @@ pub(super) fn scan_staged_diff( } if line.starts_with("@@") { - state.flush_hunk(multiline_detectors, &mut findings); + state.flush_hunk(multiline_detectors, &mut findings, &mut scratch.budget)?; state.hunk_start = parse_hunk_new_start(&line); state.next_line_number = state.hunk_start; state.in_hunk = true; @@ -321,36 +389,94 @@ pub(super) fn scan_staged_diff( } if line.starts_with("diff ") { - state.flush_hunk(multiline_detectors, &mut findings); + state.flush_hunk(multiline_detectors, &mut findings, &mut scratch.budget)?; state.in_hunk = false; state.current_path = None; + state.current_oids.clear(); continue; } + if let Some(index) = line.strip_prefix("index ") { + state.current_oids = index + .split_whitespace() + .next() + .unwrap_or("") + .split("..") + .filter(|oid| { + (oid.len() == 40 || oid.len() == 64) + && oid.bytes().all(|byte| byte.is_ascii_hexdigit()) + && oid.bytes().any(|byte| byte != b'0') + }) + .map(str::to_string) + .collect(); + } + if let Some(target) = line.strip_prefix("+++ ") { - state.select_path(target, exclude_patterns, excluded_baseline, base_dir); + state.select_path( + target, + exclude_patterns, + excluded_baseline, + base_dir, + policy.scan_lockfiles, + ); + if let Some(path) = state.current_path.as_ref() { + for oid in state.current_oids.iter().rev().take(1) { + let output = std::process::Command::new("git") + .current_dir(base_dir) + .args(["cat-file", "-s", oid]) + .output() + .map_err(|source| ScannerError::RunGitCatFile { source })?; + let size = String::from_utf8_lossy(&output.stdout) + .trim() + .parse::(); + if !output.status.success() || size.map_or(true, |size| size > policy.max_bytes) + { + state.unscannable_files.push(path.clone()); + state.current_path = None; + break; + } + } + } continue; } if let Some(marker) = line.strip_prefix("Binary files ") { - state.record_binary_marker(marker, exclude_patterns, excluded_baseline, base_dir); + state.record_binary_marker( + marker, + exclude_patterns, + excluded_baseline, + base_dir, + policy.scan_lockfiles, + ); + } + if state.scanned_files.len() + + state.excluded_files.len() + + state.unscannable_from_diff.len() + + state.history_blobs.len() + > MAX_PATHS + { + return Err(ScannerError::ResourceLimit { + reason: "Git input count exceeds the scan budget".to_string(), + }); } } - state.flush_hunk(multiline_detectors, &mut findings); + state.flush_hunk(multiline_detectors, &mut findings, &mut scratch.budget)?; let metadata = ScanMetadata { files_scanned: state.scanned_files.len(), total_lines: state.total_lines, excluded_files: state.excluded_files, - unscannable_files: Vec::new(), + unscannable_files: state.unscannable_files, suppressed_by_baseline: 0, + ..Default::default() }; Ok(StagedScan { findings, metadata, unscannable_from_diff: state.unscannable_from_diff, + history_blobs: state.history_blobs.into_iter().collect(), }) } @@ -362,6 +488,7 @@ pub(super) struct StagedScan { pub(super) findings: Vec, pub(super) metadata: ScanMetadata, pub(super) unscannable_from_diff: Vec, + pub(super) history_blobs: Vec<(String, String)>, } /// Resolves the staged blob object id for a repository-relative path. @@ -405,49 +532,79 @@ pub(super) fn scan_index_blobs( max_bytes: Option, multiline_detectors: &[&Detector], line_detectors: &[&Detector], +) -> Result<(Vec, usize, Vec), ScannerError> { + let mut blobs = Vec::new(); + let mut skipped = Vec::new(); + for path in paths { + match staged_blob_oid(path)? { + Some(oid) => blobs.push((path.clone(), oid)), + None => skipped.push(path.clone()), + } + } + let (findings, lines, blob_skips) = scan_history_blobs( + Path::new("."), + &blobs, + max_bytes, + multiline_detectors, + line_detectors, + )?; + skipped.extend(blob_skips); + Ok((findings, lines, skipped)) +} + +pub(super) fn scan_history_blobs( + repo_root: &Path, + blobs: &[(String, String)], + max_bytes: Option, + multiline_detectors: &[&Detector], + line_detectors: &[&Detector], ) -> Result<(Vec, usize, Vec), ScannerError> { let context = LineScanContext::new(line_detectors); let mut findings = Vec::new(); let mut total_lines = 0; let mut skipped = Vec::new(); - for path in paths { - let Some(oid) = staged_blob_oid(path)? else { - skipped.push(path.clone()); - continue; + for (path, oid) in blobs { + let mut command = std::process::Command::new("git"); + command + .current_dir(repo_root) + .args(["cat-file", "blob", oid]); + let bytes = match scan_git_output( + command, + |reason| ScannerError::ResourceLimit { reason }, + |reader| limits::read_bounded(reader, path, max_bytes.unwrap_or(MAX_INPUT_BYTES)), + |source| ScannerError::RunGitCatFile { source }, + ) { + Ok(bytes) => bytes, + Err(ScannerError::ResourceLimit { .. }) => { + skipped.push(path.clone()); + continue; + } + Err(error) => return Err(error), }; - let output = std::process::Command::new("git") - .args(["cat-file", "blob", &oid]) - .output() - .map_err(|source| ScannerError::RunGitCatFile { source })?; - if !output.status.success() { - skipped.push(path.clone()); - continue; - } - // Over the size cap: skipped as unscannable, never silently clean. - if max_bytes.is_some_and(|cap| output.stdout.len() as u64 > cap) { - skipped.push(path.clone()); - continue; - } // A UTF-16 blob (a Windows-written .env is the common case) is full // of NUL bytes; a byte-order mark identifies it, so decode and scan // the text instead of skipping it as binary. - if let Some(text) = crate::scanner::lines::decode_utf16_bom(&output.stdout) { - let (blob_findings, blob_lines) = - scan_content(&text, path, multiline_detectors, &context); - findings.extend(blob_findings); - total_lines += blob_lines; - continue; - } + let decoded = crate::scanner::lines::decode_utf16_bom(&bytes); // Genuinely binary content (NUL bytes) is skipped, matching file mode. - if output.stdout.contains(&0) { + if decoded.is_none() && bytes.contains(&0) { skipped.push(path.clone()); continue; } - let content = String::from_utf8_lossy(&output.stdout); + let content = decoded + .map(std::borrow::Cow::Owned) + .unwrap_or_else(|| String::from_utf8_lossy(&bytes)); let (blob_findings, blob_lines) = - scan_content(&content, path, multiline_detectors, &context); + match scan_content(&content, path, multiline_detectors, &context) { + Ok(result) => result, + Err(ScannerError::ResourceLimit { .. }) => { + skipped.push(path.clone()); + continue; + } + Err(error) => return Err(error), + }; findings.extend(blob_findings); + limits::check_findings(&findings)?; total_lines += blob_lines; } @@ -707,14 +864,12 @@ mod tests { #[test] fn test_parse_binary_marker_path_unquotes_and_takes_post_image() { assert_eq!( - parse_binary_marker_path( - "Binary files \"a/one and two.bin\" and \"b/one and two.bin\" differ" - ), - "one and two.bin" + parse_binary_marker_path("\"a/one and two.bin\" and \"b/one and two.bin\" differ"), + Some(("one and two.bin".to_string(), false)) ); assert_eq!( - parse_binary_marker_path("Binary files /dev/null and b/plain.bin differ"), - "plain.bin" + parse_binary_marker_path("/dev/null and b/plain.bin differ"), + Some(("plain.bin".to_string(), false)) ); } } diff --git a/src/utils.rs b/src/utils.rs index 4d8f528..f556f69 100644 --- a/src/utils.rs +++ b/src/utils.rs @@ -58,28 +58,77 @@ pub fn display_path(path: &Path) -> String { } } -/// Writes a report file readable only by its owner. -/// -/// `File::create` uses 0666 & ~umask, i.e. world-readable by default, and a -/// report can carry matched text when `--show-secrets` is set. The mode is -/// also forced on an existing file, whose old (possibly world-readable) -/// permissions would otherwise survive the rewrite. +/// Atomically replaces a file with an owner-readable report. pub fn write_to_file(path: &str, content: &str) -> Result<()> { + atomic_write(Path::new(path), content.as_bytes()) +} + +/// Uses a same-directory temporary file. The caller must control the parent +/// directory and its ancestors; this is not protection against ancestor races. +pub(crate) fn atomic_write(path: &Path, content: &[u8]) -> Result<()> { + use std::fs; + use std::io::{Error, ErrorKind}; + use std::sync::atomic::{AtomicU64, Ordering}; + static NEXT: AtomicU64 = AtomicU64::new(0); + let parent = path + .parent() + .filter(|parent| !parent.as_os_str().is_empty()) + .unwrap_or_else(|| Path::new(".")); + if fs::symlink_metadata(parent)?.file_type().is_symlink() || is_world_writable(parent) { + return Err(Error::new( + ErrorKind::PermissionDenied, + "Output parent must be a trusted directory", + )); + } + let check_destination = || -> Result<()> { + match fs::symlink_metadata(path) { + Ok(metadata) if !metadata.file_type().is_file() => Err(Error::new( + ErrorKind::InvalidInput, + "Output destination must be a regular file, not a symlink", + )), + Ok(_) => Ok(()), + Err(error) if error.kind() == ErrorKind::NotFound => Ok(()), + Err(error) => Err(error), + } + }; + check_destination()?; let mut options = std::fs::OpenOptions::new(); - options.write(true).create(true).truncate(true); + options.write(true).create_new(true); #[cfg(unix)] { use std::os::unix::fs::OpenOptionsExt; options.mode(0o600); } - let mut file = options.open(path)?; - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - file.set_permissions(std::fs::Permissions::from_mode(0o600))?; + for _ in 0..128 { + let temporary = parent.join(format!( + ".keywatch-write-{}-{}", + std::process::id(), + NEXT.fetch_add(1, Ordering::Relaxed) + )); + if temporary.file_name() == path.file_name() { + continue; + } + let mut file = match options.open(&temporary) { + Ok(file) => file, + Err(error) if error.kind() == ErrorKind::AlreadyExists => continue, + Err(error) => return Err(error), + }; + let result = (|| { + file.write_all(content)?; + file.sync_all()?; + drop(file); + check_destination()?; + fs::rename(&temporary, path) + })(); + if result.is_err() { + let _ = fs::remove_file(&temporary); + } + return result; } - file.write_all(content.as_bytes())?; - Ok(()) + Err(Error::new( + ErrorKind::AlreadyExists, + "Cannot reserve an output temporary file", + )) } #[cfg(unix)] diff --git a/templates/pre-push.sh b/templates/pre-push.sh index 68af175..611084c 100644 --- a/templates/pre-push.sh +++ b/templates/pre-push.sh @@ -151,10 +151,10 @@ scan_pushed_refs() { fi ;; esac - "$KEYWATCH_BIN" scan --git-history --rev-range "$range" --exit-mode critical --no-config-discovery < /dev/null + "$KEYWATCH_BIN" scan --git-history --rev-range "$range" --exit-mode critical --no-config-discovery --fail-on-unscannable < /dev/null scan_status=$? if [ "$scan_status" -eq 1 ]; then - echo "ERROR: Secret detected in $local_ref. Run '$KEYWATCH_BIN scan --git-history --rev-range $range --no-config-discovery' to inspect." >&2 + echo "ERROR: Scan blocked $local_ref because of findings or incomplete coverage. Run '$KEYWATCH_BIN scan --git-history --rev-range $range --no-config-discovery' to inspect." >&2 status=1 elif [ "$scan_status" -ne 0 ]; then echo "Error: $KEYWATCH_BIN scan failed for $local_ref (exit code: $scan_status)" >&2 diff --git a/tests/accuracy_corpus.toml b/tests/accuracy_corpus.toml new file mode 100644 index 0000000..b05106a --- /dev/null +++ b/tests/accuracy_corpus.toml @@ -0,0 +1,123 @@ +# These credentials are synthetic. These cases do not measure provider validity. +[[cases]] +name = "Typed password" +input = 'const DB_PASSWORD: &str = "OldHarmlessFixture783!";' +detector = "PasswordDetector" +[[cases]] +name = "Lifetime password" +input = '''const PWD: &'static str = "OldHarmlessFixture783!";''' +detector = "PasswordDetector" +[[cases]] +name = "JSON password" +input = '{"password":"hunter2"}' +detector = "PasswordDetector" +[[cases]] +name = "JSON token" +input = '{"api_key":"aB3xK9mQ2pR7/q7"}' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Bare apikey" +input = 'apikey="aB3xK9mQ2pR7/q7"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Bare accesskey" +input = 'accesskey="aB3xK9mQ2pR7/q7"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Bare securitykey" +input = 'securitykey="aB3xK9mQ2pR7/q7"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Slash and plus token" +input = 'api_key="aB3xK9mQ2pR7/+z8"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Quoted snake credential" +input = 'api_key="q7m2_c9v8_x5z1"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Quoted kebab credential" +input = 'api_key="1234-5678-9012"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Passphrase" +input = 'PWD=Correct-Horse-Battery-Staple' +detector = "PasswordDetector" +[[cases]] +name = "Extended placeholder" +input = 'PASSWORD="dummy7!"' +detector = "PasswordDetector" +[[cases]] +name = "Embedded placeholder" +input = 'PWD="notdummy9!"' +detector = "PasswordDetector" +[[cases]] +name = "Extended token placeholder" +input = 'api_key="changemeA7q9Z2"' +detector = "GenericKeyValueDetector" +[[cases]] +name = "Quoted code expression" +input = 'password="config.db_password.clone()"' +detector = "PasswordDetector" +[[cases]] +name = "AWS key" +input = 'AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP' +detector = "AWSKeyDetector" +[[cases]] +name = "Payment underscored key" +input = 'api_key_live_aB3dE6gH9j' +detector = "PaymentGatewayKeyDetector" +[[cases]] +name = "Payment bare key" +input = 'apikey_live_aB3dE6gH9j' +detector = "PaymentGatewayKeyDetector" +[[cases]] +name = "Hyphenated service account" +input = 'service-account-key="aB3xK9mQ2pR7nH6vD8sJ4"' +detector = "ServiceAccountKeyDetector" +[[cases]] +name = "Bare service account" +input = 'serviceaccountkey="aB3xK9mQ2pR7nH6vD8sJ4"' +detector = "ServiceAccountKeyDetector" +[[cases]] +name = "Exact password placeholder" +input = 'password="dummy"' +[[cases]] +name = "Exact token placeholder" +input = 'api_key="changeme"' +[[cases]] +name = "Exact documentation placeholder" +input = 'token="your-api-key-here"' +[[cases]] +name = "Empty typed password" +input = 'const DB_PASSWORD: &str = "";' +[[cases]] +name = "Empty JSON password" +input = '{"password":""}' +[[cases]] +name = "Proto declaration" +input = 'string password = 2;' +[[cases]] +name = "Unquoted snake reference" +input = 'token: request_data_v2' +[[cases]] +name = "Unquoted camel reference" +input = 'secret = globalState' +[[cases]] +name = "Unquoted constant reference" +input = 'config_key = VAULT_BACKEND_PATH' +[[cases]] +name = "Unquoted function reference" +input = 'password = config.db_password.clone()' +[[cases]] +name = "Shell directory reference" +input = 'docker run -v $PWD:/app example' +[[cases]] +name = "Loopback address" +input = '127.0.0.1' +[[cases]] +name = "Example email" +input = 'example@example.com' +[[cases]] +name = "Base64-like identifier" +input = 'PaymentMethodServiceEligibilityRequest' diff --git a/tests/accuracy_corpus_tests.rs b/tests/accuracy_corpus_tests.rs new file mode 100644 index 0000000..6444f09 --- /dev/null +++ b/tests/accuracy_corpus_tests.rs @@ -0,0 +1,69 @@ +use key_watch::{cli::ScanArgs, scanner::run_scan}; +use serde::Deserialize; +use std::fs; + +#[derive(Deserialize)] +struct Corpus { + cases: Vec, +} + +#[derive(Deserialize)] +struct Case { + name: String, + input: String, + detector: Option, +} + +#[test] +fn labeled_credentials_and_noncredentials_match_expected_scan_results() { + let corpus: Corpus = toml::from_str(include_str!("accuracy_corpus.toml")).unwrap(); + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("fixture.txt"); + let mut true_positive = 0; + let mut true_negative = 0; + let mut false_positive = 0; + let mut false_negative = 0; + let mut failures = Vec::new(); + for case in corpus.cases { + fs::write(&path, &case.input).unwrap(); + let (findings, metadata) = run_scan( + &ScanArgs { + paths: vec![path.to_str().unwrap().to_string()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }, + None, + ) + .unwrap(); + assert!( + metadata.is_complete(), + "{} has incomplete coverage", + case.name + ); + match case.detector { + Some(detector) + if findings + .iter() + .any(|finding| finding.detector_name == detector) => + { + true_positive += 1 + } + Some(detector) => { + false_negative += 1; + failures.push(format!("{} misses {detector}", case.name)); + } + None if findings.is_empty() => true_negative += 1, + None => { + false_positive += 1; + failures.push(format!("{} reports {} findings", case.name, findings.len())); + } + } + } + let precision = true_positive as f64 / (true_positive + false_positive) as f64; + let recall = true_positive as f64 / (true_positive + false_negative) as f64; + println!( + "Synthetic case results: TP={true_positive}, TN={true_negative}, FP={false_positive}, FN={false_negative}, precision={precision:.3}, recall={recall:.3}" + ); + assert!(failures.is_empty(), "{}", failures.join("\n")); +} diff --git a/tests/baseline_tests.rs b/tests/baseline_tests.rs index bf480f5..08834be 100644 --- a/tests/baseline_tests.rs +++ b/tests/baseline_tests.rs @@ -33,6 +33,121 @@ fn run_update_baseline(cwd: &Path, baseline_path: &Path) { ); } +fn check_credential_changes_against_baseline(staged: bool) { + let directory = tempdir().expect("create fixture directory"); + let repository = directory.path().join("repository"); + fs::create_dir(&repository).expect("create fixture repository"); + let file = repository.join("credentials.txt"); + let baseline = directory.path().join("baseline.json"); + let report = directory.path().join("report.json"); + if staged { + let output = Command::new("git") + .arg("--version") + .output() + .expect("Git is required for staged baseline tests"); + assert!(output.status.success(), "git --version failed"); + assert!( + Command::new("git") + .args(["init", "--quiet"]) + .current_dir(&repository) + .status() + .expect("initialize fixture repository") + .success() + ); + } + let stage_file = || { + if staged { + assert!( + Command::new("git") + .args(["add", "--", "credentials.txt"]) + .current_dir(&repository) + .status() + .expect("stage fixture") + .success() + ); + } + }; + let scan = |update: bool| { + let mut command = Command::new(env!("CARGO_BIN_EXE_key-watch")); + command.current_dir(&repository).args([ + "scan", + "credentials.txt", + "--no-config-discovery", + "--no-baseline-discovery", + "--show-secrets", + "--baseline", + ]); + command.arg(&baseline); + if staged { + command.arg("--staged"); + } + if update { + command.arg("--update-baseline"); + } else { + command.arg("--output").arg(&report); + } + command.output().expect("scan baseline fixture") + }; + for (original, changed, detector_name, expected_match) in [ + ( + r#"const DB_PASSWORD: &str = "OldHarmlessFixture783!";"#, + r#"const DB_PASSWORD: &str = "DifferentHarmlessFixture629!";"#, + "PasswordDetector", + r#"PASSWORD: &str = "DifferentHarmlessFixture629!""#, + ), + ( + r#"api_key="aB3xK9mQ2pR7/q7""#, + r#"api_key="aB3xK9mQ2pR7/z8""#, + "GenericKeyValueDetector", + r#"api_key="aB3xK9mQ2pR7/z8""#, + ), + ] { + if baseline.exists() { + fs::remove_file(&baseline).expect("reset fixture baseline"); + } + fs::write(&file, original).expect("write original credential"); + stage_file(); + let output = scan(true); + assert!(output.status.success(), "{output:?}"); + + for content in [original.to_string(), format!("\n\n{original}")] { + fs::write(&file, content).expect("move unchanged credential"); + stage_file(); + let output = scan(false); + assert!(output.status.success(), "{output:?}"); + let data: serde_json::Value = + serde_json::from_slice(&fs::read(&report).expect("read report")) + .expect("parse report"); + assert!(data["findings"].as_array().unwrap().is_empty()); + assert_eq!(data["suppressed_by_baseline"], 1); + } + + fs::write(&file, format!("\n\n{changed}")).expect("change credential"); + stage_file(); + let output = scan(false); + assert_eq!(output.status.code(), Some(1), "{output:?}"); + let data: serde_json::Value = + serde_json::from_slice(&fs::read(&report).expect("read changed report")) + .expect("parse changed report"); + assert_eq!(data["suppressed_by_baseline"], 0); + assert!(data["findings"].as_array().unwrap().iter().any(|finding| { + finding["plugin_name"] == detector_name + && finding["matched_content"] == expected_match + && finding["line_number"] == 3 + })); + } +} + +#[test] +fn test_filesystem_baseline_suppresses_moved_values_but_not_changed_credentials() { + check_credential_changes_against_baseline(false); +} + +#[test] +fn test_staged_baseline_suppresses_moved_values_but_not_changed_credentials() { + check_credential_changes_against_baseline(true); +} + fn make_finding(file: &str, line: usize, ftype: &str, content: &str, detector: &str) -> Finding { Finding { file_path: file.to_string(), diff --git a/tests/detector_tests.rs b/tests/detector_tests.rs index cb96fc3..377db2f 100644 --- a/tests/detector_tests.rs +++ b/tests/detector_tests.rs @@ -416,41 +416,6 @@ fn reported_by(line: &str) -> Vec { .collect() } -#[test] -fn test_credit_card_requires_issuer_prefix_and_luhn() { - for card in [ - "4111111111111111", // Visa - "5500 0000 0000 0004", // Mastercard, space separated - "4111-1111-1111-1111", // dash separated - "378282246310005", // Amex - ] { - assert!( - reported_by(card).contains(&"CreditCardDetector".to_string()), - "should detect card: {card}" - ); - } - - for card in ["6500000000000002", "6441111111111117"] { - assert!( - reported_by(card).contains(&"CreditCardDetector".to_string()), - "should detect Discover card: {card}" - ); - } - - for not_a_card in [ - "4111111111111112", // Visa prefix, fails Luhn - "1234567890123456", // no issuer prefix - "6411111111111111", // 641x is neither Discover nor UnionPay - "index aabbcc0..1111111 100644", // spans two unrelated numbers - "timestamp = 1700000000123", - ] { - assert!( - !reported_by(not_a_card).contains(&"CreditCardDetector".to_string()), - "should not detect card in: {not_a_card}" - ); - } -} - #[test] fn test_phone_number_requires_separator_or_country_code() { for phone in ["call 415-123-4567", "(415) 123-4567", "+1 415 123 4567"] { @@ -560,6 +525,8 @@ fn test_email_allowlists_documentation_domains_but_not_real_ones() { "alice@example.org", "reply-to: noreply@github.com", "author: 12345+user@users.noreply.github.com", + "sender: example@acme.io", + "qa: qa@test.com", ] { assert!( !reported_by(email).contains(&"EmailDetector".to_string()), @@ -614,6 +581,130 @@ fn test_checksum_prefix_is_allowlisted_in_random_string() { ); } +#[test] +fn test_declarations_are_not_password_findings() { + for line in [ + "optional SecretString password = 2;", + "pwd = 3; // secondary credential", + "# sasl_password = \"\"", + "const password = `Cypress@${uniqueSuffix}`;", + "fn build(password: &str) -> Secret {", + ] { + assert!( + !reported_by(line).contains(&"PasswordDetector".to_string()), + "{line} must not report a password" + ); + } + assert!( + reported_by("const db_pwd: &str = \"s3cr3tV4lue\"") + .contains(&"PasswordDetector".to_string()), + "a typed literal that holds a value is still reported" + ); +} + +#[test] +fn test_unquoted_references_are_not_generic_findings_but_literals_report() { + for line in [ + "secret = globalState", + "Token = actualTokens", + "apiKey = authDetails", + "token: request_data_v2", + "config_key = VAULT_BACKEND_PATH", + "auth = transformers", + ] { + assert!( + !reported_by(line).contains(&"GenericKeyValueDetector".to_string()), + "{line} must not report a generic secret" + ); + } + for line in [ + "api_key = \"kJ8s2mQ94xhTr3p\"", + "client_secret = \"deadbeefcafe12345678\"", + "record_key = \"audit_record\"", + "token = \"openai-api-key\"", + "_KEY = \"__array_state\"", + ] { + assert!( + reported_by(line).contains(&"GenericKeyValueDetector".to_string()), + "{line} is a literal value and must still report" + ); + } +} + +#[test] +fn test_identifier_and_payload_shapes_are_not_base64_findings() { + for line in [ + "let request = PaymentMethodServiceEligibilityRequest::default();", + "let mapping = OutgoingWebhookRetryProcessTrackerMapping::new();", + "let class = Ljava/security/MessageDigest;", + "let path = com/michaelkeevildown/9096cd3aac9029c4e6e05588448a8841;", + "let payload = eyJ2ZXJzaW9uIjoiRUNfdjEiLCJ0eXBlIjoiY2FyZCI7", + ] { + assert!( + !reported_by(line).contains(&"Base64Detector".to_string()), + "{line} must not report base64" + ); + } + assert!( + reported_by("let blob = zR366ckK8GIf3BG7sVI6u/9751z4OvBHZMM9JFWa7Bx/RCPQ8aeM+iJoqf9auuQm;") + .contains(&"Base64Detector".to_string()), + "word-less base64 with a plus sign is still reported" + ); +} + +#[test] +fn test_quoted_identifiers_are_not_random_string_findings() { + for line in [ + r#"topic = "internal-service-event-bus-topic-name""#, + r#"title = "createUserProfileSettingsRegressionTest""#, + r#"reason = "AUTHENTICATION_ATTEMPTED_BUT_NOT_SUCCESSFUL""#, + r#"name = "internationalizationconfigurationmanager""#, + r#"checksum = "SHA-256=0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef""#, + r#"id = "6d8b0a1e-6f5f-4d67-9d1e-2f0b6a1c9e33""#, + r#"digest = "5baa61e4c9b93f3f0682250b6cf8331b7ee68fd8""#, + r#"payload = "eyJ2ZXJzaW9uIjoiRUNfdjEiLCJ0eXBlIjoiY2FyZCI7""#, + ] { + assert!( + !reported_by(line).contains(&"RandomString".to_string()), + "{line} must not report a random string" + ); + } + assert!( + reported_by(r#"token = "A1b2C3d4E5f6G7h8I9j0K1l2M3n4O5p6Q7r8S9t0""#) + .contains(&"RandomString".to_string()), + "a mixed-case quoted value with digits is still reported" + ); +} + +#[test] +fn test_certificate_mentions_in_prose_are_not_findings() { + let prose = "// Text-based PEM encoded certificate (starts with -----BEGIN CERTIFICATE-----)"; + assert!( + !reported_by(prose).contains(&"CertificateDetector".to_string()), + "a marker mentioned inside prose must not report" + ); + assert!( + reported_by(" -----BEGIN CERTIFICATE-----").contains(&"CertificateDetector".to_string()), + "a marker at the start of a line is still reported" + ); +} + +#[test] +fn test_placeholder_jwt_signatures_are_not_findings() { + assert!( + !reported_by("eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.abc") + .contains(&"JWTokenDetector".to_string()), + "a short placeholder signature must not report" + ); + assert!( + reported_by( + "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.abcdefghijklmnopqrstuvwxyz0123456789ABCDEFG" + ) + .contains(&"JWTokenDetector".to_string()), + "a full-length signature is still reported" + ); +} + #[test] fn test_supabase_service_role_key_requires_service_role_claim() { // The claim is JSON inside the base64url payload, so its encoded bytes @@ -814,7 +905,7 @@ fn test_adyen_username_and_bare_rzp_prefix_do_not_report() { #[test] fn test_ip_address_validates_every_octet() { - for ip in ["10.0.0.1", "192.168.1.1", "255.255.255.255", "0.0.0.0"] { + for ip in ["10.0.0.1", "192.168.1.1", "255.255.255.255", "172.16.0.1"] { assert!( reported_by(ip).contains(&"IPAddressDetector".to_string()), "valid address must report: {ip}" @@ -831,6 +922,12 @@ fn test_ip_address_validates_every_octet() { "invalid address must not report: {not_an_ip}" ); } + for placeholder in ["127.0.0.1", "127.9.9.9", "0.0.0.0"] { + assert!( + !reported_by(placeholder).contains(&"IPAddressDetector".to_string()), + "loopback and unspecified addresses must not report: {placeholder}" + ); + } } #[test] @@ -995,7 +1092,7 @@ fn test_every_format_detector_fires_on_a_realistic_sample() { ), ( "JWTokenDetector", - "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.abc", + "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.abcdefghijklmnopqrstuvwxyz0123456789ABCDEFG", ), ("SSHPrivateKeyDetector", "-----BEGIN RSA PRIVATE KEY-----"), ("DatabaseURLDetector", "postgres://user:pass@host/db"), diff --git a/tests/hardening_tests.rs b/tests/hardening_tests.rs new file mode 100644 index 0000000..2dfb56d --- /dev/null +++ b/tests/hardening_tests.rs @@ -0,0 +1,620 @@ +use std::fs; +use std::path::Path; +use std::process::{Command, Output}; +use tempfile::TempDir; + +fn git(repo: &Path, args: &[&str]) { + let output = Command::new("git") + .current_dir(repo) + .args([ + "-c", + "core.hooksPath=/dev/null", + "-c", + "commit.gpgsign=false", + ]) + .args(args) + .output() + .expect("Git is required for hardening tests"); + assert!(output.status.success(), "{output:?}"); +} + +fn repository() -> TempDir { + let dir = tempfile::tempdir().unwrap(); + git(dir.path(), &["init", "--quiet"]); + git(dir.path(), &["config", "user.name", "Fixture"]); + git(dir.path(), &["config", "user.email", "fixture@example.com"]); + dir +} + +fn scan(repo: &Path, args: &[&str]) -> (Output, serde_json::Value) { + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(repo) + .args([ + "scan", + "--no-config-discovery", + "--no-baseline-discovery", + "--verbose", + ]) + .args(args) + .output() + .unwrap(); + let report = serde_json::from_slice(&output.stdout) + .unwrap_or_else(|_| panic!("expected JSON report: {output:?}")); + (output, report) +} + +#[test] +fn git_configuration_cannot_change_exclusion_paths_or_line_numbers() { + let dir = repository(); + fs::write( + dir.path().join("credentials.env"), + "first\nsecond\nthird\nfourth\nfifth\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Fixture"]); + fs::write( + dir.path().join("credentials.env"), + "changed\nsecond\nthird\nfourth\nAWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["config", "diff.srcPrefix", "other/"]); + git(dir.path(), &["config", "diff.dstPrefix", "vendor/"]); + git(dir.path(), &["config", "diff.interHunkContext", "10"]); + let (output, report) = scan(dir.path(), &["--staged", "--exclude", "vendor/**"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "AWSKeyDetector" + && finding["file_path"] == "credentials.env" + && finding["line_number"] == 5 + }) + ); +} + +#[test] +fn incomplete_scans_never_report_pass_and_explicit_failure_overrides_severity_policy() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("binary.dat"), b"data\0data").unwrap(); + for mode in ["strict", "critical", "always"] { + let (output, report) = scan( + dir.path(), + &[".", "--exit-mode", mode, "--fail-on-unscannable"], + ); + assert_eq!(output.status.code(), Some(1), "{mode}"); + assert_eq!(report["status"], "INCOMPLETE"); + assert_eq!(report["coverage"], "INCOMPLETE"); + assert_eq!(report["finding_status"], "PASS"); + } +} + +#[test] +fn shallow_history_is_disclosed_and_fails_closed_when_requested() { + let dir = repository(); + fs::write( + dir.path().join("credentials.env"), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Old fixture"]); + fs::write(dir.path().join("credentials.env"), "clean\n").unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Clean fixture"]); + let clone_parent = tempfile::tempdir().unwrap(); + let clone = clone_parent.path().join("shallow"); + git( + clone_parent.path(), + &[ + "clone", + "--quiet", + "--depth", + "1", + &format!("file://{}", dir.path().display()), + clone.to_str().unwrap(), + ], + ); + let (output, report) = scan( + &clone, + &[ + "--git-history", + "--fail-on-unscannable", + "--exit-mode", + "critical", + ], + ); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "INCOMPLETE"); + assert!( + report["coverage_warnings"] + .as_array() + .unwrap() + .iter() + .any(|warning| warning.as_str().unwrap().contains("shallow")) + ); +} + +#[test] +fn historical_text_marked_binary_is_scanned_from_its_committed_blob() { + let dir = repository(); + fs::write(dir.path().join(".gitattributes"), "credentials.env -diff\n").unwrap(); + fs::write( + dir.path().join("credentials.env"), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Secret fixture"]); + fs::write(dir.path().join("credentials.env"), "clean\n").unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Clean fixture"]); + let (output, report) = scan(dir.path(), &["--git-history", "--fail-on-unscannable"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "AWSKeyDetector" + && finding["file_path"] == "credentials.env" + }) + ); + assert_eq!(report["unscannable"]["count"], 0); +} + +#[test] +fn incomplete_scan_does_not_replace_an_existing_baseline() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("binary.dat"), b"data\0data").unwrap(); + let baseline = dir.path().join("accepted.json"); + let original = b"{\"version\":\"1.0\",\"entries\":[]}"; + fs::write(&baseline, original).unwrap(); + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "binary.dat", + "--baseline", + baseline.to_str().unwrap(), + "--no-config-discovery", + "--update-baseline", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert_eq!(fs::read(baseline).unwrap(), original); +} + +#[test] +fn input_limits_apply_to_files_and_staged_text_without_overflow() { + let dir = repository(); + let content = format!( + "{}AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + "# ordinary\n".repeat(110_000) + ); + fs::write(dir.path().join("large.env"), content).unwrap(); + git(dir.path(), &["add", "."]); + for mode in [vec!["large.env"], vec!["--staged"]] { + let mut args = mode; + args.extend(["--max-file-size", "1", "--fail-on-unscannable"]); + let (output, report) = scan(dir.path(), &args); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "INCOMPLETE"); + assert_eq!(report["unscannable"]["count"], 1); + } + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "large.env", + "--no-config-discovery", + "--max-file-size", + "17592186044416", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains("size")); +} + +#[test] +fn cross_line_rules_and_line_anchors_do_not_depend_on_regex_spelling() { + let dir = tempfile::tempdir().unwrap(); + fs::write( + dir.path().join("input.txt"), + format!( + "ordinary\nBEGIN\n{}END\nINT_ABCDEFGHIJKLMNOPQRST\n", + "ordinary\n".repeat(1100) + ), + ) + .unwrap(); + for (pattern, expected) in [ + (r"BEGIN\n(?:ordinary\n)*END", "BEGIN\n"), + (r"BEGIN[\s\S]*?END", "BEGIN\n"), + (r"(?s:BEGIN.*?END)", "BEGIN\n"), + (r"^(?PINT_[A-Z]{20})$", "INT_ABCDEFGHIJKLMNOPQRST"), + (r"(?i-s)^INT_[A-Z]{20}$", "INT_ABCDEFGHIJKLMNOPQRST"), + ] { + let config = dir.path().join("rule.toml"); + fs::write(&config, format!("[[rules]]\nname = 'CrossLineFixture'\nfinding_type = 'Fixture'\npattern = '{pattern}'\n")).unwrap(); + let (output, report) = scan( + dir.path(), + &[ + "input.txt", + "--show-secrets", + "--config", + config.to_str().unwrap(), + ], + ); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "CrossLineFixture" + && finding["matched_content"] + .as_str() + .unwrap() + .starts_with(expected) + }), + "{pattern}" + ); + } +} + +#[test] +fn oversized_lines_fail_instead_of_truncating_and_reporting_clean() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("long.txt"), "x".repeat(1024 * 1024 + 1)).unwrap(); + let (output, report) = scan(dir.path(), &["long.txt", "--fail-on-unscannable"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["status"], "INCOMPLETE"); +} + +#[cfg(unix)] +#[test] +fn report_and_baseline_writes_do_not_follow_destination_symlinks() { + use std::os::unix::fs::symlink; + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("clean.txt"), "ordinary\n").unwrap(); + let target = dir.path().join("owned.txt"); + let original = b"{\"version\":\"1.0\",\"entries\":[]}"; + fs::write(&target, original).unwrap(); + let link = dir.path().join("output.json"); + symlink(&target, &link).unwrap(); + for args in [ + vec!["-o", link.to_str().unwrap()], + vec!["--baseline", link.to_str().unwrap(), "--update-baseline"], + ] { + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args(["scan", "clean.txt", "--no-config-discovery"]) + .args(args) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert_eq!(fs::read(&target).unwrap(), original); + assert!( + fs::symlink_metadata(&link) + .unwrap() + .file_type() + .is_symlink() + ); + } +} + +#[test] +fn unknown_override_names_fail_instead_of_silently_changing_policy() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("clean.txt"), "ordinary\n").unwrap(); + fs::write( + dir.path().join("config.toml"), + "[overrides.MisspelledDetector]\nenabled = false\n", + ) + .unwrap(); + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "clean.txt", + "--no-config-discovery", + "--config", + "config.toml", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains("MisspelledDetector")); +} + +#[test] +fn duplicate_detector_names_are_configuration_errors() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("clean.txt"), "ordinary\n").unwrap(); + for (name, config) in [ + ( + "AWSKeyDetector", + "[[rules]]\nname = 'AWSKeyDetector'\npattern = 'fixture'\nfinding_type = 'Fixture'\n", + ), + ( + "RepeatedFixture", + "[[rules]]\nname = 'RepeatedFixture'\npattern = 'fixture'\nfinding_type = 'Fixture'\n[[rules]]\nname = 'RepeatedFixture'\npattern = 'other'\nfinding_type = 'Fixture'\n", + ), + ] { + fs::write(dir.path().join("config.toml"), config).unwrap(); + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "clean.txt", + "--no-config-discovery", + "--config", + "config.toml", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains(name)); + } +} + +#[cfg(unix)] +#[test] +fn skipped_symlinks_make_directory_coverage_incomplete() { + use std::os::unix::fs::symlink; + let dir = tempfile::tempdir().unwrap(); + symlink("missing.txt", dir.path().join("linked.txt")).unwrap(); + let (output, report) = scan(dir.path(), &[".", "--fail-on-unscannable"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["status"], "INCOMPLETE"); + assert_eq!(report["unscannable"]["count"], 1); +} + +#[test] +fn nonfinite_entropy_thresholds_are_configuration_errors() { + let dir = tempfile::tempdir().unwrap(); + fs::write(dir.path().join("clean.txt"), "ordinary\n").unwrap(); + for threshold in ["nan", "inf", "-inf"] { + fs::write( + dir.path().join("config.toml"), + format!("[[rules]]\nname = 'EntropyFixture'\npattern = 'fixture'\nfinding_type = 'Fixture'\nentropy = {threshold}\n"), + ) + .unwrap(); + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "clean.txt", + "--no-config-discovery", + "--config", + "config.toml", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2), "{threshold}: {output:?}"); + assert!(String::from_utf8_lossy(&output.stderr).contains("EntropyFixture")); + } +} + +#[test] +fn binary_filenames_with_separator_text_cannot_become_lockfile_exclusions() { + let dir = repository(); + let name = "credentials and Cargo.lock"; + fs::write( + dir.path().join(".gitattributes"), + "\"credentials and Cargo.lock\" -diff\n", + ) + .unwrap(); + fs::write( + dir.path().join(name), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + let (output, report) = scan(dir.path(), &["--staged"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| finding["file_path"] == name + && finding["plugin_name"] == "AWSKeyDetector") + ); + git(dir.path(), &["commit", "--quiet", "-m", "Secret fixture"]); + git(dir.path(), &["rm", "--quiet", name]); + git(dir.path(), &["commit", "--quiet", "-m", "Deleted fixture"]); + let (output, report) = scan(dir.path(), &["--git-history"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| finding["file_path"] == name + && finding["plugin_name"] == "AWSKeyDetector") + ); +} + +#[test] +fn git_control_character_filenames_preserve_staged_and_deleted_history_paths() { + let dir = repository(); + let mut attributes = String::new(); + let mut names = Vec::new(); + for (escape, character) in [ + ('a', '\u{7}'), + ('b', '\u{8}'), + ('f', '\u{c}'), + ('v', '\u{b}'), + ] { + let name = format!("credentials{character}.env"); + attributes.push_str(&format!("\"credentials\\{escape}.env\" -diff\n")); + fs::write( + dir.path().join(&name), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + names.push(name); + } + fs::write(dir.path().join(".gitattributes"), attributes).unwrap(); + git(dir.path(), &["add", "."]); + let (output, report) = scan(dir.path(), &["--staged", "--fail-on-unscannable"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "COMPLETE"); + for name in &names { + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "AWSKeyDetector" && finding["file_path"] == *name + }), + "Missing staged path: {name:?}" + ); + } + git(dir.path(), &["commit", "--quiet", "-m", "Add fixture"]); + for name in &names { + fs::remove_file(dir.path().join(name)).unwrap(); + } + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Delete fixture"]); + let (output, report) = scan( + dir.path(), + &[ + "--git-history", + "--rev-range", + "HEAD~1..HEAD", + "--fail-on-unscannable", + ], + ); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "COMPLETE"); + for name in names { + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "AWSKeyDetector" && finding["file_path"] == name + }), + "Missing deleted path: {name:?}" + ); + } +} + +#[test] +fn decoded_multiline_findings_use_the_encoded_source_line() { + let dir = tempfile::tempdir().unwrap(); + fs::write( + dir.path().join("input.txt"), + "ordinary\nb3JkaW5hcnkKQkVHSU4KRU5ECg==\n", + ) + .unwrap(); + fs::write( + dir.path().join("rule.toml"), + "[[rules]]\nname = 'DecodedFixture'\nfinding_type = 'Fixture'\npattern = 'BEGIN\\nEND'\n", + ) + .unwrap(); + let (_, report) = scan(dir.path(), &["input.txt", "--config", "rule.toml"]); + let finding = report["findings"] + .as_array() + .unwrap() + .iter() + .find(|finding| finding["plugin_name"] == "DecodedFixture") + .unwrap(); + assert_eq!(finding["line_number"], 2); +} + +#[test] +fn staged_size_limit_applies_to_the_post_image_not_the_previous_blob() { + let dir = repository(); + fs::write(dir.path().join("input.txt"), "ordinary\n".repeat(140_000)).unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Large fixture"]); + fs::write( + dir.path().join("input.txt"), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + let (output, report) = scan( + dir.path(), + &["--staged", "--max-file-size", "1", "--fail-on-unscannable"], + ); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "COMPLETE"); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| finding["plugin_name"] == "AWSKeyDetector") + ); +} + +#[test] +fn lockfile_inclusion_is_explicit_and_consistent_across_modes() { + let dir = repository(); + fs::write( + dir.path().join("Cargo.lock"), + "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP\n", + ) + .unwrap(); + git(dir.path(), &["add", "."]); + for mode in ["Cargo.lock", "--staged"] { + let (output, report) = scan(dir.path(), &[mode]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!(report["excluded"]["count"], 1); + let (output, report) = scan(dir.path(), &[mode, "--scan-lockfiles"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| finding["plugin_name"] == "AWSKeyDetector") + ); + } + git(dir.path(), &["commit", "--quiet", "-m", "Lockfile fixture"]); + let (output, report) = scan(dir.path(), &["--git-history", "--scan-lockfiles"]); + assert_eq!(output.status.code(), Some(1)); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| finding["plugin_name"] == "AWSKeyDetector") + ); +} + +#[test] +fn report_rule_fingerprint_is_stable_and_changes_with_detector_policy() { + let dir = repository(); + fs::write(dir.path().join("clean.txt"), "ordinary\n").unwrap(); + let (_, original) = scan(dir.path(), &["clean.txt"]); + let (_, repeated) = scan(dir.path(), &["clean.txt"]); + assert_eq!( + original["detector_fingerprint"], + repeated["detector_fingerprint"] + ); + fs::write( + dir.path().join("policy.toml"), + "[overrides.PasswordDetector]\nseverity = 'LOW'\n", + ) + .unwrap(); + let (_, changed) = scan(dir.path(), &["clean.txt", "--config", "policy.toml"]); + assert_ne!( + original["detector_fingerprint"], + changed["detector_fingerprint"] + ); +} diff --git a/tests/hooks_tests.rs b/tests/hooks_tests.rs index 3159311..23c2101 100644 --- a/tests/hooks_tests.rs +++ b/tests/hooks_tests.rs @@ -527,7 +527,7 @@ fn test_pre_push_uses_named_remote_push_url_when_argv_url_is_absent() { assert_eq!( fs::read_to_string(&marker).expect("read scanner args"), format!( - "scan --git-history --rev-range {remote_sha}..{local_sha} --exit-mode critical --no-config-discovery\n" + "scan --git-history --rev-range {remote_sha}..{local_sha} --exit-mode critical --no-config-discovery --fail-on-unscannable\n" ) ); fs::remove_dir_all(&temp_dir).expect("cleanup temp dir"); @@ -623,7 +623,7 @@ fn test_pre_push_scans_unnormalizable_remote_when_no_filters_exist() { assert_eq!( fs::read_to_string(&marker).expect("read scanner args"), format!( - "scan --git-history --rev-range {local_sha} --exit-mode critical --no-config-discovery\n" + "scan --git-history --rev-range {local_sha} --exit-mode critical --no-config-discovery --fail-on-unscannable\n" ), "a new ref (all-zero remote sha) must scan the full reachable history" ); @@ -969,3 +969,49 @@ fn test_pre_push_scans_only_the_pushed_range_end_to_end() { fs::remove_dir_all(&temp_dir).expect("cleanup temp dir"); } + +#[cfg(unix)] +#[test] +fn test_pre_push_blocks_incomplete_history_without_findings() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let git = real_git_path(); + let run_git = |args: &[&str]| { + let output = std::process::Command::new(&git) + .args([ + "-c", + "core.hooksPath=/dev/null", + "-c", + "commit.gpgsign=false", + ]) + .args(args) + .current_dir(directory.path()) + .output() + .expect("run git"); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).trim().to_string() + }; + run_git(&["init", "--quiet"]); + run_git(&["config", "user.email", "fixture@example.com"]); + run_git(&["config", "user.name", "Fixture"]); + fs::write(directory.path().join("image.dat"), b"ordinary\0binary").unwrap(); + run_git(&["add", "image.dat"]); + run_git(&["commit", "--quiet", "-m", "Binary fixture"]); + let tip = run_git(&["rev-parse", "HEAD"]); + let hook = generate_pre_push_hook(&hook_install_args(HookType::PrePush, None, None, None)); + let output = run_hook_with_packaged_keywatch( + &hook, + directory.path(), + &format!("refs/heads/main {tip} refs/heads/main {ZERO_SHA}\n"), + ); + assert_eq!( + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!(String::from_utf8_lossy(&output.stderr).contains("incomplete coverage")); +} diff --git a/tests/report_tests.rs b/tests/report_tests.rs index 9f475a1..30641ac 100644 --- a/tests/report_tests.rs +++ b/tests/report_tests.rs @@ -17,6 +17,7 @@ fn test_create_report() { excluded_files: vec![], unscannable_files: vec![], suppressed_by_baseline: 0, + ..Default::default() }; let report = create_report(findings, metadata, "0.5s".to_string(), false) @@ -46,6 +47,7 @@ fn test_report_with_findings() { excluded_files: vec![], unscannable_files: vec![], suppressed_by_baseline: 0, + ..Default::default() }; let report = create_report(findings, metadata, "0.1s".to_string(), false) @@ -77,6 +79,7 @@ fn test_create_report_includes_excluded_files_and_plugin_metadata() { excluded_files: vec!["ignored.log".to_string(), "vendor/secrets.txt".to_string()], unscannable_files: vec![], suppressed_by_baseline: 0, + ..Default::default() }; let report = create_report(findings, metadata, "1.2s".to_string(), false) @@ -112,6 +115,7 @@ fn test_create_sarif_report_uses_camel_case_fields_and_hides_matched_content() { excluded_files: vec!["skip.log".to_string()], unscannable_files: vec!["blob.bin".to_string()], suppressed_by_baseline: 3, + ..Default::default() }; let sarif = create_sarif_report(findings, metadata, "2026-08-01T00:00:00Z".to_string()) @@ -132,31 +136,14 @@ fn test_create_sarif_report_uses_camel_case_fields_and_hides_matched_content() { assert!(driver.get("semantic_version").is_none()); let properties = &json["runs"][0]["properties"]; - let property_keys: Vec<&str> = properties - .as_object() - .expect("run properties object") - .keys() - .map(String::as_str) - .collect(); - assert_eq!( - property_keys, - vec![ - "excludedFiles", - "filesScanned", - "scanTime", - "status", - "suppressedByBaseline", - "totalLines", - "unscannableFiles", - ], - "run scan counts must serialize in deterministic key order" - ); assert_eq!(properties["filesScanned"], 5); assert_eq!(properties["totalLines"], 120); assert_eq!(properties["excludedFiles"], 1); assert_eq!(properties["unscannableFiles"], 1); assert_eq!(properties["suppressedByBaseline"], 3); - assert_eq!(properties["status"], "fail"); + assert_eq!(properties["status"], "incomplete"); + assert_eq!(properties["coverage"], "incomplete"); + assert_eq!(properties["findingStatus"], "fail"); let result = &json["runs"][0]["results"][0]; assert_eq!(result["ruleId"], "AWS Key"); @@ -235,6 +222,7 @@ fn test_create_sarif_report_maps_all_severities_to_expected_levels() { excluded_files: vec![], unscannable_files: vec![], suppressed_by_baseline: 0, + ..Default::default() }; let sarif = create_sarif_report(findings, metadata, "2026-08-01T00:00:00Z".to_string()) @@ -319,6 +307,29 @@ fn test_get_severity_counts_groups_high_medium_low() { ); } +#[test] +fn sarif_locations_encode_filename_characters_instead_of_uri_fragments() { + let finding = Finding { + file_path: "src/secret #é%.rs".to_string(), + line_number: 1, + finding_type: "Credential".to_string(), + severity: Severity::High, + matched_content: "synthetic".to_string(), + detector_name: "FixtureDetector".to_string(), + }; + let report = create_sarif_report( + vec![finding], + ScanMetadata::default(), + "fixture".to_string(), + ) + .unwrap(); + let json = parse_json(&report); + assert_eq!( + json["runs"][0]["results"][0]["locations"][0]["physicalLocation"]["artifactLocation"]["uri"], + "src/secret%20%23%C3%A9%25.rs" + ); +} + #[test] fn test_redact_shows_no_prefix_for_short_matches() { // Below eight characters a four-character prefix would reveal most or diff --git a/tests/scanner_tests.rs b/tests/scanner_tests.rs index 2a0d572..7781c09 100644 --- a/tests/scanner_tests.rs +++ b/tests/scanner_tests.rs @@ -85,6 +85,193 @@ fn detectors_config_path() -> PathBuf { Path::new(env!("CARGO_MANIFEST_DIR")).join("detectors.toml") } +#[test] +fn test_typed_password_findings_include_the_complete_literal() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.rs"); + for declaration in [ + r#"const DB_PASSWORD: &str = "OldHarmlessFixture783!";"#, + r#"const DB_PASSWORD: &'static str = "OldHarmlessFixture783!";"#, + r#"const DB_PASSWORD: &str = "Harmless\"Fixture783!";"#, + ] { + fs::write(&file, declaration).expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan typed password"); + let password = findings + .iter() + .find(|finding| finding.detector_name == "PasswordDetector") + .expect("report the typed password"); + assert_eq!( + password.matched_content, + declaration + .strip_prefix("const DB_") + .unwrap() + .trim_end_matches(';') + ); + assert_eq!(password.line_number, 1); + } +} + +#[test] +fn test_password_assignments_include_complete_quoted_and_bare_values() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.txt"); + for assignment in [ + r#"pwd="Harmless\"Fixture783!""#, + r#"pwd='Harmless\'Fixture783!'"#, + "pwd=Correct-Horse-Battery-Staple", + r#"PWD: &str = "OldHarmlessFixture783!""#, + r#"PWD: &'static str = "OldHarmlessFixture783!""#, + r#"PWD: "Correct-Horse-Battery-Staple""#, + ] { + fs::write(&file, assignment).expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan password"); + assert!(findings.iter().any(|finding| { + finding.detector_name == "PasswordDetector" && finding.matched_content == assignment + })); + } +} + +#[test] +fn test_empty_password_literals_and_shell_directory_references_do_not_report() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.txt"); + for content in [ + r#"const DB_PASSWORD: &str = "";"#, + r#"const DB_PASSWORD: &'static str = "";"#, + r#"const PWD: &'static str = "";"#, + r#""password": """#, + "password = ''", + r#"docker --volume "$PWD:/app""#, + ] { + fs::write(&file, content).expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan noncredential"); + assert!( + !findings + .iter() + .any(|finding| finding.detector_name == "PasswordDetector"), + "{content} must not report a password: {findings:?}" + ); + } +} + +#[test] +fn test_generic_credentials_include_complete_values_and_key_aliases() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.txt"); + for assignment in [ + r#"api_key="aB3xK9mQ2pR7/q7""#, + r#"api_key='aB3xK9mQ2pR7+q7='"#, + "api_key=aB3xK9mQ2pR7/q7", + r#"apikey = "aB3xK9mQ2pR7""#, + r#"accesskey = "aB3xK9mQ2pR7""#, + r#"securitykey = "aB3xK9mQ2pR7""#, + r#""api_key": "aB3xK9mQ2pR7/q7""#, + r#""apikey": "aB3xK9mQ2pR7+q7=""#, + r#"api_key="wJalrXUtnFEMI/aB3xK9mQ2pR7""#, + ] { + fs::write(&file, assignment).expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan credential"); + let credential = findings + .iter() + .find(|finding| finding.detector_name == "GenericKeyValueDetector") + .expect("report the generic credential"); + assert_eq!( + credential.matched_content, + assignment.trim_start_matches('"') + ); + assert_eq!(credential.line_number, 1); + } +} + +#[test] +fn test_json_credentials_on_one_line_report_separate_complete_values() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.json"); + fs::write( + &file, + r#"{"password":"hunter2", "api_key":"aB3xK9mQ2pR7/q7"}"#, + ) + .expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan JSON credentials"); + for (name, matched_content) in [ + ("PasswordDetector", r#"password":"hunter2""#), + ("GenericKeyValueDetector", r#"api_key":"aB3xK9mQ2pR7/q7""#), + ] { + assert!(findings.iter().any(|finding| { + finding.detector_name == name + && finding.matched_content == matched_content + && finding.line_number == 1 + })); + } +} + +#[test] +fn test_credential_literals_do_not_borrow_identifier_or_placeholder_exemptions() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.env"); + for (assignment, detector_name) in [ + (r#"api_key = "q7m2_c9v8_x5z1""#, "GenericKeyValueDetector"), + (r#"api_key = "1234-5678-9012""#, "GenericKeyValueDetector"), + (r#"api_key = "changemeA7q9Z2""#, "GenericKeyValueDetector"), + ( + r#"api_key = "your-aB3xK9mQ2pR7""#, + "GenericKeyValueDetector", + ), + ("PWD=Correct-Horse-Battery-Staple", "PasswordDetector"), + (r#"PASSWORD="dummy7!""#, "PasswordDetector"), + (r#"PWD="notdummy9!""#, "PasswordDetector"), + ( + r#"PASSWORD="config.db_password.clone()""#, + "PasswordDetector", + ), + ] { + fs::write(&file, assignment).expect("write fixture"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let (findings, _) = run_scan(&args, None).expect("scan credential literal"); + assert!( + findings.iter().any(|finding| { + finding.detector_name == detector_name && finding.matched_content == assignment + }), + "report the complete credential literal: {assignment}" + ); + } +} + fn run_git_history_scan(current_dir: &Path, extra_args: &[&str]) -> Result { Command::new(env!("CARGO_BIN_EXE_key-watch")) .args(["scan", "--git-history"]) From c6f86119a16b2c66e391713d248473161bddf540 Mon Sep 17 00:00:00 2001 From: PiX <69745008+pixincreate@users.noreply.github.com> Date: Wed, 7 Oct 2026 14:48:47 +0530 Subject: [PATCH 2/3] scanner: Fix CI and review findings Use private test directories and match complete credential expressions. Bound MongoDB matches and share Git object-size queries. Apply line limits to file content, including final carriage returns. Shorten the README and remove review documents. Refresh the baseline with reviewed synthetic fixtures. Assisted-by: GPT-6.1 Sol Signed-off-by: PiX <69745008+pixincreate@users.noreply.github.com> --- .keywatch-baseline.json | 371 +++++++++++++++++++++++------ README.md | 346 +++++---------------------- detectors.toml | 41 ++-- docs/plans/2026-10-06-hardening.md | 165 ------------- docs/security-and-validation.md | 148 ------------ src/scanner.rs | 18 +- src/scanner/lines.rs | 26 +- src/scanner/staged.rs | 301 +++++++++++++++++++---- tests/baseline_tests.rs | 24 ++ tests/hardening_tests.rs | 134 +++++++++++ tests/scanner_tests.rs | 114 +++++++++ tests/utils_tests.rs | 48 ++-- 12 files changed, 964 insertions(+), 772 deletions(-) delete mode 100644 docs/plans/2026-10-06-hardening.md delete mode 100644 docs/security-and-validation.md diff --git a/.keywatch-baseline.json b/.keywatch-baseline.json index 06d3539..975b102 100644 --- a/.keywatch-baseline.json +++ b/.keywatch-baseline.json @@ -45,7 +45,7 @@ }, { "file_path": "detectors.toml", - "line_number": 883, + "line_number": 870, "finding_type": "Base64 Encoded String", "matched_content_hash": "e7e83ac014759b0de9f4e42e44daa4ae4820947de0c4436363b752b701dd147f", "plugin_name": "Base64Detector" @@ -262,91 +262,91 @@ }, { "file_path": "src/detector.rs", - "line_number": 188, + "line_number": 190, "finding_type": "Random String", "matched_content_hash": "b072b74dff00d183f00c1f247fd961d781e865fdf8409f66ed8b6ee324f7fa16", "plugin_name": "RandomString" }, { "file_path": "src/detector.rs", - "line_number": 188, + "line_number": 190, "finding_type": "Base64 Encoded String", "matched_content_hash": "629c198cd173e43cfaf3fc9b505d06104466a1e966d5491b8c82d5c251b13f67", "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 232, + "line_number": 234, "finding_type": "Base64 Encoded String", "matched_content_hash": "21860d0774458a5952aeaabd9c216ac03212211ed5191516ba2e6bcfd9f73d05", "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 242, + "line_number": 244, "finding_type": "Base64 Encoded String", "matched_content_hash": "69abc601294453463474ddc77dcbcf003202a1dfa2e1f886d880151676c42e5f", "plugin_name": "Base64Detector" }, { "file_path": "src/detector.rs", - "line_number": 362, + "line_number": 370, "finding_type": "Random String", "matched_content_hash": "6bb761274b9fe9cb8eaad9e1c7a0a519c6aa21f619a88bf7c10b6ddc41476822", "plugin_name": "RandomString" }, { "file_path": "src/detector.rs", - "line_number": 362, + "line_number": 370, "finding_type": "Base64 Encoded String", "matched_content_hash": "55aed33555bb9537c54581d477cb02670b2dc79458c39e02fd8796ccb3aa9cc2", "plugin_name": "Base64Detector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 539, + "line_number": 551, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 539, + "line_number": 551, "finding_type": "Generic Key/Secret", "matched_content_hash": "f6bd37622d846ac435a7b7dcbde2347d59494dfd3c3a787604e8b91491a39c95", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 595, + "line_number": 607, "finding_type": "SSH Private Key", "matched_content_hash": "678a65e8968aabff441076ae306e13d4fc85b1d36d8036a10d2100c1dc40d251", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 595, + "line_number": 607, "finding_type": "Private Key Content", "matched_content_hash": "f91082d1cbd2032b5ea19f2bbf6b3f88e02c12ae320fc9e08e6dc0f73d82bdd1", "plugin_name": "PrivateKeyDetector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 595, + "line_number": 607, "finding_type": "Base64 Encoded String", "matched_content_hash": "3dde06bf268892d4210f4b0bf1402ecc6a8ad1e015f8204cdc31667762572ef5", "plugin_name": "Base64Detector" }, { "file_path": "src/scanner/lines.rs", - "line_number": 618, + "line_number": 630, "finding_type": "Password", "matched_content_hash": "f260ab98a91b6cf1495f7d0048606e54f4ea955195e06f0523068b49a9611b44", "plugin_name": "PasswordDetector" }, { "file_path": "tests/baseline_tests.rs", - "line_number": 168, + "line_number": 192, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" @@ -1102,7 +1102,7 @@ }, { "file_path": "tests/exit_tests.rs", - "line_number": 706, + "line_number": 32, "finding_type": "Generic Key/Secret", "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", "plugin_name": "GenericKeyValueDetector" @@ -1158,14 +1158,14 @@ }, { "file_path": "tests/report_tests.rs", - "line_number": 100, + "line_number": 103, "finding_type": "Generic Key/Secret", "matched_content_hash": "01c80fce098d3bb4634fe8110c31070d44fc509417b970a68733152f066a0089", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/report_tests.rs", - "line_number": 329, + "line_number": 340, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" @@ -1186,119 +1186,119 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 296, + "line_number": 410, "finding_type": "AWS Access Key", "matched_content_hash": "3f733150de7916d4778298d7f90493889c38b76876b80c058e439851ce60cb2b", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 297, + "line_number": 411, "finding_type": "Generic Key/Secret", "matched_content_hash": "f6bd37622d846ac435a7b7dcbde2347d59494dfd3c3a787604e8b91491a39c95", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 300, + "line_number": 414, "finding_type": "SendGrid API Key", "matched_content_hash": "324af8a2810ec90cd74766d9fa88ddb6957d311ffa77a7a4f1e533f230e6745c", "plugin_name": "SendGridAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 300, + "line_number": 414, "finding_type": "Base64 Encoded String", "matched_content_hash": "c281ba6691c2e2772aebdbd6a7a05c869f7627d7c69eab07edfa130cee7bda70", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 301, + "line_number": 415, "finding_type": "Base64 Encoded String", "matched_content_hash": "703296baa6f0ae75d7b4c6e41b9908603d1273c9db9a28d6fefd4e23d8f3c8b5", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 301, + "line_number": 415, "finding_type": "OpenAI API Key", "matched_content_hash": "7a0d70456feea263871762c537e5e6141866b456eadc938eb9840e59eb08d1a9", "plugin_name": "OpenAIAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 344, + "line_number": 458, "finding_type": "Stripe API Key", "matched_content_hash": "c371dd8bd98e42a547fdc2378597152357423ba782bbb8ecbbafd0508ec8b825", "plugin_name": "StripeAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 374, + "line_number": 488, "finding_type": "Generic Key/Secret", "matched_content_hash": "4c8b5530b17fef887a9b327f69a38f1958e4b656f2b1c54b1d426320ead690ff", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 374, + "line_number": 488, "finding_type": "AWS Secret Access Key", "matched_content_hash": "2c770dc5ac4de63cd2cb89e604080a34ce55dd946586ba7e0d961b31e321403f", "plugin_name": "AWSSecretKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 374, + "line_number": 488, "finding_type": "Base64 Encoded String", "matched_content_hash": "332a14e95304a48a4ef671073cc98f0fc4046c61042ac20287bfa22b37c27d01", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 375, + "line_number": 489, "finding_type": "Generic Key/Secret", "matched_content_hash": "6c6d557cc63eda114745424f991b2019fdfde2844412a1f58fb5527f0e0bef9f", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 414, + "line_number": 528, "finding_type": "SSH Private Key", "matched_content_hash": "678a65e8968aabff441076ae306e13d4fc85b1d36d8036a10d2100c1dc40d251", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 414, + "line_number": 528, "finding_type": "Private Key Content", "matched_content_hash": "b94c156724c086556087773aad998c0791dc79172caa4b3184a7330cfa7c1d76", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 414, + "line_number": 528, "finding_type": "Base64 Encoded String", "matched_content_hash": "4587ecda3342a38feff73eef527c2582add9f98a892b185b54b470bb91bc631e", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 416, + "line_number": 530, "finding_type": "SSH Private Key", "matched_content_hash": "ca54604a9ed82ad96e5f5d68450002ef41112a7058469bbbdde94f83b267dca5", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 416, + "line_number": 530, "finding_type": "Private Key Content", "matched_content_hash": "43531827dd6142886cfeb10207f046021a4eb6c575828583ad2cb20d9430c72f", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 417, + "line_number": 531, "finding_type": "Base64 Encoded String", "matched_content_hash": "131407c23861d5a739223f303ce0b62808fc0dabe0301622d9836dc3c4999fcb", "plugin_name": "Base64Detector" @@ -1312,7 +1312,7 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 455, + "line_number": 569, "finding_type": "Email Address", "matched_content_hash": "48f63a76aa3c93efd693e0d5a960f5fe3ddd697e6a280f812bf678de80106a25", "plugin_name": "EmailDetector" @@ -1368,154 +1368,154 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 854, + "line_number": 968, "finding_type": "Aadhaar Card Number", "matched_content_hash": "73e8a0879ef4b4acf94b2c188c0b620e4dd198bef454f6270118d63fd87ff4a3", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 854, + "line_number": 968, "finding_type": "Aadhaar Card Number", "matched_content_hash": "4714ac4f70659987acae4a4760aedecd3924449f855e3be8115a51eb1ad35f93", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 854, + "line_number": 968, "finding_type": "Aadhaar Card Number", "matched_content_hash": "c941e88b8d0be7288dcc792a8ecfae22482b8a83ea4494f1ca0e36293009730b", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 882, + "line_number": 996, "finding_type": "Voter ID (EPIC)", "matched_content_hash": "e530110a549dd903837a0e2af06b406b01367b4572d07bffee9387334c027f20", "plugin_name": "VoterIDDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 882, + "line_number": 996, "finding_type": "Voter ID (EPIC)", "matched_content_hash": "8d6bf65189e5bdf52955b4a1592eb9c2fb0560709152317a1dabc36971c38f30", "plugin_name": "VoterIDDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 906, + "line_number": 1020, "finding_type": "PAN Card Number", "matched_content_hash": "75d7a58ed9996980fee805fb623d81193cdb5e9efddf2fa12b595171ca16530e", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 906, + "line_number": 1020, "finding_type": "PAN Card Number", "matched_content_hash": "4dd22fad4bf96036ceea29ca253be8d72851cc289d7e694c9b208b7370812ee6", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 930, + "line_number": 1044, "finding_type": "ABHA Health ID", "matched_content_hash": "25b618101c3bfdb4438894aa540e108cd150903bd24da57ac24f0d4174c0fe7e", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 930, + "line_number": 1044, "finding_type": "ABHA Health ID", "matched_content_hash": "397d0bc36d77ccfc6f74c2eaa6615bb75c3fe3bdc3df26e6f53f413cfcdacdb3", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 955, + "line_number": 1069, "finding_type": "Aadhaar Card Number", "matched_content_hash": "8d7ab03162973f8bc2baf7c8556af862841d23c4d66f7f18f8cf5c5777f8b263", "plugin_name": "AadhaarCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 955, + "line_number": 1069, "finding_type": "ABHA Health ID", "matched_content_hash": "89deb2b58721d6303daa4a62f2dee30b9a388afb905fc4d3595ec88ca875fb15", "plugin_name": "ABHADetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 955, + "line_number": 1069, "finding_type": "PAN Card Number", "matched_content_hash": "18fe123b5e00bbed7228c4ededc8a4e6e0bfd59169af97ec2e62b595eac11dde", "plugin_name": "PANCardDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1060, + "line_number": 1174, "finding_type": "Random String", "matched_content_hash": "cbe1ce18b874bc08437699d863cd422c49e104462e7d5ac6bd390acc0d7c973a", "plugin_name": "RandomString" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1060, + "line_number": 1174, "finding_type": "Google API Key", "matched_content_hash": "aec6660855470791a5b4f5a6c1ac4b61e61d2b942b1cdec400fb7811743798d0", "plugin_name": "GoogleAPIKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1107, + "line_number": 1221, "finding_type": "Password", "matched_content_hash": "67ef748345ad7f183084a7449ec05906e648c882400b9d940f2fadec23a7b197", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1202, + "line_number": 1316, "finding_type": "AWS Access Key", "matched_content_hash": "05c0aace2b76ca255ed3a7a953016d981477226dccc3b0e709d00174c8bc48b5", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1845, + "line_number": 1959, "finding_type": "Generic Key/Secret", "matched_content_hash": "97894929682fc00219686473cbcfa3731d73b23e88ffa7c19a511c1bbfa18aa5", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 1845, + "line_number": 1959, "finding_type": "AWS Secret Access Key", "matched_content_hash": "58db90a4f2acf80492ed15e73ad77de6c84b9aace839d821931aa0474c898386", "plugin_name": "AWSSecretKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2049, + "line_number": 2163, "finding_type": "AWS Access Key", "matched_content_hash": "357b7fb7890985d4c94a43012d1f7aefe25757f36d8810388357993bb38bd8e7", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2301, + "line_number": 2415, "finding_type": "AWS Access Key", "matched_content_hash": "0cbae582394e61dc81d9946ed87ec44f4c83cf1167181f62cd1454eb5e2e5469", "plugin_name": "AWSKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2602, + "line_number": 2716, "finding_type": "Password", "matched_content_hash": "6121378fbc25183476235d474ee367f7dbd0bbe3d96642daaeb483b5e9108cdb", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2716, + "line_number": 2830, "finding_type": "Base64 Encoded String", "matched_content_hash": "37f93597692f90e64e009b94ce81237f0f5a6d03ffce8085e91814efe71e4431", "plugin_name": "Base64Detector" @@ -1536,42 +1536,42 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2857, + "line_number": 2971, "finding_type": "Base64 Encoded String", "matched_content_hash": "f6c41d46bc806e05582dd7504058eb98e514473b4637fcbd3f1f46a12d6a1399", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2810, + "line_number": 2924, "finding_type": "Generic Key/Secret", "matched_content_hash": "e1a8fe4e1ccad188f7428469d1d0f7c55ce10a6290696ac32472f6ff40c2d2b6", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2864, + "line_number": 2978, "finding_type": "Base64 Encoded String", "matched_content_hash": "de00639949784d409b3d3e7b866432f64ca1fe35e047b312a012bbc0a19501b9", "plugin_name": "Base64Detector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2906, + "line_number": 3020, "finding_type": "SSH Private Key", "matched_content_hash": "1b01887f477e98dd56ec542c12433dee8c323176bd532c05163823079263ba31", "plugin_name": "SSHPrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2906, + "line_number": 3020, "finding_type": "Private Key Content", "matched_content_hash": "d3022d6b263b30e0ab3e0d77220f81e0bebfcc0b723f84f77d2d82d306d22e50", "plugin_name": "PrivateKeyDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 2906, + "line_number": 3020, "finding_type": "Base64 Encoded String", "matched_content_hash": "81623b8aeb6fe8c68c0a53b2d7a12ecf1d044b3ac15886fdaced0eb177f385dc", "plugin_name": "Base64Detector" @@ -2068,101 +2068,318 @@ }, { "file_path": "tests/scanner_tests.rs", - "line_number": 244, + "line_number": 252, "finding_type": "Generic Key/Secret", "matched_content_hash": "62673b45fa4d97d4e3914584c611fc886c814691240b08da38c11b6155a50d1b", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 245, + "line_number": 253, "finding_type": "Generic Key/Secret", "matched_content_hash": "f8abdc480c8cf97c232ae37084a8dd211daa71b7c8d15fbdcdf7456d3cd4bdde", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 247, + "line_number": 255, "finding_type": "Generic Key/Secret", "matched_content_hash": "5698e9b921268822471d96185883282d0aee44c9f40e39863abfd5334f004a11", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 250, + "line_number": 258, "finding_type": "Password", "matched_content_hash": "d8880112f59bc42ba1980daec1c8ea00037a09cd98e8e9d36ccd8e3c635d5fa0", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 251, + "line_number": 259, "finding_type": "Password", "matched_content_hash": "1e5e3c3f3c82f9c1819e26230e521021d0faef82800732b02a90ffc079ef49f3", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 252, + "line_number": 260, "finding_type": "Password", "matched_content_hash": "b852778a6601bce63299473637d31fc855da0175928516bea4c59ac3df9700b4", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 254, + "line_number": 262, "finding_type": "Password", "matched_content_hash": "b2c167e2202c4fe897b118e21d5758cf6147186b276660620d9c85e462f5a307", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 455, + "line_number": 569, "finding_type": "Password", "matched_content_hash": "3021414aa2f536cb3b4b850cc0573e71e97596e81014e0e62d1abf7785a2f998", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 525, + "line_number": 639, "finding_type": "Password", "matched_content_hash": "406dd360d9af78b059bb954a52e93435b61065bb2fe4d742acef80ae4223c549", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 526, + "line_number": 640, "finding_type": "Password", "matched_content_hash": "5492a757450efbf6b362c4e37a438efcf02ffeebf1992acb4d364f33adfaf04e", "plugin_name": "PasswordDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 647, + "line_number": 761, "finding_type": "Generic Key/Secret", "matched_content_hash": "450344b31602c3e15912c9ca9f0504aa45cd6d2816934c07f8bdf07434266f1d", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 648, + "line_number": 762, "finding_type": "Generic Key/Secret", "matched_content_hash": "84a0a3964fed0da6095ff2ebc1ec44efb3c973fdf5c8728e4838be7e232ff47a", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 684, + "line_number": 798, "finding_type": "Generic Key/Secret", "matched_content_hash": "158e13610e827eb0e4e836966cad520a90a938d98b85b95b38c9294103dfe20f", "plugin_name": "GenericKeyValueDetector" }, { "file_path": "tests/scanner_tests.rs", - "line_number": 723, + "line_number": 837, "finding_type": "Generic Key/Secret", "matched_content_hash": "4580776c2c5ad71d3dd4cfade7fb51fa6b5a06315aaa5704d925d64c0293b7b0", "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 105, + "finding_type": "Password", + "matched_content_hash": "56b8a99f8a0aa3ed080e1311f4353e91a1ca3b916b1b0290da828b9b4aa2e229", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 106, + "finding_type": "Password", + "matched_content_hash": "981dfbfcd56d4e3aabfbaf512a6ffdca038948cb2ecfb903e89013c19de3320d", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 111, + "finding_type": "Password", + "matched_content_hash": "fe5bfb02cefd4b93b5ba286d30ff60330a939e2a4ccdacc958209249092195a6", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 112, + "finding_type": "Password", + "matched_content_hash": "9052eb8c5e99faacd122fea6a1c92c3fefdaaba46787730dc89e6b7ca2f8fc32", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 117, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "2de21c4091c54cf02c615b46c4e9d21a5d5eba7958f04cafae410c5fbe456737", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 118, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "a89f7d9ca14ac7f9336b5017f03a3e89e68944d6f7236d2eb3767a13f0838c5f", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 123, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "d3ce6308f6cac61ad7e47944df0e215545386debcd0ad39fba6f3580a050e799", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/baseline_tests.rs", + "line_number": 124, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "4a1c421454bc96b889f5b49bddd941860bf2ed43d2cb0bb8d9c42e297298a41b", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/detector_tests.rs", + "line_number": 1101, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "aa01165935d3f8891e4beb2392f0c5cdd5963b5fbe041e4ec9a3fb65d01e098c", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/hardening_tests.rs", + "line_number": 571, + "finding_type": "Random String", + "matched_content_hash": "9dc10ed3eb1e7bcb06c2c09a76596d866a5803d014fe835d79fa0e7957eea6f2", + "plugin_name": "RandomString" + }, + { + "file_path": "tests/hardening_tests.rs", + "line_number": 571, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "fd8a0c0a6c91024974a81cc2be4b69d9e1c98c964263e5ca33db017468b7d019", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 245, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "a89f7d9ca14ac7f9336b5017f03a3e89e68944d6f7236d2eb3767a13f0838c5f", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 249, + "finding_type": "Generic Key/Secret", + "matched_content_hash": "25af5ff0875ef2a462cd8807adcb9aa9757d7eb9ad03c3065f33f72960efe2b3", + "plugin_name": "GenericKeyValueDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 294, + "finding_type": "Password", + "matched_content_hash": "3a949ba4ed81754babdbc8cc08b08cd7b09a7a95a302e3ce69777c9cac339366", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 295, + "finding_type": "Password", + "matched_content_hash": "9cc19b7267dc4c1053fc302a335ae13a2bab49d9ff3c13a2e01c052c9e1c9fe2", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 296, + "finding_type": "Password", + "matched_content_hash": "81decd1f129eb101fb89672183795baf6f5e1995e163efada3f9279608988bf8", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 297, + "finding_type": "Password", + "matched_content_hash": "dc7355be013f53a3070b34a272d13e238808918dab44c3091c123e1608b549f8", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 298, + "finding_type": "Password", + "matched_content_hash": "dbd1ec9df020e777fe8e670ba659f6cfa91c581d0a6bf9d2c5b5b2c49977e9fd", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 299, + "finding_type": "Password", + "matched_content_hash": "759818c0d45eed5e33376102ef5641f40fcd871bc6b16b8f2e46e1f63db0b2dd", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 301, + "finding_type": "Password", + "matched_content_hash": "5482a1e1a1a0b361aecb98ab20bd0f83b9e9b583818b11010a079948edf22ddf", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 305, + "finding_type": "Password", + "matched_content_hash": "028ac867b369f798afb8fc70d763279027d036b34e1a75edca22fba01a245920", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 308, + "finding_type": "Password", + "matched_content_hash": "a59fe552bf6ecb169f5cf9193a448a8feb12a89f1f9b7a453c85b802f1b15a45", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 309, + "finding_type": "Password", + "matched_content_hash": "e332335607736d9e6a65fc87ac3524e0867f25dd3262e8f765553008bd0a3809", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 321, + "finding_type": "Password", + "matched_content_hash": "8d2f7259b1a57e143802b9718f71629d696ed9ef6a6f8101c7a609052f160fe6", + "plugin_name": "PasswordDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 339, + "finding_type": "Database URL", + "matched_content_hash": "9022a5f459f8f46f46c8f71e7b711fec15d0cc8af703bf3df0f26346ec2f5391", + "plugin_name": "DatabaseURLDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 340, + "finding_type": "Database URL", + "matched_content_hash": "07bf27b525beaf6aef5a3c0b88ee13f2820bc3109c2ac6c041ffa65717813c42", + "plugin_name": "DatabaseURLDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 341, + "finding_type": "Database URL", + "matched_content_hash": "f6b9195f92dcb7b4b2e1bef7c3f88f728cb33f8575a134b2c998556eb29a1401", + "plugin_name": "DatabaseURLDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 359, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "d3ce6308f6cac61ad7e47944df0e215545386debcd0ad39fba6f3580a050e799", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 360, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "3ac1fa4539cf477b9b0d1f38295384d4cdab175f09e8270fed98dd9570f47e42", + "plugin_name": "MongoDBConnectionStringDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 360, + "finding_type": "Email Address", + "matched_content_hash": "9d759cc12219630b78bb6872f2b0cd139565c81db69a110204c628a934a229f6", + "plugin_name": "EmailDetector" + }, + { + "file_path": "tests/scanner_tests.rs", + "line_number": 361, + "finding_type": "MongoDB Connection String", + "matched_content_hash": "9884dac55e743d48ff35fd9de24adbbcea35ced20ce8793b7e985e8dc729a638", + "plugin_name": "MongoDBConnectionStringDetector" } ] } diff --git a/README.md b/README.md index 2ba5776..d95db5c 100644 --- a/README.md +++ b/README.md @@ -1,353 +1,135 @@ # KeyWatch -KeyWatch scans files, directories, and git repositories for secrets such as API keys, tokens, passwords, and private keys. -It runs as a command-line tool, as a git hook, as a GitHub Action, and as a container image. - -See `CHANGELOG.md` for differences between this source tree and published releases. -Do not assume an installed binary or hook contains unreleased changes. +KeyWatch scans files and Git repositories for API keys, passwords, tokens, and private keys. +You can run it from the command line, Git hooks, GitHub Actions, or a container. ## Install -Install with cargo: - ```sh cargo install key-watch -key-watch --version -``` - -Or download a binary from GitHub Releases, place it on your `PATH`, and make it executable: - -```sh -mkdir -p ~/.local/bin -mv ~/Downloads/key-watch ~/.local/bin/key-watch -chmod +x ~/.local/bin/key-watch -~/.local/bin/key-watch --version -``` - -Building from source requires Rust 1.85 or later. - -The command is `key-watch`. -To use the shorter aliases `keywatch` and `kw`, add this line to your shell configuration file: - -```sh -eval "$(key-watch init bash)" # or: zsh, fish, posix ``` -## Scan from the command line +You can also download a binary from [GitHub Releases](https://github.com/pixincreate/KeyWatch/releases) and put it on your `PATH`. +To build from source, use Rust 1.85 or later. -Scan a file, a directory, or standard input: +## Scan ```sh -key-watch scan secrets.txt # one file -key-watch scan . # a directory tree -cat secrets.txt | key-watch scan --stdin +key-watch scan . # Scan a directory +key-watch scan secrets.txt # Scan a file +key-watch scan --staged # Scan staged changes +key-watch scan --git-history # Scan local Git history +key-watch scan --git-history --rev-range main..HEAD # Scan a commit range +cat secrets.txt | key-watch scan --stdin # Scan standard input ``` -Scan a git repository: - -```sh -key-watch scan --staged # only the lines staged for commit -key-watch scan --git-history # every commit on every branch -key-watch scan --git-history --rev-range abc123..def456 # a commit range -``` +KeyWatch prints each finding with its file and line number. +Console output always redacts matched text. +Reports redact matched text unless you pass `--show-secrets`. -Control the output: +Write a report with: ```sh -key-watch scan . --verbose # print the full JSON report -key-watch scan . --output report.json # write the report to a file -key-watch scan . --format sarif --output report.sarif # write SARIF to a file +key-watch scan . --output report.json +key-watch scan . --format sarif --output report.sarif ``` -By default, KeyWatch prints one line per finding with the file, line number, and a redacted preview. -Reports never contain the full matched text unless you pass `--show-secrets`. - -### Scan options - -| Option | Purpose | -| ------------------------- | ----------------------------------------------------------------------------------------------------------------- | -| `--exclude ` | Skip paths that match these comma-separated glob patterns | -| `--exit-mode ` | Set the finding policy: `strict` fails on any finding, `critical` fails on HIGH or CRITICAL findings, `always` ignores findings | -| `--fail-on-unscannable` | Fail on incomplete coverage in every exit mode; incomplete scans cannot update baselines | -| `--baseline ` | Use a specific baseline file | -| `--no-baseline-discovery` | Do not look for a baseline file automatically | -| `--update-baseline` | Record the current findings in the baseline instead of reporting them | -| `--prune-baseline` | With `--update-baseline`, also remove baseline entries that no longer match anything | -| `--config ` | Use a specific `.keywatch.toml` configuration file | -| `--trusted-detectors` | Ignore a `detectors.toml` supplied by the scanned repository; use only built-in or operator rules | -| `--no-repo-config` | Do not look for `.keywatch.toml` in the scanned tree; an explicit `--config` still loads | -| `--no-config-discovery` | Shorthand for `--trusted-detectors` plus `--no-repo-config`; the installed hooks pass it | -| `--show-secrets` | Include the full matched text in reports | -| `--max-file-size ` | Lower the 16 MiB input ceiling; larger inputs are unscannable | -| `--scan-lockfiles` | Include lockfiles excluded by the default noise policy | - -Notes: - -- Lockfiles such as `Cargo.lock`, `package-lock.json`, `pnpm-lock.yaml`, and `yarn.lock` are skipped by default to reduce checksum findings. - Lockfiles can contain credentials; use `--scan-lockfiles` when your policy requires them. -- `--staged` reads the content you staged with `git add`, not the files on disk. - A secret that is staged but already removed from the working copy is still found. - A secret whose lines were staged in separate commits can span change hunks the diff never shows together; run `key-watch scan .` on the tree to catch that case. -- `--git-history` scans every branch and tag. - Use `--rev-range` to scan only a range of commits. - Shallow history produces incomplete coverage; fetch complete history before using a history gate. - KeyWatch does not fetch history automatically. - Text that Git renders as binary is scanned from its committed blobs, including deleted versions. -- A scan path that does not exist, is a symbolic link, or cannot be read is an error. - The scan never reports a clean result for input it could not read. - Recursive scans do not follow symbolic links; skipped links and unreadable entries produce incomplete coverage unless you exclude their paths. -- Files that start with a UTF-16 byte-order mark are decoded and scanned. - Other files that contain NUL bytes are treated as binary and reported as unscannable. -- Base64 runs of 24 or more characters are decoded, and the decoded text is scanned as well. - An encoded credential is reported at the line that contains it. -- GitHub tokens are checked against their built-in checksum, so lookalike strings do not appear in results. - -JSON reports separate `finding_status` from `coverage`. -The combined `status` is `INCOMPLETE` when requested inputs cannot be scanned completely, even when no finding exists. -Reports include the scanner version and a fingerprint of the effective detector definitions. -This fingerprint identifies rules; it does not authenticate the binary or describe every exclusion and suppression. -Report and baseline files use atomic replacement and reject destination symlinks or unsafe immediate parent directories. -Use directories whose parents and ancestors you control. - -Each logical input has a 16 MiB byte ceiling and a 1 MiB line ceiling. -Scans also limit path records, finding counts, and retained finding text. -Exceeding a limit produces incomplete coverage or a runtime error, not a clean report. -See [Security and validation](docs/security-and-validation.md) for limits, measured workloads, and remaining risks. - ### Exit codes -| Code | Meaning | -| ---- | ----------------------------------------------------------------- | -| 0 | Finding policy passes and no explicit coverage failure applies | -| 1 | Finding policy fails, or coverage is incomplete with `--fail-on-unscannable` | -| 2 | Invalid input, configuration error, or runtime error | +- `0`: The scan passes your finding policy. +- `1`: The scan finds a secret, or coverage is incomplete with `--fail-on-unscannable`. +- `2`: The command has an input, configuration, or runtime error. -## Git hooks +The default policy fails on any finding. +Use `--exit-mode critical` to fail only on HIGH or CRITICAL findings. +For CI, add `--fail-on-unscannable` so skipped inputs also fail the scan. -KeyWatch installs two git hooks: +### Limits -- The **pre-commit** hook scans the lines you staged. - A secret in staged content blocks the commit. - Findings in lines you did not change never block a commit. -- The **pre-push** hook scans the commits you are about to push. - It runs in `critical` exit mode, so HIGH and CRITICAL findings block the push; MEDIUM and LOW findings are reported but do not block. - Incomplete coverage also blocks the push. - Uncommitted files never block a push. +- Git scans inspect changes, not complete repository snapshots. + History scans need complete local history. +- Lockfiles are skipped by default. + Add `--scan-lockfiles` to include them. +- Each input is limited to 16 MiB, and each line to 1 MiB. + Binary files and skipped symbolic links make coverage incomplete. +- Finding and path limits can stop large scans with an error. -Install and remove hooks inside a repository: +## Handle false positives -```sh -key-watch hook install pre-commit -key-watch hook install pre-push -key-watch hook uninstall pre-commit -key-watch hook uninstall pre-push -``` - -Add `--global` to install or remove a hook for every repository on the machine: +Review findings before you accept them in a baseline: ```sh -key-watch hook install pre-commit --global -key-watch hook uninstall pre-commit --global -``` - -### Hook options - -| Option | Applies to | Purpose | -| ------------------------ | ---------- | ------------------------------------------- | -| `--exclude ` | pre-commit | Skip staged paths that match these patterns | -| `--allowed-repos ` | pre-push | Allow pushes only to these repositories | -| `--blocked-repos ` | pre-push | Block pushes to these repositories | - -### How hooks behave - -- Hooks always use the built-in detector rules. - A repository cannot weaken its own scan by committing a modified detector file. -- Hooks respect a committed baseline file and `keywatch:ignore` markers. -- KeyWatch refuses to overwrite or remove a hook file it did not install. -- The first push of a branch scans the full history of that branch, because every commit on it is new to the remote. - Review old findings before accepting any in a baseline. -- A global install sets `core.hooksPath` in your git configuration. - Git then ignores each repository's own `.git/hooks` scripts. - To keep a repository's own hooks instead, run `git config core.hooksPath .git/hooks` inside that repository. - The KeyWatch hook then no longer runs there. - -## Baselines - -A baseline records findings you have reviewed and accepted, so later scans report only new findings. -The baseline file stores unsalted hashes of matched text, not the plaintext secrets. -An attacker can guess low-entropy values offline from these hashes. -Review this disclosure risk before committing a baseline. -Protect baseline changes and inline ignore markers through code review. - -```sh -# Record the current findings key-watch scan . --update-baseline - -# Later scans report only new findings -key-watch scan . ``` -KeyWatch finds a committed `.keywatch-baseline.json` automatically. -You do not need to pass `--baseline` on every scan. +Later scans use `.keywatch-baseline.json` automatically. +The baseline stores hashes, but weak values can still be guessed from them. -Detector corrections can change the matched text used for a fingerprint without changing the baseline file format. -An entry created from a truncated match does not suppress a corrected, complete match. -Review findings that reappear after an upgrade before you run `--update-baseline`. -Do not accept them only to restore a clean scan. -An incomplete scan cannot create, merge, or prune a baseline. - -## Ignore a single line - -Add `keywatch:ignore` to a line to suppress findings on that line: +To ignore one line, add `keywatch:ignore`: ```sh password = 'known-test-password' # keywatch:ignore ``` -## Configuration +## Configure -Place a `.keywatch.toml` file in the repository root to add rules, disable detectors, or exclude paths: +Put a `.keywatch.toml` file in your repository root: ```toml exclude = ["target/**"] -[[rules]] -name = "InternalToken" -pattern = "INT_[A-Za-z0-9]{32}" -finding_type = "Internal Token" -severity = "HIGH" - [overrides.EmailDetector] enabled = false ``` -Unknown keys in the configuration file are rejected, so a misspelled key cannot silently weaken a scan. -Unknown detector override names and duplicate detector names are also rejected before configuration changes apply. +You can also add custom rules. +Run `key-watch scan --help` for all scan options. +Use `--no-config-discovery` when you do not trust the repository's rules or configuration. +This flag does not disable baselines or inline ignore markers. -## GitHub Action +## Git hooks -```yaml -name: Secret scan +```sh +key-watch hook install pre-commit +key-watch hook install pre-push +``` -on: - pull_request: - push: +The pre-commit hook scans staged changes. +The pre-push hook scans pushed commits and blocks HIGH or CRITICAL findings and incomplete coverage. +Use `hook uninstall` to remove a hook. -permissions: - contents: read +## GitHub Action +```yaml jobs: - keywatch: + secrets: runs-on: ubuntu-latest + permissions: + contents: read steps: - uses: actions/checkout@v7 - - id: keywatch - uses: pixincreate/KeyWatch@v3 + - uses: pixincreate/KeyWatch@v3 with: paths: "." - exit-mode: strict ``` -The Action installs a released KeyWatch binary, verifies its checksum, and writes a JSON report. -It fails on incomplete coverage, including reports from binaries that only expose unscannable file counts. -It supports Linux x64 and macOS runners. -Pin an exact release tag or commit SHA when you need a fixed version. -Checksums from the same release detect corruption; they do not independently authenticate a compromised release publisher. - -| Input | Default | Purpose | -| ----------- | ---------------------- | -------------------------------------------------------------------- | -| `version` | Action release version | Exact KeyWatch release to install | -| `paths` | `.` | Space-separated paths or globs to scan | -| `args` | empty | Extra scanner arguments; Action-managed options cannot be overridden | -| `exit-mode` | `strict` | `strict`, `critical`, or `always` | -| `output` | temporary file | Path for the JSON report | -| `config` | empty | Path to a trusted `.keywatch.toml` | -| `verbose` | `false` | Deprecated; enabling it is rejected to keep secrets out of logs | - -The Action exposes `findings-count` and `exit-code` as step outputs. - -## Container image - -```sh -docker pull ghcr.io/pixincreate/keywatch:3 -docker run --rm --volume "$PWD:/workspace:ro" ghcr.io/pixincreate/keywatch:3 scan . -``` - -Images are tagged `x.y.z`, `x.y`, `x`, and `latest`. -Use an exact version tag for reproducible results. -The image runs as a non-root user. - -## Uninstall - -If you installed with cargo: +The Action downloads a released binary, not the source in your checkout. +See [CHANGELOG.md](CHANGELOG.md) for release changes. -```sh -cargo uninstall key-watch -``` - -If you installed a binary manually, delete it from your `PATH` directory: +## Container ```sh -rm -f ~/.local/bin/key-watch +docker run --rm -v "$PWD:/workspace:ro" ghcr.io/pixincreate/keywatch:3 scan . ``` -In both cases, remove the `key-watch init` line from your shell configuration file if you added one. - -## Architecture - -KeyWatch is a single Rust binary. -`main.rs` starts the program and maps every validation, configuration, or runtime failure to exit code 2. -Scans exit with code 0 or 1. -Separate modules own detector loading, scanning, baselines, reports, and hooks. - -### Modules and adapters - -![KeyWatch CLI module and adapter architecture](docs/architecture/cli-modules.svg) - -Green boxes are internal modules. -Blue boxes are entry and output boundaries. -Yellow boxes are external adapters such as git and the installed hook scripts, which call `key-watch scan` themselves. - -### Scan pipeline - -Path scans collect files and process batches of four files in parallel. -Stdin and Git blobs use bounded complete inputs. -Git text scans inspect added lines and bounded addition hunks. -Multiline matching scans each complete bounded input or hunk. -Reports separate findings from coverage status. -`--update-baseline` writes the baseline instead of producing a report, but rejects incomplete coverage. - -### Detector and configuration trust - -![KeyWatch detector and configuration trust boundaries](docs/architecture/detector-config-trust.svg) - -Detector rules and repository configuration are separate systems. -External detector sources take precedence, and the compiled-in rules are the fallback. -Trusted scans use embedded detector rules and still honor configuration explicitly selected with `--config`. -Repository baselines and inline suppressions remain policy inputs that require review. - -### Core data types - -- **Detector** — one named rule: pattern, finding type, severity, optional keywords, entropy threshold, allowlist, and validator. -- **Finding** — one detected secret: file path, line number, finding type, severity, matched content, and the detector that produced it. -- **Severity** — `Critical`, `High`, `Medium`, `Low`. -- **KeywatchConfig** — parsed `.keywatch.toml`: custom rules, per-detector overrides, and exclude patterns. -- **Baseline** — versioned fingerprint entries that filter out known findings. -- **ScanMetadata** — files scanned, total lines, skipped files, coverage warnings, and the effective detector fingerprint. - -The diagram sources are in `docs/architecture/*.d2`. -After editing them, run `scripts/render-diagrams.sh render` with D2 v0.7.1, or `scripts/render-diagrams.sh check` to detect stale images. - ## Development ```sh -cargo build --release -cargo test -cargo fmt -cargo clippy +cargo test --all-features --all-targets +cargo fmt --check +cargo clippy --all-targets -- -D warnings ``` ## License -KeyWatch is licensed under the GPL-3.0-only license. -See [LICENSE](LICENSE). +[GPL-3.0-only](LICENSE). diff --git a/detectors.toml b/detectors.toml index 067a8f7..dd11f0c 100644 --- a/detectors.toml +++ b/detectors.toml @@ -109,7 +109,7 @@ severity = "HIGH" name = "PasswordDetector" # Each value branch captures the complete value. Typed Rust assignments must # match the literal after `=`, not the type annotation before it. -pattern = '''(?i)(?:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*(?-i)(?:"((?:\\[^\r\n]|[^"\\\r\n])*)"|'((?:\\[^\r\n]|[^'\\\r\n])*)'|([^\s"'`;,\r\n]+)(?:[ \t]*;)?)''' +pattern = '''(?i)(?:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*(?-i)(?:"((?:\\[^\r\n]|[^"\\\r\n])*)"|'((?:\\[^\r\n]|[^'\\\r\n])*)'|(\$\{\{[^\r\n]*)|([^\s"'`;,\r\n]+)(?:[ \t]*;)?)''' finding_type = "Password" severity = "HIGH" keywords = ["password", "passwd", "pwd"] @@ -117,25 +117,28 @@ keywords = ["password", "passwd", "pwd"] # Makefiles), never a password literal. allowlist = [ "^\\$PWD:", + # Only a complete, unquoted secret reference is exempt. Literal suffixes + # and fallback values remain part of the match and must report. + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*[:=][ \t]*\$\{\{[ \t]*secrets\.[A-Za-z_][A-Za-z0-9_]*[ \t]*\}\}[ \t]*$''', # Rust expressions are plumbing, not literals: type paths (`Secret::new`), # generics (`Secret`), constructor calls (`Some(...)`) and field or # method access (`config.password.clone()`). Quoted literals and bare # alphanumeric values stay reported. - "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*(?:::[A-Za-z_][A-Za-z0-9_]*)+$", - "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*<[^\"'\\s]*>$", - "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*\\([^\"'\\s]*\\)$", - "[:=][ \\t]*[A-Za-z_][A-Za-z0-9_]*(?:\\.[A-Za-z_][A-Za-z0-9_]*(?:\\([^\"'\\s]*\\))?)+$", + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*[A-Za-z_][A-Za-z0-9_]*(?:::[A-Za-z_][A-Za-z0-9_]*)+$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*[A-Za-z_][A-Za-z0-9_]*<[^"'\s]*>$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*[A-Za-z_][A-Za-z0-9_]*\([^"'\s]*\)$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*(?:\([^"'\s]*\))?)+$''', # A bare CamelCase value is a Rust type (`password: String,`), matching the # GenericKeyValueDetector rule. Digits or symbols keep it reported. - "[:=][ \\t]*[A-Z][a-z]+(?:[A-Z][a-z]*)*$", + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*[A-Z][a-z]+(?:[A-Z][a-z]*)*$''', # Suppress only complete, selected documentation placeholders. # Extended values such as `dummy7!` must still report. - '''[:=][ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', # Match empty literals completely before suppressing them. - '''[:=][ \t]*(?:""|'')$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*(?::[ \t]*&[ \t]*(?:'[A-Za-z_][A-Za-z0-9_]*[ \t]+)?str[ \t]*=|[:=])[ \t]*(?:""|'')$''', # Proto field numbers and Rust parameter types do not hold literals. - "[:=][ \\t]*\\d+[ \\t]*;$", - "[:=][ \\t]*&str\\)?$", + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*[:=][ \t]*\d+[ \t]*;$''', + '''^(?i:password|passwd|\$?pwd)["']?[ \t]*[:=][ \t]*&str\)?$''', ] [[detectors]] @@ -209,18 +212,18 @@ entropy = 2.5 # A bare CamelCase value (no digits, no symbols) is a Rust type path in a # field declaration (`token: PaymentTokenData,`), not a credential either. allowlist = [ - "[:=]\\s*[a-z]+(?:_[a-z]+)+$", - "[:=]\\s*[A-Z][a-z]+(?:[A-Z][a-z]*)*$", + "^[^:=]+[:=]\\s*[a-z]+(?:_[a-z]+)+$", + "^[^:=]+[:=]\\s*[A-Z][a-z]+(?:[A-Z][a-z]*)*$", # Suppress only complete, selected documentation placeholders. - '''[:=][ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', + '''^[^:=]+[:=][ \t]*(?:"(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)"|'(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx)'|(?:changeme|changeit|your-api-key-here|replace-me-please|placeholder|dummy|xxxxx))$''', # Match the complete AWS documentation example, not a shared prefix. - '''[:=][ \t]*(?:"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"|'wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY'|wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY)$''', + '''^[^:=]+[:=][ \t]*(?:"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"|'wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY'|wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY)$''', # Unquoted code references remain heuristics. Quoted snake-case and # kebab-case values can hold credentials and must not borrow these rules. - "[:=]\\s*_*[a-z][a-z0-9]*(?:_[a-z][a-z0-9]*)+$", - "[:=]\\s*[a-z]{10,}$", - "[:=]\\s*[a-z][a-z]*(?:[A-Z][a-z]*)+$", - "[:=]\\s*[A-Z][A-Z0-9]*(?:_+[A-Z0-9]+)+$", + "^[^:=]+[:=]\\s*_*[a-z][a-z0-9]*(?:_[a-z][a-z0-9]*)+$", + "^[^:=]+[:=]\\s*[a-z]{10,}$", + "^[^:=]+[:=]\\s*[a-z][a-z]*(?:[A-Z][a-z]*)+$", + "^[^:=]+[:=]\\s*[A-Z][A-Z0-9]*(?:_+[A-Z0-9]+)+$", ] [[detectors]] @@ -416,7 +419,7 @@ keywords = ["AccountKey", "DefaultEndpointsProtocol"] [[detectors]] name = "MongoDBConnectionStringDetector" -pattern = "mongodb(?:\\+srv)?://[^:]+:[^@]+@[^/]+" +pattern = '''mongodb(?:\+srv)?://[^:/@?#"'`\s\\]+:[^/@?#"'`\s\\]+@(?:\[[0-9A-Fa-f:.]+\]|[A-Za-z0-9_.-]+)(?::[0-9]+)?''' finding_type = "MongoDB Connection String" severity = "HIGH" keywords = ["mongodb://", "mongodb+srv://"] diff --git a/docs/plans/2026-10-06-hardening.md b/docs/plans/2026-10-06-hardening.md deleted file mode 100644 index c10b51a..0000000 --- a/docs/plans/2026-10-06-hardening.md +++ /dev/null @@ -1,165 +0,0 @@ -# KeyWatch hardening plan - -## Goal and constraints - -Close the verified detection and scan-coverage gaps before making production-readiness claims. -Keep the CLI and module architecture. -Preserve the existing false-positive changes, except for exemptions that hide credentials. -Do not add project-specific fixture exemptions or dependencies. -Do not commit, publish, or replace the installed binary without permission. - -## Stage 1: Complete credential matches - -Status: Implemented and tested. -Repository CI remains blocked by baseline drift. - -1. Add failing scanner tests for complete typed password values and token suffixes. -2. Fix `PasswordDetector` and `GenericKeyValueDetector` patterns and keyword coverage. -3. Support quoted JSON keys and multiple credential fields on one line. -4. Remove quoted snake-case and kebab-case exemptions. -5. Restrict placeholder exemptions to complete, selected documentation values. -6. Test baseline suppression through the executable in filesystem and staged modes. -7. Update conflicting test expectations and baseline compatibility documentation. -8. Run formatting, lint, debug tests, release tests, and the Rust 1.85 build check. -9. Inspect baseline drift without automatically accepting findings. - -### Files - -- `detectors.toml` -- `tests/scanner_tests.rs` -- `tests/baseline_tests.rs` -- `tests/detector_tests.rs` -- `README.md` -- `CHANGELOG.md` - -### Approved acceptance tests - -1. Typed Rust passwords and tokens containing `/` or `+` include the complete value. -2. Changing a password or token suffix produces a finding despite an existing baseline. -3. Moving an unchanged credential to another line remains suppressed by its baseline. -4. JSON credentials and bare `apikey`, `accesskey`, and `securitykey` names report. -5. Extended placeholders, passphrases, and quoted snake-case or kebab-case credentials report. -6. Exact selected placeholders and unquoted code references remain suppressed. - -Keep detector names, severity levels, entropy thresholds, custom-rule behavior, and baseline format unchanged. -Corrected match text can change fingerprints. -Do not use legacy prefix fingerprints to suppress complete matches. -Review affected findings before updating a baseline. -This stage does not provide a complete Rust or JSON parser. - -### Validation results - -- `cargo test --locked --all-features --all-targets` exits with status 0. - All 285 debug tests pass. -- `cargo test --locked --release --all-features --all-targets` exits with status 0. - All 285 release tests pass. -- `cargo clippy --locked --all-features --all-targets -- -D warnings` exits with status 0. -- `cargo fmt --all --check` and `rustup run nightly cargo fmt --all --check` exit with status 0. -- `rustup run 1.85.0 cargo check --locked --offline --all-features --all-targets` exits with status 0. -- `git diff --check` exits with status 0. - -Baseline inspection uses temporary copies and the repository's configured exclusions. -The scan reports 71 unbaselined findings with 62 distinct baseline identities. -These findings occur in test and workflow files. -The temporary merge also refreshes 49 existing locations. -The temporary prune removes 38 stale identities. -The repository baseline remains unchanged by this stage. -Review the findings before accepting any baseline changes. -The repository self-scan and baseline drift checks do not pass until this work is resolved. - -## Stage 2: Git coverage and scan outcomes - -Status: Implemented and tested through executable Git and report regressions. - -Define predictable Git prefixes and hunk context. -Disclose shallow history without fetching automatically. -Address historical binary blobs and incomplete-scan exit behavior. -Separate detection results from coverage status. -Test exclusion paths, line attribution, shallow clones, and binary attributes through executable scans. - -## Stage 3: Resource limits - -Status: Implemented and tested. -Synthetic workload measurements are recorded in `docs/security-and-validation.md`. -Production-environment measurements remain a deployment requirement. -A whole-workspace measurement reaches the finding budget and exits with status 2 before producing a report. -Do not claim that this source tree can complete every large-repository scan. - -Use checked size conversion and consistent byte limits across scan modes. -Bound long lines, chunks, decoded content, and retained findings. -Measure memory and execution time on representative repositories. -Do not claim large-project readiness without those measurements. - -## Stage 4: Storage and configuration safety - -Status: Implemented and tested. - -Protect report and baseline writes against symlinks and partial writes. -Reject unknown override names. -Disclose skipped paths and define lockfile policy. -Document offline guessing risks for low-entropy baseline values. - -## Stage 5: Accuracy and deployment evidence - -Status: Implemented and tested within the documented scope. -The labeled corpus, benchmark script, report provenance, hook checks, and Action checks are included. -Published-binary authentication and untested platforms remain operational requirements. - -Expand the labeled positive and negative corpus with paired examples. -Measure precision and recall without claiming universal detection. -Review suppression ownership, release provenance, and supported platforms. -Document unresolved coverage limits and the need for additional security controls. - -You approved implementation of stages 2 through 5, with review after completion. -Changes to PII defaults, provider verification, release automation, or installation are not approved by this plan. - -## Final review points - -Review `docs/security-and-validation.md` for coverage, limits, trust boundaries, and measured accuracy. -The installed binary and existing installed hooks remain unchanged. -No commits, publication, dependency changes, or automatic baseline acceptance occur during this work. -The source changes preserve baseline format `1.0` but change some finding identities. -The JSON status can be `INCOMPLETE`, and explicit coverage failure overrides every severity exit mode. -Report and baseline writes require safe, operator-controlled directories. -Configuration rejects unknown overrides, duplicate detector names, and non-finite entropy thresholds. - -### Final validation evidence - -The following checks exit with status 0: - -- `cargo test --locked --all-features --all-targets --quiet`: All 307 debug tests pass. -- `cargo test --locked --release --all-features --all-targets --quiet`: All 307 release tests pass. -- `cargo clippy --locked --all-features --all-targets -- -D warnings`. -- `rustup run nightly cargo fmt --all --check`. -- `rustup run 1.85.0 cargo check --locked --offline --all-features --all-targets`. -- `git diff --check`. -- `uv run --no-project --no-config python -B scripts/action_validation/validate.py`. -- `cargo audit --no-fetch --json`: The cached database reports zero vulnerabilities and no warnings. - -The staged and history regressions include deleted binary paths with control characters. -Configuration tests reject non-finite thresholds before they can affect detection or report fingerprints. -The two documented false-positive corpus residuals remain unchanged. -The 34-case labeled corpus passes, but does not establish real-world precision or recall. - -### Baseline review before draft PR preparation - -The final baseline inspection uses temporary copies, trusted rules, and the repository's configured exclusions. -The self-scan exits with status 1, reports complete coverage, and finds 103 unbaselined occurrences. -These occurrences have 87 distinct baseline identities in tests, detector definitions, workflows, and the benchmark script. -The temporary merge changes the entry count from 225 to 312 and refreshes 63 existing locations. -The temporary prune removes 38 stale identities and leaves 274 entries. -The repository baseline remains unchanged by the hardening stages. -Review these findings before accepting them. -The repository self-scan and baseline drift checks remain blocked until that review is complete. - -### Draft PR preparation - -You approve review and baselining of confirmed synthetic fixtures before creating the draft PR. -The reviewed fixture update changes the baseline from 225 to 309 entries. -It includes test credentials, benchmark credentials, and the CI test token. -The required installed-scanner check of staged files exits with status 0. -The source-built repository self-scan still exits with status 1 and reports complete coverage. -Three findings remain: one workflow expression and two detector-definition matches. -Those findings are not accepted by the fixture baseline update. -CI self-scan and baseline drift checks remain blocked for your review. -The whole-workspace resource-budget limitation also remains unresolved. diff --git a/docs/security-and-validation.md b/docs/security-and-validation.md deleted file mode 100644 index 7f255fc..0000000 --- a/docs/security-and-validation.md +++ /dev/null @@ -1,148 +0,0 @@ -# Security and validation - -## Intended use - -Use KeyWatch as an additional secret-detection control, not as proof that a repository contains no credentials. -Regex rules, entropy checks, and identifier exemptions can produce false positives and false negatives. -Provider verification is not implemented, and the scanner does not determine whether a credential is active. -The installed binary, published Action binary, container image, and hook templates can differ from this source tree. -Check their versions and release notes before you deploy them. - -## Trust boundaries - -Repository content, Git configuration, filenames, diffs, and object data are untrusted inputs. -Operator-selected configuration controls rules, exclusions, and severity. -Baseline entries and `keywatch:ignore` markers suppress findings. -Require review for these suppressions and for changes to scan policy. -Hooks ignore repository detector configuration but still trust repository baselines and inline ignore markers. -This does not create a tamper-proof gate against a malicious contributor. - -For CI gates, use trusted detector rules, an operator-owned configuration, and reviewed suppressions. -Use `--fail-on-unscannable` to reject incomplete coverage. -Supply a trusted baseline explicitly when the repository baseline is not an approved policy source. -Protect configuration and baseline files through repository access controls and code review. -Do not expose privileged CI credentials to code from untrusted pull requests. - -## Findings, coverage, and limits - -Reports distinguish detected findings from scan coverage. -`INCOMPLETE` means some requested content or history could not be scanned. -It does not mean that the unavailable content contains a secret. -`COMPLETE` means the scanner completes its selected scope within its implemented capabilities. -It does not establish universal credential detection or include deliberately excluded files. -Review exclusions and coverage warnings with the findings. - -The scanner enforces these limits: - -| Resource | Limit | -| --- | --- | -| Raw logical input or Git addition hunk | 16 MiB | -| Line | 1 MiB | -| Path records | 100,000 | -| Findings per input and aggregate checks | 10,000 | -| Retained finding text and metadata per input and aggregate checks | 32 MiB | -| Concurrent filesystem batch | Four files | -| Retained Git standard error | 64 KiB | - -`--max-file-size` lowers the raw input ceiling in filesystem, stdin, staged, and history modes. -Larger values do not raise the hard ceiling. -Limits produce incomplete coverage or a runtime error. -Partial input is not accepted as a clean scan or used to update a baseline. -Batch buffers, decoded content, detector definitions, paths, and reports also consume memory. -The retained-text limit is not a process memory limit. -Execution has no built-in wall-clock deadline; enforce a CI job timeout. - -Multiline matching uses bounded complete inputs rather than fixed overlap windows. -Git text scans still inspect additions, not complete snapshots of every historical text file. -Credentials split across separate hunks or commits can therefore be missed. -Run a filesystem scan of the checked-out tree as well as the required Git scan. -Text rendered as binary uses actual Git blobs, including historical and deleted versions. -True binary content remains unscannable. -Shallow history remains incomplete until you fetch full history outside the scanner. -Archives, arbitrary obfuscation, recursive encoding, and complete language parsing are not supported. - -## Secret storage and output - -Reports redact matched content unless you explicitly use `--show-secrets`. -Do not upload unredacted reports to shared CI artifacts or logs. -The process still holds matched content in memory while scanning. -Restrict access to the scanner process and its environment. - -Baseline hashes use domain-separated SHA-256 without a salt or keyed secret. -An attacker who obtains a baseline can test guesses for weak passwords offline. -Review this disclosure risk before publishing a baseline. -Rotate leaked credentials; adding a baseline entry does not remove a leak. - -Report and baseline writes use private same-directory temporary files and atomic replacement. -They reject destination symlinks and unsafe immediate parent directories. -Control the parent directory and all ancestors. -This is not a defense against an attacker who can replace those ancestors. -The implementation does not promise directory-sync durability after power loss. - -## Accuracy evidence - -`tests/accuracy_corpus.toml` contains 34 synthetic cases: 20 credentials and 14 noncredentials. -The scanner reports the expected detector for every positive case and no findings for every negative case. -The measured case-level results are 20 true positives, 14 true negatives, zero false positives, and zero false negatives. -Precision and recall are both 1.000 on this corpus only. -These results do not measure real-world accuracy or prove that other provider formats are covered. -The separate false-positive corpus retains its two documented residual findings. -Expand the corpus when real reports reveal another failure class. -Add both the unwanted finding and a similar credential that must remain detectable. - -## Synthetic workload measurements - -Run the benchmark with an optimized binary: - -```sh -cargo build --locked --release -uv run --no-project --no-config python -B scripts/benchmark.py target/release/key-watch -``` - -The script removes its generated files after execution. -The measured run uses macOS and the release build of this source tree. -The ordinary workload contains 2,000 files, 400,000 lines, and 9,600,000 bytes. - -| Workload | Seconds | Peak resident bytes | Exit | Coverage | -| --- | ---: | ---: | ---: | --- | -| Many small files | 0.533 | 49,119,232 | 0 | Complete | -| Oversized line | 0.244 | 34,996,224 | 1 | Incomplete | -| Finding budget exceeded | 0.136 | 32,948,224 | 1 | Incomplete | -| 100,000 rejected multiline matches | 0.210 | 31,784,960 | 0 | Complete | - -These are single-run measurements, not throughput guarantees or comparisons with other scanners. -No production CI environment, Windows runtime, or container runtime benchmark is measured here. -Measure your repository and CI environment before you rely on its scan duration or memory use. - -### Existing repository measurement - -A local Rust workspace scan excludes `target/**` and `**/node_modules/**` and disables configuration and baseline discovery. -It reaches the finding or retained-text budget after 20.328 seconds, with 126,812,160 peak resident bytes. -The command exits with status 2 and does not produce a report. -`--exit-mode always` does not suppress this runtime failure. -This measurement does not establish complete coverage of that workspace. -A source partition completes 340 files and 355,501 lines in 1.992 seconds, with 81,838,080 peak resident bytes. -Its report has complete coverage and 102 findings; the finding status is `FAIL`. -The command exits with status 0 because it uses `--exit-mode always`. -These findings are not labeled, so this measurement does not establish accuracy. -Baseline filtering occurs after scanning and does not bypass the raw-finding budget. -Use reviewed scan partitions and measure them before deployment. -Do not treat a budget failure as a clean result or increase limits without measuring the effect. - -## Deployment and release evidence - -JSON and SARIF reports contain a fingerprint of the effective detector definitions. -JSON also identifies the scanner version; SARIF identifies it in the tool driver. -The fingerprint covers patterns, keywords, allowlists, validators, entropy thresholds, and severity. -It is not a signature, binary attestation, or complete record of exclusions and baseline policy. -Record the scanned commit, command, configuration, baseline, and binary digest with your CI evidence. - -Download checksums from the same release detect corruption, not compromise of the release publisher. -Pin Action references and container digests when you require reproducible inputs. -The Action supports Linux x64 and macOS, not Windows or Linux ARM64. -Reinstall hooks after changing the binary and hook templates. -Local tests do not validate an unpublished binary through the public release download path. -Release provenance, independently authenticated binaries, and platform deployment tests remain separate operational requirements. - -Do not claim that this project is an independently audited or complete production security gate. -Use additional secret-management controls, credential rotation, and review processes. diff --git a/src/scanner.rs b/src/scanner.rs index ed93ad0..d62eee5 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -202,7 +202,9 @@ fn scan_git_history( command, |stderr| ScannerError::GitLogNonZero { stderr }, |reader| { - scan_staged_diff_with_limit( + let mut sizes = staged::GitObjectSizes::new(&repo_root)?; + let mut object_size = |oid: &str| sizes.size(oid); + let result = scan_staged_diff_with_limit( reader, &exclude_patterns, excluded_baseline, @@ -212,8 +214,11 @@ fn scan_git_history( staged::DiffScanPolicy { max_bytes: limits::input_limit(args.max_file_size)?, scan_lockfiles: args.scan_lockfiles, + object_size: &mut object_size, }, - ) + )?; + sizes.finish()?; + Ok(result) }, |source| ScannerError::RunGitLog { source }, )?; @@ -275,7 +280,9 @@ fn scan_staged( command, |stderr| ScannerError::GitDiffNonZero { stderr }, |reader| { - scan_staged_diff_with_limit( + let mut sizes = staged::GitObjectSizes::new(&repo_root)?; + let mut object_size = |oid: &str| sizes.size(oid); + let result = scan_staged_diff_with_limit( reader, &exclude_patterns, excluded_baseline, @@ -285,8 +292,11 @@ fn scan_staged( staged::DiffScanPolicy { max_bytes: limits::input_limit(args.max_file_size)?, scan_lockfiles: args.scan_lockfiles, + object_size: &mut object_size, }, - ) + )?; + sizes.finish()?; + Ok(result) }, |source| ScannerError::RunGitDiff { source }, )?; diff --git a/src/scanner/lines.rs b/src/scanner/lines.rs index 12eb5bd..c95577e 100644 --- a/src/scanner/lines.rs +++ b/src/scanner/lines.rs @@ -18,10 +18,8 @@ fn is_inline_suppressed(lowered_line: &str) -> bool { lowered_line.contains(INLINE_SUPPRESS) } -/// Reads one line into `raw_line` (cleared first), returning `false` at end -/// of stream. Strips a trailing `\n` and `\r`. The caller decodes lossily: -/// one invalid byte must not abort a scan. Shared by the stream and staged -/// parsers so terminator handling exists in exactly one place. +/// Reads a bounded patch line. Allows one prefix byte and a CRLF terminator. +/// The parser checks content length after removing the patch prefix. pub(super) fn read_raw_line( reader: &mut ReaderType, path: &str, @@ -43,7 +41,7 @@ pub(super) fn read_raw_line( .iter() .position(|byte| *byte == b'\n') .map_or(available.len(), |position| position + 1); - if take > MAX_LINE_BYTES.saturating_sub(raw_line.len()) { + if take > (MAX_LINE_BYTES + 3).saturating_sub(raw_line.len()) { return Err(ScannerError::ResourceLimit { reason: format!("Line exceeds {MAX_LINE_BYTES} bytes: {path}"), }); @@ -62,12 +60,18 @@ pub(super) fn read_raw_line( if raw_line.last() == Some(&b'\n') { raw_line.pop(); } - if raw_line.last() == Some(&b'\r') { - raw_line.pop(); - } Ok(true) } +pub(super) fn check_line_length(line: &str, path: &str) -> Result<(), ScannerError> { + if line.len() > MAX_LINE_BYTES { + return Err(ScannerError::ResourceLimit { + reason: format!("Line exceeds {MAX_LINE_BYTES} bytes: {path}"), + }); + } + Ok(()) +} + /// Lowercases `src` into `buf` without allocating a fresh string per line. /// ASCII input (the overwhelming majority of scanned bytes) takes a /// byte-per-byte fast path; the Unicode path is char-by-char and an order of @@ -439,11 +443,7 @@ pub(super) fn scan_content( )?; for (line_idx, line) in content.lines().enumerate() { - if line.len() > MAX_LINE_BYTES { - return Err(ScannerError::ResourceLimit { - reason: format!("Line exceeds {MAX_LINE_BYTES} bytes: {path}"), - }); - } + check_line_length(line, path)?; total_lines += 1; scan_line_detectors( line, diff --git a/src/scanner/staged.rs b/src/scanner/staged.rs index e2b60d3..ed046d4 100644 --- a/src/scanner/staged.rs +++ b/src/scanner/staged.rs @@ -7,13 +7,137 @@ use crate::report::{Finding, ScanMetadata}; use crate::scanner::ScannerError; use crate::scanner::files::{is_baseline_file, is_default_excluded_file, matches_exclude_patterns}; use crate::scanner::lines::{ - LineScanContext, LineScratch, read_raw_line, scan_content, scan_line_detectors, - scan_multiline_chunk, + LineScanContext, LineScratch, check_line_length, read_raw_line, scan_content, + scan_line_detectors, scan_multiline_chunk, }; use glob::Pattern; -use std::io::{BufRead, BufReader}; +use std::io::{BufRead, BufReader, Read, Write}; use std::path::{Path, PathBuf}; +fn drain_git_stderr(stderr: Option) -> std::thread::JoinHandle { + std::thread::spawn(move || { + let mut buffer = Vec::new(); + if let Some(mut stderr) = stderr { + let mut chunk = [0u8; 8192]; + while let Ok(count) = stderr.read(&mut chunk) { + if count == 0 { + break; + } + let retained = count.min((64 * 1024usize).saturating_sub(buffer.len())); + buffer.extend_from_slice(&chunk[..retained]); + } + } + String::from_utf8_lossy(&buffer).into_owned() + }) +} + +/// One bounded request-response session for text post-image size checks. +pub(super) struct GitObjectSizes { + child: std::process::Child, + stdin: Option, + stdout: BufReader, + stderr: Option>, + cache: std::collections::BTreeMap>, + finished: bool, +} + +impl GitObjectSizes { + pub(super) fn new(repo_root: &Path) -> Result { + let mut child = std::process::Command::new("git") + .current_dir(repo_root) + .args(["cat-file", "--batch-check"]) + .stdin(std::process::Stdio::piped()) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()) + .spawn() + .map_err(|source| ScannerError::RunGitCatFile { source })?; + let stdin = child.stdin.take(); + let stdout = child.stdout.take().expect("piped Git stdout"); + let stderr = drain_git_stderr(child.stderr.take()); + Ok(Self { + child, + stdin, + stdout: BufReader::new(stdout), + stderr: Some(stderr), + cache: std::collections::BTreeMap::new(), + finished: false, + }) + } + + pub(super) fn size(&mut self, oid: &str) -> Result, ScannerError> { + if let Some(size) = self.cache.get(oid) { + return Ok(*size); + } + if self.cache.len() >= MAX_PATHS { + return Err(ScannerError::ResourceLimit { + reason: "Git object count exceeds the scan budget".to_string(), + }); + } + let stdin = self.stdin.as_mut().expect("open Git size session"); + writeln!(stdin, "{oid}") + .and_then(|()| stdin.flush()) + .map_err(|source| ScannerError::RunGitCatFile { source })?; + let mut response = Vec::new(); + Read::by_ref(&mut self.stdout) + .take(257) + .read_until(b'\n', &mut response) + .map_err(|source| ScannerError::RunGitCatFile { source })?; + let size = parse_object_size_response(oid, &response)?; + self.cache.insert(oid.to_string(), size); + Ok(size) + } + + pub(super) fn finish(mut self) -> Result<(), ScannerError> { + self.stdin.take(); + let status = self + .child + .wait() + .map_err(|source| ScannerError::GitProcess { source })?; + self.finished = true; + let stderr = self.stderr.take().unwrap().join().unwrap_or_default(); + if !status.success() { + return Err(ScannerError::ResourceLimit { + reason: format!( + "Git object size lookup failed: {}", + summarize_git_stderr(&stderr) + ), + }); + } + Ok(()) + } +} + +impl Drop for GitObjectSizes { + fn drop(&mut self) { + self.stdin.take(); + if !self.finished { + let _ = self.child.kill(); + let _ = self.child.wait(); + } + if let Some(stderr) = self.stderr.take() { + let _ = stderr.join(); + } + } +} + +fn parse_object_size_response(oid: &str, response: &[u8]) -> Result, ScannerError> { + let invalid = || ScannerError::ResourceLimit { + reason: "Invalid Git object size response".to_string(), + }; + if response.len() > 256 || response.last() != Some(&b'\n') { + return Err(invalid()); + } + let response = std::str::from_utf8(response).map_err(|_| invalid())?; + let fields: Vec<_> = response.trim_end_matches('\n').split(' ').collect(); + match fields.as_slice() { + [returned_oid, "missing"] if *returned_oid == oid => Ok(None), + [returned_oid, "blob", size] if *returned_oid == oid => { + size.parse::().map(Some).map_err(|_| invalid()) + } + _ => Err(invalid()), + } +} + /// Runs `command`, feeds its stdout to `scan`, and reaps the child process. /// /// Both git-backed scan modes share this so the process lifetime is handled @@ -35,22 +159,7 @@ pub(super) fn scan_git_output( let stdout = child.stdout.take().ok_or(ScannerError::CaptureGitStdout)?; // Drained on its own thread: git can fill the stderr pipe (a full usage // dump) while this process is still reading stdout, deadlocking both. - let stderr = child.stderr.take(); - let stderr_reader = std::thread::spawn(move || { - use std::io::Read; - let mut buffer = Vec::new(); - if let Some(mut stderr) = stderr { - let mut chunk = [0u8; 8192]; - while let Ok(count) = stderr.read(&mut chunk) { - if count == 0 { - break; - } - let retained = count.min((64 * 1024usize).saturating_sub(buffer.len())); - buffer.extend_from_slice(&chunk[..retained]); - } - } - String::from_utf8_lossy(&buffer).into_owned() - }); + let stderr_reader = drain_git_stderr(child.stderr.take()); let scanned = scan(BufReader::new(stdout)); if scanned.is_err() { let _ = child.kill(); @@ -172,6 +281,7 @@ struct StagedDiffState { next_line_number: usize, hunk_start: usize, hunk_added: Vec, + pending_line: Option<(String, bool)>, hunk_bytes: usize, total_lines: usize, scanned_files: std::collections::BTreeSet, @@ -179,10 +289,31 @@ struct StagedDiffState { unscannable_from_diff: Vec, unscannable_files: Vec, current_oids: Vec, + post_image_oid: Option, history_blobs: std::collections::BTreeSet<(String, String)>, } impl StagedDiffState { + fn flush_pending_line( + &mut self, + preserve_cr: bool, + context: &LineScanContext<'_>, + scratch: &mut LineScratch, + findings: &mut Vec, + ) -> Result<(), ScannerError> { + let Some((mut content, added)) = self.pending_line.take() else { + return Ok(()); + }; + if !preserve_cr && content.ends_with('\r') { + content.pop(); + } + check_line_length(&content, "")?; + if added { + self.handle_added_line(&content, context, scratch, findings)?; + } + Ok(()) + } + /// Scans the buffered added lines of the hunk that just ended, or the /// whole diff when the stream ends. fn flush_hunk( @@ -322,6 +453,7 @@ pub(super) fn scan_staged_diff( multiline_detectors: &[&Detector], line_detectors: &[&Detector], ) -> Result { + let mut object_size = |_oid: &str| Ok(Some(0)); scan_staged_diff_with_limit( reader, exclude_patterns, @@ -332,13 +464,15 @@ pub(super) fn scan_staged_diff( DiffScanPolicy { max_bytes: MAX_INPUT_BYTES, scan_lockfiles: false, + object_size: &mut object_size, }, ) } -pub(super) struct DiffScanPolicy { +pub(super) struct DiffScanPolicy<'a> { pub max_bytes: u64, pub scan_lockfiles: bool, + pub object_size: &'a mut dyn FnMut(&str) -> Result, ScannerError>, } pub(super) fn scan_staged_diff_with_limit( @@ -348,7 +482,7 @@ pub(super) fn scan_staged_diff_with_limit( base_dir: &Path, multiline_detectors: &[&Detector], line_detectors: &[&Detector], - policy: DiffScanPolicy, + policy: DiffScanPolicy<'_>, ) -> Result { let context = LineScanContext::new(line_detectors); let mut findings = Vec::new(); @@ -370,19 +504,30 @@ pub(super) fn scan_staged_diff_with_limit( } let line = String::from_utf8_lossy(&raw_line); + if state.in_hunk && line.starts_with('\\') { + // A no-newline marker preserves a final CR as file content. + check_line_length(&line[1..], "")?; + state.flush_pending_line(true, &context, &mut scratch, &mut findings)?; + continue; + } + state.flush_pending_line(false, &context, &mut scratch, &mut findings)?; if state.in_hunk { if let Some(content) = line.strip_prefix('+') { - state.handle_added_line(content, &context, &mut scratch, &mut findings)?; + state.pending_line = Some((content.to_string(), true)); continue; } - if line.starts_with('-') || line.starts_with('\\') { + if let Some(content) = line.strip_prefix('-') { + state.pending_line = Some((content.to_string(), false)); continue; } } + let line = line.strip_suffix('\r').unwrap_or(&line); + check_line_length(line, "")?; + if line.starts_with("@@") { state.flush_hunk(multiline_detectors, &mut findings, &mut scratch.budget)?; - state.hunk_start = parse_hunk_new_start(&line); + state.hunk_start = parse_hunk_new_start(line); state.next_line_number = state.hunk_start; state.in_hunk = true; continue; @@ -393,20 +538,29 @@ pub(super) fn scan_staged_diff_with_limit( state.in_hunk = false; state.current_path = None; state.current_oids.clear(); + state.post_image_oid = None; continue; } if let Some(index) = line.strip_prefix("index ") { - state.current_oids = index + let oids: Vec<_> = index .split_whitespace() .next() .unwrap_or("") .split("..") - .filter(|oid| { - (oid.len() == 40 || oid.len() == 64) - && oid.bytes().all(|byte| byte.is_ascii_hexdigit()) - && oid.bytes().any(|byte| byte != b'0') - }) + .collect(); + let valid = |oid: &str| { + (oid.len() == 40 || oid.len() == 64) + && oid.bytes().all(|byte| byte.is_ascii_hexdigit()) + && oid.bytes().any(|byte| byte != b'0') + }; + state.post_image_oid = match oids.as_slice() { + [_, new] if valid(new) => Some((*new).to_string()), + _ => None, + }; + state.current_oids = oids + .into_iter() + .filter(|oid| valid(oid)) .map(str::to_string) .collect(); } @@ -420,21 +574,13 @@ pub(super) fn scan_staged_diff_with_limit( policy.scan_lockfiles, ); if let Some(path) = state.current_path.as_ref() { - for oid in state.current_oids.iter().rev().take(1) { - let output = std::process::Command::new("git") - .current_dir(base_dir) - .args(["cat-file", "-s", oid]) - .output() - .map_err(|source| ScannerError::RunGitCatFile { source })?; - let size = String::from_utf8_lossy(&output.stdout) - .trim() - .parse::(); - if !output.status.success() || size.map_or(true, |size| size > policy.max_bytes) - { - state.unscannable_files.push(path.clone()); - state.current_path = None; - break; - } + let size = match state.post_image_oid.as_deref() { + Some(oid) => (policy.object_size)(oid)?, + None => None, + }; + if size.is_none_or(|size| size > policy.max_bytes) { + state.unscannable_files.push(path.clone()); + state.current_path = None; } } continue; @@ -461,6 +607,7 @@ pub(super) fn scan_staged_diff_with_limit( } } + state.flush_pending_line(false, &context, &mut scratch, &mut findings)?; state.flush_hunk(multiline_detectors, &mut findings, &mut scratch.budget)?; let metadata = ScanMetadata { @@ -617,12 +764,68 @@ mod tests { use crate::scanner::test_support::make_test_detector as make_detector; use std::io::Cursor; + #[test] + fn object_size_responses_require_matching_blob_identity_and_bounded_sizes() { + for oid in ["1".repeat(40), "2".repeat(64)] { + let response = format!("{oid} blob 1048576\n"); + assert_eq!( + parse_object_size_response(&oid, response.as_bytes()).unwrap(), + Some(1048576) + ); + let missing = format!("{oid} missing\n"); + assert_eq!( + parse_object_size_response(&oid, missing.as_bytes()).unwrap(), + None + ); + for response in [ + String::new(), + format!("{oid} blob 1"), + format!("{oid} blob 18446744073709551616\n"), + format!("{} blob 1\n", "3".repeat(40)), + format!("{oid} tree 1\n"), + format!("{oid} blob 1 extra\n"), + format!("{}\n", "1".repeat(256)), + ] { + assert!( + parse_object_size_response(&oid, response.as_bytes()).is_err(), + "{response:?}" + ); + } + } + } + + #[test] + fn missing_or_invalid_post_image_ids_cannot_skip_size_limits() { + let detector = make_detector("Line", r"SECRET_\w+", "Test", "HIGH"); + for index in [ + "", + "index 1111111111111111111111111111111111111111..invalid 100644", + "index 1111111111111111111111111111111111111111..0000000000000000000000000000000000000000 100644", + ] { + let diff = format!( + "diff --git a/input.env b/input.env\n{index}\n--- a/input.env\n+++ b/input.env\n@@ -0,0 +1 @@\n+SECRET_VALUE\n" + ); + let result = scan_staged_diff( + Cursor::new(diff), + &[], + None, + Path::new("."), + &[], + &[&detector], + ) + .unwrap(); + assert!(!result.metadata.is_complete(), "{index}"); + assert_eq!(result.metadata.unscannable_files, ["input.env"]); + assert!(result.findings.is_empty()); + } + } + #[test] fn test_scan_staged_diff_preserves_plus_prefixed_added_content() { let detector = make_detector("Line", r"SECRET_\w+", "Test", "HIGH"); let line_detectors = vec![&detector]; let diff = "diff --git a/notes.txt b/notes.txt\n\ - index aabbcc0..ddeeff1 100644\n\ + index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n\ --- /dev/null\n\ +++ b/notes.txt\n\ @@ -0,0 +1,3 @@\n\ @@ -692,11 +895,13 @@ mod tests { let detector = make_detector("Line", r"SECRET_\w+", "Test", "HIGH"); let line_detectors = vec![&detector]; let diff = "diff --git a/first.txt b/first.txt\n\ + index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n\ --- a/first.txt\n\ +++ b/first.txt\n\ @@ -0,0 +7 @@\n\ +SECRET_A\n\ diff --git a/second.txt b/second.txt\n\ + index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n\ --- a/second.txt\n\ +++ b/second.txt\n\ @@ -0,0 +2 @@\n\ @@ -730,7 +935,7 @@ mod tests { let detector = make_detector("Line", r"SECRET_\w+", "Test", "HIGH"); let line_detectors = vec![&detector]; let diff = "diff --git a/img.png b/img.png\n\ - index aabbcc0..ddeeff1 100644\n\ + index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n\ Binary files a/img.png and b/img.png differ\n"; let staged = scan_staged_diff( @@ -790,6 +995,7 @@ mod tests { let line_detectors = vec![&detector]; let mut diff: Vec = Vec::new(); diff.extend_from_slice(b"diff --git a/legacy.csv b/legacy.csv\n"); + diff.extend_from_slice(b"index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n"); diff.extend_from_slice(b"--- a/legacy.csv\n"); diff.extend_from_slice(b"+++ b/legacy.csv\n"); diff.extend_from_slice(b"@@ -0,0 +1,2 @@\n"); @@ -819,6 +1025,7 @@ mod tests { let detector = make_detector("Block", r"(?s)BEGIN KEY.*END KEY", "Test", "HIGH"); let multiline_detectors = vec![&detector]; let diff = "diff --git a/key.pem b/key.pem\n\ + index 1111111111111111111111111111111111111111..2222222222222222222222222222222222222222 100644\n\ --- a/key.pem\n\ +++ b/key.pem\n\ @@ -0,0 +5,3 @@\n\ diff --git a/tests/baseline_tests.rs b/tests/baseline_tests.rs index 08834be..9ab608d 100644 --- a/tests/baseline_tests.rs +++ b/tests/baseline_tests.rs @@ -101,6 +101,30 @@ fn check_credential_changes_against_baseline(staged: bool) { "GenericKeyValueDetector", r#"api_key="aB3xK9mQ2pR7/z8""#, ), + ( + "password: ${{ secrets.FIXTURE_TOKEN }}Literal783!", + "password: ${{ secrets.FIXTURE_TOKEN }}Literal629!", + "PasswordDetector", + "password: ${{ secrets.FIXTURE_TOKEN }}Literal629!", + ), + ( + "password: ${{ secrets.FIXTURE_TOKEN }}: dummy", + "password: ${{ secrets.FIXTURE_TOKEN }}: placeholder", + "PasswordDetector", + "password: ${{ secrets.FIXTURE_TOKEN }}: placeholder", + ), + ( + "api_key = aB3xK9mQ2pR7=dummy", + "api_key = aB3xK9mQ2pR7=placeholder", + "GenericKeyValueDetector", + "api_key = aB3xK9mQ2pR7=placeholder", + ), + ( + "mongodb://user:Fixture783!@localhost:27017", + "mongodb://user:Fixture629!@localhost:27017", + "MongoDBConnectionStringDetector", + "mongodb://user:Fixture629!@localhost:27017", + ), ] { if baseline.exists() { fs::remove_file(&baseline).expect("reset fixture baseline"); diff --git a/tests/hardening_tests.rs b/tests/hardening_tests.rs index 2dfb56d..2b79986 100644 --- a/tests/hardening_tests.rs +++ b/tests/hardening_tests.rs @@ -562,6 +562,140 @@ fn staged_size_limit_applies_to_the_post_image_not_the_previous_blob() { ); } +#[test] +fn line_limits_count_content_not_git_prefixes_or_line_endings() { + use std::io::Write; + use std::process::Stdio; + + const LIMIT: usize = 1024 * 1024; + let credential = "AWS_ACCESS_KEY_ID=AKIAABCDEFGHIJKLMNOP"; + for ending in ["\r", "", "\n", "\r\n"] { + for extra_byte in [0, 1] { + let dir = repository(); + let content = format!( + "+{}{credential}{ending}", + "#".repeat(LIMIT - credential.len() - 1 - usize::from(ending == "\r") + extra_byte) + ); + fs::write(dir.path().join("boundary.env"), &content).unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Boundary fixture"]); + for arguments in [ + vec!["boundary.env", "--fail-on-unscannable"], + vec!["--git-history", "--fail-on-unscannable"], + ] { + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "--no-config-discovery", + "--no-baseline-discovery", + "--verbose", + ]) + .args(arguments) + .output() + .unwrap(); + check_boundary_scan(&output, extra_byte == 0); + } + git(dir.path(), &["rm", "--cached", "boundary.env"]); + git( + dir.path(), + &["commit", "--quiet", "-m", "Remove boundary fixture"], + ); + git(dir.path(), &["add", "boundary.env"]); + let output = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .current_dir(dir.path()) + .args([ + "scan", + "--staged", + "--no-config-discovery", + "--no-baseline-discovery", + "--verbose", + "--fail-on-unscannable", + ]) + .output() + .unwrap(); + check_boundary_scan(&output, extra_byte == 0); + + let mut child = Command::new(env!("CARGO_BIN_EXE_key-watch")) + .args([ + "scan", + "--stdin", + "--no-config-discovery", + "--no-baseline-discovery", + "--verbose", + ]) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + child + .stdin + .take() + .unwrap() + .write_all(content.as_bytes()) + .unwrap(); + check_boundary_scan(&child.wait_with_output().unwrap(), extra_byte == 0); + } + } +} + +fn check_boundary_scan(output: &Output, accepted: bool) { + if accepted { + assert_eq!(output.status.code(), Some(1), "{output:?}"); + let report: serde_json::Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(report["coverage"], "COMPLETE"); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "AWSKeyDetector" && finding["line_number"] == 1 + }) + ); + } else if let Ok(report) = serde_json::from_slice::(&output.stdout) { + assert_eq!(output.status.code(), Some(1)); + assert_eq!(report["coverage"], "INCOMPLETE"); + } else { + assert_eq!(output.status.code(), Some(2), "{output:?}"); + assert!(String::from_utf8_lossy(&output.stderr).contains("Line exceeds")); + } +} + +#[test] +fn git_preserves_unterminated_carriage_return_content() { + let dir = repository(); + fs::write(dir.path().join("ending.env"), "TRAILING\r").unwrap(); + fs::write( + dir.path().join("rules.toml"), + "[[rules]]\nname = 'CarriageReturnFixture'\nfinding_type = 'Fixture'\npattern = 'TRAILING\\r$'\n", + ).unwrap(); + git(dir.path(), &["add", "ending.env"]); + for mode in ["ending.env", "--staged", "--git-history"] { + if mode == "--git-history" { + git(dir.path(), &["commit", "--quiet", "-m", "CR fixture"]); + } + let (output, report) = scan( + dir.path(), + &[mode, "--config", "rules.toml", "--show-secrets"], + ); + assert_eq!(output.status.code(), Some(1), "{mode}: {output:?}"); + assert_eq!(report["coverage"], "COMPLETE"); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["plugin_name"] == "CarriageReturnFixture" + && finding["matched_content"] == "TRAILING\r" + }), + "{mode}: {report}" + ); + } +} + #[test] fn lockfile_inclusion_is_explicit_and_consistent_across_modes() { let dir = repository(); diff --git a/tests/scanner_tests.rs b/tests/scanner_tests.rs index 7781c09..13d7877 100644 --- a/tests/scanner_tests.rs +++ b/tests/scanner_tests.rs @@ -241,6 +241,14 @@ fn test_credential_literals_do_not_borrow_identifier_or_placeholder_exemptions() let file = directory.path().join("credentials.env"); for (assignment, detector_name) in [ (r#"api_key = "q7m2_c9v8_x5z1""#, "GenericKeyValueDetector"), + ( + "api_key = aB3xK9mQ2pR7=placeholder", + "GenericKeyValueDetector", + ), + ( + "api_key = aB3xK9mQ2pR7=globalState", + "GenericKeyValueDetector", + ), (r#"api_key = "1234-5678-9012""#, "GenericKeyValueDetector"), (r#"api_key = "changemeA7q9Z2""#, "GenericKeyValueDetector"), ( @@ -272,6 +280,112 @@ fn test_credential_literals_do_not_borrow_identifier_or_placeholder_exemptions() } } +#[test] +fn test_workflow_password_references_do_not_hide_literal_values() { + let directory = tempfile::tempdir().unwrap(); + let file = directory.path().join("workflow.yml"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + for (assignment, should_report) in [ + ("password: ${{ secrets.GITHUB_TOKEN }}", false), + ("PASSWORD = ${{secrets.DATABASE_PASSWORD}}", false), + (r#"password: "${{ secrets.GITHUB_TOKEN }}""#, true), + ("password: ${{ secrets.GITHUB_TOKEN }}Literal783!", true), + ("password: ${{ secrets.GITHUB_TOKEN }} Literal783!", true), + ("password: ${{ secrets.GITHUB_TOKEN }}: dummy", true), + ( + "password: ${{ secrets.GITHUB_TOKEN }}=config.password", + true, + ), + ( + "password: ${{ secrets.GITHUB_TOKEN || 'Literal783!' }}", + true, + ), + ("password: ${{ 'Literal783!' }}", true), + ("password: ${{ secrets.GITHUB_TOKEN", true), + ] { + fs::write(&file, assignment).unwrap(); + let (findings, _) = run_scan(&args, None).unwrap(); + let password = findings + .iter() + .find(|finding| finding.detector_name == "PasswordDetector"); + if should_report { + assert_eq!(password.unwrap().matched_content, assignment); + } else { + assert!( + password.is_none(), + "Reference reports a password: {assignment}" + ); + } + } +} + +#[test] +fn test_mongodb_scheme_references_do_not_consume_unrelated_source() { + let directory = tempfile::tempdir().unwrap(); + let file = directory.path().join("source.txt"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + for source in [ + "keywords = [\"mongodb://\", \"mongodb+srv://\"]\npattern = \"user:Fixture783!@localhost\"", + "mongodb://\nuser:Fixture783!@localhost", + "mongodb://user:\nFixture783!@localhost", + "mongodb://user:Fixture783!@\nlocalhost", + ] { + fs::write(&file, source).unwrap(); + let (findings, _) = run_scan(&args, None).unwrap(); + assert!( + !findings + .iter() + .any(|finding| { finding.detector_name == "MongoDBConnectionStringDetector" }), + "Unrelated source reports a MongoDB credential: {source}" + ); + } +} + +#[test] +fn test_mongodb_credentials_stop_at_authority_boundaries() { + let directory = tempfile::tempdir().unwrap(); + let file = directory.path().join("connections.txt"); + let authorities = [ + "mongodb://user:Fixture783!@localhost:27017", + "mongodb+srv://user:p%40ss%3Aword%2F783@cluster.example.net", + "mongodb://user:Fixture783!@[::1]:27017", + ]; + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let mut expected = authorities; + expected.sort_unstable(); + for suffix in ["", "/db?authSource=admin"] { + for separator in [",", " ", "\", \""] { + let source = authorities + .map(|authority| format!("{authority}{suffix}")) + .join(separator); + fs::write(&file, source).unwrap(); + let (findings, _) = run_scan(&args, None).unwrap(); + let mut mongodb: Vec<_> = findings + .iter() + .filter(|finding| finding.detector_name == "MongoDBConnectionStringDetector") + .map(|finding| finding.matched_content.as_str()) + .collect(); + mongodb.sort_unstable(); + assert_eq!(mongodb, expected); + } + } +} + fn run_git_history_scan(current_dir: &Path, extra_args: &[&str]) -> Result { Command::new(env!("CARGO_BIN_EXE_key-watch")) .args(["scan", "--git-history"]) diff --git a/tests/utils_tests.rs b/tests/utils_tests.rs index 1f9b36f..e8b01a4 100644 --- a/tests/utils_tests.rs +++ b/tests/utils_tests.rs @@ -1,27 +1,17 @@ use key_watch::utils::write_to_file; -use std::env::temp_dir; use std::fs; -use std::time::{SystemTime, UNIX_EPOCH}; - -fn unique_temp_file(name: &str) -> std::path::PathBuf { - let timestamp = SystemTime::now() - .duration_since(UNIX_EPOCH) - .expect("System time should be after Unix epoch") - .as_millis(); - temp_dir().join(format!("key_watch_{name}_{timestamp}.txt")) -} +use tempfile::tempdir; #[test] fn test_write_to_file() { - let temp_file = unique_temp_file("test_output"); + let directory = tempdir().expect("create private output directory"); + let temp_file = directory.path().join("report.txt"); let content = "Temporary content written to file."; let path_str = temp_file.to_str().unwrap(); write_to_file(path_str, content).expect("Failed to write to file"); let read_back = fs::read_to_string(path_str).expect("Failed to read back"); assert_eq!(read_back, content, "Content should match"); - - fs::remove_file(temp_file).expect("Cleanup"); } #[test] @@ -32,14 +22,24 @@ fn test_portable_config_loading() { assert!(!detectors.is_empty(), "Should load at least one detector"); } +#[test] +fn test_write_to_file_replaces_existing_output() { + let directory = tempdir().expect("create private output directory"); + let output = directory.path().join("report.txt"); + fs::write(&output, "original content").unwrap(); + + write_to_file(output.to_str().unwrap(), "replacement content").unwrap(); + assert_eq!(fs::read_to_string(output).unwrap(), "replacement content"); +} + #[cfg(unix)] #[test] fn test_write_to_file_tightens_existing_permissions() { use std::os::unix::fs::PermissionsExt; - // `mode(0o600)` only applies at creation; a pre-existing world-readable - // report must be tightened before new content is written into it. - let temp_file = unique_temp_file("existing_output"); + // Atomic replacement must not preserve world-readable permissions. + let directory = tempdir().expect("create private output directory"); + let temp_file = directory.path().join("report.txt"); let path_str = temp_file.to_str().unwrap(); fs::write(path_str, "old content").expect("create pre-existing file"); let mut perms = fs::metadata(path_str).expect("stat file").permissions(); @@ -47,6 +47,7 @@ fn test_write_to_file_tightens_existing_permissions() { fs::set_permissions(path_str, perms).expect("set 0644"); write_to_file(path_str, "new content").expect("rewrite file"); + assert_eq!(fs::read_to_string(path_str).unwrap(), "new content"); let mode = fs::metadata(path_str) .expect("stat file") @@ -57,6 +58,19 @@ fn test_write_to_file_tightens_existing_permissions() { 0o600, "rewritten reports must be readable only by their owner" ); +} + +#[cfg(unix)] +#[test] +fn test_write_to_file_rejects_world_writable_parent_without_changing_output() { + use std::os::unix::fs::PermissionsExt; + + let directory = tempdir().expect("create output directory"); + let output = directory.path().join("report.txt"); + fs::write(&output, "original content").unwrap(); + fs::set_permissions(directory.path(), fs::Permissions::from_mode(0o777)).unwrap(); - let _ = fs::remove_file(path_str); + let error = write_to_file(output.to_str().unwrap(), "replacement content").unwrap_err(); + assert_eq!(error.kind(), std::io::ErrorKind::PermissionDenied); + assert_eq!(fs::read_to_string(output).unwrap(), "original content"); } From 159f8f724542764b166ee98dee2b7c32d4a55450 Mon Sep 17 00:00:00 2001 From: PiX <69745008+pixincreate@users.noreply.github.com> Date: Wed, 7 Oct 2026 21:45:54 +0530 Subject: [PATCH 3/3] scanner: Fix coverage and output edge cases Count filesystem visits across all operands and preserve findings when Git patches include gitlinks. Set output permissions before publication so a restrictive umask cannot remove owner access. Limit control-character filename fixtures to Unix and keep decoding checks portable. Add paired regressions for reported false positives. Assisted-by: GPT-6.1 Sol Signed-off-by: PiX <69745008+pixincreate@users.noreply.github.com> --- src/scanner.rs | 5 +- src/scanner/files.rs | 5 +- src/scanner/staged.rs | 49 +++++++++++++++++-- src/utils.rs | 5 ++ tests/accuracy_corpus.toml | 15 ++++++ tests/hardening_tests.rs | 96 +++++++++++++++++++++++++++++++++++++- tests/scanner_tests.rs | 35 ++++++++++++++ tests/utils_tests.rs | 38 +++++++++++++++ 8 files changed, 241 insertions(+), 7 deletions(-) diff --git a/src/scanner.rs b/src/scanner.rs index d62eee5..cfa1182 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -396,12 +396,14 @@ fn scan_filesystem( // Explicit operands are validated strictly: a typo'd path or an operand // the scanner will not read (symlink, device, FIFO) must not produce a // silent "No secrets found" pass. + let mut visited_paths = 0; for path_str in &args.paths { - if target_paths.len() + unlistable_dirs.len() >= limits::MAX_PATHS { + if visited_paths >= limits::MAX_PATHS { return Err(ScannerError::ResourceLimit { reason: "Input count exceeds the scan budget".to_string(), }); } + visited_paths += 1; let path = Path::new(path_str); let metadata = match fs::symlink_metadata(path) { Ok(metadata) => metadata, @@ -444,6 +446,7 @@ fn scan_filesystem( path_str, &mut unlistable_dirs, &exclude_patterns, + &mut visited_paths, )?; } else { return Err(ScannerError::ScanPathUnsupported { diff --git a/src/scanner/files.rs b/src/scanner/files.rs index e7adf89..3cbc8a2 100644 --- a/src/scanner/files.rs +++ b/src/scanner/files.rs @@ -110,6 +110,7 @@ pub(super) fn collect_files( root: &str, unlistable_dirs: &mut Vec, exclude_patterns: &[Pattern], + visited_paths: &mut usize, ) -> Result<(), ScannerError> { // A directory that cannot be listed hides everything beneath it; record // it as unscannable instead of silently reporting a clean scan, so @@ -125,12 +126,12 @@ pub(super) fn collect_files( } }; for entry in entries { - if targets.len() + unlistable_dirs.len() + directories.len() >= super::limits::MAX_PATHS - { + if *visited_paths >= super::limits::MAX_PATHS { return Err(ScannerError::ResourceLimit { reason: "Filesystem input count exceeds the scan budget".to_string(), }); } + *visited_paths += 1; let entry = match entry { Ok(entry) => entry, Err(_) => { diff --git a/src/scanner/staged.rs b/src/scanner/staged.rs index ed046d4..fd49107 100644 --- a/src/scanner/staged.rs +++ b/src/scanner/staged.rs @@ -131,8 +131,11 @@ fn parse_object_size_response(oid: &str, response: &[u8]) -> Result, let fields: Vec<_> = response.trim_end_matches('\n').split(' ').collect(); match fields.as_slice() { [returned_oid, "missing"] if *returned_oid == oid => Ok(None), - [returned_oid, "blob", size] if *returned_oid == oid => { - size.parse::().map(Some).map_err(|_| invalid()) + [returned_oid, kind, size] + if *returned_oid == oid && matches!(*kind, "blob" | "commit" | "tree" | "tag") => + { + let size = size.parse::().map_err(|_| invalid())?; + Ok((*kind == "blob").then_some(size)) } _ => Err(invalid()), } @@ -777,12 +780,21 @@ mod tests { parse_object_size_response(&oid, missing.as_bytes()).unwrap(), None ); + for kind in ["commit", "tree", "tag"] { + let response = format!("{oid} {kind} 166\n"); + assert_eq!( + parse_object_size_response(&oid, response.as_bytes()).unwrap(), + None + ); + } for response in [ String::new(), format!("{oid} blob 1"), format!("{oid} blob 18446744073709551616\n"), format!("{} blob 1\n", "3".repeat(40)), - format!("{oid} tree 1\n"), + format!("{oid} unknown 1\n"), + format!("{oid} tree invalid\n"), + format!("{} commit 1\n", "3".repeat(40)), format!("{oid} blob 1 extra\n"), format!("{}\n", "1".repeat(256)), ] { @@ -1066,6 +1078,19 @@ mod tests { Some("plain.txt") ); assert_eq!(parse_diff_target_path("/dev/null"), None); + for (escape, character) in [ + ('a', '\u{7}'), + ('b', '\u{8}'), + ('f', '\u{c}'), + ('v', '\u{b}'), + ] { + let quoted = format!("\"b/credentials\\{escape}.env\""); + let expected = format!("credentials{character}.env"); + assert_eq!( + parse_diff_target_path("ed).as_deref(), + Some(expected.as_str()) + ); + } } #[test] @@ -1078,5 +1103,23 @@ mod tests { parse_binary_marker_path("/dev/null and b/plain.bin differ"), Some(("plain.bin".to_string(), false)) ); + for (escape, character) in [ + ('a', '\u{7}'), + ('b', '\u{8}'), + ('f', '\u{c}'), + ('v', '\u{b}'), + ] { + let previous = format!("\"a/credentials\\{escape}.env\""); + let current = format!("\"b/credentials\\{escape}.env\""); + let expected = format!("credentials{character}.env"); + assert_eq!( + parse_binary_marker_path(&format!("{previous} and {current} differ")), + Some((expected.clone(), false)) + ); + assert_eq!( + parse_binary_marker_path(&format!("{previous} and /dev/null differ")), + Some((expected, true)) + ); + } } } diff --git a/src/utils.rs b/src/utils.rs index f556f69..9b45dfe 100644 --- a/src/utils.rs +++ b/src/utils.rs @@ -114,6 +114,11 @@ pub(crate) fn atomic_write(path: &Path, content: &[u8]) -> Result<()> { Err(error) => return Err(error), }; let result = (|| { + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + file.set_permissions(fs::Permissions::from_mode(0o600))?; + } file.write_all(content)?; file.sync_all()?; drop(file); diff --git a/tests/accuracy_corpus.toml b/tests/accuracy_corpus.toml index b05106a..f4afae7 100644 --- a/tests/accuracy_corpus.toml +++ b/tests/accuracy_corpus.toml @@ -121,3 +121,18 @@ input = 'example@example.com' [[cases]] name = "Base64-like identifier" input = 'PaymentMethodServiceEligibilityRequest' +[[cases]] +name = "Public Visa test number" +input = '{"card_number":"4111111111111111"}' +[[cases]] +name = "Public Amex test number" +input = '{"card_number":"378282246310005"}' +[[cases]] +name = "Public Mastercard test number" +input = '{"card_number":"5555555555554444"}' +[[cases]] +name = "Unquoted access token reference" +input = 'access_token: router_data_v2,' +[[cases]] +name = "Base64-like type in a declaration" +input = 'let response: PaymentMethodServiceEligibilityRequest;' diff --git a/tests/hardening_tests.rs b/tests/hardening_tests.rs index 2b79986..a94b5ad 100644 --- a/tests/hardening_tests.rs +++ b/tests/hardening_tests.rs @@ -443,7 +443,7 @@ fn binary_filenames_with_separator_text_cannot_become_lockfile_exclusions() { && finding["plugin_name"] == "AWSKeyDetector") ); } - +#[cfg(unix)] #[test] fn git_control_character_filenames_preserve_staged_and_deleted_history_paths() { let dir = repository(); @@ -696,6 +696,100 @@ fn git_preserves_unterminated_carriage_return_content() { } } +#[test] +fn gitlinks_are_unscannable_without_hiding_ordinary_file_findings() { + let dir = repository(); + fs::write(dir.path().join("credentials.env"), "ordinary text\n").unwrap(); + git(dir.path(), &["add", "."]); + git(dir.path(), &["commit", "--quiet", "-m", "Initial fixture"]); + let revision = Command::new("git") + .current_dir(dir.path()) + .args(["rev-parse", "HEAD"]) + .output() + .unwrap(); + assert!(revision.status.success()); + let oid = String::from_utf8(revision.stdout).unwrap(); + let credential = ["AKIA", "ABCDEFGHIJKLMNOP"].concat(); + fs::write( + dir.path().join("credentials.env"), + format!("AWS_ACCESS_KEY_ID={credential}\n"), + ) + .unwrap(); + git(dir.path(), &["add", "credentials.env"]); + git( + dir.path(), + &[ + "update-index", + "--add", + "--cacheinfo", + "160000", + oid.trim(), + "dependency", + ], + ); + for history in [false, true] { + if history { + git(dir.path(), &["commit", "--quiet", "-m", "Gitlink fixture"]); + } + let args = if history { + vec![ + "--git-history", + "--rev-range", + "HEAD~1..HEAD", + "--fail-on-unscannable", + ] + } else { + vec!["--staged", "--fail-on-unscannable"] + }; + let (output, report) = scan(dir.path(), &args); + assert_eq!(output.status.code(), Some(1), "{output:?}"); + assert_eq!(report["coverage"], "INCOMPLETE"); + assert_eq!(report["unscannable"]["count"], 1); + assert_eq!(report["unscannable"]["sample"][0], "dependency"); + assert!( + report["findings"] + .as_array() + .unwrap() + .iter() + .any(|finding| { + finding["file_path"] == "credentials.env" + && finding["plugin_name"] == "AWSKeyDetector" + }) + ); + } +} + +#[test] +fn directory_visit_budget_is_shared_across_explicit_roots() { + let dir = tempfile::tempdir().unwrap(); + let roots = [dir.path().join("first"), dir.path().join("second")]; + for root in &roots { + fs::create_dir(root).unwrap(); + for index in 0..500 { + fs::create_dir(root.join(format!("empty-{index}"))).unwrap(); + } + let (output, report) = scan(dir.path(), &[root.to_str().unwrap()]); + assert!(output.status.success(), "{output:?}"); + assert_eq!(report["coverage"], "COMPLETE"); + } + let mut command = Command::new(env!("CARGO_BIN_EXE_key-watch")); + command.current_dir(dir.path()).args([ + "scan", + "--no-config-discovery", + "--no-baseline-discovery", + "--exit-mode", + "always", + ]); + for root in &roots { + for _ in 0..100 { + command.arg(root); + } + } + let output = command.output().unwrap(); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + assert!(String::from_utf8_lossy(&output.stderr).contains("scan budget")); +} + #[test] fn lockfile_inclusion_is_explicit_and_consistent_across_modes() { let dir = repository(); diff --git a/tests/scanner_tests.rs b/tests/scanner_tests.rs index 13d7877..be6f7a4 100644 --- a/tests/scanner_tests.rs +++ b/tests/scanner_tests.rs @@ -3039,3 +3039,38 @@ fn test_json_escaped_private_key_is_detected() { fs::remove_dir_all(&test_dir).expect("cleanup"); } + +#[test] +fn test_token_references_are_not_literals_but_quoted_values_report() { + let directory = tempfile::tempdir().expect("create fixture directory"); + let file = directory.path().join("credentials.txt"); + let args = ScanArgs { + paths: vec![file.to_string_lossy().into_owned()], + no_config_discovery: true, + no_baseline_discovery: true, + ..Default::default() + }; + let reference = "access_token: router_data_v2"; + fs::write(&file, reference).expect("write reference"); + let (findings, metadata) = run_scan(&args, None).expect("scan reference"); + assert!(metadata.is_complete()); + assert!( + findings.is_empty(), + "a reference is not a credential: {findings:?}" + ); + + let (key, value) = reference.split_once(':').expect("reference assignment"); + let literal = format!("{key}: {:?}", value.trim()); + fs::write(&file, &literal).expect("write literal"); + let (findings, metadata) = run_scan(&args, None).expect("scan literal"); + assert!(metadata.is_complete()); + let credential = findings + .iter() + .find(|finding| finding.detector_name == "GenericKeyValueDetector") + .expect("a quoted value must not inherit the reference exemption"); + assert_eq!( + credential.matched_content, + format!("token: {:?}", value.trim()) + ); + assert_eq!(credential.line_number, 1); +} diff --git a/tests/utils_tests.rs b/tests/utils_tests.rs index e8b01a4..edbcf52 100644 --- a/tests/utils_tests.rs +++ b/tests/utils_tests.rs @@ -74,3 +74,41 @@ fn test_write_to_file_rejects_world_writable_parent_without_changing_output() { assert_eq!(error.kind(), std::io::ErrorKind::PermissionDenied); assert_eq!(fs::read_to_string(output).unwrap(), "original content"); } + +#[cfg(unix)] +#[test] +fn reports_remain_owner_readable_under_a_restrictive_umask() { + use std::os::unix::fs::PermissionsExt; + use std::process::Command; + + let directory = tempdir().unwrap(); + let input = directory.path().join("ordinary.txt"); + fs::write(&input, "ordinary text\n").unwrap(); + for existing in [false, true] { + let report = directory.path().join(format!("report-{existing}.json")); + if existing { + fs::write(&report, "original content").unwrap(); + } + let output = Command::new("sh") + .args(["-c", "umask 0777; exec \"$@\"", "keywatch-umask"]) + .arg(env!("CARGO_BIN_EXE_key-watch")) + .args([ + "scan", + "--no-config-discovery", + "--no-baseline-discovery", + "--output", + ]) + .arg(&report) + .arg(&input) + .output() + .unwrap(); + assert!(output.status.success(), "{output:?}"); + assert_eq!( + fs::metadata(&report).unwrap().permissions().mode() & 0o777, + 0o600 + ); + let content: serde_json::Value = + serde_json::from_slice(&fs::read(report).unwrap()).unwrap(); + assert_eq!(content["status"], "PASS"); + } +}