From 6ea785471ddeb7aebf29ae7bac06941f1a7443fd Mon Sep 17 00:00:00 2001 From: HackingGate Date: Fri, 9 Oct 2026 23:00:40 +0900 Subject: [PATCH 1/4] An ideographic zero reads with the letter after it, and a lookalike on a template comment is refused with its reason Part of #334, items 1, 2 and 3. Item 4 follows after #333. 1 and 2, ruled together. U+3007 IDEOGRAPHIC NUMBER ZERO is Han with a Latin O for its UTS #39 skeleton. lookalikes in src/guard/unicode.rs kept it in a Latin word only when every letter so far was drawn as Latin, so a kanji or kana run before U+3007 + `K` split the word after U+3007 and the disguise passed. It now reads such a letter with the letter AFTER it (new reading_scripts, resolved from the right): before a Latin letter it joins that Latin word wherever it sits, so kana + U+3007 + `K` + kana, U+4EF6 + U+3007 + `K` and U+4E8C + U+3007 + `K` are each refused; before a kanji it is the numeral zero and stays in the kanji run, so `API` + U+3007 + U+4EF6 and U+5168 U+3007 U+4EF6 pass; with no letter after it, it reads with the letter before it (`HELL` + U+3007 is refused). Residual: U+3007 + `GB` and the U+3007 U+3007 placeholder before a product name are refused, and `allow = ["U+3007"]` admits them; a Latin word ending in U+3007 run straight into a kanji passes. docs/REFERENCE.md says so in place of "stays one word". 3. comment_line_note says "The first non-blank line", and takes the rule's own finding test as a predicate. The commit-msg path of prevent_unusual_unicode now goes through the new unusual_unicode_in_message, which attaches the note when a lookalike is on the comment-opened candidate; the title seam does not, since every line of a title is a headline. The lookalike pass is split out of unusual_findings into lookalike_findings. Tests: an_ideographic_zero_inside_a_latin_word_is_a_lookalike (three disguise lines, `HELL` + U+3007, and three more kanji numerals passing), an_ideographic_zero_reads_with_the_letter_after_it, the_note_names_the_first_non_blank_line, a_lookalike_on_a_comment_opened_first_line_carries_the_note, and in tests/guard_cli.rs an_ideographic_zero_before_latin_is_refused_and_its_allowance_admits_it. No function or test is removed. Claude-Session: https://claude.ai/code/session_01QJkWXa2HxJ6q9TMNAGSNv4 --- docs/REFERENCE.md | 23 +++++-- src/guard/message.rs | 145 ++++++++++++++++++++++++++++++++++--------- src/guard/unicode.rs | 136 +++++++++++++++++++++++++++++++--------- tests/guard_cli.rs | 24 +++++++ 4 files changed, 261 insertions(+), 67 deletions(-) diff --git a/docs/REFERENCE.md b/docs/REFERENCE.md index 32635b6..941d04a 100644 --- a/docs/REFERENCE.md +++ b/docs/REFERENCE.md @@ -1449,9 +1449,18 @@ file guard bans: `API` or `README` run straight into a Japanese or Chinese subject is not a mixed word and passes. The exception is a letter at that boundary which is itself drawn like the other side's script: U+3007 IDEOGRAPHIC NUMBER ZERO is - Han with a Latin `O` for its skeleton, so `g` + U+3007 + `od` stays one word - and is refused, while a kanji numeral carrying it with no Latin letter beside - it passes. A skeleton made only of characters no script owns + Han with a Latin `O` for its skeleton. It is read with the letter after it, + as both languages read it: before a Latin letter it joins that Latin word, + whatever comes before it, so `g` + U+3007 + `od`, U+3007 + `K`, and a kanji + or kana run + U+3007 + `K` are each refused; before a kanji it is the numeral + zero and stays in the kanji run, so a kanji numeral carrying it and `API` + + U+3007 + U+4EF6 ("API, zero items") pass; with no letter after it, it is read + with the letter before it (`HELL` + U+3007 is refused). The cost is that + U+3007 written as a numeral or a placeholder right before Latin letters -- + U+3007 + `GB`, or the U+3007 U+3007 placeholder before a product name -- is + refused as well; the report names U+3007, and `allow = ["U+3007"]` admits + it. And a Latin word ending in U+3007 run straight into a kanji passes. A + skeleton made only of characters no script owns (the kanji for "one" skeletons to the katakana prolonged sound mark) is not "drawn like" any script. @@ -1473,9 +1482,11 @@ git's cleanup, and whether a line opening with `core.commentChar` (default `#`) is stripped depends on the cleanup mode (`git commit -m "#42 Fix"` keeps it, the editor strips it), so where the first such line opens with the comment character, the first line that does not is judged as well. Both rules read it -that way, so a `commit.template` whose first line is a non-ASCII comment is -refused under `ascii-only-commit-subject` although the editor would strip it; -the report says why, and a template opening with an ASCII line passes. A pushed commit was +that way, so a `commit.template` whose first non-blank line is a non-ASCII +comment is refused under `ascii-only-commit-subject`, and one whose first +non-blank line carries a lookalike word under `prevent-unusual-unicode`, +although the editor would strip it; either report says why, and a template +opening with an ASCII line passes. A pushed commit was already cleaned up, and its first line is its subject. A **title** a shim collects is a subject as well: `gh pr merge --subject` is one outright, and a pull-request title becomes one when the forge squash-merges it, on a branch no diff --git a/src/guard/message.rs b/src/guard/message.rs index 73023d9..6a4001d 100644 --- a/src/guard/message.rs +++ b/src/guard/message.rs @@ -283,22 +283,27 @@ fn unusual_findings(label: &str, text: &str, allowed: &[char], headlines: &[usiz .map_or_else(|| String::from("UNKNOWN"), |name| name.to_string()), )); } - if !headlines.contains(&index) { - continue; + if headlines.contains(&index) { + findings.extend(lookalike_findings(label, index, line, allowed)); } - for lookalike in crate::guard::unicode::lookalikes(line) { - if admitted_by_allowance(lookalike.character, allowed) { - continue; - } - findings.push(format!( + } + findings +} + +/// The lookalike pass over one subject line, numbered from zero. +fn lookalike_findings(label: &str, index: usize, line: &str, allowed: &[char]) -> Vec { + crate::guard::unicode::lookalikes(line) + .into_iter() + .filter(|lookalike| !admitted_by_allowance(lookalike.character, allowed)) + .map(|lookalike| { + format!( "{label}:{}:{}: {} in the SUBJECT LINE", index + 1, lookalike.column + 1, lookalike.describe(), - )); - } - } - findings + ) + }) + .collect() } pub(crate) fn prevent_ai_author(request: &Request<'_>) -> Result> { @@ -330,13 +335,36 @@ fn admitted_by_allowance(character: char, allowed: &[char]) -> bool { /// knows which line is the subject. pub(crate) fn prevent_unusual_unicode(request: &Request<'_>) -> Result> { for (label, text, subjects) in headed_messages(request)? { - if let Some(refusal) = unusual_unicode_over(request.rule, &label, &text, &subjects)? { + if let Some(refusal) = unusual_unicode_in_message(request.rule, &label, &text, &subjects)? { return Ok(Some(refusal)); } } Ok(None) } +/// One message at a git hook, with the lines [`subject_lines`] made its +/// candidates; the report carries [`comment_line_note`] when a lookalike was +/// found on a candidate that opens with the comment character. +fn unusual_unicode_in_message( + rule: &crate::config::Rule, + label: &str, + text: &str, + subjects: &[usize], +) -> Result> { + let Some(mut refusal) = unusual_unicode_over(rule, label, text, subjects)? else { + return Ok(None); + }; + let allowed = allowances(rule)?; + refusal + .report + .push_str(comment_line_note(subjects, |commented| { + text.split('\n') + .nth(commented) + .is_some_and(|line| !lookalike_findings("", commented, line, &allowed).is_empty()) + })); + Ok(Some(refusal)) +} + /// A commit subject line that is not printable ASCII. /// /// A house style, not a security check, and opt in: no bundled set declares @@ -359,7 +387,9 @@ pub(crate) fn ascii_only_commit_subject(request: &Request<'_>) -> Result) -> Result &'static str { +/// comment -- one carrying a lookalike word, or any non-ASCII letter under +/// `ascii-only-commit-subject` -- is refused, and the report says so where it +/// fires rather than leaving the author to reverse-engineer it. Only the git +/// hooks attach it: they are the seam whose subjects come from +/// [`subject_lines`], and only there is a second candidate a comment. +fn comment_line_note(subjects: &[usize], found_on: impl Fn(usize) -> bool) -> &'static str { let [commented, _, ..] = subjects else { return ""; }; - if non_ascii_in_subject("", text, allowed, &[*commented]).is_empty() { + if !found_on(*commented) { return ""; } - "\n\nThe first line opens with the comment character and was judged as a subject: \ - `git commit -m` records such a line, and this hook cannot see whether the editor will \ - strip it. If it is a `commit.template` comment, open the template with an ASCII line." + "\n\nThe first non-blank line opens with the comment character and was judged as a \ + subject: `git commit -m` records such a line, and this hook cannot see whether the \ + editor will strip it. If it is a `commit.template` comment, open the template with an ASCII line." } fn non_ascii_in_subject( @@ -742,21 +775,73 @@ mod tests { found.iter().all(|finding| finding.starts_with("m:1:")), "{found:?}" ); - assert!(comment_line_note(&text, &[], &subjects).contains("commit.template")); + assert!(ascii_note(&text).contains("commit.template")); // An ASCII comment first keeps the Japanese one out of the candidates. let template = format!("# Summary\n# {JAPANESE}{KANA}\nFix the parser\n"); let kept = subject_findings(&template, &[]); assert_eq!(kept, Vec::::new()); // And the note stays out of a report on an ordinary subject. - let plain = "Fix \u{2014} the parser\n"; - assert_eq!( - comment_line_note(plain, &[], &subject_lines(plain, Some('#'))), - "" + assert_eq!(ascii_note("Fix \u{2014} the parser\n"), ""); + assert_eq!(ascii_note("# Summary\nFix \u{2014} the parser\n"), ""); + } + + /// The note `ascii-only-commit-subject` would attach to this message. + fn ascii_note(text: &str) -> &'static str { + let subjects = subject_lines(text, Some('#')); + comment_line_note(&subjects, |commented| { + !non_ascii_in_subject("", text, &[], &[commented]).is_empty() + }) + } + + #[test] + fn the_note_names_the_first_non_blank_line() { + // Leading blank lines are skipped, so the comment is not the first + // line, and the note says which one it means. + let text = format!("\n\n# {JAPANESE}{KANA}\nFix the parser\n"); + assert_eq!(subject_lines(&text, Some('#')), vec![2, 3]); + assert!( + ascii_note(&text).contains("The first non-blank line opens with the comment character"), + "{}", + ascii_note(&text) ); - let under = "# Summary\nFix \u{2014} the parser\n"; - assert_eq!( - comment_line_note(under, &[], &subject_lines(under, Some('#'))), - "" + } + + #[test] + fn a_lookalike_on_a_comment_opened_first_line_carries_the_note() { + let rule = loaded_rule("message-comment-note", "prevent-unusual-unicode", "[]").unwrap(); + let judged = |text: &str| { + unusual_unicode_in_message(&rule, "m", text, &subject_lines(text, Some('#'))).unwrap() + }; + let commented = judged("# the c\u{0430}che template\nFix the parser\n") + .expect("a lookalike on the comment-opened candidate passed"); + assert!(commented.report.contains("m:1:"), "{}", commented.report); + assert!( + commented + .report + .contains("The first non-blank line opens with the comment character"), + "{}", + commented.report + ); + // The note stays out of a report on an ordinary subject, and out of + // one whose finding is on the line under the comment. + for text in [ + "Fix the c\u{0430}che\n", + "# Summary\nFix the c\u{0430}che\n", + ] { + let refusal = judged(text).expect("the lookalike subject passed"); + assert!( + !refusal.report.contains("comment character"), + "{text:?}: {}", + refusal.report + ); + } + // An invisible on the comment line is refused on every line, subject + // or not, so it is no reason for the note. + let invisible = judged("# a pa\u{200D}rser\nFix the parser\n").expect("ZWJ passed"); + assert!( + !invisible.report.contains("comment character"), + "{}", + invisible.report ); } diff --git a/src/guard/unicode.rs b/src/guard/unicode.rs index 9ece93f..4cbac9b 100644 --- a/src/guard/unicode.rs +++ b/src/guard/unicode.rs @@ -265,54 +265,95 @@ impl Lookalike { /// run into a kanji compound is two single-script words, not one mixed one, /// and a script boundary between Latin and kanji is not a lookalike. /// -/// Except where the letter at the boundary is itself drawn as the other +/// Except for a letter at that boundary which is itself drawn as the other /// side's script. U+3007 IDEOGRAPHIC NUMBER ZERO is Han, and its skeleton is -/// a Latin `O`: in `g` + U+3007 + `od` it is the disguise the rule exists -/// for, and splitting there would leave three single-script words and no -/// finding. So a crossing letter stays in the word when it is drawn as the -/// word's script, or when every letter so far is drawn as the crossing one's -/// (U+3007 + `K`), and `lookalikes_in_word` judges the word whole. A kanji -/// compound carrying U+3007 with no Latin letter beside it never crosses. +/// a Latin `O`. Such a letter is read with the letter AFTER it, which is how +/// both languages read it: before `K` it is the `O` of `OK`, and before a +/// counter kanji (U+4EF6, U+5E74) it is the numeral zero. So it joins the +/// Latin word when a Latin letter follows it, wherever the run before it +/// came from -- `g` + U+3007 + `od`, U+3007 + `K`, and U+4EF6 + U+3007 + `K` +/// each carry the Latin word U+3007 + `K` or the whole `g` + U+3007 + `od`, +/// and `lookalikes_in_word` judges it whole. Followed by a kanji it stays in +/// the kanji run (`API` + U+3007 + U+4EF6 is `API` and a numeral). With no +/// letter after it, it is read with the letter before it. A kanji compound +/// carrying U+3007 with no Latin letter after it never crosses. pub(crate) fn lookalikes(line: &str) -> Vec { let mut found = Vec::new(); let mut word: Vec = Vec::new(); let mut start = 0usize; - // The script the word so far is written in, once a letter has said so. A - // mark carries no script and never decides it, and neither does a letter - // kept in the word as another script's disguise. - let mut current: Option