diff --git a/docs/REFERENCE.md b/docs/REFERENCE.md index 32635b6..163ff52 100644 --- a/docs/REFERENCE.md +++ b/docs/REFERENCE.md @@ -1449,9 +1449,31 @@ file guard bans: `API` or `README` run straight into a Japanese or Chinese subject is not a mixed word and passes. The exception is a letter at that boundary which is itself drawn like the other side's script: U+3007 IDEOGRAPHIC NUMBER ZERO is - Han with a Latin `O` for its skeleton, so `g` + U+3007 + `od` stays one word - and is refused, while a kanji numeral carrying it with no Latin letter beside - it passes. A skeleton made only of characters no script owns + Han with a Latin `O` for its skeleton, and a run of it is read by two rules + in order, looking at the nearest letter after the run and before it in the + same word; marks and letters no script owns (such as the prolonged sound + mark U+30FC) are skipped in that search. (1) Before a Latin letter it joins + that Latin word whatever comes before it, so `g` + U+3007 + `od`, U+3007 + + `K`, and a kanji or kana run + U+3007 + `K` (a kanji numeral included) are + refused in a word where Latin wins or ties the script vote (see below). + (2) Otherwise it is read with the letter before it when it is drawn as + that letter's script: after a Latin letter it is part of the Latin + word, so `TOD` + U+3007 + U+4E00 U+89A7, `TOD` + U+3007 + kana, `HELL` + + U+3007 U+3007, and `g` + U+3007 + a Cyrillic `o` are refused; after a kanji, + a kana, a Hangul letter, a Cyrillic or Greek letter, a digit, or nothing it + stays Han, so kanji numerals, + `2026` + U+5E74 U+4E00 U+3007 U+6708, and the U+3007 U+3007 placeholder + before kana or kanji pass. A U+3007 read as Latin counts as a Latin letter + when the word's script is decided, so a Cyrillic or Greek letter beside it + is named when Latin wins or ties that vote. In a word where Cyrillic or + Greek letters are the majority (a Cyrillic-spelled `exec` + U+3007 + `t`) + U+3007 is not named: the same majority-vote limit lets an all-Cyrillic + lookalike of `execot` pass. The cost is U+3007 meant as a numeral or a + placeholder right next to Latin letters: U+3007 + `GB`, the U+3007 U+3007 + placeholder before a Latin word, and a Latin word + U+3007 + a counter (`API` + U+3007 + U+4EF6, + "API, zero items") are refused. The report names U+3007, and + `allow = ["U+3007"]` admits it. A + skeleton made only of characters no script owns (the kanji for "one" skeletons to the katakana prolonged sound mark) is not "drawn like" any script. @@ -1473,9 +1495,11 @@ git's cleanup, and whether a line opening with `core.commentChar` (default `#`) is stripped depends on the cleanup mode (`git commit -m "#42 Fix"` keeps it, the editor strips it), so where the first such line opens with the comment character, the first line that does not is judged as well. Both rules read it -that way, so a `commit.template` whose first line is a non-ASCII comment is -refused under `ascii-only-commit-subject` although the editor would strip it; -the report says why, and a template opening with an ASCII line passes. A pushed commit was +that way, so a `commit.template` whose first non-blank line is a non-ASCII +comment is refused under `ascii-only-commit-subject`, and one whose first +non-blank line carries a lookalike word under `prevent-unusual-unicode`, +although the editor would strip it; either report says why, and a template +opening with an ASCII line passes. A pushed commit was already cleaned up, and its first line is its subject. A **title** a shim collects is a subject as well: `gh pr merge --subject` is one outright, and a pull-request title becomes one when the forge squash-merges it, on a branch no diff --git a/src/guard/message.rs b/src/guard/message.rs index 73023d9..6a4001d 100644 --- a/src/guard/message.rs +++ b/src/guard/message.rs @@ -283,22 +283,27 @@ fn unusual_findings(label: &str, text: &str, allowed: &[char], headlines: &[usiz .map_or_else(|| String::from("UNKNOWN"), |name| name.to_string()), )); } - if !headlines.contains(&index) { - continue; + if headlines.contains(&index) { + findings.extend(lookalike_findings(label, index, line, allowed)); } - for lookalike in crate::guard::unicode::lookalikes(line) { - if admitted_by_allowance(lookalike.character, allowed) { - continue; - } - findings.push(format!( + } + findings +} + +/// The lookalike pass over one subject line, numbered from zero. +fn lookalike_findings(label: &str, index: usize, line: &str, allowed: &[char]) -> Vec { + crate::guard::unicode::lookalikes(line) + .into_iter() + .filter(|lookalike| !admitted_by_allowance(lookalike.character, allowed)) + .map(|lookalike| { + format!( "{label}:{}:{}: {} in the SUBJECT LINE", index + 1, lookalike.column + 1, lookalike.describe(), - )); - } - } - findings + ) + }) + .collect() } pub(crate) fn prevent_ai_author(request: &Request<'_>) -> Result> { @@ -330,13 +335,36 @@ fn admitted_by_allowance(character: char, allowed: &[char]) -> bool { /// knows which line is the subject. pub(crate) fn prevent_unusual_unicode(request: &Request<'_>) -> Result> { for (label, text, subjects) in headed_messages(request)? { - if let Some(refusal) = unusual_unicode_over(request.rule, &label, &text, &subjects)? { + if let Some(refusal) = unusual_unicode_in_message(request.rule, &label, &text, &subjects)? { return Ok(Some(refusal)); } } Ok(None) } +/// One message at a git hook, with the lines [`subject_lines`] made its +/// candidates; the report carries [`comment_line_note`] when a lookalike was +/// found on a candidate that opens with the comment character. +fn unusual_unicode_in_message( + rule: &crate::config::Rule, + label: &str, + text: &str, + subjects: &[usize], +) -> Result> { + let Some(mut refusal) = unusual_unicode_over(rule, label, text, subjects)? else { + return Ok(None); + }; + let allowed = allowances(rule)?; + refusal + .report + .push_str(comment_line_note(subjects, |commented| { + text.split('\n') + .nth(commented) + .is_some_and(|line| !lookalike_findings("", commented, line, &allowed).is_empty()) + })); + Ok(Some(refusal)) +} + /// A commit subject line that is not printable ASCII. /// /// A house style, not a security check, and opt in: no bundled set declares @@ -359,7 +387,9 @@ pub(crate) fn ascii_only_commit_subject(request: &Request<'_>) -> Result) -> Result &'static str { +/// comment -- one carrying a lookalike word, or any non-ASCII letter under +/// `ascii-only-commit-subject` -- is refused, and the report says so where it +/// fires rather than leaving the author to reverse-engineer it. Only the git +/// hooks attach it: they are the seam whose subjects come from +/// [`subject_lines`], and only there is a second candidate a comment. +fn comment_line_note(subjects: &[usize], found_on: impl Fn(usize) -> bool) -> &'static str { let [commented, _, ..] = subjects else { return ""; }; - if non_ascii_in_subject("", text, allowed, &[*commented]).is_empty() { + if !found_on(*commented) { return ""; } - "\n\nThe first line opens with the comment character and was judged as a subject: \ - `git commit -m` records such a line, and this hook cannot see whether the editor will \ - strip it. If it is a `commit.template` comment, open the template with an ASCII line." + "\n\nThe first non-blank line opens with the comment character and was judged as a \ + subject: `git commit -m` records such a line, and this hook cannot see whether the \ + editor will strip it. If it is a `commit.template` comment, open the template with an ASCII line." } fn non_ascii_in_subject( @@ -742,21 +775,73 @@ mod tests { found.iter().all(|finding| finding.starts_with("m:1:")), "{found:?}" ); - assert!(comment_line_note(&text, &[], &subjects).contains("commit.template")); + assert!(ascii_note(&text).contains("commit.template")); // An ASCII comment first keeps the Japanese one out of the candidates. let template = format!("# Summary\n# {JAPANESE}{KANA}\nFix the parser\n"); let kept = subject_findings(&template, &[]); assert_eq!(kept, Vec::::new()); // And the note stays out of a report on an ordinary subject. - let plain = "Fix \u{2014} the parser\n"; - assert_eq!( - comment_line_note(plain, &[], &subject_lines(plain, Some('#'))), - "" + assert_eq!(ascii_note("Fix \u{2014} the parser\n"), ""); + assert_eq!(ascii_note("# Summary\nFix \u{2014} the parser\n"), ""); + } + + /// The note `ascii-only-commit-subject` would attach to this message. + fn ascii_note(text: &str) -> &'static str { + let subjects = subject_lines(text, Some('#')); + comment_line_note(&subjects, |commented| { + !non_ascii_in_subject("", text, &[], &[commented]).is_empty() + }) + } + + #[test] + fn the_note_names_the_first_non_blank_line() { + // Leading blank lines are skipped, so the comment is not the first + // line, and the note says which one it means. + let text = format!("\n\n# {JAPANESE}{KANA}\nFix the parser\n"); + assert_eq!(subject_lines(&text, Some('#')), vec![2, 3]); + assert!( + ascii_note(&text).contains("The first non-blank line opens with the comment character"), + "{}", + ascii_note(&text) ); - let under = "# Summary\nFix \u{2014} the parser\n"; - assert_eq!( - comment_line_note(under, &[], &subject_lines(under, Some('#'))), - "" + } + + #[test] + fn a_lookalike_on_a_comment_opened_first_line_carries_the_note() { + let rule = loaded_rule("message-comment-note", "prevent-unusual-unicode", "[]").unwrap(); + let judged = |text: &str| { + unusual_unicode_in_message(&rule, "m", text, &subject_lines(text, Some('#'))).unwrap() + }; + let commented = judged("# the c\u{0430}che template\nFix the parser\n") + .expect("a lookalike on the comment-opened candidate passed"); + assert!(commented.report.contains("m:1:"), "{}", commented.report); + assert!( + commented + .report + .contains("The first non-blank line opens with the comment character"), + "{}", + commented.report + ); + // The note stays out of a report on an ordinary subject, and out of + // one whose finding is on the line under the comment. + for text in [ + "Fix the c\u{0430}che\n", + "# Summary\nFix the c\u{0430}che\n", + ] { + let refusal = judged(text).expect("the lookalike subject passed"); + assert!( + !refusal.report.contains("comment character"), + "{text:?}: {}", + refusal.report + ); + } + // An invisible on the comment line is refused on every line, subject + // or not, so it is no reason for the note. + let invisible = judged("# a pa\u{200D}rser\nFix the parser\n").expect("ZWJ passed"); + assert!( + !invisible.report.contains("comment character"), + "{}", + invisible.report ); } diff --git a/src/guard/unicode.rs b/src/guard/unicode.rs index 9ece93f..9b19f72 100644 --- a/src/guard/unicode.rs +++ b/src/guard/unicode.rs @@ -265,54 +265,147 @@ impl Lookalike { /// run into a kanji compound is two single-script words, not one mixed one, /// and a script boundary between Latin and kanji is not a lookalike. /// -/// Except where the letter at the boundary is itself drawn as the other +/// Except for a letter at that boundary which is itself drawn as the other /// side's script. U+3007 IDEOGRAPHIC NUMBER ZERO is Han, and its skeleton is -/// a Latin `O`: in `g` + U+3007 + `od` it is the disguise the rule exists -/// for, and splitting there would leave three single-script words and no -/// finding. So a crossing letter stays in the word when it is drawn as the -/// word's script, or when every letter so far is drawn as the crossing one's -/// (U+3007 + `K`), and `lookalikes_in_word` judges the word whole. A kanji -/// compound carrying U+3007 with no Latin letter beside it never crosses. +/// a Latin `O`. A run of such letters is read by `reading_scripts`, by two +/// rules in order: +/// +/// 1. Before a letter of the script it is drawn as, it joins that word, +/// whatever came before it: U+3007 + `K`, `g` + U+3007 + `od`, U+4EF6 + +/// U+3007 + `K`, and U+4E8C + U+3007 + `K`, because U+3007 + `K` is the +/// disguise even after a kanji numeral. +/// 2. Otherwise it is read with the letter before it, when it is drawn as +/// that letter's script: `TOD` + U+3007 + U+4E00 U+89A7, `HELL` + U+3007 +/// U+3007, and `g` + U+3007 + a Cyrillic `o` are refused. After a kanji, +/// a kana, a Hangul letter, a Cyrillic or Greek letter, a digit, or +/// nothing, it stays Han, so a kanji numeral carrying it never crosses. +/// +/// `lookalikes_in_word` then judges each word whole, counting each letter as +/// the script it is read as. A U+3007 read as Latin is refused, and a +/// Cyrillic or Greek letter beside it named, only when Latin wins or ties the +/// word's script vote. In a Cyrillic- or Greek-majority word (a +/// Cyrillic-spelled `exec` + U+3007 + `t`) U+3007 is not named, by the same +/// majority-vote limit that lets an all-Cyrillic lookalike of `execot` pass. pub(crate) fn lookalikes(line: &str) -> Vec { let mut found = Vec::new(); let mut word: Vec = Vec::new(); + let mut voices: Vec> = Vec::new(); let mut start = 0usize; - // The script the word so far is written in, once a letter has said so. A - // mark carries no script and never decides it, and neither does a letter - // kept in the word as another script's disguise. - let mut current: Option