diff --git a/.changepacks/changepack_log_Hn4kRtBqWm2ZsVxLdOeYc.json b/.changepacks/changepack_log_Hn4kRtBqWm2ZsVxLdOeYc.json index 33f5f300..e3a038c2 100644 --- a/.changepacks/changepack_log_Hn4kRtBqWm2ZsVxLdOeYc.json +++ b/.changepacks/changepack_log_Hn4kRtBqWm2ZsVxLdOeYc.json @@ -1 +1 @@ -{"changes":{"libs/braillify/Cargo.toml":"Patch","packages/c/Cargo.toml":"Patch","packages/dotnet/Braillify/Braillify.csproj":"Patch","packages/dotnet/BraillifyNet/BraillifyNet.csproj":"Patch","packages/go/Cargo.toml":"Patch","packages/node/package.json":"Patch","packages/python/pyproject.toml":"Patch","packages/ruby/Cargo.toml":"Patch"},"note":"Apply the National Institute of Korean Language rulings on the grade-1 symbol and the ever contraction, read a whole-run groupsign as letters where the rules require it, separate a dash that opens the text, and join a spaced hyphen that is not a subtraction sign, and close the editorial gap inside a bracket; corpus accuracy 455,975 of 467,121 sentences.","date":"2026-09-21T04:00:00.0000000Z"} \ No newline at end of file +{"changes":{"libs/braillify/Cargo.toml":"Minor","packages/c/Cargo.toml":"Minor","packages/dotnet/Braillify/Braillify.csproj":"Minor","packages/dotnet/BraillifyNet/BraillifyNet.csproj":"Minor","packages/go/Cargo.toml":"Minor","packages/jvm/build.gradle.kts":"Minor","packages/node/package.json":"Minor","packages/python/pyproject.toml":"Minor","packages/ruby/Cargo.toml":"Minor"},"note":"Apply the National Institute of Korean Language rulings on the grade-1 symbol, the ever contraction, circled capitals and the n-ary product, read a whole-run groupsign as letters where the rules require it, separate a dash that opens the text, join a spaced hyphen that is not a subtraction sign, and close the editorial gap inside a bracket; report which rule wrote every braille cell; write chemical formulas, reaction equations, structural formulas, electron dot formulas, ring compounds of every shape, fused rings and pedigrees by the science articles; write them in the spatial form in the science_spatial context, and write the units that follow a number inside a formula; let every binding take a reading context such as science; read text with no Hangul and no context as English, writing science notation for it only in the science context; space a colon that runs into Korean content or closes a title in running text, fold the grave accent, dot above and heavy vertical line into the symbols they stand for, keep a Korean word's bracketed figures and a slash after a Korean word out of the math route, split a colon and close a bracket gap before the later rules see the words, write a lone reference mark as an asterisk, read a plus after a number as a superscript only in a grade, tell a tone mark from a middle dot, write a run of superscripts after one sign, write nothing for a LaTeX null right delimiter, keep Roman exclamations, acronym lists, unit ranges and open Roman glosses out of the math route, follow the punctuation spacing of the Korean orthography appendix, abbreviate the Article 18 words after punctuation, and keep units, grades and Korean-number punctuation in the right Roman section; corpus accuracy 456,626 of 467,121 sentences.","date":"2026-09-21T04:00:00.0000000Z"} \ No newline at end of file diff --git a/Cargo.lock b/Cargo.lock index 1f0c0ed1..1d76e7bb 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -194,7 +194,7 @@ checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" [[package]] name = "braillify" -version = "2.1.2" +version = "2.2.0" dependencies = [ "anyhow", "assert_cmd", @@ -219,14 +219,14 @@ dependencies = [ [[package]] name = "braillify-c" -version = "0.1.2" +version = "0.2.0" dependencies = [ "braillify", ] [[package]] name = "braillify-go" -version = "2.0.0" +version = "2.0.1" dependencies = [ "braillify", ] @@ -991,6 +991,9 @@ version = "0.1.0" dependencies = [ "braillify", "console_error_panic_hook", + "rstest", + "serde", + "serde_json", "wasm-bindgen", "wasm-bindgen-test", ] diff --git a/apps/landing/src/app/RuleTrace.tsx b/apps/landing/src/app/RuleTrace.tsx new file mode 100644 index 00000000..867ebffd --- /dev/null +++ b/apps/landing/src/app/RuleTrace.tsx @@ -0,0 +1,316 @@ +'use client' + +import { Box, Flex, Text, VStack } from '@devup-ui/react' + +/** 한 번에 그리는 규칙 행의 최대 개수. 키 입력마다 다시 그리므로 상한을 둔다. */ +const MAX_VISIBLE_RULES = 120 + +/** `kind` 원문 → 화면 표기. 모르는 값이면 원문을 그대로 보여준다. */ +const KIND_LABEL: Record = { + korean: '한글', + jamo: '자모', + token: '기호', + math: '수학', + 'english-ueb': '영어', + emitter: '구조', +} + +/** + * 규칙이 하나도 잡히지 않는 경로별 설명. 점역 자체는 정상이므로 + * "적용된 규칙이 없다"고 읽히면 안 된다. + */ +const NO_RULE_NOTICE: Record = { + 'english-ueb': + '이 낱말은 축약 규칙 탐색이 아니라 낱말 기호표로 점역되어, 규칙 단위로 나눌 수 없습니다.', +} + +/** 출력 점자의 일부를 만들어 낸 규칙 하나. WASM 객체를 평범한 값으로 옮긴 것. */ +export interface TraceRule { + section: string + standardRef: string + name: string + description: string + kind: string + start: number + end: number + braille: string +} + +/** + * 한 번의 점역 결과 스냅샷. + * + * - `idle` — 입력이 없거나 WASM이 아직 로드되지 않음. 아무것도 그리지 않는다. + * - `failed` — 점역이 실패함. 출력 상자가 이미 사유를 보여주므로 목록은 숨긴다. + * - `ok` — 점역 성공. 목록을 그린다. + */ +export interface TraceSnapshot { + status: 'idle' | 'ok' | 'failed' + braille: string + rules: TraceRule[] + /** 규칙이 설명하는 출력 칸 수. */ + attributed: number + /** 전체 출력 칸 수. `attributed`보다 크면 추적되지 않은 칸이 있다는 뜻이다. */ + total: number + /** 입력을 처리한 엔진: `korean` | `english-ueb` | `math`. */ + path: string +} + +export const IDLE_TRACE: TraceSnapshot = { + status: 'idle', + braille: '', + rules: [], + attributed: 0, + total: 0, + path: '', +} + +export const FAILED_TRACE: TraceSnapshot = { + status: 'failed', + braille: '점역할 수 없는 문자가 있습니다.', + rules: [], + attributed: 0, + total: 0, + path: '', +} + +interface RuleSpanJson { + section: string + subsection: string + standard_ref: string + name: string + description: string + kind: string + start: number + end: number + braille: string +} + +interface TraceResultJson { + braille: string + rules: RuleSpanJson[] + attributed: number + total: number + path: string +} + +/** + * 점역 결과를 JSON으로 받는다. WASM 쪽은 문자열 하나만 넘기므로 해제할 핸들이 + * 없고, 렌더 중에 WASM 메모리를 다시 읽는 일도 없다. + */ +export function readTrace(json: string): TraceSnapshot { + const result = JSON.parse(json) as TraceResultJson + return { + status: 'ok', + braille: result.braille, + attributed: result.attributed, + total: result.total, + path: result.path, + rules: result.rules.map((span) => ({ + section: span.section, + standardRef: span.standard_ref, + name: span.name, + description: span.description, + kind: span.kind, + start: span.start, + end: span.end, + braille: span.braille, + })), + } +} + +/** + * 항 번호 표기. `-`는 규정 항이 없는 구조 출력이라 `제N항`으로 꾸며내지 않는다. + * + * 영어는 한국 점자 규정이 아니라 UEB 규정을 따르므로 `제N항`이 아닌 `§N` 표기를 + * 쓴다. 수학은 같은 규정 안의 별도 장이라 한글 제N항과 번호가 겹치므로 `수학`을 + * 붙여 구분한다. 번호 체계가 다른 규정을 같은 꼴로 적으면 출처를 잘못 읽게 된다. + * + * 규칙이 도는 엔진과 규칙이 구현하는 규정의 계열은 서로 다를 수 있다. 동그라미 + * 숫자는 수식 안에서 만나도 한글 제64항이고, 수학 제64항은 햇(단위 벡터)이다. + * 그래서 계열은 엔진(`kind`)이 아니라 규칙이 밝힌 출처(`standardRef`)에서 읽는다. + * + * 한 판단이 여러 항에 걸칠 때는 `, `로 이어 적는다. 국립국어원 회신(2026-09-21)이 + * 그런 경우 항을 하나만 고르지 말고 "다 적으라"고 했다. + */ +function sectionLabel( + section: string, + kind: string, + standardRef: string, +): string | null { + if (section === '-') return null + + const series = standardRef.includes('한글 제') + ? '한글 ' + : standardRef.includes('수학 제') + ? '수학 ' + : kind === 'math' + ? '수학 ' + : '' + + if (kind === 'english-ueb' && series === '') { + return section + .split(', ') + .map((one) => `§${one}`) + .join('·') + } + return section + .split(', ') + .map((one) => `${series}제${one}항`) + .join('·') +} + +/** 반열린 구간 `[start, end)`를 1부터 세는 사람 기준 표기로 옮긴다. */ +function rangeLabel(start: number, end: number): string { + if (end - start <= 1) return `${start + 1}번째 칸` + return `${start + 1}–${end}번째 칸` +} + +/** 빈 칸(U+2800)도 눈에 보이도록 칸 하나를 칩으로 그린다. */ +function BrailleCells({ braille }: { braille: string }) { + return ( + + {Array.from(braille).map((cell, index) => ( + + {cell} + + ))} + + ) +} + +function RuleRow({ rule }: { rule: TraceRule }) { + const section = sectionLabel(rule.section, rule.kind, rule.standardRef) + + return ( + + + {section ? ( + + {section} + + ) : null} + + + + {rule.name} + + + {rule.description} · {KIND_LABEL[rule.kind] ?? rule.kind} + + + + + + + {rangeLabel(rule.start, rule.end)} + + + ) +} + +/** 점역 결과를 만들어 낸 규칙 목록. 추적은 아직 부분적이라 덮인 범위를 함께 밝힌다. */ +export function RuleTrace({ trace }: { trace: TraceSnapshot }) { + if (trace.status !== 'ok') return null + + const isPartial = trace.attributed < trace.total + const visibleRules = trace.rules.slice(0, MAX_VISIBLE_RULES) + const hiddenCount = trace.rules.length - visibleRules.length + const emptyNotice = + trace.rules.length > 0 + ? null + : (NO_RULE_NOTICE[trace.path] ?? + '이 입력에 대해 기록된 규칙이 아직 없습니다.') + + return ( + + + + 적용 규칙 + + {trace.total > 0 ? ( + + {trace.attributed}/{trace.total}칸 추적됨 + + ) : null} + + + {isPartial ? ( + + 출력 {trace.total}칸 가운데 {trace.attributed}칸만 규칙으로 + 설명됩니다. 나머지 {trace.total - trace.attributed}칸은 아직 규칙 + 추적이 붙지 않은 부분입니다. + + ) : null} + {emptyNotice ? ( + + {emptyNotice} + + ) : ( + + {visibleRules.map((rule, index) => ( + + ))} + + )} + {hiddenCount > 0 ? ( + + 규칙 {hiddenCount}개는 목록에 표시하지 않았습니다. + + ) : null} + + ) +} diff --git a/apps/landing/src/app/Trans.tsx b/apps/landing/src/app/Trans.tsx index f02474ef..eea6e844 100644 --- a/apps/landing/src/app/Trans.tsx +++ b/apps/landing/src/app/Trans.tsx @@ -1,28 +1,41 @@ 'use client' import { VStack } from '@devup-ui/react' -import { useEffect, useState } from 'react' +import { useEffect, useMemo, useState } from 'react' import { DemoArrow } from './DemoArrow' import { DemoHeading } from './DemoHeading' +import { + FAILED_TRACE, + IDLE_TRACE, + readTrace, + RuleTrace, + type TraceSnapshot, +} from './RuleTrace' import { TransInput } from './TransInput' +type Translate = (input: string) => TraceSnapshot + +const idleTranslate: Translate = () => IDLE_TRACE + export function Trans() { const [input, setInput] = useState('') - const [translateToUnicode, setTranslateToUnicode] = useState< - (input: string) => string - >(() => () => '') + const [translate, setTranslate] = useState(() => idleTranslate) useEffect(() => { import('braillify').then((mod) => { - setTranslateToUnicode(() => (input: string) => { + setTranslate(() => (text: string) => { + if (text.length === 0) return IDLE_TRACE try { - return mod.translateToUnicode(input) + return readTrace(mod.translateToUnicodeWithTrace(text)) } catch (e) { console.error(e) - return '점역할 수 없는 문자가 있습니다.' + return FAILED_TRACE } }) }) - }, [input]) + }, []) + + // 한 번의 점역으로 점자 출력과 규칙 목록을 모두 얻는다. + const trace = useMemo(() => translate(input), [translate, input]) const [inputFocused, setInputFocused] = useState(false) const [translationFocused, setTranslationFocused] = useState(false) @@ -60,9 +73,10 @@ export function Trans() { focusPlaceholder="⠕⠈⠥⠄⠝⠀⠨⠎⠢⠱⠁⠚⠂⠀⠉⠗⠬⠶⠮⠀⠕⠃⠐⠱⠁⠚⠗⠨⠍⠠⠝⠬⠖" isFocused={translationFocused} readOnly - value={translateToUnicode(input)} + value={trace.braille} /> + ) } diff --git a/libs/braillify/AGENTS.md b/libs/braillify/AGENTS.md index 296475da..b2a5656e 100644 --- a/libs/braillify/AGENTS.md +++ b/libs/braillify/AGENTS.md @@ -37,13 +37,16 @@ src/ ``` Input text ↓ DocumentIR::parse() (tokenize into Word/Space/Mode tokens) - ↓ TokenRuleEngine::apply_all() (token-level rules by phase) - │ ├── LatexMergeRule (merge $...$ across spaces) - │ ├── LatexFractionRule (detect $\frac{}{})$) - │ ├── LatexMathRule (strip LaTeX → math notation) - │ ├── InlineFractionRule (detect N/N inline fractions) - │ ├── MathExpressionTokenRule (detect & encode math expressions) - │ └── ...other token rules + ↓ TokenRuleEngine::apply_all() (token-level rules by phase, then priority; + │ a rule returning Noop hands the word to the + │ next rule of the same phase) + │ ├── Normalization LatexMergeRule ($...$ across spaces), science notation, + │ │ bracket gap, colon/semicolon split, leading asterisk, … + │ ├── FractionDetection MathExpressionTokenRule (math, LaTeX, \frac → Token::Fraction) + │ ├── WordShortcut WordShortcutRule + │ ├── ModeEntry DigitalNotationRule (URL, e-mail) + │ ├── UppercasePassage UppercasePassageRule (capitals word/passage) + │ └── PostWord middle dot, tilde, hyphen, asterisk, English-dominant wrap ↓ emit() (character-level encoding) ├── Token::Word → RuleEngine (BrailleRule trait, char-by-char) ├── Token::Space → braille space byte @@ -233,12 +236,12 @@ cargo fmt && cargo clippy # Format + lint bun test test_cases/ # JSON integrity checks (packages/node/pkg 빌드 필요) ``` -규정 fixture 는 `test_cases/{korean,math,english}/*.json` 이고, 그와 별개로 +규정 fixture 는 `test_cases/{korean,math,english,science}/*.json` 이고, 그와 별개로 `test_cases/{2021,2022,2023,2024,2025}_corpus/sentence_*.json` 에 국립국어원 한국어-한국점자 병렬 말뭉치 46만 7121문장이 들어 있다. 말뭉치는 `rule_map.json` 에서 `benchmark: true` 로 표시되어 **pass/fail 에 들어가지 않고 정확도만 보고**한다. -**Current status: 규정 fixture 5141/5141 (100%), 말뭉치 455,975/467,121 (97.61%).** +**Current status: 규정 fixture 5284/5284 (100%, `limitation` 없음), 말뭉치 456,626/467,121 (97.75%).** ⚠️ `roman_marker_bench` 는 **어절 수가 맞는 문장만** 센다. 띄어쓰기를 바꾸면 비교 모집단 자체가 움직이므로 네 수치를 그대로 빼서 비교하면 안 된다. 실제로 diff --git a/libs/braillify/src/encoder.rs b/libs/braillify/src/encoder.rs index 81fe77ad..b45f9fa9 100644 --- a/libs/braillify/src/encoder.rs +++ b/libs/braillify/src/encoder.rs @@ -1,8 +1,10 @@ use std::borrow::Cow; +use crate::korean_char::JamoSpans; use crate::rules; use crate::rules::context::EncodingMode; use crate::rules::token::{Token, WordMeta, WordToken}; +use crate::rules::trace::{TokenOrigins, TracePath, TraceSink}; pub struct Encoder { pub(crate) is_english: bool, @@ -128,19 +130,22 @@ impl Encoder { rules::token_rules::math_expression::MathExpressionTokenRule, )); token_engine.register(Box::new( - rules::token_rules::latex_fraction::LatexFractionRule, + rules::token_rules::word_shortcut::WordShortcutRule, )); token_engine.register(Box::new( - rules::token_rules::inline_fraction::InlineFractionRule, + rules::token_rules::digital_notation::DigitalNotationRule, )); token_engine.register(Box::new( - rules::token_rules::word_shortcut::WordShortcutRule, + rules::token_rules::structural_formula::StructuralFormulaRule, )); token_engine.register(Box::new( - rules::token_rules::roman_numeral::RomanNumeralRule, + rules::token_rules::cell_notation::CellNotationRule, )); token_engine.register(Box::new( - rules::token_rules::digital_notation::DigitalNotationRule, + rules::token_rules::chemical_formula::ChemicalFormulaRule, + )); + token_engine.register(Box::new( + rules::token_rules::dental_formula::DentalFormulaRule, )); token_engine.register(Box::new( rules::token_rules::uppercase_passage::UppercasePassageRule, @@ -168,7 +173,7 @@ impl Encoder { )); token_engine.register(Box::new(rules::token_rules::spacing::AsteriskSpacingRule)); token_engine.register(Box::new( - rules::token_rules::spacing::KoreanAuxiliaryVerbSpacingRule, + rules::token_rules::spacing::LeadingAsteriskSpacingRule, )); token_engine.register(Box::new( rules::token_rules::english_dominant_korean_wrap::EnglishDominantKoreanWrapRule, @@ -219,14 +224,28 @@ impl Encoder { self.math_mode_active = active; } - fn encode_via_ir(&mut self, text: &str, result: &mut Vec) -> Result<(), String> { - self.encode_via_ir_with_transform(text, result, |_, _| Ok(())) + pub(crate) fn char_rule_registry(&mut self) -> Vec<&'static rules::RuleMeta> { + self.rule_engine.registry() + } + + pub(crate) fn token_rule_registry(&mut self) -> Vec<&'static rules::RuleMeta> { + self.token_engine.registry() + } + + fn encode_via_ir( + &mut self, + text: &str, + result: &mut Vec, + trace: Option>, + ) -> Result<(), String> { + self.encode_via_ir_with_transform(text, result, trace, |_, _| Ok(())) } fn encode_via_ir_with_transform( &mut self, text: &str, result: &mut Vec, + trace: Option>, transform: F, ) -> Result<(), String> where @@ -235,8 +254,14 @@ impl Encoder { let mut ir = rules::token::DocumentIR::parse(text, self.english_indicator); ir.state.matrix_context_active = self.matrix_context_active; ir.state.math_mode_active = self.math_mode_active; + ir.state.korean_context_active = self.default_mode == Some(EncodingMode::Korean); + ir.state.science_context_active = + self.default_mode.is_some_and(EncodingMode::reads_science); + ir.state.jamo_spans = trace.is_some().then(Box::::default); + // 과학 글도 국어 점자 문장이므로 모드 스택은 국어로 둔다. if let Some(mode) = self.default_mode + && !mode.reads_science() && mode != ir.state.current_mode() { while ir.state.pop_mode().is_some() {} @@ -254,7 +279,14 @@ impl Encoder { } let state_before_token_rules = ir.state.clone(); - self.token_engine.apply_all(&mut ir.tokens, &mut ir.state)?; + if trace.is_some() { + rules::math::begin_collection(); + } + let mut origins = trace + .is_some() + .then(|| TokenOrigins::seeded(ir.tokens.len())); + self.token_engine + .apply_all_tracked(&mut ir.tokens, &mut ir.state, origins.as_mut())?; let mode_stack_after_token_rules = ir.state.mode_stack.clone(); // 제39항 영-한 wrap 활성화 신호는 token 단계의 결정이며 emit 단계에서도 // 유효해야 한다. mode_stack과 함께 보존한다. @@ -266,8 +298,13 @@ impl Encoder { ir.state.english_dominant_no_indicator = no_indicator_after_token_rules; transform(text, &mut ir.tokens)?; - let output = rules::emit::emit(&mut ir, &mut self.rule_engine)?; - result.extend(output); + // `transform` injects formatting tokens without origin tracking, so the + // side table no longer lines up with the stream and must be dropped. + let origins = origins.filter(|o| o.len() == ir.tokens.len()); + + let output = rules::emit::emit(&mut ir, &mut self.rule_engine, trace, origins.as_ref()); + rules::math::end_collection(); + result.extend(output?); self.is_english = ir.state.is_english; self.triple_big_english = ir.state.triple_big_english; @@ -278,6 +315,15 @@ impl Encoder { } pub fn encode(&mut self, text: &str, result: &mut Vec) -> Result<(), String> { + self.encode_traced(text, result, None) + } + + pub(crate) fn encode_traced( + &mut self, + text: &str, + result: &mut Vec, + mut trace: Option>, + ) -> Result<(), String> { // UEB Grade-2 path: pure-English input (no Korean, UEB-eligible, no // explicit mode) is encoded by the unified English engine. It returns // `Some` only when it fully handles the input; otherwise we fall through @@ -295,12 +341,30 @@ impl Encoder { // contains `-`, `(`, `,`, `.` is NOT blocked (that over-broad reading // of the math detector would swallow `child-ish-ly`, `with(er)`, …). && !crate::rules::english_ueb::is_math_owned(text) - && let Some(bytes) = crate::rules::english_ueb::try_encode(text) { - result.extend(bytes); - return Ok(()); + let encoded = if trace.is_some() { + crate::rules::english_ueb::try_encode_traced(text) + } else { + crate::rules::english_ueb::try_encode(text).map(|cells| (cells, Vec::new())) + }; + if let Some((bytes, spans)) = encoded { + let output_base = result.len(); + result.extend(bytes); + if let Some(sink) = trace.as_mut() { + let token_index = sink.token_index() as usize; + for (rule, output) in spans { + sink.record_span( + rule, + token_index, + output_base + output.start as usize..output_base + output.end as usize, + ); + } + sink.trace.set_path(TracePath::EnglishUeb); + } + return Ok(()); + } } - self.encode_via_ir(text, result) + self.encode_via_ir(text, result, trace) } pub fn encode_with_formatting( @@ -313,7 +377,7 @@ impl Encoder { return self.encode(text, result); } - self.encode_via_ir_with_transform(text, result, |source, tokens| { + self.encode_via_ir_with_transform(text, result, None, |source, tokens| { inject_formatting_tokens(source, spans, tokens) }) } diff --git a/libs/braillify/src/english_logic.rs b/libs/braillify/src/english_logic.rs index 30b4b242..78f59f18 100644 --- a/libs/braillify/src/english_logic.rs +++ b/libs/braillify/src/english_logic.rs @@ -117,6 +117,17 @@ pub(crate) fn begins_korean_mode_number(chars: impl Iterator) -> bo true } +/// 제33항 — 로마자와 한글 사이의 쉼표·쌍점: 한글이 바로 붙은 수(`1초에`, `27개`, +/// 제51항 시각 `22:35에`)는 한글 쪽이다. +pub(crate) fn opens_korean_number(chars: impl Iterator) -> bool { + let mut chars = chars.peekable(); + let mut saw_digit = false; + while let Some(ch) = chars.next_if(|ch| ch.is_ascii_digit() || matches!(ch, ',' | '.' | ':')) { + saw_digit |= ch.is_ascii_digit(); + } + saw_digit && chars.next().is_some_and(utils::is_korean_char) +} + /// Returns whether `index` is an ampersand inside a complete sequence of /// non-empty ASCII-letter segments joined by `&`. Korean rule 35 allows the /// resulting Roman text to continue directly into digits and later Roman @@ -417,11 +428,16 @@ pub(crate) fn should_render_symbol_as_english( remaining_words.first().and_then(|w| w.chars().next()) }; - // A non-English closing enclosure is a hard Roman-section boundary. The + // A non-English enclosure mark is a hard Roman-section boundary. The // look-behind helpers deliberately skip punctuation for attached UEB runs, // but must not reach through that boundary and pull a following version or - // identifier mark (`(XBB).1.5`) back into the closed Roman section. - if !is_english && prev_char.is_some_and(|ch| matches!(ch, ')' | ']' | '}')) { + // identifier mark (`(XBB).1.5`, `OPS(.837)`) back into the closed section. + let opens_after_opening = matches!(symbol, '(' | '[' | '{'); + if !is_english + && prev_char.is_some_and(|ch| { + matches!(ch, ')' | ']' | '}') || (matches!(ch, '(' | '[' | '{') && !opens_after_opening) + }) + { return false; } @@ -545,9 +561,25 @@ pub(crate) fn should_render_symbol_as_english( { false } + // 제29항: 로마자 사이의 느낌표·물음표(`Wow! Perfect`)는 로마자 구간 안에 있다. + '!' | '?' => { + is_english + && prev_ascii_letter_or_digit(word_chars, index) + && next_ascii_letter_or_digit(word_chars, index, remaining_words) + } '/' | '@' | '#' | '.' | '_' | ':' => { let prev_ascii = prev_ascii_letter_or_digit(word_chars, index); - let next_ascii = next_ascii_letter_or_digit(word_chars, index, remaining_words); + let next_ascii = next_ascii_letter_or_digit(word_chars, index, remaining_words) + && !(symbol == ':' + && prev_char.is_some_and(|ch| ch.is_ascii_alphanumeric()) + && word_chars.get(index + 1).map_or_else( + || { + remaining_words + .first() + .is_some_and(|word| opens_korean_number(word.chars())) + }, + |_| opens_korean_number(word_chars[index + 1..].iter().copied()), + )); // 제33항 [다만] — 앞에 로마자가 없이 숫자만 온 빗금은 단위를 가르는 // 기호이지 제74항 디지털 표기의 일부가 아니다(`17.1/km`). 구간을 열지 @@ -774,6 +806,77 @@ mod tests { ); } + /// 제29항 — 닫힌 로마자 구간 뒤 한글 괄호를 넘어 소수점을 끌어오지 않는다. + #[rstest::rstest] + #[case::after_an_opening_bracket("OPS(.837)", '.', 4, false)] + #[case::after_a_closing_bracket("(XBB).1", '.', 5, false)] + #[case::between_roman_letters("a.b", '.', 1, true)] + #[case::bracket_after_a_bracket("((W)", '(', 1, true)] + fn a_dot_after_a_korean_bracket_stays_korean( + #[case] input: &str, + #[case] symbol: char, + #[case] index: usize, + #[case] expected: bool, + ) { + let word: Vec = input.chars().collect(); + assert_eq!( + should_render_symbol_as_english(true, false, false, &[], symbol, &word, index, &[]), + expected + ); + } + + /// 제33항 — 로마자 뒤 쌍점 다음이 한글이 붙은 수이면 한글 쌍점이다. + #[rstest::rstest] + #[case::korean_number_in_the_next_word("Bq:", &["1초에"], false)] + #[case::korean_number_in_the_same_word("Bq:1초에", &[], false)] + #[case::roman_word_follows("Wat:", &["Arun"], true)] + #[case::bare_number_follows("Pt:", &["3"], true)] + #[case::ratio_ending_in_korean("52:24:4이고", &[], false)] + #[case::time_ending_in_korean("22:35에", &[], false)] + #[case::standalone_colon(":", &["2021을"], true)] + fn a_colon_before_a_korean_number_is_korean( + #[case] input: &str, + #[case] remaining: &[&str], + #[case] expected: bool, + ) { + let word: Vec = input.chars().collect(); + let colon = word.iter().position(|ch| *ch == ':').unwrap(); + assert_eq!( + should_render_symbol_as_english(true, true, false, &[], ':', &word, colon, remaining), + expected + ); + } + + /// 제29항 — 로마자 사이의 느낌표·물음표는 로마자 구간을 닫지 않는다. + #[rstest::rstest] + #[case::exclamation_before_next_word("Wow!", '!', true, &["Busan"], true)] + #[case::question_before_next_word("Wow?", '?', true, &["Perfect"], true)] + #[case::before_korean("Wow!", '!', true, &["나"], false)] + #[case::at_text_end("Wow!", '!', true, &[], false)] + #[case::outside_roman_section("Wow!", '!', false, &["Busan"], false)] + fn exclamation_between_roman_words_stays_in_the_section( + #[case] input: &str, + #[case] symbol: char, + #[case] is_english: bool, + #[case] remaining: &[&str], + #[case] expected: bool, + ) { + let word: Vec = input.chars().collect(); + assert_eq!( + should_render_symbol_as_english( + true, + is_english, + false, + &[], + symbol, + &word, + word.len() - 1, + remaining, + ), + expected + ); + } + #[rstest::rstest] #[case::roman_led_chain("CV3-AD685", 3, false, true)] #[case::roman_led_numeric_chain("N-79-20", 4, false, true)] diff --git a/libs/braillify/src/fraction.rs b/libs/braillify/src/fraction.rs index 78b85144..50bc056f 100644 --- a/libs/braillify/src/fraction.rs +++ b/libs/braillify/src/fraction.rs @@ -35,16 +35,6 @@ pub fn encode_fraction(numerator: &str, denominator: &str) -> Result, St Ok(result) } -pub fn encode_fraction_in_context(numerator: &str, denominator: &str) -> Result, String> { - let mut result = vec![60]; - result.extend(encode_number_string(numerator, "fraction numerator")?); - result.push(56); - result.push(12); - result.push(60); - result.extend(encode_number_string(denominator, "fraction denominator")?); - Ok(result) -} - pub fn encode_mixed_fraction( whole: &str, numerator: &str, @@ -289,26 +279,6 @@ mod tests { assert!(result.unwrap_err().contains("denominator")); } - #[test] - fn test_encode_fraction_in_context_simple() { - let result = encode_fraction_in_context("2", "3").unwrap(); - assert_eq!(result, vec![60, 3, 56, 12, 60, 9]); - } - - #[test] - fn test_encode_fraction_in_context_invalid_numerator() { - let result = encode_fraction_in_context("x", "3"); - assert!(result.is_err()); - assert!(result.unwrap_err().contains("numerator")); - } - - #[test] - fn test_encode_fraction_in_context_invalid_denominator() { - let result = encode_fraction_in_context("2", "y"); - assert!(result.is_err()); - assert!(result.unwrap_err().contains("denominator")); - } - #[test] fn test_encode_mixed_fraction_simple() { let result = encode_mixed_fraction("3", "1", "6").unwrap(); diff --git a/libs/braillify/src/korean_char.rs b/libs/braillify/src/korean_char.rs index 65ee8406..4b018bbe 100644 --- a/libs/braillify/src/korean_char.rs +++ b/libs/braillify/src/korean_char.rs @@ -3,26 +3,88 @@ use crate::{ char_struct::KoreanChar, jauem::{choseong::encode_choseong, jongseong::encode_jongseong}, moeum::jungsong::encode_jungsong, + rules::trace::JamoRule, split::split_korean_jauem, utils::build_char, }; +/// Where syllable composition reports the article behind each stretch of cells. +/// +/// Implemented twice so the choice is made at compile time: [`NoSpans`] makes +/// every report vanish, leaving the untraced encoder byte-identical to one with +/// no tracing code at all. A runtime flag instead leaves a branch per jamo on the +/// hottest path in the library, which measured 1-5% slower. +trait SpanSink { + fn report(&mut self, rule: JamoRule, start: usize, end: usize); +} + +struct NoSpans; + +impl SpanSink for NoSpans { + #[inline(always)] + fn report(&mut self, _rule: JamoRule, _start: usize, _end: usize) {} +} + +/// The article behind each stretch of one syllable's cells. +/// +/// Composition takes a different branch depending on which abbreviations exist +/// for the syllable, and each branch cites a different article. Collecting the +/// spans is what lets a caller report 제3항 for a 받침 instead of one composite +/// entry for the whole character. +#[derive(Debug, Default, Clone)] +pub struct JamoSpans { + spans: Vec<(JamoRule, core::ops::Range)>, +} + +impl JamoSpans { + pub fn drain(&mut self) -> impl Iterator)> + '_ { + self.spans.drain(..) + } +} + +impl SpanSink for JamoSpans { + fn report(&mut self, rule: JamoRule, start: usize, end: usize) { + if start < end { + self.spans.push((rule, start as u32..end as u32)); + } + } +} + /// 합성 종성(예: ㄳ→ㄱ+ㅅ) 두 번째 자모가 있으면 인코딩 후 result에 추가한다. -fn extend_compound_jongseong(jong1: Option, result: &mut Vec) -> Result<(), String> { +fn extend_compound_jongseong( + jong1: Option, + result: &mut Vec, + spans: &mut S, +) -> Result<(), String> { if let Some(code) = jong1 { + let start = result.len(); let bytes = encode_jongseong(code)?; result.extend(bytes); + spans.report(JamoRule::Jongseong, start, result.len()); } Ok(()) } pub fn encode_korean_char(korean: &KoreanChar) -> Result, String> { + encode_syllable(korean, &mut NoSpans) +} + +pub fn encode_korean_char_with_spans( + korean: &KoreanChar, + spans: &mut JamoSpans, +) -> Result, String> { + encode_syllable(korean, spans) +} + +fn encode_syllable(korean: &KoreanChar, spans: &mut S) -> Result, String> { let mut result = Vec::new(); let (cho0, cho1) = split_korean_jauem(korean.cho)?; if cho1.is_some() { // 쌍자음이라는 뜻, 초성은 반드시 쌍자음이다. result.push(32); + spans.report(JamoRule::DoubleChoseong, result.len() - 1, result.len()); } + let vowel = JamoRule::for_vowel(korean.jung); if let Some(jong) = korean.jong { let (jong0, jong1) = split_korean_jauem(jong)?; if let Ok(code) = @@ -30,40 +92,62 @@ pub fn encode_korean_char(korean: &KoreanChar) -> Result, String> { { // 초성 자체를 결합 if cho0 != 'ㅇ' { + let start = result.len(); result.push(encode_choseong(cho0)?); + spans.report(JamoRule::Choseong, start, result.len()); } + let start = result.len(); result.extend(code); - extend_compound_jongseong(jong1, &mut result)?; + spans.report(vowel, start, result.len()); + extend_compound_jongseong(jong1, &mut result, spans)?; } else if let Ok(code) = char_shortcut::encode_char_shortcut(build_char(cho0, korean.jung, Some(jong0))) { + let start = result.len(); result.extend(code); - extend_compound_jongseong(jong1, &mut result)?; + spans.report(JamoRule::Shortcut, start, result.len()); + extend_compound_jongseong(jong1, &mut result, spans)?; } else if let Ok(code) = char_shortcut::encode_char_shortcut(build_char(cho0, korean.jung, None)) { + let start = result.len(); result.extend(code); + spans.report(JamoRule::Shortcut, start, result.len()); // 종성 자체를 결합 + let start = result.len(); result.extend(encode_jongseong(jong)?); + spans.report(JamoRule::Jongseong, start, result.len()); } else { // shortcut 이 없으므로 초성, 중성, 종성 모두 결합 if cho0 != 'ㅇ' { + let start = result.len(); result.push(encode_choseong(cho0)?); + spans.report(JamoRule::Choseong, start, result.len()); } + let start = result.len(); result.extend(encode_jungsong(korean.jung)?); + spans.report(vowel, start, result.len()); + let start = result.len(); result.extend(encode_jongseong(jong)?); + spans.report(JamoRule::Jongseong, start, result.len()); } } else if let Ok(code) = char_shortcut::encode_char_shortcut(build_char(cho0, korean.jung, None)) { + let start = result.len(); result.extend(code); + spans.report(JamoRule::Shortcut, start, result.len()); } else { // shortcut 이 없으므로 초성 중성, 모두 결합 if cho0 != 'ㅇ' { + let start = result.len(); result.push(encode_choseong(cho0)?); + spans.report(JamoRule::Choseong, start, result.len()); } + let start = result.len(); result.extend(encode_jungsong(korean.jung)?); + spans.report(vowel, start, result.len()); } Ok(result) diff --git a/libs/braillify/src/lib.rs b/libs/braillify/src/lib.rs index 54e8fe93..04377015 100644 --- a/libs/braillify/src/lib.rs +++ b/libs/braillify/src/lib.rs @@ -182,6 +182,11 @@ mod test_helpers { } pub use encoder::Encoder; +use rules::trace::TraceSink; +pub use rules::trace::{ + EmitterRule, RuleId, RuleKind, RuleOutcome, Trace, TraceEvent, TracePath, registered_rules, + rule_meta, +}; thread_local! { static ENCODER_CACHE: RefCell> = const { RefCell::new(None) }; @@ -452,10 +457,16 @@ fn normalize_pure_roman_compatibility_units<'a>(text: Cow<'a, str>) -> Cow<'a, s /// already defines: U+FF01–U+FF5E are the fullwidth forms of ASCII `!`–`~` /// (`%`, `m`, `&`), U+30FB/U+FF65/U+2027 are CJK spellings of the 가운뎃점 `·` /// (제50항), U+301C is the wave-dash form of the 물결표 `~` (제49항), and U+00B4 -/// is a typed acute accent standing for the 아포스트로피 `'` (제61항). The +/// is a typed acute accent standing for the 아포스트로피 `'` (제61항). The grave +/// accent U+0060 is typed for the same quotation mark (제49항), U+02D9 DOT ABOVE +/// between two letters for the 가운뎃점 (제50항; after `"` it is the 제56항 input +/// notation), and U+2503 is a heavy-weight 세로선 `|` (제71항). The /// zero-width marks U+200B–U+200D and U+FEFF carry no print at all, so they are /// dropped like the soft hyphen. /// +/// U+212B ANGSTROM SIGN is the canonical equivalent (NFC) of `Å`, the unit +/// symbol of 제69항 [붙임 2] and 과학 제30항. +/// /// U+FF1A `:` and U+FF03 `#` are excluded: the standard gives those fullwidth /// glyphs their own meanings — the 옛한글 장음 표시 of 제27항 and the 기수 기호 of /// 수학 제65항 — so they are not print variants of ASCII `:` and `#`. @@ -471,11 +482,20 @@ fn parenthesized_number_expansion(c: char) -> Option { (1..=20).contains(&value).then(|| format!("({value})")) } -fn may_normalize_print_variant(c: char) -> bool { +fn carries_no_print(c: char) -> bool { matches!( c, - '\u{02DA}' | '\u{2010}' | '\u{2011}' | '\u{2043}' | '\u{00AD}' | '\u{00B0}' | '²' | '³' - ) || is_foldable_fullwidth(c) + '\u{00AD}' | '\u{200B}' | '\u{200C}' | '\u{200D}' | '\u{FEFF}' | '\u{FE00}'..='\u{FE0F}' + ) +} + +fn may_normalize_print_variant(c: char) -> bool { + carries_no_print(c) + || matches!( + c, + '\u{02DA}' | '\u{2010}' | '\u{2011}' | '\u{2043}' | '\u{00B0}' | '²' | '³' | '\u{212B}' + ) + || is_foldable_fullwidth(c) || parenthesized_number_expansion(c).is_some() || matches!( c, @@ -486,10 +506,10 @@ fn may_normalize_print_variant(c: char) -> bool { | '\u{2A2F}' | '\u{301C}' | '\u{00B4}' - | '\u{200B}' - | '\u{200C}' - | '\u{200D}' - | '\u{FEFF}' + | '`' + | '\u{02D9}' + | '\u{2503}' + | '\u{00B7}' ) } @@ -551,6 +571,16 @@ fn normalize_print_variants<'a>(text: Cow<'a, str>) -> Cow<'a, str> { continue; } } + // 문장 부호 제21항 [붙임 1·2]: 가운데에 찍은 세 점 이상은 줄임표다. + let dot_run = chars[index..] + .iter() + .take_while(|c| **c == '\u{00B7}') + .count(); + if dot_run >= 3 { + out.push('…'); + index += dot_run; + continue; + } match ch { '\u{02DA}' => out.push('\u{00B0}'), '\u{2010}' | '\u{2011}' | '\u{2043}' => out.push('-'), @@ -558,12 +588,24 @@ fn normalize_print_variants<'a>(text: Cow<'a, str>) -> Cow<'a, str> { out.push('\u{00B7}'); } '\u{2A2F}' => out.push('\u{00D7}'), + '\u{212B}' => out.push('\u{00C5}'), _ if parenthesized_number_expansion(ch).is_some() => { out.push_str(&parenthesized_number_expansion(ch).unwrap_or_default()); } '\u{301C}' => out.push('~'), - '\u{00B4}' => out.push('\''), - '\u{00AD}' | '\u{200B}' | '\u{200C}' | '\u{200D}' | '\u{FEFF}' => {} + '\u{00B4}' | '`' => out.push('\''), + '\u{2503}' => out.push('|'), + '\u{02D9}' + if index + .checked_sub(1) + .is_some_and(|previous| chars[previous].is_alphanumeric()) + && chars + .get(index + 1) + .is_some_and(|next| next.is_alphanumeric()) => + { + out.push('\u{00B7}'); + } + _ if carries_no_print(ch) => {} _ if is_foldable_fullwidth(ch) => { out.push(char::from_u32(ch as u32 - 0xFEE0).unwrap_or(ch)); } @@ -581,11 +623,21 @@ fn normalize_print_variants<'a>(text: Cow<'a, str>) -> Cow<'a, str> { /// (`轉輪륜王` → `⠊⠸⠩⠱⠒…`) 그 문맥은 건너뛴다. 옛한글임을 알리는 표시는 홀로 쓴 /// 자모(`洪ㄱ字`), 방점(`·갈`, `中國·귁`), 한자 뒤에 곧바로 붙인 독음(`君군`, /// `轉輪륜王`의 `輪륜`), 그리고 한글이 하나도 없는 한자만의 표기(`榮養`)다. +/// 방점은 음절 앞에 선다. 낱말 사이의 가운뎃점(`5·18`)은 방점이 아니다 — 국립국어원 +/// 회신 5: 중세국어에는 가운뎃점이 쓰이지 않는다. fn is_middle_korean_hanja_context(chars: &[char]) -> bool { let has_old_jamo = chars.iter().any(|c| { matches!(*c, '\u{3131}'..='\u{318E}' | '\u{1100}'..='\u{11FF}' | '\u{E000}'..='\u{F8FF}') }); - let has_tone_mark = chars.iter().any(|c| matches!(*c, '\u{00B7}' | '\u{FF1A}')); + let has_tone_mark = chars.iter().enumerate().any(|(index, c)| { + matches!(*c, '\u{00B7}' | '\u{FF1A}') + && chars + .get(index + 1) + .is_some_and(|next| matches!(*next, '\u{AC00}'..='\u{D7A3}')) + && index.checked_sub(1).is_none_or(|previous| { + chars[previous].is_whitespace() || hanja::is_hanja(chars[previous]) + }) + }); let has_modern_hangul = chars.iter().any(|c| matches!(*c, '\u{AC00}'..='\u{D7A3}')); let has_gloss = chars.iter().enumerate().any(|(index, c)| { hanja::reading(*c).is_some_and(|reading| { @@ -1061,24 +1113,45 @@ fn decompose_accented_latin<'a>(text: Cow<'a, str>) -> Cow<'a, str> { /// 제37항 — 입력이 (공백을 제외하고) 전부 ASCII 로마자(알파벳)로만 이루어진 /// "고립된 로마자 구간"인지 판별한다. 이런 입력은 국어 점자 문맥(context:korean)에서 /// 로마자표 ⠴ … 종료표 ⠲로 감싼다. `%p`(제69항 단위표)처럼 비알파벳 기호가 섞인 -/// 입력은 로마자 구간이 아니므로 제외된다. +/// 입력은 로마자 구간이 아니므로 제외된다. 그리스 문자만으로 된 입력도 국어 문장 +/// 안에서는 로마자표와 종료표로 감싼다(제31항). 로마자와 섞인 `μm` 은 단위 +/// 기호(제69항 [붙임 1])라 따로 적는다. fn is_isolated_roman_section(text: &str) -> bool { - let mut has_letter = false; - for ch in text.chars() { - if ch == ' ' { - continue; - } - if ch.is_ascii_alphabetic() { - has_letter = true; - } else { - return false; - } - } - has_letter + let letters: Vec = text.chars().filter(|ch| *ch != ' ').collect(); + !letters.is_empty() + && (letters.iter().all(char::is_ascii_alphabetic) + || letters.iter().all(|ch| matches!(ch, 'Α'..='Ω' | 'α'..='ω'))) } /// Encode text to braille with explicit options. pub fn encode_with_options(text: &str, options: &EncodeOptions) -> Result, String> { + encode_with_options_traced(text, options, None) +} + +/// Encode `text` and report which rules produced which output cells. +/// +/// Read [`Trace::path`] before drawing conclusions from partial attribution: +/// each engine reports only the rule families it instruments. +pub fn encode_with_trace(text: &str) -> Result<(Vec, Trace), String> { + encode_with_options_and_trace(text, &EncodeOptions::default()) +} + +/// [`encode_with_trace`] with an explicit encoding mode. +pub fn encode_with_options_and_trace( + text: &str, + options: &EncodeOptions, +) -> Result<(Vec, Trace), String> { + let mut trace = Trace::default(); + let cells = encode_with_options_traced(text, options, Some(&mut trace))?; + trace.set_output_len(cells.len() as u32); + Ok((cells, trace)) +} + +fn encode_with_options_traced( + text: &str, + options: &EncodeOptions, + mut trace: Option<&mut Trace>, +) -> Result, String> { use crate::rules::context::EncodingMode; // PDF 수학 — Mathematical Alphanumeric 변형(italic/bold/script 등)을 ASCII로 @@ -1091,6 +1164,93 @@ pub fn encode_with_options(text: &str, options: &EncodeOptions) -> Result 0 { + cells.push(255); + } + let paragraph_options = if formula_paragraph(paragraph) { + &science_options + } else { + options + }; + cells.extend(encode_with_options(paragraph, paragraph_options)?); + } + mark_trace_path(&mut trace, TracePath::KoreanRules); + return Ok(cells); + } + } + if options.default_mode == Some(EncodingMode::Science) + && let Some(cells) = crate::rules::science::ring::encode(text) + .or_else(|| crate::rules::science::weather::encode(text)) + .or_else(|| crate::rules::science::circuit::encode(text)) + { + mark_trace_path(&mut trace, TracePath::KoreanRules); + return Ok(cells); + } + if spatial && let Some(cells) = crate::rules::science::bond_lines::encode(text) { + mark_trace_path(&mut trace, TracePath::KoreanRules); + return Ok(cells); + } + if reads_science_shapes + && let Some(cells) = crate::rules::science::diagram::encode(text, spatial) + { + mark_trace_path(&mut trace, TracePath::KoreanRules); + return Ok(cells); + } + if reads_science_shapes + && let Some(cells) = crate::rules::science::quantity::encode(text, |segment| { + encode_with_options(segment, options) + }) + { + mark_trace_path(&mut trace, TracePath::MathExpression); + return Ok(cells); + } + // 과학 제4·7항 — 화학식은 로마자 낱말도 수식도 아니다. 영어·수학 경로와 글꼴 + // 정규화를 건너뛰어, 토큰 단계의 화학식 규칙이 강조(제7항 5)까지 그대로 본다. + let chemistry = + reads_science_shapes && crate::rules::science::formula::owns_text(text, science); + // 한글 제69항 — 한글 없이 단위 기호 글자(`㎜Hg`, `㎾h`)로 적힌 글은 로마자 낱말이 + // 아니라 단위다. 정규화가 글자를 `mm`·`kW` 로 풀기 전에 그 신호를 잡아 둔다. + let unit_glyphs = science + && !has_korean + && text + .chars() + .any(crate::rules::korean::rule_69::is_compatibility_unit_presentation); // Content-routed English must be considered before math normalization. The // legacy math path decomposes accented Latin for Korean math 제65항, which turns // UEB §4.2 modified letters (`Rhône`, `Hwǣr`) into combining-mark sequences and @@ -1099,11 +1259,17 @@ pub fn encode_with_options(text: &str, options: &EncodeOptions) -> Result Result Result Result Result { + mark_trace_path(&mut trace, TracePath::MathExpression); + return Ok(bytes); + } + Err(_) => { + if let (Some(sink), Some(mark)) = (trace.as_deref_mut(), mark) { + sink.rollback_to(mark); + } + } } } @@ -1365,22 +1563,79 @@ pub fn encode_with_options(text: &str, options: &EncodeOptions) -> Result>()); let wrap_roman_section = matches!(options.default_mode, Some(EncodingMode::English)) - || (matches!(options.default_mode, Some(EncodingMode::Korean)) + || ((unit_glyphs + || science_roman_section + || matches!(options.default_mode, Some(EncodingMode::Korean))) && is_isolated_roman_section(text)); if wrap_roman_section && !result.is_empty() { result.insert(0, 52); result.push(50); + if let Some(sink) = trace { + sink.shift_output(1); + let end = result.len() as u32; + for output in [0..1, end - 1..end] { + sink.push(TraceEvent { + rule: RuleId::emitter(EmitterRule::RomanSectionMarker), + outcome: RuleOutcome::Consumed, + token_index: 0, + word_chars: 0..0, + output, + }); + } + } } Ok(result) }) } +/// Run a UEB entry point, recording its rule spans when a trace is collected. +/// +/// Both closures must invoke the SAME entry point: `traced` differs only by +/// collecting spans around it. Crossing them would make a traced encode take a +/// different route than an untraced one and silently change output. +fn encode_ueb( + text: &str, + trace: &mut Option<&mut Trace>, + untraced: impl FnOnce(&str) -> Option>, + traced: impl FnOnce(&str) -> Option<(Vec, Vec)>, +) -> Option> { + let Some(sink) = trace.as_deref_mut() else { + return untraced(text); + }; + let (bytes, spans) = traced(text)?; + for (rule, output) in spans { + sink.push(TraceEvent { + rule, + outcome: RuleOutcome::Consumed, + token_index: 0, + word_chars: 0..0, + output, + }); + } + Some(bytes) +} + +fn mark_trace_path(trace: &mut Option<&mut Trace>, path: TracePath) { + if let Some(sink) = trace.as_deref_mut() { + sink.set_path(path); + } +} + /// Encode text with explicit formatting spans. pub fn encode_with_formatting(text: &str, spans: &[FormattingSpan]) -> Result, String> { if spans.is_empty() { @@ -1427,6 +1682,488 @@ pub fn encode_to_braille_font(text: &str) -> Result { .collect::()) } +/// [`encode`] with the text read in a named context, for input whose print +/// shape alone does not decide the rule (`44+XX` is a Roman section in plain +/// Korean text but a stand-apart expression in science). +/// +/// The names are the ones the test fixtures use: `korean`, `english`, `math`, +/// `number`, `middle_korean`, `object_symbol`, `ipa`, `science`, and +/// `science_spatial` (science, writing diagrams in the spatial form). +pub fn encode_in_context(text: &str, context: &str) -> Result, String> { + let mode = context + .parse::() + .map_err(|()| format!("unknown context: {context}"))?; + encode_with_options( + text, + &EncodeOptions { + default_mode: Some(mode), + }, + ) +} + +/// Unicode version of [`encode_in_context`]. +pub fn encode_to_unicode_in_context(text: &str, context: &str) -> Result { + Ok(encode_in_context(text, context)? + .iter() + .map(|c| unicode::encode_unicode(*c)) + .collect()) +} + +/// Braille-font version of [`encode_in_context`]. +pub fn encode_to_braille_font_in_context(text: &str, context: &str) -> Result { + Ok(encode_in_context(text, context)? + .iter() + .map(|c| unicode::encode_unicode(*c)) + .collect()) +} + +#[cfg(test)] +mod trace_tests { + use super::*; + use crate::rules::context::EncodingMode; + + fn korean_mode() -> EncodeOptions { + EncodeOptions { + default_mode: Some(EncodingMode::Korean), + } + } + + #[rstest::rstest] + #[case::korean_syllables("안녕")] + #[case::korean_abbreviation("그래서")] + #[case::english_prose("hello")] + #[case::latex_fraction("$\\frac{3}{4}$")] + #[case::mixed_sentence("가나다 라마")] + #[case::roman_inside_korean("가 ABC")] + #[case::digits("2024년")] + fn tracing_leaves_the_encoded_cells_unchanged(#[case] input: &str) { + let plain = encode(input).expect("input must encode"); + let (traced, _) = encode_with_trace(input).expect("input must encode"); + + assert_eq!(plain, traced); + } + + #[rstest::rstest] + #[case::korean("안녕", TracePath::KoreanRules)] + #[case::mixed("가나다 라마", TracePath::KoreanRules)] + #[case::english("hello", TracePath::EnglishUeb)] + fn path_names_the_engine_that_ran(#[case] input: &str, #[case] expected: TracePath) { + let (_, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!(trace.path(), expected); + } + + #[test] + fn korean_syllables_attribute_every_cell_to_a_registered_rule() { + let (cells, trace) = encode_with_trace("안녕").expect("input must encode"); + + assert_eq!(trace.attributed_cells(), cells.len() as u32); + assert_eq!(trace.unattributed_cells(), 0); + assert!( + trace.events().iter().all(|e| e.rule.meta().is_some()), + "every recorded id resolves against the registry" + ); + } + + /// 약자 abbreviation is a token-level rewrite whose cells never reach the + /// character engine, so it is attributed through the token-origin side table + /// rather than by the character rule loop. + #[test] + fn token_rule_output_is_attributed_to_the_token_engine() { + let (cells, trace) = encode_with_trace("그래서").expect("input must encode"); + + assert!(!cells.is_empty(), "the abbreviation still encodes"); + assert_eq!(trace.attributed_cells(), cells.len() as u32); + assert!( + trace + .events() + .iter() + .all(|e| e.rule.kind() == Some(RuleKind::Token)), + "abbreviation cells come from a token rule: {:?}", + trace.events() + ); + } + + /// A number written straight against an ASCII unit is emitted as one piece, + /// so its cells are credited in the emitter rather than by the character + /// loop that attributes the digits and the letters separately. + #[rstest::rstest] + #[case::centimetre("3cm")] + #[case::kilogram("5kg")] + fn a_measured_quantity_is_credited_to_the_measurement_rule(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert!(!cells.is_empty(), "the measurement still encodes"); + assert!( + trace.events().iter().any(|e| e + .rule + .meta() + .is_some_and(|m| m.name == "measurement_symbols")), + "the measurement cells name their rule: {:?}", + trace.events() + ); + } + + /// UEB picks contractions by a cell-minimising search, so only the winning + /// path may be credited. Every recorded range must therefore land inside the + /// output and name a UEB rule. + #[rstest::rstest] + #[case::uncontracted("hello")] + #[case::sentence("the child was here")] + #[case::accented("naive")] + fn english_attributes_only_the_selected_contraction_path(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!(trace.path(), TracePath::EnglishUeb); + assert!(trace.attributed_cells() > 0, "UEB now names its rules"); + assert!( + trace.events().iter().all(|event| { + // Inter-word blanks belong to the emitter, not to a UEB rule. + matches!( + event.rule.kind(), + Some(RuleKind::EnglishUeb | RuleKind::Emitter) + ) && event.output.start < event.output.end + && event.output.end as usize <= cells.len() + }), + "{:?}", + trace.events() + ); + } + + /// A whole-word sign is a table lookup rather than a contraction search, so + /// it is recorded where the table is consulted. Its section is known exactly, + /// so the cells are named rather than left unexplained. + #[rstest::rstest] + #[case::alphabetic_wordsign("knowledge", "10.1")] + #[case::shortform("about", "10.9")] + #[case::lower_wordsign("enough", "10.5")] + fn english_wordsigns_name_their_section(#[case] input: &str, #[case] section: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!(trace.attributed_cells(), cells.len() as u32); + let sections: Vec<&str> = trace + .events() + .iter() + .filter_map(|event| event.rule.meta().map(|meta| meta.section)) + .collect(); + assert!( + sections.contains(§ion), + "expected §{section} among {sections:?}" + ); + } + + #[rstest::rstest] + #[case::korean("가나다 라마")] + #[case::korean_prose("나는 학교에 간다")] + #[case::english("the child was here")] + #[case::mixed_numbers("2024년 제12항")] + #[case::math_plain("3+4=7")] + #[case::math_variables("$x^2+y^2=z^2$")] + #[case::math_function("$\\sin x$")] + #[case::latex_fraction("$\\frac{3}{4}$")] + #[case::ueb_capital_then_digits("A1")] + #[case::ueb_capital_digits_and_decimal("Q50 2.2d")] + // 제35항 numeric bridge resuming into a lowercase a-j letter: UEB 6.5.2 makes + // the emitter write a continuation cell there, and it must name itself. + #[case::roman_number_bridge_into_low_letter("가나 (1c) 다라")] + fn every_output_cell_is_accounted_for(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!( + trace.attributed_cells(), + cells.len() as u32, + "unattributed cells in {input:?} with output {cells:?}: {:?}", + trace.events() + ); + } + + #[test] + fn capitalised_spelled_word_keeps_letter_attribution() { + let (cells, trace) = encode_with_trace("MP3").expect("input must encode"); + + assert_eq!(trace.attributed_cells(), cells.len() as u32); + let letter_cells = trace + .events() + .iter() + .filter(|event| event.rule.meta().is_some_and(|meta| meta.section == "4.1")) + .map(|event| event.output.end - event.output.start) + .sum::(); + assert_eq!(letter_cells, 2, "only M and P belong to the letter rule"); + } + + #[rstest::rstest] + #[case::ethene_hydration("C_{2}H_{4}(g) + H_{2}O(g) -> C_{2}H_{5}OH(g)")] + #[case::aluminium_ion("Al3+(aq) + 3e- -> Al(s)")] + #[case::water_formation("2H_{2}(g) + O_{2}(g) -> 2H_{2}O(g)")] + fn chemical_equation_cells_are_accounted_for(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!( + trace.attributed_cells(), + cells.len() as u32, + "unattributed cells in {input:?} with output {cells:?}: {:?}", + trace.events() + ); + } + + /// The two paths that still leave cells unexplained, pinned to their exact + /// numbers so the gap cannot widen unnoticed and any narrowing is visible. + /// + /// Both are mode indicators rather than content: the Roman indicator the + /// emitter writes ahead of a Roman run, and the numeric/symbol cells the UEB + /// engine writes outside its contraction search. + #[rstest::rstest] + #[case::roman_in_korean("가영이는 Los Angeles에 산다")] + #[case::numbers_and_symbols("50% & 3 items")] + #[case::measurement("3kg 5%")] + #[case::acronym_with_digit("MP3 player")] + fn indicator_and_numeric_cells_are_accounted_for(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!( + trace.attributed_cells(), + cells.len() as u32, + "unattributed cells in {input:?}: {:?}", + trace.events() + ); + } + + /// Every rule must name a cell range that is really its own, so a cell may + /// never be claimed by two rules at once. + #[rstest::rstest] + #[case::korean("안녕하세요")] + #[case::mixed("가영이는 Los Angeles에 산다")] + #[case::measurement("3kg 5%")] + #[case::english("the child was here")] + #[case::math("3+4=7")] + #[case::chemical("C_{2}H_{4}(g) + H_{2}O(g) -> C_{2}H_{5}OH(g)")] + #[case::inline_chemical("$C_{2}H_{4}$(g)+$H_{2}O$(g)→$C_{2}H_{5}OH$(g)")] + fn no_cell_is_claimed_twice(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + let mut claims = vec![0u32; cells.len()]; + for event in trace.events() { + for cell in event.output.clone() { + claims[cell as usize] += 1; + } + } + + assert!( + claims.iter().all(|count| *count == 1), + "cells claimed {claims:?} times in {input:?}: {:?}", + trace.events() + ); + } + + /// A math expression reaches the emitter as one pre-encoded run, so without + /// the math engine's own spans it would report only the token rule that + /// detected it. + #[rstest::rstest] + #[case::sum("3+4=7", "1")] + #[case::superscript("$x^2$", "18")] + #[case::function("$\\sin x$", "47")] + fn math_expressions_name_their_math_article(#[case] input: &str, #[case] section: &str) { + let (_, trace) = encode_with_trace(input).expect("input must encode"); + + let sections: Vec<&str> = trace + .events() + .iter() + .filter(|event| event.rule.kind() == Some(RuleKind::Math)) + .filter_map(|event| event.rule.meta().map(|meta| meta.section)) + .collect(); + + assert!( + sections.contains(§ion), + "expected 수학 제{section}항 among {sections:?}" + ); + } + + #[rstest::rstest] + #[case::korean("안녕하세요")] + #[case::mixed("가나다 라마 ABC")] + #[case::numbers("제12항 3개")] + fn every_recorded_range_lies_inside_the_output(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + for event in trace.events() { + assert!( + event.output.end as usize <= cells.len(), + "{event:?} runs past {} cells", + cells.len() + ); + assert!(event.output.start <= event.output.end); + } + } + + /// 제37항 inserts the Roman indicator at cell 0 *after* encoding, so every + /// range recorded before that insertion points one cell short unless shifted. + #[test] + fn roman_wrap_shifts_recorded_ranges_onto_the_right_cells() { + let (cells, trace) = + encode_with_options_and_trace("ABC", &korean_mode()).expect("input must encode"); + + assert_eq!(cells.first(), Some(&52), "제37항 로마자표"); + let spans: Vec<&[u8]> = trace + .events() + .iter() + .map(|event| &cells[event.output.start as usize..event.output.end as usize]) + .collect(); + assert!( + spans.contains(&&[1u8, 3, 9][..]), + "one span must render A, B, C; got {spans:?}" + ); + } + + /// The encoder is cached per thread, so a leaked sink would make the second + /// trace of the same input differ from the first. + #[test] + fn trace_does_not_bleed_across_calls_on_the_cached_encoder() { + let (_, before) = encode_with_trace("안녕").expect("input must encode"); + let _ = encode_with_trace("hello").expect("input must encode"); + let (_, after) = encode_with_trace("안녕").expect("input must encode"); + + assert_eq!(before, after); + } + + /// 안 = ㅇ + ㅏ + ㄴ, so its two cells are the 제6항 vowel and the 제3항 받침 + /// rather than one composite entry for the syllable. + #[test] + fn syllable_cells_name_their_own_article() { + let (cells, trace) = encode_with_trace("안녕").expect("input must encode"); + + let article_at = |cell: u32| { + trace + .rules_at_cell(cell) + .first() + .and_then(|rule| rule.meta()) + .map(|meta| (meta.section, meta.name)) + }; + + assert_eq!(article_at(0), Some(("6", "syllable_jungseong"))); + assert_eq!(article_at(1), Some(("3", "syllable_jongseong"))); + assert!(trace.rules_at_cell(cells.len() as u32).is_empty()); + } + + #[rstest::rstest] + #[case::vowel("안녕", "6")] + #[case::final_consonant("안녕", "3")] + #[case::double_initial("깎다", "2")] + #[case::initial_consonant("라", "1")] + fn syllable_composition_cites_jamo_articles(#[case] input: &str, #[case] section: &str) { + let (_, trace) = encode_with_trace(input).expect("input must encode"); + + let sections: Vec<&str> = trace + .events() + .iter() + .filter(|event| event.rule.kind() == Some(RuleKind::Jamo)) + .filter_map(|event| event.rule.meta().map(|meta| meta.section)) + .collect(); + + assert!( + sections.contains(§ion), + "expected 제{section}항 among {sections:?}" + ); + } + + /// 수학 제32·33항's 합동/기하 glyphs (`△`, `→`, `□`, `≅`) pull the whole string + /// onto the math route, which encodes *before* the token pipeline and so + /// carries its own sink instead of the emitter's origin table. Without that + /// sink the expression would encode with nothing recorded at all. + #[rstest::rstest] + #[case::congruence_triangle("△ABC")] + #[case::implication("p → q")] + #[case::relation("A≅B")] + fn a_whole_route_math_expression_names_its_math_rules(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_eq!(encode(input).expect("input must encode"), cells); + assert_eq!(trace.path(), TracePath::MathExpression); + assert_eq!(trace.attributed_cells(), cells.len() as u32); + assert!( + trace + .events() + .iter() + .any(|event| event.rule.kind() == Some(RuleKind::Math)), + "the math engine must name itself: {:?}", + trace.events() + ); + } + + /// The whole-route math encoder runs speculatively: it emits cells for the + /// tokens it consumed and only then discovers a token it cannot encode, at + /// which point the Korean pipeline re-encodes the whole input. The discarded + /// cells never ship, so crediting their rules would name rules that did not + /// produce the output — and would leave two rules claiming the same cell. + #[rstest::rstest] + #[case::trailing_at_sign("△AB@")] + #[case::percent_between_operands("△A%B")] + #[case::bare_at_sign("A□@B")] + fn a_failed_math_route_credits_no_rule_for_the_cells_it_threw_away(#[case] input: &str) { + let (cells, trace) = encode_with_trace(input).expect("input must encode"); + + assert_ne!( + trace.path(), + TracePath::MathExpression, + "the math route must have failed for this case to mean anything" + ); + assert!( + trace + .events() + .iter() + .all(|event| event.rule.kind() != Some(RuleKind::Math)), + "rolled-back math rules must not survive: {:?}", + trace.events() + ); + + let mut claims = vec![0u32; cells.len()]; + for event in trace.events() { + for cell in event.output.clone() { + claims[cell as usize] += 1; + } + } + assert!( + claims.iter().all(|count| *count == 1), + "cells claimed {claims:?} times in {input:?}: {:?}", + trace.events() + ); + } + + /// `EncodingMode::English` forces the UEB engine even where content routing + /// would not pick it — a letterless `4:30` reads as a Korean-context number + /// otherwise. The forced entry point has to collect the same spans as the + /// content-routed one, or a declared-English testcase would trace as though + /// no rule had run. + #[rstest::rstest] + #[case::letterless_time("4:30")] + #[case::prose("the child")] + fn forced_english_mode_still_names_its_ueb_rules(#[case] input: &str) { + let options = EncodeOptions { + default_mode: Some(EncodingMode::English), + }; + let (cells, trace) = + encode_with_options_and_trace(input, &options).expect("input must encode"); + + assert_eq!( + encode_with_options(input, &options).expect("input must encode"), + cells + ); + assert_eq!(trace.path(), TracePath::EnglishUeb); + assert_eq!(trace.attributed_cells(), cells.len() as u32); + } + + #[test] + fn contributing_rules_lists_each_rule_once() { + let (_, trace) = encode_with_trace("가나다 라마").expect("input must encode"); + let contributing = trace.contributing_rules(); + let mut unique = contributing.clone(); + unique.sort_unstable(); + unique.dedup(); + + assert!(!contributing.is_empty()); + assert_eq!(contributing.len(), unique.len()); + } +} + #[cfg(test)] mod state_bleed_tests { use super::encode; @@ -1717,6 +2454,63 @@ mod test { )] } + /// 과학 문맥은 과학 점자 전체를 적는 문맥이다. fixture 가 밝힌 문맥과 상관없이 + /// 과학 fixture 는 모두 과학 문맥에서도 같은 답을 내야 한다. 공간 표기 형식을 + /// 밝힌 fixture 는 그 형식을 고른 과학 문맥에서 본다. + #[test] + fn every_science_fixture_holds_in_the_science_context() { + use crate::rules::context::EncodingMode; + let rule_map = load_test_case_rule_map(); + let mut checked = 0; + let mut failures = Vec::new(); + for (path, key) in collect_test_files(&rule_map) { + if !key.starts_with("science/") { + continue; + } + let filename = path.file_name().unwrap().to_string_lossy().to_string(); + let records: Vec = + serde_json::from_str(&std::fs::read_to_string(&path).unwrap()).unwrap(); + for (line, record) in records.iter().enumerate() { + if record.get("limitation").is_some() { + continue; + } + checked += 1; + let input = record["input"].as_str().unwrap(); + let answers: Vec = testcase_answer_forms(record, &filename, line) + .into_iter() + .map(|(_, _, unicode)| unicode) + .collect(); + let spatial = + record.get("context").and_then(|c| c.as_str()) == Some("science_spatial"); + let options = EncodeOptions { + default_mode: Some(if spatial { + EncodingMode::ScienceSpatial + } else { + EncodingMode::Science + }), + }; + let actual = encode_with_options(input, &options).map(|cells| { + cells + .iter() + .map(|cell| unicode::encode_unicode(*cell)) + .collect::() + }); + if !actual.as_ref().is_ok_and(|actual| answers.contains(actual)) { + failures.push(format!( + "{filename}:{line} {input:?} {answers:?} != {actual:?}" + )); + } + } + } + assert!(checked > 0, "no science fixture was found"); + assert!( + failures.is_empty(), + "{} of {checked} science fixtures differ in the science context:\n{}", + failures.len(), + failures.join("\n") + ); + } + #[derive(serde::Deserialize)] struct NiklCorpusCase { input: String, @@ -2340,6 +3134,7 @@ mod test { // asserted by `rule_64::lone_combining_square_is_no_op`). let is_only_nonemitting = s.chars().all(|c| { c == ' ' + || carries_no_print(c) || matches!( crate::char_struct::CharType::new(c), Ok(crate::char_struct::CharType::CombiningMark) @@ -2482,6 +3277,25 @@ mod coverage_targeted_tests { assert_eq!(kind.markers(), (start, end)); } + #[rstest::rstest] + #[case::tone_mark_opens_a_word("·갈 〔 刀 〕", true)] + #[case::rising_tone_opens_a_word(":돌 〔 石 〕", true)] + #[case::tone_mark_after_hanja("中國·귁", true)] + #[case::separator_between_numbers("5·18 고(故)", false)] + #[case::separator_between_words("영업·마케팅 지선(支線)", false)] + #[case::separator_after_a_bracket("최연고(80)·서정수 고(故)", false)] + fn tells_a_tone_mark_from_a_middle_dot(#[case] text: &str, #[case] middle_korean: bool) { + let chars: Vec = text.chars().collect(); + assert_eq!(is_middle_korean_hanja_context(&chars), middle_korean); + } + + #[rstest::rstest] + #[case::hanja_after_a_middle_dot("5·18 고(故)", "5·18 고(고)")] + #[case::variation_selector("\u{FE0F}가로 350mm", "가로 350mm")] + fn writes_what_the_print_shows(#[case] text: &str, #[case] same_as: &str) { + assert_eq!(encode(text), encode(same_as)); + } + /// Mathematical italic small h (U+210E) normalizes to plain 'h'. #[test] fn normalize_math_planck_h() { @@ -2864,6 +3678,7 @@ mod coverage_targeted_tests { #[case::word("but", true)] #[case::phrase_with_spaces("Table of Contents", true)] #[case::percent_unit("%p", false)] + #[case::greek_unit("Ω", true)] #[case::has_digit("abc123", false)] #[case::empty("", false)] #[case::only_space(" ", false)] @@ -3243,11 +4058,30 @@ mod print_variant_fold_coverage { #[case::hyphenation_point("\u{2027}", "\u{00B7}")] #[case::one_dot_leader("\u{2024}", "\u{00B7}")] #[case::vector_cross("\u{2A2F}", "\u{00D7}")] + #[case::angstrom_sign("\u{212B}", "\u{00C5}")] #[case::parenthesised_five("\u{2478}", "(5)")] #[case::parenthesised_twenty("\u{2487}", "(20)")] #[case::wave_dash("\u{301C}", "~")] #[case::acute_accent("\u{00B4}", "'")] + #[case::grave_accent_quote("`\u{D06C}\u{B9BC}`", "'\u{D06C}\u{B9BC}'")] + #[case::dot_above_between_words( + "\u{C601}\u{C5C5}\u{02D9}\u{B9C8}", + "\u{C601}\u{C5C5}\u{00B7}\u{B9C8}" + )] + #[case::dot_above_opening_emphasis("\"\u{02D9}\u{AC15}", "\"\u{02D9}\u{AC15}")] + #[case::dot_above_ending_a_word("\u{C601}\u{02D9}", "\u{C601}\u{02D9}")] + #[case::heavy_vertical_line("\u{2503}\u{ADF8}", "|\u{ADF8}")] + #[case::three_middle_dots( + "\u{AC00}\u{00B7}\u{00B7}\u{00B7}\u{B098}", + "\u{AC00}\u{2026}\u{B098}" + )] + #[case::six_middle_dots("\u{00B7}\u{00B7}\u{00B7}\u{00B7}\u{00B7}\u{00B7}", "\u{2026}")] + #[case::two_middle_dots_stay( + "\u{AC00}\u{00B7}\u{00B7}\u{B098}", + "\u{AC00}\u{00B7}\u{00B7}\u{B098}" + )] #[case::soft_hyphen("\u{00AD}", "")] + #[case::variation_selector("\u{FE00}", "")] #[case::zero_width_space("\u{200B}", "")] #[case::zero_width_joiner("\u{200D}", "")] #[case::byte_order_mark("\u{FEFF}", "")] @@ -3304,3 +4138,65 @@ mod print_variant_fold_coverage { ); } } + +#[cfg(test)] +mod science_context_tests { + use super::*; + + /// 과학 문맥에서 따로 선 로마자는 단위만 로마자 구간이고(제30항), 화학식과 + /// 유전자는 과학 기호로 적는다(제7·23항). + #[rstest::rstest] + #[case::formula_with_a_two_letter_element("NaCl", "⠠⠝⠁⠠⠉⠇")] + #[case::unit("HP", "⠴⠠⠠⠓⠏⠲")] + #[case::gene("AA", "⠠⠠⠁⠁")] + #[case::chromosomes_in_a_sentence("염색체는 44+XY이다.", "⠱⠢⠠⠗⠁⠰⠝⠉⠵⠀⠀⠼⠙⠙⠢⠠⠠⠭⠽⠀⠀⠕⠊⠲")] + #[case::abbreviations_keep_the_capital_phrase( + "DNA, RNA, ATP는 중요하다.", + "⠴⠠⠠⠠⠙⠝⠁⠂⠀⠗⠝⠁⠂⠀⠁⠞⠏⠠⠄⠲⠉⠵⠀⠨⠍⠶⠬⠚⠊⠲" + )] + fn writes_science_notation_apart_from_units(#[case] input: &str, #[case] expected: &str) { + assert_eq!( + encode_to_unicode_in_context(input, "science"), + Ok(expected.to_string()) + ); + assert_eq!( + encode_to_braille_font_in_context(input, "science"), + Ok(expected.to_string()) + ); + } + + /// 통일영어점자 §8.8.3 — 한글 없는 글의 화학식은 영어 점자로 적고, 과학 문맥에서만 + /// 과학 기호로 적는다. + #[rstest::rstest] + #[case::formula("HOCH₂", "⠠⠓⠠⠕⠠⠉⠠⠓⠰⠢⠼⠃", "⠠⠠⠠⠓⠕⠉⠓⠰⠼⠃⠠⠄")] + #[case::chain_of_elements("H-O-H", "⠰⠰⠠⠓⠤⠠⠕⠤⠠⠓", "⠠⠠⠠⠓⠰⠂⠕⠰⠂⠓⠠⠄")] + fn reads_hangul_free_text_as_science_only_in_its_context( + #[case] input: &str, + #[case] without_context: &str, + #[case] in_science: &str, + ) { + assert_eq!(encode_to_unicode(input), Ok(without_context.to_string())); + assert_eq!( + encode_to_unicode_in_context(input, "science"), + Ok(in_science.to_string()) + ); + } + + #[test] + fn names_the_context_it_cannot_read() { + assert_eq!( + encode_in_context("pOH", "chemistry"), + Err("unknown context: chemistry".to_string()) + ); + } + + #[rstest::rstest] + #[case::science("science", "⠴⠏⠠⠕⠠⠓")] + #[case::korean("korean", "⠴⠏⠠⠠⠕⠓⠲")] + fn reads_the_same_text_by_its_context(#[case] context: &str, #[case] expected: &str) { + assert_eq!( + encode_to_unicode_in_context("pOH", context), + Ok(expected.to_string()) + ); + } +} diff --git a/libs/braillify/src/math_symbol_shortcut.rs b/libs/braillify/src/math_symbol_shortcut.rs index 4baf0c0e..ba5cfb02 100644 --- a/libs/braillify/src/math_symbol_shortcut.rs +++ b/libs/braillify/src/math_symbol_shortcut.rs @@ -1,250 +1,546 @@ use phf::phf_map; +use crate::rules::RuleMeta; use crate::unicode::decode_unicode; -static SHORTCUT_MAP: phf::Map = phf_map! { - // PDF 한국 점자 규정 (수학) — 동그라미 숫자 ①②③④⑤⑥⑦⑧⑨⑩ - '\u{2460}' => &[decode_unicode('⠼'), decode_unicode('⠂')], // ① - '\u{2461}' => &[decode_unicode('⠼'), decode_unicode('⠆')], // ② - '\u{2462}' => &[decode_unicode('⠼'), decode_unicode('⠒')], // ③ - '\u{2463}' => &[decode_unicode('⠼'), decode_unicode('⠲')], // ④ - '\u{2464}' => &[decode_unicode('⠼'), decode_unicode('⠢')], // ⑤ - '\u{2465}' => &[decode_unicode('⠼'), decode_unicode('⠖')], // ⑥ - '\u{2466}' => &[decode_unicode('⠼'), decode_unicode('⠶')], // ⑦ - '\u{2467}' => &[decode_unicode('⠼'), decode_unicode('⠦')], // ⑧ - '\u{2468}' => &[decode_unicode('⠼'), decode_unicode('⠔')], // ⑨ - '\u{2469}' => &[decode_unicode('⠼'), decode_unicode('⠴')], // ⑩ - '+' => &[decode_unicode('⠢')], // 5 (덧셈표) - '/' => &[decode_unicode('⠸'), decode_unicode('⠌')], // _/ (분수 기호) - '\u{2212}' => &[decode_unicode('⠔')], // 9 (뺄셈표) - '\u{00D7}' => &[decode_unicode('⠡')], // * (곱셈표) - '\u{00F7}' => &[decode_unicode('⠌'), decode_unicode('⠌')], // // (나눗셈표) - '=' => &[decode_unicode('⠒'), decode_unicode('⠒')], // 33 (등호) - '>' => &[decode_unicode('⠢'), decode_unicode('⠢')], // 55 (보다크다) - '<' => &[decode_unicode('⠔'), decode_unicode('⠔')], // 99 (보다작다) - '\u{2260}' => &[decode_unicode('⠨'), decode_unicode('⠒'), decode_unicode('⠒')], // .33 (같지않다) - '\u{2265}' => &[decode_unicode('⠲'), decode_unicode('⠲')], // 44 (크거나같다) - '\u{2267}' => &[decode_unicode('⠲'), decode_unicode('⠲')], // 44 (크거나같다) - '\u{2264}' => &[decode_unicode('⠖'), decode_unicode('⠖')], // 66 (작거나같다) - '\u{2266}' => &[decode_unicode('⠖'), decode_unicode('⠖')], // 66 (작거나같다) - '\u{2252}' => &[decode_unicode('⠐'), decode_unicode('⠒'), decode_unicode('⠒')], // "33 (근삿값) - '\u{2236}' => &[decode_unicode('⠐'), decode_unicode('⠂')], // "1 (비) - '\u{2192}' => &[decode_unicode('⠒'), decode_unicode('⠕')], // 3o (오른쪽 화살표) - '\u{2190}' => &[decode_unicode('⠪'), decode_unicode('⠒')], // [3 (왼쪽 화살표) - '\u{2194}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠕')], // [3o (양쪽 화살표) - '\u{2191}' => &[decode_unicode('⠰'), decode_unicode('⠒'), decode_unicode('⠕')], // ;3o (위쪽 화살표) - '\u{2193}' => &[decode_unicode('⠘'), decode_unicode('⠒'), decode_unicode('⠕')], // ^3o (아래쪽 화살표) - '\u{21D2}' => &[decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], // 33o (항진명제) - '\u{21D4}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], // [33o (필요충분) - '\u{21C4}' => &[decode_unicode('⠪'), decode_unicode('⠶'), decode_unicode('⠕')], // [7o (동치명제) - '\u{2032}' => &[decode_unicode('⠤')], // - (프라임) - '\u{2033}' => &[decode_unicode('⠤'), decode_unicode('⠤')], // -- (더블 프라임, PDF 제17항) - '\u{2034}' => &[decode_unicode('⠤'), decode_unicode('⠤'), decode_unicode('⠤')], // --- (트리플 프라임) - '\u{00B2}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠃')], // ^#b (제곱) - '\u{00B3}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠉')], // ^#c (세제곱) - '\u{2074}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠙')], // ^#d (네제곱) - '\u{2075}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠑')], // ^#e (오제곱) - '\u{2077}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠛')], // ^#g (칠제곱) - '\u{2079}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠊')], // ^#i (구제곱) - '\u{00B9}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠁')], // ^#a (1제곱) - '\u{2070}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠚')], // ^#j (0제곱) - '\u{1D4F}' => &[decode_unicode('⠘'), decode_unicode('⠅')], // ^k (위첨자 k) - '\u{1D50}' => &[decode_unicode('⠘'), decode_unicode('⠍')], // ^m (위첨자 m) - '\u{02E3}' => &[decode_unicode('⠘'), decode_unicode('⠭')], // ^x (위첨자 x) - '\u{207D}' => &[decode_unicode('⠘'), decode_unicode('⠦')], // ^8 (위첨자 () - '\u{207E}' => &[decode_unicode('⠴')], // 0 (위첨자 )) - '\u{207F}' => &[decode_unicode('⠘'), decode_unicode('⠝')], // ^n (위첨자 n) - '\u{207B}' => &[decode_unicode('⠘'), decode_unicode('⠔')], // ^9 (위첨자 마이너스) - '\u{207A}' => &[decode_unicode('⠘'), decode_unicode('⠢')], // ^5 (위첨자 플러스) - '\u{2080}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠚')], // ;#j (아래첨자 0) - '\u{2081}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠁')], // ;#a (아래첨자 1) - '\u{2082}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠃')], // ;#b (아래첨자 2) - '\u{2083}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠉')], // ;#c (아래첨자 3) - '\u{2084}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠙')], // ;#d (아래첨자 4) - '\u{2085}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠑')], // ;#e (아래첨자 5) - '\u{2086}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠋')], // ;#f (아래첨자 6) - '\u{2087}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠛')], // ;#g (아래첨자 7) - '\u{2088}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠓')], // ;#h (아래첨자 8) - '\u{2089}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠊')], // ;#i (아래첨자 9) - '\u{208D}' => &[decode_unicode('⠰'), decode_unicode('⠦')], // ;8 (아래첨자 () - '\u{208E}' => &[decode_unicode('⠴')], // 0 (아래첨자 )) - '\u{2090}' => &[decode_unicode('⠰'), decode_unicode('⠁')], // ;a (아래첨자 a) - '\u{2098}' => &[decode_unicode('⠰'), decode_unicode('⠍')], // ;m (아래첨자 m) - '\u{2093}' => &[decode_unicode('⠰'), decode_unicode('⠭')], // ;x (아래첨자 x) - '\u{2099}' => &[decode_unicode('⠰'), decode_unicode('⠝')], // ;n (아래첨자 n) - '\u{208A}' => &[decode_unicode('⠰'), decode_unicode('⠢')], // ;5 (아래첨자 +) - '\u{2044}' => &[decode_unicode('⠌')], // / (분수 슬래시) - '\u{2500}' => &[decode_unicode('⠌')], // ─ (괘선 — PDF 제7항 분수선 기호 형태) - '\u{2E29}' => &[decode_unicode('⠄')], // open-ended right delimiter (`\right.`) - '_' => &[decode_unicode('⠠'), decode_unicode('⠤')], // 밑줄 marker (PDF 제23항 2) - '\u{0332}' => &[decode_unicode('⠠'), decode_unicode('⠤')], // ̲ (combining low line — 밑줄 결합부호) - '|' => &[decode_unicode('⠳')], // | (절댓값) - '\u{00AC}' => &[decode_unicode('⠈'), decode_unicode('⠔')], // @9 (부정) - '\u{00B0}' => &[decode_unicode('⠴'), decode_unicode('⠙')], // 0d (도) - '\u{00B1}' => &[decode_unicode('⠢'), decode_unicode('⠔')], // ± (PDF 제2항 — plus-minus) - '\u{00B7}' => &[decode_unicode('⠐')], // " (점 곱셈) - '…' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], // ,,, (줄임표) - '⋯' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], // ,,, (줄임표) - '\u{221A}' => &[decode_unicode('⠜')], // > (근호) - '\u{2224}' => &[decode_unicode('⠨'), decode_unicode('⠳')], // .\ (나누어떨어지지않는다) - '\u{2220}' => &[decode_unicode('⠹')], // ? (각) - '\u{22A5}' => &[decode_unicode('⠴'), decode_unicode('⠄')], // 0' (수직) - '\u{2225}' => &[decode_unicode('⠰'), decode_unicode('⠆')], // ;2 (평행) - '\u{2AFD}' => &[decode_unicode('⠰'), decode_unicode('⠆')], // ;2 (평행) - '\u{223D}' => &[decode_unicode('⠠'), decode_unicode('⠄')], // ,' (닮음) - '\u{2261}' => &[decode_unicode('⠶'), decode_unicode('⠶')], // 77 (합동) - '\u{221E}' => &[decode_unicode('⠿')], // = (무한대) - '\u{222B}' => &[decode_unicode('⠮')], // ! (부정적분) - '\u{222E}' => &[decode_unicode('⠾')], // ) (선적분) - '\u{222C}' => &[decode_unicode('⠮'), decode_unicode('⠮')], // !! (이중적분) - '\u{2207}' => &[decode_unicode('⠸'), decode_unicode('⠩')], // _% (델연산자) - '\u{2202}' => &[decode_unicode('⠫')], // $ (편도함수) - '\u{2208}' => &[decode_unicode('⠖')], // 6 (원소 왼쪽) - '\u{220B}' => &[decode_unicode('⠲')], // 4 (원소 오른쪽) - '\u{2209}' => &[decode_unicode('⠨'), decode_unicode('⠖')], // .6 (원소 아닌) - '\u{220C}' => &[decode_unicode('⠨'), decode_unicode('⠲')], // .4 (원소아닌 오른쪽) - '\u{2282}' => &[decode_unicode('⠖'), decode_unicode('⠂')], // 61 (부분집합 왼쪽) - '\u{2283}' => &[decode_unicode('⠐'), decode_unicode('⠲')], // "4 (부분집합 오른쪽) - '\u{2284}' => &[decode_unicode('⠨'), decode_unicode('⠖'), decode_unicode('⠂')], // .61 (부분집합 아님) - '\u{2285}' => &[decode_unicode('⠨'), decode_unicode('⠐'), decode_unicode('⠲')], // ."4 (부분집합 아님) - '\u{2205}' => &[decode_unicode('⠨'), decode_unicode('⠋')], // .f (공집합) - '\u{222A}' => &[decode_unicode('⠬')], // + (합집합) - '\u{2229}' => &[decode_unicode('⠩')], // % (교집합) - '\u{2200}' => &[decode_unicode('⠨'), decode_unicode('⠄')], // .' (모든) - '\u{2203}' => &[decode_unicode('⠨'), decode_unicode('⠢')], // .5 (존재하는) - '\u{2204}' => &[decode_unicode('⠨'), decode_unicode('⠨'), decode_unicode('⠢')], // ..5 (존재하지 않는) - '\u{2227}' => &[decode_unicode('⠹')], // ? (논리곱) - '\u{2228}' => &[decode_unicode('⠼')], // # (논리합) - '\u{22BB}' => &[decode_unicode('⠼'), decode_unicode('⠤')], // #- (배타적 논리합) - '\u{2234}' => &[decode_unicode('⠠'), decode_unicode('⠡')], // ,* (그러므로) - '\u{2235}' => &[decode_unicode('⠈'), decode_unicode('⠌')], // @/ (왜냐하면) - '\u{2248}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔')], // @9@9 (이중물결) - '\u{224A}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒')], // @9@93 (이중물결 아래줄) - '\u{2243}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒')], // @93 (물결 아래줄) - '\u{2245}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒'), decode_unicode('⠒')], // @933 (물결아래등호) - '\u{2241}' => &[decode_unicode('⠨'), decode_unicode('⠈'), decode_unicode('⠔')], // .@9 (not sim) - '\u{226E}' => &[decode_unicode('⠨'), decode_unicode('⠔'), decode_unicode('⠔')], // .99 (보다작지않다) - '\u{226F}' => &[decode_unicode('⠨'), decode_unicode('⠢'), decode_unicode('⠢')], // .55 (보다크지않다) - '\u{2270}' => &[decode_unicode('⠨'), decode_unicode('⠖'), decode_unicode('⠖')], // .66 (작거나같지않다) - '\u{2271}' => &[decode_unicode('⠨'), decode_unicode('⠲'), decode_unicode('⠲')], // .44 (크거나같지않다) - '\u{25B7}' => &[decode_unicode('⠸'), decode_unicode('⠜')], // _> (오른쪽 세모꼴) - '\u{25C1}' => &[decode_unicode('⠸'), decode_unicode('⠣')], // _< (왼쪽 세모꼴) - '\u{25A1}' => &[decode_unicode('⠸'), decode_unicode('⠶')], // _7 (네모) - '\u{25B3}' => &[decode_unicode('⠸'), decode_unicode('⠬')], // _+ (세모) - '\u{25B1}' => &[decode_unicode('⠸'), decode_unicode('⠌'), decode_unicode('⠌')], // _// (평행사변형) - '\u{23E2}' => &[decode_unicode('⠸'), decode_unicode('⠌'), decode_unicode('⠡')], // _/* (사다리꼴) - '\u{2302}' => &[decode_unicode('⠸'), decode_unicode('⠪'), decode_unicode('⠅')], // _[k (집) - '\u{2394}' => &[decode_unicode('⠸'), decode_unicode('⠪'), decode_unicode('⠕')], // _[o (기하 기호) - '\u{29BE}' => &[decode_unicode('⠸'), decode_unicode('⠴'), decode_unicode('⠴')], // _00 (원안점) - '\u{03A3}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠎')], // ,.s (총합) - '\u{2295}' => &[decode_unicode('⠸'), decode_unicode('⠢')], // _5 (동그라미 덧셈표) - '\u{2296}' => &[decode_unicode('⠸'), decode_unicode('⠔')], // _9 (동그라미 뺄셈표) - '\u{2297}' => &[decode_unicode('⠸'), decode_unicode('⠡')], // _* (동그라미 곱셈표) - '\u{2217}' => &[decode_unicode('⠸'), decode_unicode('⠣')], // _< (별표) - '\u{2218}' => &[decode_unicode('⠸'), decode_unicode('⠴')], // _0 (동그라미) - '\u{03B1}' => &[decode_unicode('⠨'), decode_unicode('⠁')], // .a (알파) - '\u{03B2}' => &[decode_unicode('⠨'), decode_unicode('⠃')], // .b (베타) - '\u{03B3}' => &[decode_unicode('⠨'), decode_unicode('⠛')], // .g (감마) - '\u{03B4}' => &[decode_unicode('⠨'), decode_unicode('⠙')], // .d (델타) - '\u{03B5}' => &[decode_unicode('⠨'), decode_unicode('⠑')], // .e (엡실론) - '\u{03B6}' => &[decode_unicode('⠨'), decode_unicode('⠵')], // .z (제타) - '\u{03B7}' => &[decode_unicode('⠨'), decode_unicode('⠱')], // .: (에타) - '\u{03B8}' => &[decode_unicode('⠨'), decode_unicode('⠹')], // .? (세타) - '\u{03B9}' => &[decode_unicode('⠨'), decode_unicode('⠊')], // .i (요타) - '\u{03BA}' => &[decode_unicode('⠨'), decode_unicode('⠅')], // .k (카파) - '\u{03BB}' => &[decode_unicode('⠨'), decode_unicode('⠇')], // .l (람다) - '\u{03BC}' => &[decode_unicode('⠨'), decode_unicode('⠍')], // .m (뮤) - '\u{03BD}' => &[decode_unicode('⠨'), decode_unicode('⠝')], // .n (뉴) - '\u{03BE}' => &[decode_unicode('⠨'), decode_unicode('⠭')], // .x (크시) - '\u{03BF}' => &[decode_unicode('⠨'), decode_unicode('⠕')], // .o (오미크론) - '\u{03C0}' => &[decode_unicode('⠨'), decode_unicode('⠏')], // .p (파이) - '\u{03C1}' => &[decode_unicode('⠨'), decode_unicode('⠗')], // .r (로) - '\u{03C3}' => &[decode_unicode('⠨'), decode_unicode('⠎')], // .s (시그마) - '\u{03C4}' => &[decode_unicode('⠨'), decode_unicode('⠞')], // .t (타우) - '\u{03C5}' => &[decode_unicode('⠨'), decode_unicode('⠥')], // .u (입실론) - '\u{03C6}' => &[decode_unicode('⠨'), decode_unicode('⠋')], // .f (피) - '\u{03C7}' => &[decode_unicode('⠨'), decode_unicode('⠯')], // .& (키) - '\u{03C8}' => &[decode_unicode('⠨'), decode_unicode('⠽')], // .y (프시) - '\u{03C9}' => &[decode_unicode('⠨'), decode_unicode('⠺')], // .w (오메가) - '\u{0391}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠁')], // ,.a (대문자 알파) - '\u{0392}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠃')], // ,.b (대문자 베타) - '\u{0393}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠛')], // ,.g (대문자 감마) - '\u{0395}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠑')], // ,.e (대문자 엡실론) - '\u{0396}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠵')], // ,.z (대문자 제타) - '\u{0397}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠱')], // ,.: (대문자 에타) - '\u{0398}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠹')], // ,.? (대문자 세타) - '\u{0399}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠊')], // ,.i (대문자 요타) - '\u{039A}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠅')], // ,.k (대문자 카파) - '\u{039B}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠇')], // ,.l (대문자 람다) - '\u{039C}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠍')], // ,.m (대문자 뮤) - '\u{039D}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠝')], // ,.n (대문자 뉴) - '\u{039E}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠭')], // ,.x (대문자 크시) - '\u{039F}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠕')], // ,.o (대문자 오미크론) - '\u{03A0}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠏')], // ,.p (대문자 파이) - '\u{03A1}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠗')], // ,.r (대문자 로) - '\u{03A4}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠞')], // ,.t (대문자 타우) - '\u{03A5}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠥')], // ,.u (대문자 입실론) - '\u{03A6}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠋')], // ,.f (대문자 피) - '\u{03A7}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠯')], // ,.& (대문자 키) - '\u{03A8}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠽')], // ,.y (대문자 프시) - '\u{03A9}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠺')], // ,.w (대문자 오메가) - '\u{0394}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠙')], // ,.d (대문자 델타) - '\u{2196}' => &[decode_unicode('⠪'), decode_unicode('⠢')], // [5 (왼쪽 위 화살표) - '\u{2197}' => &[decode_unicode('⠔'), decode_unicode('⠕')], // 9o (오른쪽 위 화살표) - '\u{2198}' => &[decode_unicode('⠢'), decode_unicode('⠕')], // 5o (오른쪽 아래 화살표) - '\u{2199}' => &[decode_unicode('⠪'), decode_unicode('⠔')], // [9 (왼쪽 아래 화살표) - '\u{21CF}' => &[decode_unicode('⠨'), decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], // .33o (함의 부정) - '\u{2135}' => &[decode_unicode('⠗'), decode_unicode('⠋')], // rf (알레프) - '\u{2206}' => &[decode_unicode('⠸'), decode_unicode('⠬')], // _+ (세모꼴) - '\u{2219}' => &[decode_unicode('⠸'), decode_unicode('⠲')], // _4 (검정 동그라미) - '\u{FF03}' => &[decode_unicode('⠸'), decode_unicode('⠹')], // _? (샤프 기호) - '\u{1D9C}' => &[decode_unicode('⠘'), decode_unicode('⠉')], // ^c (여집합) - '\u{0302}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠢')], // @@5 (결합 hat) - '\u{0304}' => &[decode_unicode('⠈'), decode_unicode('⠉')], // @c (결합 가로바) - '\u{0305}' => &[decode_unicode('⠈'), decode_unicode('⠉')], // @c (결합 윗줄) - '\u{2016}' => &[decode_unicode('⠳'), decode_unicode('⠳')], // \\ (이중 세로선) - '\u{2322}' => &[decode_unicode('⠈'), decode_unicode('⠪')], // @[ (호) - // PDF 수학 제65항 5 — 문자 위 결합 부호 (틸데) - '\u{0303}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], // @@9 (결합 틸데) - // 결합 윗 한 점 U+0307은 컨텍스트에 따라 의미가 다르다: - // - 숫자 뒤 : 순환소수 마크 (PDF 수학 제9항) → ⠈ - // - 문자 뒤 : 문자 위 한 점 (PDF 수학 제65항 5) → ⠈⠲ - // 이 SHORTCUT_MAP의 값은 숫자 뒤 기본형이고, 문자 뒤 처리는 rule_65에서 별도 분기한다. - '\u{0307}' => &[decode_unicode('⠈')], // @ (결합 윗점 - 기본/숫자 뒤) - '\u{0308}' => &[decode_unicode('⠈'), decode_unicode('⠲'), decode_unicode('⠲')], // @44 (결합 윗 두 점) - '\u{0309}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], // @@9 (결합 고리/훅) - '\u{030A}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], // @@9 (결합 윗고리) - '\u{211B}' => &[decode_unicode('⠠'), decode_unicode('⠗')], // ,R (ℛ = script R) - '~' => &[decode_unicode('⠈'), decode_unicode('⠔')], // @9 (물결 = 닮음) - '\u{0338}' => &[decode_unicode('⠨')], // . (부정 표지) - '\u{203E}' => &[decode_unicode('⠈'), decode_unicode('⠉')], // @c (선분 기호 U+203E) - '\u{20E1}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠕')], // [3O (직선 기호 U+20E1) - '\u{20D7}' => &[decode_unicode('⠒'), decode_unicode('⠕')], // 3O (반직선 기호 U+20D7) - // PDF 수학 제60항 6 — 추론 기호 ⊢/⊣/⊨/⫤ - '\u{22A2}' => &[decode_unicode('⠸'), decode_unicode('⠒')], // _3 (⊢ vdash) - '\u{22A3}' => &[decode_unicode('⠈'), decode_unicode('⠸'), decode_unicode('⠒')], // @_3 (⊣ dashv) - '\u{22A8}' => &[decode_unicode('⠘'), decode_unicode('⠸'), decode_unicode('⠒')], // ^_3 (⊨ models) - '\u{2AE4}' => &[decode_unicode('⠨'), decode_unicode('⠸'), decode_unicode('⠒')], // ._3 (⫤ Dashv) - // PDF 수학 제60항 7 — 앞선다 ≲ (보다같거나 작다 + 닮음) - '\u{2272}' => &[decode_unicode('⠔'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔')], // 99@9 (≲ lesssim) - // PDF 수학 제60항 8 — 앞서고같지않다 ≺ (보다작다) - '\u{227A}' => &[decode_unicode('⠔'), decode_unicode('⠔')], // 99 (≺ prec — same as <) - // PDF 수학 제61항 7 — 동치명제 ⇌ - '\u{21CC}' => &[decode_unicode('⠪'), decode_unicode('⠶'), decode_unicode('⠕')], // [7o (⇌ rightleftharpoons) - // PDF 수학 제23항 1 — 켤레복소수/평균값 macron ¯ - '\u{00AF}' => &[decode_unicode('⠈'), decode_unicode('⠉')], // @c (¯ macron) - // PDF 수학 제25항 — 총합 기호 ∑ (Greek capital Sigma과 동일 점형) - '\u{2211}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠎')], // ,.s - // PDF 수학 제26항 — 곱 기호 ∏ - '\u{220F}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠏')], // ,.p +#[derive(Debug, Clone, Copy)] +pub(crate) struct MathSymbolShortcut { + pub(crate) cells: &'static [u8], + pub(crate) fallback_meta: &'static RuleMeta, +} + +macro_rules! math_meta { + ($(($constant:ident, $section:literal, $name:literal, $description:literal)),+ $(,)?) => { + $( + pub(crate) static $constant: RuleMeta = RuleMeta { + section: $section, + subsection: None, + name: $name, + standard_ref: concat!("2024 Korean Braille Standard, 수학 제", $section, "항"), + description: $description, + }; + )+ + }; +} + +math_meta! { + (META_2, "2", "math_arithmetic_operator", "Arithmetic operators"), + (META_3, "3", "math_equality_symbol", "Equality symbols"), + (META_4, "4", "math_comparison_symbol", "Comparison symbols"), + (META_5, "5", "math_ratio_symbol", "Ratio and proportion symbols"), + (META_6, "6", "math_bracket_symbol", "Brackets, including the simultaneous-equation brace"), + (META_7, "7", "math_fraction_symbol", "Fraction notation"), + (META_9, "9", "math_repeating_decimal", "Repeating decimal marks"), + (META_10, "10", "math_arrow_symbol", "Arrow symbols"), + (META_13, "13", "math_greek_symbol", "Greek letters"), + (META_15, "15", "math_custom_binary_operator", "Custom binary operators"), + (META_16, "16", "math_base_subscript", "Base-notation subscripts"), + (META_17, "17", "math_prime_mark", "Prime marks"), + (META_18, "18", "math_superscript_symbol", "Superscript symbols"), + (META_19, "19", "math_subscript_symbol", "Subscript symbols"), + (META_21, "21", "math_absolute_value", "Absolute-value bars"), + (META_22, "22", "math_root_symbol", "Root symbols"), + (META_23, "23", "math_overline_symbol", "Overline and underline marks"), + (META_25, "25", "math_sigma_symbol", "Summation symbols"), + (META_27, "27", "math_divisibility_symbol", "Divisibility symbols"), + (META_28, "28", "math_norm_symbol", "Norm symbols"), + (META_30, "30", "math_dot_congruence", "Dot-congruence symbols"), + (META_31, "31", "math_asymptotic_equality", "Asymptotic equality"), + (META_32, "32", "math_congruence_symbol", "Congruence symbols"), + (META_33, "33", "math_geometric_operator", "Geometric operators"), + (META_34, "34", "math_relation_symbol", "Relation symbols and their negations"), + (META_35, "35", "math_segment_symbol", "Segment bar over two points"), + (META_36, "36", "math_arc_symbol", "Arc symbol"), + (META_37, "37", "math_line_symbol", "Bidirectional line symbols"), + (META_38, "38", "math_ray_symbol", "Ray symbols, also used for vectors"), + (META_39, "39", "math_angle_symbol", "Angle symbol"), + (META_40, "40", "math_geometric_shape", "Geometric shapes"), + (META_41, "41", "math_perpendicular_symbol", "Perpendicular symbols"), + (META_42, "42", "math_similarity_symbol", "Similarity symbols"), + (META_43, "43", "math_identity_symbol", "Identity symbols"), + (META_44, "44", "math_parallel_symbol", "Parallel symbols"), + (META_50, "50", "math_infinity_symbol", "Infinity"), + (META_53, "53", "math_derivative_product", "Product signs in derivative formulas"), + (META_54, "54", "math_partial_derivative", "Partial derivatives"), + (META_55, "55", "math_nabla_symbol", "Nabla"), + (META_56, "56", "math_integral_symbol", "Indefinite integrals"), + (META_58, "58", "math_double_integral", "Double integrals"), + (META_59, "59", "math_contour_integral", "Contour integrals"), + (META_60, "60", "math_set_symbol", "Set and inference symbols"), + (META_61, "61", "math_logic_symbol", "Logic symbols"), + (META_64, "64", "math_hat_symbol", "Hat notation"), + (META_65, "65", "math_miscellaneous_symbol", "Miscellaneous math symbols"), +} + +pub(crate) static META_KOREAN_49: RuleMeta = RuleMeta { + section: "49", + subsection: None, + name: "korean_sentence_punctuation_in_math", + standard_ref: "2024 Korean Braille Standard, 한글 제49항", + description: "Question and exclamation marks inside math input", +}; +pub(crate) static META_2_APPENDIX: RuleMeta = RuleMeta { + section: "2", + subsection: Some("붙임"), + name: "math_dot_multiplication", + standard_ref: "2024 Korean Braille Standard, 수학 제2항 [붙임]", + description: "Middle dot written as the multiplication sign", +}; +pub(crate) static META_KOREAN_51: RuleMeta = RuleMeta { + section: "51", + subsection: None, + name: "korean_colon_in_math", + standard_ref: "2024 Korean Braille Standard, 한글 제51항", + description: "Colon inside math input", +}; +/// 한글 제53항 governs the ellipsis in prose, but 수학 제12항 [붙임 1] claims it +/// back inside an expression — "쉼표는 `"`으로 적고, 줄임표는 `,,,`으로 적는다". +/// Both write ⠠⠠⠠, so only the article tells them apart, and 국립국어원 settled +/// on 2026-09-21 that an ellipsis inside a formula follows the math standard. +pub(crate) static META_12_APPENDIX_1: RuleMeta = RuleMeta { + section: "12", + subsection: Some("붙임 1"), + name: "math_ellipsis", + standard_ref: "2024 Korean Braille Standard, 수학 제12항 [붙임 1]", + description: "Ellipsis inside a mathematical expression", +}; +pub(crate) static META_KOREAN_59: RuleMeta = RuleMeta { + section: "59", + subsection: None, + name: "korean_semicolon_in_math", + standard_ref: "2024 Korean Braille Standard, 한글 제59항", + description: "Semicolon inside math input", +}; +pub(crate) static META_KOREAN_64: RuleMeta = RuleMeta { + section: "64", + subsection: None, + name: "korean_enclosed_number_in_math", + standard_ref: "2024 Korean Braille Standard, 한글 제64항", + description: "Circled numbers inside math input", +}; +pub(crate) static META_KOREAN_69_APPENDIX_2: RuleMeta = RuleMeta { + section: "69", + subsection: Some("붙임 2"), + name: "korean_degree_symbol_in_math", + standard_ref: "2024 Korean Braille Standard, 한글 제69항 [붙임 2]", + description: "Degree sign inside math input", +}; +pub(crate) static META_SCIENCE_29: RuleMeta = RuleMeta { + section: "29", + subsection: None, + name: "science_proportion_symbol", + standard_ref: "2024 Korean Braille Standard, 과학 제29항", + description: "Proportionality sign", +}; + +pub(crate) static MATH_SYMBOL_VARIANT_METAS: &[&RuleMeta] = &[ + &META_2, + &META_4, + &META_5, + &META_SCIENCE_29, + &META_7, + &META_9, + &META_10, + &META_13, + &META_15, + &META_16, + &META_17, + &META_18, + &META_19, + &META_21, + &META_22, + &META_23, + &META_25, + &META_27, + &META_28, + &META_30, + &META_31, + &META_32, + &META_33, + &META_34, + &META_35, + &META_36, + &META_37, + &META_38, + &META_39, + &META_40, + &META_41, + &META_42, + &META_43, + &META_44, + &META_50, + &META_53, + &META_54, + &META_55, + &META_56, + &META_58, + &META_59, + &META_60, + &META_61, + &META_64, + &META_65, + &META_2_APPENDIX, + &META_12_APPENDIX_1, + &META_KOREAN_64, + &META_KOREAN_69_APPENDIX_2, + &META_6, +]; + +macro_rules! shortcut_map { + ($($meta:expr => { $($symbol:expr => $cells:expr),+ $(,)? }),+ $(,)?) => { + phf_map! { + $($( + $symbol => MathSymbolShortcut { + cells: $cells, + fallback_meta: $meta, + }, + )+)+ + } + }; +} + +static SHORTCUT_MAP: phf::Map = shortcut_map! { + &META_KOREAN_64 => { + '\u{2460}' => &[decode_unicode('⠼'), decode_unicode('⠂')], + '\u{2461}' => &[decode_unicode('⠼'), decode_unicode('⠆')], + '\u{2462}' => &[decode_unicode('⠼'), decode_unicode('⠒')], + '\u{2463}' => &[decode_unicode('⠼'), decode_unicode('⠲')], + '\u{2464}' => &[decode_unicode('⠼'), decode_unicode('⠢')], + '\u{2465}' => &[decode_unicode('⠼'), decode_unicode('⠖')], + '\u{2466}' => &[decode_unicode('⠼'), decode_unicode('⠶')], + '\u{2467}' => &[decode_unicode('⠼'), decode_unicode('⠦')], + '\u{2468}' => &[decode_unicode('⠼'), decode_unicode('⠔')], + '\u{2469}' => &[decode_unicode('⠼'), decode_unicode('⠴')], + }, + &META_2 => { + '+' => &[decode_unicode('⠢')], + '\u{2212}' => &[decode_unicode('⠔')], + '\u{00D7}' => &[decode_unicode('⠡')], + '\u{2A09}' => &[decode_unicode('⠡')], + '\u{00F7}' => &[decode_unicode('⠌'), decode_unicode('⠌')], + '\u{00B1}' => &[decode_unicode('⠢'), decode_unicode('⠔')], + }, + &META_7 => { + '/' => &[decode_unicode('⠸'), decode_unicode('⠌')], + '\u{2500}' => &[decode_unicode('⠌')], + }, + &META_3 => { + '=' => &[decode_unicode('⠒'), decode_unicode('⠒')], + '\u{2260}' => &[decode_unicode('⠨'), decode_unicode('⠒'), decode_unicode('⠒')], + '\u{2252}' => &[decode_unicode('⠐'), decode_unicode('⠒'), decode_unicode('⠒')], + '\u{2248}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔')], + }, + &META_4 => { + '>' => &[decode_unicode('⠢'), decode_unicode('⠢')], + '<' => &[decode_unicode('⠔'), decode_unicode('⠔')], + '\u{2265}' => &[decode_unicode('⠲'), decode_unicode('⠲')], + '\u{2267}' => &[decode_unicode('⠲'), decode_unicode('⠲')], + '\u{2264}' => &[decode_unicode('⠖'), decode_unicode('⠖')], + '\u{2266}' => &[decode_unicode('⠖'), decode_unicode('⠖')], + '\u{226E}' => &[decode_unicode('⠨'), decode_unicode('⠔'), decode_unicode('⠔')], + '\u{226F}' => &[decode_unicode('⠨'), decode_unicode('⠢'), decode_unicode('⠢')], + '\u{2270}' => &[decode_unicode('⠨'), decode_unicode('⠖'), decode_unicode('⠖')], + '\u{2271}' => &[decode_unicode('⠨'), decode_unicode('⠲'), decode_unicode('⠲')], + }, + &META_5 => { + '\u{2236}' => &[decode_unicode('⠐'), decode_unicode('⠂')], + }, + &META_SCIENCE_29 => { + '\u{221D}' => &[decode_unicode('⠬'), decode_unicode('⠒')], + }, + &META_38 => { + '\u{20D7}' => &[decode_unicode('⠒'), decode_unicode('⠕')], + }, + &META_37 => { + '\u{20E1}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠕')], + }, + &META_10 => { + '\u{2192}' => &[decode_unicode('⠒'), decode_unicode('⠕')], + '\u{27F6}' => &[decode_unicode('⠒'), decode_unicode('⠕')], + '\u{2194}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠕')], + '\u{2190}' => &[decode_unicode('⠪'), decode_unicode('⠒')], + '\u{2191}' => &[decode_unicode('⠰'), decode_unicode('⠒'), decode_unicode('⠕')], + '\u{2193}' => &[decode_unicode('⠘'), decode_unicode('⠒'), decode_unicode('⠕')], + '\u{21D2}' => &[decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], + '\u{21D4}' => &[decode_unicode('⠪'), decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], + '\u{2196}' => &[decode_unicode('⠪'), decode_unicode('⠢')], + '\u{2197}' => &[decode_unicode('⠔'), decode_unicode('⠕')], + '\u{2198}' => &[decode_unicode('⠢'), decode_unicode('⠕')], + '\u{2199}' => &[decode_unicode('⠪'), decode_unicode('⠔')], + }, + &META_61 => { + '\u{21C4}' => &[decode_unicode('⠪'), decode_unicode('⠶'), decode_unicode('⠕')], + '\u{21CC}' => &[decode_unicode('⠪'), decode_unicode('⠶'), decode_unicode('⠕')], + '\u{00AC}' => &[decode_unicode('⠈'), decode_unicode('⠔')], + '\u{2200}' => &[decode_unicode('⠨'), decode_unicode('⠄')], + '\u{2203}' => &[decode_unicode('⠨'), decode_unicode('⠢')], + '\u{2204}' => &[decode_unicode('⠨'), decode_unicode('⠨'), decode_unicode('⠢')], + '\u{2227}' => &[decode_unicode('⠹')], + '\u{2228}' => &[decode_unicode('⠼')], + '\u{22BB}' => &[decode_unicode('⠼'), decode_unicode('⠤')], + '~' => &[decode_unicode('⠈'), decode_unicode('⠔')], + }, + &META_17 => { + '\u{2032}' => &[decode_unicode('⠤')], + '\u{2033}' => &[decode_unicode('⠤'), decode_unicode('⠤')], + '\u{2034}' => &[decode_unicode('⠤'), decode_unicode('⠤'), decode_unicode('⠤')], + }, + &META_18 => { + '\u{00B2}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠃')], + '\u{00B3}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠉')], + '\u{2074}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠙')], + '\u{2075}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠑')], + '\u{2077}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠛')], + '\u{2079}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠊')], + '\u{00B9}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠁')], + '\u{2070}' => &[decode_unicode('⠘'), decode_unicode('⠼'), decode_unicode('⠚')], + '\u{1D4F}' => &[decode_unicode('⠘'), decode_unicode('⠅')], + '\u{1D50}' => &[decode_unicode('⠘'), decode_unicode('⠍')], + '\u{02E3}' => &[decode_unicode('⠘'), decode_unicode('⠭')], + '\u{207D}' => &[decode_unicode('⠘'), decode_unicode('⠦')], + '\u{207E}' => &[decode_unicode('⠴')], + '\u{207F}' => &[decode_unicode('⠘'), decode_unicode('⠝')], + '\u{207B}' => &[decode_unicode('⠘'), decode_unicode('⠔')], + '\u{207A}' => &[decode_unicode('⠘'), decode_unicode('⠢')], + }, + &META_16 => { + '\u{2080}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠚')], + '\u{2081}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠁')], + '\u{2082}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠃')], + '\u{2083}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠉')], + '\u{2084}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠙')], + '\u{2085}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠑')], + '\u{2086}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠋')], + '\u{2087}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠛')], + '\u{2088}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠓')], + '\u{2089}' => &[decode_unicode('⠰'), decode_unicode('⠼'), decode_unicode('⠊')], + '\u{208D}' => &[decode_unicode('⠰'), decode_unicode('⠦')], + '\u{208E}' => &[decode_unicode('⠴')], + }, + &META_19 => { + '\u{2090}' => &[decode_unicode('⠰'), decode_unicode('⠁')], + '\u{2098}' => &[decode_unicode('⠰'), decode_unicode('⠍')], + '\u{2093}' => &[decode_unicode('⠰'), decode_unicode('⠭')], + '\u{2099}' => &[decode_unicode('⠰'), decode_unicode('⠝')], + '\u{208A}' => &[decode_unicode('⠰'), decode_unicode('⠢')], + }, + &META_6 => { + '\u{2E29}' => &[decode_unicode('⠄')], + }, + &META_34 => { + '\u{0338}' => &[decode_unicode('⠨')], + '\u{211B}' => &[decode_unicode('⠠'), decode_unicode('⠗')], + '\u{2241}' => &[decode_unicode('⠨'), decode_unicode('⠈'), decode_unicode('⠔')], + }, + &META_60 => { + '\u{1D9C}' => &[decode_unicode('⠘'), decode_unicode('⠉')], + }, + &META_61 => { + '\u{21CF}' => &[decode_unicode('⠨'), decode_unicode('⠒'), decode_unicode('⠒'), decode_unicode('⠕')], + }, + &META_7 => { + '\u{2044}' => &[decode_unicode('⠌')], + }, + &META_23 => { + '_' => &[decode_unicode('⠠'), decode_unicode('⠤')], + '\u{0332}' => &[decode_unicode('⠠'), decode_unicode('⠤')], + '\u{0304}' => &[decode_unicode('⠈'), decode_unicode('⠉')], + '\u{0305}' => &[decode_unicode('⠈'), decode_unicode('⠉')], + '\u{00AF}' => &[decode_unicode('⠈'), decode_unicode('⠉')], + }, + &META_21 => { + '|' => &[decode_unicode('⠳')], + }, + &META_KOREAN_69_APPENDIX_2 => { + '\u{00B0}' => &[decode_unicode('⠴'), decode_unicode('⠙')], + }, + &META_2_APPENDIX => { + '\u{00B7}' => &[decode_unicode('⠐')], + }, + &META_12_APPENDIX_1 => { + '…' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], + '⋯' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], + }, + &META_22 => { + '\u{221A}' => &[decode_unicode('⠜')], + }, + &META_27 => { + '\u{2223}' => &[decode_unicode('⠳')], + '\u{2224}' => &[decode_unicode('⠨'), decode_unicode('⠳')], + }, + &META_39 => { + '\u{2220}' => &[decode_unicode('⠹')], + }, + &META_41 => { + '\u{22A5}' => &[decode_unicode('⠴'), decode_unicode('⠄')], + }, + &META_44 => { + '\u{2225}' => &[decode_unicode('⠰'), decode_unicode('⠆')], + '\u{2AFD}' => &[decode_unicode('⠰'), decode_unicode('⠆')], + }, + &META_42 => { + '\u{223D}' => &[decode_unicode('⠠'), decode_unicode('⠄')], + }, + &META_43 => { + '\u{2261}' => &[decode_unicode('⠶'), decode_unicode('⠶')], + }, + &META_50 => { + '\u{221E}' => &[decode_unicode('⠿')], + }, + &META_56 => { + '\u{222B}' => &[decode_unicode('⠮')], + }, + &META_59 => { + '\u{222E}' => &[decode_unicode('⠾')], + }, + &META_58 => { + '\u{222C}' => &[decode_unicode('⠮'), decode_unicode('⠮')], + }, + &META_55 => { + '\u{2207}' => &[decode_unicode('⠸'), decode_unicode('⠩')], + }, + &META_54 => { + '\u{2202}' => &[decode_unicode('⠫')], + }, + &META_60 => { + '\u{2208}' => &[decode_unicode('⠖')], + '\u{220B}' => &[decode_unicode('⠲')], + '\u{2209}' => &[decode_unicode('⠨'), decode_unicode('⠖')], + '\u{220C}' => &[decode_unicode('⠨'), decode_unicode('⠲')], + '\u{2282}' => &[decode_unicode('⠖'), decode_unicode('⠂')], + '\u{2283}' => &[decode_unicode('⠐'), decode_unicode('⠲')], + '\u{2284}' => &[decode_unicode('⠨'), decode_unicode('⠖'), decode_unicode('⠂')], + '\u{2285}' => &[decode_unicode('⠨'), decode_unicode('⠐'), decode_unicode('⠲')], + '\u{2205}' => &[decode_unicode('⠨'), decode_unicode('⠋')], + '\u{222A}' => &[decode_unicode('⠬')], + '\u{2229}' => &[decode_unicode('⠩')], + '\u{22A2}' => &[decode_unicode('⠸'), decode_unicode('⠒')], + '\u{22A3}' => &[decode_unicode('⠈'), decode_unicode('⠸'), decode_unicode('⠒')], + '\u{22A8}' => &[decode_unicode('⠘'), decode_unicode('⠸'), decode_unicode('⠒')], + '\u{2AE4}' => &[decode_unicode('⠨'), decode_unicode('⠸'), decode_unicode('⠒')], + '\u{2272}' => &[decode_unicode('⠔'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔')], + '\u{227A}' => &[decode_unicode('⠔'), decode_unicode('⠔')], + }, + &META_65 => { + '\u{2234}' => &[decode_unicode('⠠'), decode_unicode('⠡')], + '\u{2235}' => &[decode_unicode('⠈'), decode_unicode('⠌')], + '\u{2135}' => &[decode_unicode('⠗'), decode_unicode('⠋')], + '\u{FF03}' => &[decode_unicode('⠸'), decode_unicode('⠹')], + '\u{0303}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], + '\u{0308}' => &[decode_unicode('⠈'), decode_unicode('⠲'), decode_unicode('⠲')], + '\u{0309}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], + '\u{030A}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠔')], + }, + &META_30 => { + '\u{224A}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒')], + }, + &META_31 => { + '\u{2243}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒')], + }, + &META_32 => { + '\u{2245}' => &[decode_unicode('⠈'), decode_unicode('⠔'), decode_unicode('⠒'), decode_unicode('⠒')], + }, + &META_33 => { + '\u{25B7}' => &[decode_unicode('⠸'), decode_unicode('⠜')], + '\u{25C1}' => &[decode_unicode('⠸'), decode_unicode('⠣')], + }, + &META_40 => { + '\u{25A1}' => &[decode_unicode('⠸'), decode_unicode('⠶')], + '\u{25B3}' => &[decode_unicode('⠸'), decode_unicode('⠬')], + '\u{25B1}' => &[decode_unicode('⠸'), decode_unicode('⠌'), decode_unicode('⠌')], + '\u{23E2}' => &[decode_unicode('⠸'), decode_unicode('⠌'), decode_unicode('⠡')], + '\u{2302}' => &[decode_unicode('⠸'), decode_unicode('⠪'), decode_unicode('⠅')], + '\u{2394}' => &[decode_unicode('⠸'), decode_unicode('⠪'), decode_unicode('⠕')], + '\u{29BE}' => &[decode_unicode('⠸'), decode_unicode('⠴'), decode_unicode('⠴')], + '\u{2206}' => &[decode_unicode('⠸'), decode_unicode('⠬')], + '\u{2219}' => &[decode_unicode('⠸'), decode_unicode('⠲')], + }, + &META_25 => { + '\u{2211}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠎')], + }, + &META_15 => { + '\u{2295}' => &[decode_unicode('⠸'), decode_unicode('⠢')], + '\u{2296}' => &[decode_unicode('⠸'), decode_unicode('⠔')], + '\u{2297}' => &[decode_unicode('⠸'), decode_unicode('⠡')], + '\u{2217}' => &[decode_unicode('⠸'), decode_unicode('⠣')], + '\u{2218}' => &[decode_unicode('⠸'), decode_unicode('⠴')], + }, + &META_13 => { + '\u{03B1}' => &[decode_unicode('⠨'), decode_unicode('⠁')], + '\u{03B2}' => &[decode_unicode('⠨'), decode_unicode('⠃')], + '\u{03B3}' => &[decode_unicode('⠨'), decode_unicode('⠛')], + '\u{03B4}' => &[decode_unicode('⠨'), decode_unicode('⠙')], + '\u{03B5}' => &[decode_unicode('⠨'), decode_unicode('⠑')], + '\u{03B6}' => &[decode_unicode('⠨'), decode_unicode('⠵')], + '\u{03B7}' => &[decode_unicode('⠨'), decode_unicode('⠱')], + '\u{03B8}' => &[decode_unicode('⠨'), decode_unicode('⠹')], + '\u{03B9}' => &[decode_unicode('⠨'), decode_unicode('⠊')], + '\u{03BA}' => &[decode_unicode('⠨'), decode_unicode('⠅')], + '\u{03BB}' => &[decode_unicode('⠨'), decode_unicode('⠇')], + '\u{03BC}' => &[decode_unicode('⠨'), decode_unicode('⠍')], + '\u{03BD}' => &[decode_unicode('⠨'), decode_unicode('⠝')], + '\u{03BE}' => &[decode_unicode('⠨'), decode_unicode('⠭')], + '\u{03BF}' => &[decode_unicode('⠨'), decode_unicode('⠕')], + '\u{03C0}' => &[decode_unicode('⠨'), decode_unicode('⠏')], + '\u{03C1}' => &[decode_unicode('⠨'), decode_unicode('⠗')], + '\u{03C3}' => &[decode_unicode('⠨'), decode_unicode('⠎')], + '\u{03C4}' => &[decode_unicode('⠨'), decode_unicode('⠞')], + '\u{03C5}' => &[decode_unicode('⠨'), decode_unicode('⠥')], + '\u{03C6}' => &[decode_unicode('⠨'), decode_unicode('⠋')], + '\u{03C7}' => &[decode_unicode('⠨'), decode_unicode('⠯')], + '\u{03C8}' => &[decode_unicode('⠨'), decode_unicode('⠽')], + '\u{03C9}' => &[decode_unicode('⠨'), decode_unicode('⠺')], + '\u{0391}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠁')], + '\u{0392}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠃')], + '\u{0393}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠛')], + '\u{0394}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠙')], + '\u{0395}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠑')], + '\u{0396}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠵')], + '\u{0397}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠱')], + '\u{0398}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠹')], + '\u{0399}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠊')], + '\u{039A}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠅')], + '\u{039B}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠇')], + '\u{039C}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠍')], + '\u{039D}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠝')], + '\u{039E}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠭')], + '\u{039F}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠕')], + '\u{03A0}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠏')], + '\u{03A1}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠗')], + '\u{03A3}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠎')], + '\u{03A4}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠞')], + '\u{03A5}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠥')], + '\u{03A6}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠋')], + '\u{03A7}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠯')], + '\u{03A8}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠽')], + '\u{03A9}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠺')], + '\u{2126}' => &[decode_unicode('⠠'), decode_unicode('⠨'), decode_unicode('⠺')], + }, + &META_35 => { + '\u{203E}' => &[decode_unicode('⠈'), decode_unicode('⠉')], + }, + &META_36 => { + '\u{2322}' => &[decode_unicode('⠈'), decode_unicode('⠪')], + }, + &META_64 => { + '\u{0302}' => &[decode_unicode('⠈'), decode_unicode('⠈'), decode_unicode('⠢')], + }, + &META_28 => { + '\u{2016}' => &[decode_unicode('⠳'), decode_unicode('⠳')], + }, + &META_9 => { + '\u{0307}' => &[decode_unicode('⠈')], + }, }; pub fn encode_char_math_symbol_shortcut(text: char) -> Result<&'static [u8], String> { - if let Some(code) = SHORTCUT_MAP.get(&text) { - Ok(code) - } else { - Err("Invalid math symbol character".to_string()) - } + math_symbol_shortcut(text).map(|shortcut| shortcut.cells) +} + +pub(crate) fn math_symbol_shortcut(text: char) -> Result<&'static MathSymbolShortcut, String> { + SHORTCUT_MAP + .get(&text) + .ok_or_else(|| "Invalid math symbol character".to_string()) } pub fn is_math_symbol_char(text: char) -> bool { @@ -255,6 +551,168 @@ pub fn is_math_symbol_char(text: char) -> bool { mod test { use super::*; + /// Every character the table can encode names the article that grants it + /// those cells. Nothing is exempt: a symbol the standard does not define is + /// absent from the table rather than present with an unknown article. + #[test] + fn every_shortcut_declares_a_real_article() { + let missing = SHORTCUT_MAP + .entries() + .find(|(_, shortcut)| shortcut.fallback_meta.section == "?"); + + assert!(missing.is_none(), "shortcut without article: {missing:?}"); + } + + /// 한글 제50항's 가운뎃점 is two cells, ⠐⠆. The single ⠐ this table gives the + /// same character is 수학 제2항 [붙임] — "점으로 표현된 곱셈 기호는 `"`으로 + /// 적는다" — so a middle dot met inside a formula is multiplication, not the + /// punctuation mark it looks like. + #[test] + fn a_middle_dot_in_a_formula_is_the_multiplication_sign() { + let dot = SHORTCUT_MAP[&'\u{00B7}']; + + assert_eq!(dot.fallback_meta.section, "2"); + assert_eq!(dot.fallback_meta.subsection, Some("붙임")); + assert_eq!(dot.cells, [decode_unicode('⠐')]); + assert_ne!( + dot.cells, + crate::symbol_shortcut::encode_char_symbol_shortcut('\u{00B7}').unwrap() + ); + } + + /// An ellipsis writes ⠠⠠⠠ whether it falls in prose or in a formula, so the + /// article is the only thing that separates them: 수학 제12항 [붙임 1] inside + /// an expression, 한글 제53항 outside it. + #[test] + fn an_ellipsis_in_a_formula_cites_the_math_article() { + let ellipsis = SHORTCUT_MAP[&'…']; + + assert_eq!(ellipsis.fallback_meta.section, "12"); + assert_eq!(ellipsis.fallback_meta.subsection, Some("붙임 1")); + assert_eq!( + ellipsis.cells, + [ + decode_unicode('⠠'), + decode_unicode('⠠'), + decode_unicode('⠠') + ] + ); + } + + /// 제6항 1 lists 연립식 괄호 as `7'` and closes it with `,7`. LaTeX writes + /// the opening half as `\left\{ ... \right.`, so the sentinel standing for + /// `\right.` carries the brace's second cell and belongs to that article — + /// it is not a delimiter that prints nothing. + #[test] + fn the_simultaneous_equation_brace_cites_article_6() { + assert_eq!(SHORTCUT_MAP[&'⸩'].fallback_meta.section, "6"); + assert_eq!(SHORTCUT_MAP[&'⸩'].cells, [decode_unicode('⠄')]); + } + + /// 국립국어원 ruled on 2026-09-21 that the n-ary product cannot be + /// transcribed: the standard never mentions it. Its cells are those of + /// Greek capital pi, which makes borrowing them look reasonable and is + /// exactly why the table must not carry it. + #[test] + fn the_n_ary_product_is_not_transcribable() { + assert!(!SHORTCUT_MAP.contains_key(&'∏')); + assert!(encode_char_math_symbol_shortcut('∏').is_err()); + } + + /// Each of these was identified by matching its cells against the notation + /// printed in the standard, not by searching for the character itself: + /// `⠌` is the 분수표 of 제7항 1, `⠠⠗`/`⠨⠈⠔` are 관계가있다/관계가없다 of + /// 제34항, `⠘⠉` is 여집합 of 제60항 5, `⠨⠒⠒⠕` is 항진명제의 부정 of 제61항 4. + #[rstest::rstest] + #[case::fraction_slash('⁄', "7")] + #[case::script_r('ℛ', "34")] + #[case::not_similar('≁', "34")] + #[case::negation_overlay('\u{0338}', "34")] + #[case::superscript_c('ᶜ', "60")] + #[case::not_implies('⇏', "61")] + fn cell_matched_shortcuts_name_their_article(#[case] symbol: char, #[case] section: &str) { + assert_eq!(SHORTCUT_MAP[&symbol].fallback_meta.section, section); + } + + /// 제35항 to 제39항 run 선분 `@c`, 호 `@[`, 직선 `[3O`, 반직선 `3O`, 각 `?`, + /// one article each and in that order. Three of these marks sat one article + /// away from the one that defines them, which nothing caught because the + /// cells were right either way. The overline is the segment bar of 제35항, + /// not 제36항's arc; the two-headed arrow above a pair is 제37항's line, not + /// a ray; and the single-headed one is 제38항's ray, which its 붙임 also + /// lends to vectors, rather than 제39항's angle. + #[rstest::rstest] + #[case::segment_bar('\u{203E}', "35")] + #[case::arc('\u{2322}', "36")] + #[case::line_above('\u{20E1}', "37")] + #[case::ray_above('\u{20D7}', "38")] + #[case::angle('\u{2220}', "39")] + fn geometry_marks_cite_the_article_that_defines_them( + #[case] symbol: char, + #[case] section: &str, + ) { + assert_eq!(SHORTCUT_MAP[&symbol].fallback_meta.section, section); + } + + /// 제10항 lists the arrows together — right, left, up, down and the four + /// diagonals — so an arrow standing between operands belongs there. The ray + /// of 제38항 is the mark drawn above a pair of points, which Unicode spells + /// as a combining character, not as the arrow one types between them. + #[rstest::rstest] + #[case::right('\u{2192}')] + #[case::long_right('\u{27F6}')] + #[case::both_ways('\u{2194}')] + #[case::left('\u{2190}')] + #[case::up('\u{2191}')] + #[case::down('\u{2193}')] + #[case::upper_left('\u{2196}')] + fn a_standing_arrow_belongs_to_the_arrow_article(#[case] symbol: char) { + assert_eq!(SHORTCUT_MAP[&symbol].fallback_meta.section, "10"); + assert_ne!( + SHORTCUT_MAP[&symbol].fallback_meta.section, + SHORTCUT_MAP[&'\u{20D7}'].fallback_meta.section + ); + } + + /// 제23항 gives the bar over a variable — 켤레 복소수 and 평균값 — the same + /// `@c` cells as 제35항's segment bar, so the two are told apart by code + /// point alone: a combining or spacing macron marks a variable, while the + /// overline spans a pair of points. + #[rstest::rstest] + #[case::combining_macron('\u{0304}')] + #[case::combining_overline('\u{0305}')] + #[case::spacing_macron('\u{00AF}')] + fn a_bar_over_a_variable_stays_with_article_23(#[case] symbol: char) { + assert_eq!(SHORTCUT_MAP[&symbol].fallback_meta.section, "23"); + assert_eq!(SHORTCUT_MAP[&symbol].cells, SHORTCUT_MAP[&'\u{203E}'].cells); + } + + /// A second code point for a symbol the standard already defines means the + /// same thing, so it takes the same cells and the same article. Chemistry + /// writes its reaction arrow long and its product sign n-ary; the ohm sign + /// is stronger still, being canonically equivalent to capital omega, so + /// Unicode itself forbids treating the two as different characters. + #[rstest::rstest] + #[case::long_rightwards_arrow('\u{27F6}', '\u{2192}')] + #[case::n_ary_times('\u{2A09}', '\u{00D7}')] + #[case::ohm_sign('\u{2126}', '\u{03A9}')] + fn a_glyph_variant_matches_the_symbol_it_varies(#[case] variant: char, #[case] base: char) { + assert_eq!(SHORTCUT_MAP[&variant].cells, SHORTCUT_MAP[&base].cells); + assert_eq!( + SHORTCUT_MAP[&variant].fallback_meta.section, + SHORTCUT_MAP[&base].fallback_meta.section + ); + } + + /// 제27항 writes 나누어떨어진다 as `\` and negates it to `.\`, so the plain + /// sign is the negated one without its leading dot. + #[test] + fn divides_is_the_undotted_form_of_does_not_divide() { + let divides = SHORTCUT_MAP[&'\u{2223}'].cells; + let does_not = SHORTCUT_MAP[&'\u{2224}'].cells; + assert_eq!(does_not, [decode_unicode('⠨'), divides[0]]); + } + /// `is_math_symbol_char` true 케이스 — 연산자/그리스/집합/미적분 기호 전체. #[rstest::rstest] // basic operators diff --git a/libs/braillify/src/rules/context.rs b/libs/braillify/src/rules/context.rs index 0f26671c..f388fc6f 100644 --- a/libs/braillify/src/rules/context.rs +++ b/libs/braillify/src/rules/context.rs @@ -45,6 +45,20 @@ pub enum EncodingMode { /// `[ ]`는 ⠐⠘⠷ … ⠘⠾, `/ /`는 ⠐⠘⠌ … ⠘⠌으로 묶는다. /// 음운 기호(ə, ː, θ, ŋ, æ 등)는 국제음성기호 점자 변환표에 따라 점역한다. Ipa, + /// 과학 점자로 적는 국어 글. 묵자 모양만으로는 과학 기호인지 알 수 없는 글을 + /// 과학 규정으로 읽는다 — 연산 기호가 든 로마자 식(제6항), 유전자(제23항), + /// 치식(제26항), 단위 속 화학식(제30항 [붙임]). + Science, + /// [`Self::Science`] 와 같되, 구조식과 전자 점식을 공간 표기 형식으로 적는다. + /// 제9항·제15항은 같은 묵자를 기호 표기와 공간 표기 가운데 어느 쪽으로도 적게 한다. + ScienceSpatial, +} + +impl EncodingMode { + /// 과학 점자로 읽는 문맥인가. + pub fn reads_science(self) -> bool { + matches!(self, Self::Science | Self::ScienceSpatial) + } } impl std::str::FromStr for EncodingMode { @@ -62,6 +76,8 @@ impl std::str::FromStr for EncodingMode { "middle_korean" => Ok(Self::MiddleKorean), "object_symbol" => Ok(Self::ObjectSymbol), "ipa" => Ok(Self::Ipa), + "science" => Ok(Self::Science), + "science_spatial" => Ok(Self::ScienceSpatial), _ => Err(()), } } @@ -111,10 +127,24 @@ pub struct EncoderState { /// Explicit math mode (`context = math` in fixtures/API options). /// Keeps parentheses in math form even when their contents include Hangul. pub math_mode_active: bool, + /// Explicit Korean context (`context = korean`): the text sits in a Korean + /// sentence even when it carries no Hangul, as the unit table of 과학 제30항 + /// does (`mH₂O` → ⠴⠍⠠⠓⠰⠼⠃⠠⠕). + pub korean_context_active: bool, + /// Explicit science context (`context = science`). + pub science_context_active: bool, /// 짝맞춤 작은따옴표(`‘…’`) 추적: `‘`를 만나면 +1, 닫음 `’`로 -1. /// 0보다 크면 현재 위치는 paired closing 위치이므로 `’`를 `⠴⠄`로 emit. /// 0이면 standalone apostrophe로 `⠄` 한 셀만 emit. (PDF 제61항) pub unmatched_open_single_quotes: i32, + /// Per-article spans the last Korean syllable produced, handed from + /// `RuleKorean` to the trace recorder in [`super::engine`]. + /// + /// `None` whenever no trace is being collected, which is both the signal to + /// rules that this work is unwanted and the reason the untraced encoder pays + /// only one pointer for the feature — this struct is carried by `&mut` + /// through the per-character loop, so its size is on the hot path. + pub jamo_spans: Option>, } impl EncoderState { @@ -136,10 +166,17 @@ impl EncoderState { doc_summary: DocumentSummary::default(), matrix_context_active: false, math_mode_active: false, + korean_context_active: false, + science_context_active: false, unmatched_open_single_quotes: 0, + jamo_spans: None, } } + pub fn reads_science_shapes(&self) -> bool { + self.english_indicator || self.korean_context_active || self.science_context_active + } + /// Get the current encoding mode (top of stack, default Korean). pub fn current_mode(&self) -> EncodingMode { self.mode_stack @@ -249,21 +286,18 @@ mod tests { use super::*; use std::str::FromStr; - #[test] - fn encoding_mode_from_str_all_variants() { - assert_eq!(EncodingMode::from_str("korean"), Ok(EncodingMode::Korean)); - assert_eq!(EncodingMode::from_str("english"), Ok(EncodingMode::English)); - assert_eq!(EncodingMode::from_str("math"), Ok(EncodingMode::Math)); - assert_eq!(EncodingMode::from_str("number"), Ok(EncodingMode::Number)); - assert_eq!( - EncodingMode::from_str("middle_korean"), - Ok(EncodingMode::MiddleKorean) - ); - assert_eq!( - EncodingMode::from_str("object_symbol"), - Ok(EncodingMode::ObjectSymbol) - ); - assert_eq!(EncodingMode::from_str("ipa"), Ok(EncodingMode::Ipa)); + #[rstest::rstest] + #[case::korean("korean", EncodingMode::Korean)] + #[case::english("english", EncodingMode::English)] + #[case::math("math", EncodingMode::Math)] + #[case::number("number", EncodingMode::Number)] + #[case::middle_korean("middle_korean", EncodingMode::MiddleKorean)] + #[case::object_symbol("object_symbol", EncodingMode::ObjectSymbol)] + #[case::ipa("ipa", EncodingMode::Ipa)] + #[case::science("science", EncodingMode::Science)] + #[case::science_spatial("science_spatial", EncodingMode::ScienceSpatial)] + fn encoding_mode_from_str_all_variants(#[case] name: &str, #[case] mode: EncodingMode) { + assert_eq!(EncodingMode::from_str(name), Ok(mode)); } #[test] diff --git a/libs/braillify/src/rules/emit.rs b/libs/braillify/src/rules/emit.rs index 0f2a98ad..a9399798 100644 --- a/libs/braillify/src/rules/emit.rs +++ b/libs/braillify/src/rules/emit.rs @@ -1,4 +1,4 @@ -use crate::char_struct::{CharType, KoreanChar}; +use crate::char_struct::{CharType, KoreanChar}; use crate::english_logic; use crate::fraction; use crate::rules::context::{EncoderState, RuleContext}; @@ -6,6 +6,7 @@ use crate::rules::engine::RuleEngine; use crate::rules::korean::rule_29::{ENGLISH_CONTINUATION, ROMAN_INDICATOR, ROMAN_TERMINATOR}; use crate::rules::korean::rule_69::parse_numeric_ascii_unit_prefix; use crate::rules::roman_mode; +use crate::rules::trace::{EmitterRule, RuleId, TokenOrigins, TraceSink}; use crate::rules::traits::Phase; use super::token::{DocumentIR, ModeEvent, SpaceKind, Token, WordToken}; @@ -435,7 +436,12 @@ fn is_math_operator_space_suppression<'a>(tokens: &'a [Token<'a>], space_idx: us false } -pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result, String> { +pub fn emit( + ir: &mut DocumentIR, + char_engine: &mut RuleEngine, + mut trace: Option>, + origins: Option<&TokenOrigins>, +) -> Result, String> { let mut result = Vec::new(); let word_texts = if ir.tokens.len() > 1 { collect_word_texts(&ir.tokens) @@ -463,12 +469,22 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result &ir.tokens, context, &mut result, + trace.as_mut().map(|sink| sink.at_token(idx)), )?; word_index += 1; } Token::Space(SpaceKind::Regular) => { if !is_math_operator_space_suppression(&ir.tokens, idx) { + let start = result.len(); result.push(0); + record_token_span( + &mut trace, + origins, + idx, + &result, + start, + RuleId::emitter(EmitterRule::WordSpace), + ); } } Token::Mode(event) => { @@ -497,10 +513,20 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result ir.state.roman_section_is_english_context = roman_section_has_english_phrase_context(&ir.tokens, idx); } + let start = result.len(); enter_roman_before_ueb_prefix(&ir.tokens, idx, event, &mut ir.state, &mut result); emit_mode_event(event, &mut ir.state, &mut result); + record_token_span( + &mut trace, + origins, + idx, + &result, + start, + RuleId::emitter(EmitterRule::UndeclaredTokenOutput), + ); } Token::Fraction(frac) => { + let start = result.len(); if let Some(ref w) = frac.whole { result.extend(fraction::encode_mixed_fraction( w, @@ -514,6 +540,14 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result )?); } ir.state.is_number = true; + record_token_span( + &mut trace, + origins, + idx, + &result, + start, + RuleId::emitter(EmitterRule::UndeclaredTokenOutput), + ); } Token::PreEncoded(bytes) => { // 제39항 한글 wrap 점형은 영어 모드를 자동으로 휴면(⠸⠷)·재개(⠸⠾)시킨다. @@ -524,7 +558,16 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result } else if bytes.as_slice() == HANGUL_WRAP_END_BYTES { roman_mode::set_section_open_keeping_number_chain(&mut ir.state, true); } + let start = result.len(); result.extend(bytes); + record_token_span( + &mut trace, + origins, + idx, + &result, + start, + RuleId::emitter(EmitterRule::UndeclaredTokenOutput), + ); } } } @@ -540,6 +583,62 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result Ok(result) } +/// Attribute `start..end` to the rule that produced token `idx`. +/// +/// A math expression arrives here as one pre-encoded run, so its own rules would +/// be hidden behind the token rule that detected it. When the run is one the math +/// engine produced, its per-rule spans replace the single token-rule span. +/// Close the Roman section and attribute whatever terminator it wrote. +/// +/// The emitter decides section boundaries from the token stream, so these cells +/// never pass through a character rule and would otherwise be the one part of a +/// Korean/Roman sentence left unexplained. +fn close_roman_section_traced( + result: &mut Vec, + state: &mut EncoderState, + all_tokens: &[Token<'_>], + token_index: usize, + trace: &mut Option>, +) { + let start = result.len(); + close_roman_section(result, state, all_tokens, token_index); + if let Some(sink) = trace.as_mut() + && result.len() > start + { + sink.record_span( + RuleId::emitter(EmitterRule::RomanSectionMarker), + token_index, + start..result.len(), + ); + } +} + +fn record_token_span( + trace: &mut Option>, + origins: Option<&TokenOrigins>, + idx: usize, + result: &[u8], + start: usize, + fallback: RuleId, +) { + let Some(sink) = trace.as_mut() else { + return; + }; + let end = result.len(); + if start == end { + return; + } + if let Some(spans) = crate::rules::math::spans_for(&result[start..end]) { + for (rule, offset, len) in spans { + let span_start = start + offset as usize; + sink.record_span(rule, idx, span_start..span_start + len as usize); + } + return; + } + let rule = origins.and_then(|o| o.get(idx)).unwrap_or(fallback); + sink.record_span(rule, idx, start..end); +} + fn collect_word_texts<'tokens, 'source>(tokens: &'tokens [Token<'source>]) -> Vec<&'tokens str> { let mut word_texts = Vec::with_capacity(tokens.len().div_ceil(2)); @@ -679,9 +778,8 @@ fn spaced_ampersand_connects_roman_words(tokens: &[Token<'_>], ampersand_index: return false; } - tokens + tokens[ampersand_index + 1..] .iter() - .skip(ampersand_index + 1) .find_map(|token| match token { Token::Space(_) | Token::Mode(_) => None, Token::Word(word) => Some( @@ -697,7 +795,7 @@ fn spaced_ampersand_connects_roman_words(tokens: &[Token<'_>], ampersand_index: /// Rule 29 keeps consecutive Roman/number text in one section even across /// print spaces. A separated enclosure continues that section only when the -/// *complete* enclosure is Roman/number text. This distinguishes +/// *complete* enclosure is Roman text with no Hangul. This distinguishes /// `GRI (Global Reporting Initiative)` from `Poison (모래성)` and from a mixed /// gloss such as `TVB (Television - 전시광파유한공사)`. fn separated_symbol_continues_roman_section(tokens: &[Token<'_>], token_index: usize) -> bool { @@ -727,25 +825,30 @@ fn separated_symbol_continues_roman_section(tokens: &[Token<'_>], token_index: u return true; } - // Rule 35: punctuation may introduce a numeric continuation (`'23`). - if next_word + let group = next_word .chars - .iter() - .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) - .is_some_and(char::is_ascii_digit) + .first() + .and_then(|opening| matching_group_close(*opening).map(|closing| (*opening, closing))); + + // Rule 35: punctuation may introduce a numeric continuation (`'23`). A + // spaced opening bracket instead starts a group of its own: a number in it + // (`KT (32,750원)`, `C (55)`) is not joined to the Roman word before. + if group.is_none() + && next_word + .chars + .iter() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(char::is_ascii_digit) { return true; } - let Some(opening) = next_word.chars.first().copied() else { - return false; - }; - let Some(closing) = matching_group_close(opening) else { + let Some((opening, closing)) = group else { return false; }; let mut depth = 0usize; - let mut saw_roman_or_number = false; + let mut saw_roman = false; let mut saw_korean = false; for token in tokens.iter().skip(token_index + 1) { match token { @@ -761,12 +864,12 @@ fn separated_symbol_continues_roman_section(tokens: &[Token<'_>], token_index: u // and the function returns as soon as that level closes. depth -= 1; if depth == 0 { - return saw_roman_or_number && !saw_korean; + return saw_roman && !saw_korean; } continue; } if depth > 0 { - saw_roman_or_number |= ch.is_ascii_alphanumeric(); + saw_roman |= ch.is_ascii_alphabetic(); saw_korean |= crate::utils::is_korean_char(ch); } } @@ -891,6 +994,7 @@ fn apply_core_encoding_rules( remaining_words: &[&str], prev_word: &str, result: &mut Vec, + trace: Option>, ) -> Result { let mut ctx = RuleContext { word_chars, @@ -906,7 +1010,7 @@ fn apply_core_encoding_rules( state, result, }; - engine.apply_phase(Phase::CoreEncoding, &mut ctx) + engine.apply_phase(Phase::CoreEncoding, &mut ctx, trace) } #[allow(clippy::too_many_arguments)] @@ -924,6 +1028,7 @@ fn apply_inter_character_rules( remaining_words: &[&str], prev_word: &str, result: &mut Vec, + trace: Option>, ) -> Result { let mut ctx = RuleContext { word_chars, @@ -939,9 +1044,10 @@ fn apply_inter_character_rules( state, result, }; - engine.apply_phase(Phase::InterCharacter, &mut ctx) + engine.apply_phase(Phase::InterCharacter, &mut ctx, trace) } +#[allow(clippy::too_many_arguments)] fn emit_word( word: &WordToken, token_index: usize, @@ -950,6 +1056,7 @@ fn emit_word( all_tokens: &[Token], context: WordContext<'_>, result: &mut Vec, + mut trace: Option>, ) -> Result<(), String> { let prev_word = context.prev_word; let remaining_words = context.remaining_words; @@ -982,7 +1089,15 @@ fn emit_word( } let mut encoded = crate::encode(&numeric)?; encoded.extend(unit); + let start = result.len(); result.extend(encoded); + if let Some(sink) = trace.as_mut() { + sink.record_span( + crate::rules::trace::korean_rule_id("measurement_symbols"), + token_index, + start..result.len(), + ); + } roman_mode::set_section_open_keeping_number_chain(state, continues_roman_section); return Ok(()); } @@ -1010,7 +1125,17 @@ fn emit_word( } // English entry (제28/35/39항) — 로마자표/연속표 emit + 영어 모드 전환. + let roman_open_start = result.len(); roman_mode::enter_english_if_starting(state, word_chars, has_ascii_alphabetic, result); + if let Some(sink) = trace.as_mut() + && result.len() > roman_open_start + { + sink.record_span( + RuleId::emitter(EmitterRule::RomanSectionMarker), + token_index, + roman_open_start..result.len(), + ); + } let first_ascii_index = word_chars.iter().position(|c| c.is_ascii_alphabetic()); let ascii_starts_at_beginning = matches!(first_ascii_index, Some(0)); @@ -1087,7 +1212,13 @@ fn emit_word( } else if english_logic::should_force_terminator_before_symbol(*sym) || !english_logic::should_skip_terminator_for_symbol(*sym) { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced( + result, + state, + all_tokens, + token_index, + &mut trace, + ); } else { roman_mode::exit_english( state, @@ -1096,7 +1227,13 @@ fn emit_word( } } _ => { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced( + result, + state, + all_tokens, + token_index, + &mut trace, + ); } } } @@ -1111,7 +1248,15 @@ fn emit_word( // a capital indicator or a lowercase k-z cell is sufficient // for every other Roman letter class. if matches!(*c, 'a'..='j') { + let bridge_start = result.len(); result.push(crate::rules::korean::rule_29::ENGLISH_CONTINUATION); + if let Some(sink) = trace.as_mut() { + sink.record_span( + RuleId::emitter(EmitterRule::RomanSectionMarker), + token_index, + bridge_start..result.len(), + ); + } } roman_mode::resume_english_from_roman_number_chain(state); } @@ -1188,6 +1333,7 @@ fn emit_word( remaining_words, prev_word, result, + trace.as_mut().map(TraceSink::reborrow), )?; is_number = state.is_number; is_big_english = state.is_big_english; @@ -1217,6 +1363,7 @@ fn emit_word( remaining_words, prev_word, result, + trace.as_mut().map(TraceSink::reborrow), )?; is_number = state.is_number; is_big_english = state.is_big_english; @@ -1242,7 +1389,7 @@ fn emit_word( // 영어 주도 문서: 영어 단어 사이의 종료표 ⠲ 모두 생략하고 영어 모드를 유지. } else if state.english_indicator && state.is_english { if remaining_words.is_empty() { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced(result, state, all_tokens, token_index, &mut trace); } else if let Some(next_word) = remaining_words.first() { let ascii_letters = next_word .chars() @@ -1284,7 +1431,13 @@ fn emit_word( // print has whitespace first (`Poison (모래성)`), // Rule 29 closes the Roman run before that space. if next_word_is_separated && !separated_continuation { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced( + result, + state, + all_tokens, + token_index, + &mut trace, + ); } else if separated_continuation && sym == '&' { // A standalone ampersand joining Roman words is // itself part of the current Roman section. @@ -1297,7 +1450,13 @@ fn emit_word( } else if english_logic::should_force_terminator_before_symbol(sym) || !english_logic::should_skip_terminator_for_symbol(sym) { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced( + result, + state, + all_tokens, + token_index, + &mut trace, + ); } else { roman_mode::exit_english( state, @@ -1306,11 +1465,17 @@ fn emit_word( } } _ => { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced( + result, + state, + all_tokens, + token_index, + &mut trace, + ); } } } else { - close_roman_section(result, state, all_tokens, token_index); + close_roman_section_traced(result, state, all_tokens, token_index, &mut trace); } } } @@ -1382,10 +1547,7 @@ mod tests { crate::rules::token_rules::emphasis_ring::EmphasisRingRule, )); engine.register(Box::new( - crate::rules::token_rules::latex_fraction::LatexFractionRule, - )); - engine.register(Box::new( - crate::rules::token_rules::inline_fraction::InlineFractionRule, + crate::rules::token_rules::math_expression::MathExpressionTokenRule, )); engine.register(Box::new( crate::rules::token_rules::word_shortcut::WordShortcutRule, @@ -1424,7 +1586,7 @@ mod tests { .apply_all(&mut ir.tokens, &mut ir.state) .unwrap(); ir.state = state_before_token_rules; - let emitted = emit(&mut ir, &mut engine).unwrap(); + let emitted = emit(&mut ir, &mut engine, None, None).unwrap(); let expected = encode(text).unwrap(); assert_eq!( emitted, expected, @@ -1623,6 +1785,8 @@ mod tests { #[case::nested_complete_group("nested", true)] #[case::non_textual_group_body("non_text", false)] #[case::unclosed_group("unclosed", false)] + #[case::number_group("number", false)] + #[case::apostrophe_year("year", true)] fn separated_symbol_requires_a_complete_roman_group( #[case] scenario: &str, #[case] expected: bool, @@ -1650,6 +1814,16 @@ mod tests { Token::Space(SpaceKind::Regular), word_token("(Beta"), ], + "number" => vec![ + word_token("KT"), + Token::Space(SpaceKind::Regular), + word_token("(55)"), + ], + "year" => vec![ + word_token("Class"), + Token::Space(SpaceKind::Regular), + word_token("'23"), + ], _ => unreachable!("unknown fixture"), }; @@ -1664,7 +1838,8 @@ mod tests { let mut ir = DocumentIR::parse("ABC/한글", true); let mut engine = make_char_engine(); - let output = emit(&mut ir, &mut engine).expect("mixed Roman/Korean word must encode"); + let output = + emit(&mut ir, &mut engine, None, None).expect("mixed Roman/Korean word must encode"); assert!(output.contains(&crate::unicode::decode_unicode('⠲'))); assert!(!ir.state.is_english); @@ -1693,6 +1868,7 @@ mod tests { remaining_words: &remaining_words, }, &mut result, + None, ) .expect("Roman word must encode"); @@ -1748,7 +1924,7 @@ mod tests { state: EncoderState::new(false), }; let mut engine = make_char_engine(); - let out = emit(&mut ir, &mut engine).unwrap(); + let out = emit(&mut ir, &mut engine, None, None).unwrap(); assert_eq!(out, vec![52, 48, 32, 32, 32, 32, 32, 32, 4, 48]); } @@ -1770,7 +1946,7 @@ mod tests { }; let mut engine = make_char_engine(); - let out = emit(&mut ir, &mut engine).unwrap(); + let out = emit(&mut ir, &mut engine, None, None).unwrap(); assert!(out.starts_with(&[52, 32, 32])); } @@ -1792,7 +1968,7 @@ mod tests { }; let mut engine = make_char_engine(); - let out = emit(&mut ir, &mut engine).unwrap(); + let out = emit(&mut ir, &mut engine, None, None).unwrap(); assert!(out.starts_with(&[52, 32, 32])); assert_eq!(out.iter().filter(|byte| **byte == 52).count(), 1); @@ -1827,7 +2003,7 @@ mod tests { }; let mut engine = make_char_engine(); - let out = emit(&mut ir, &mut engine).unwrap(); + let out = emit(&mut ir, &mut engine, None, None).unwrap(); assert_eq!(out.iter().filter(|byte| **byte == 52).count(), 1); } @@ -2057,7 +2233,7 @@ mod tests { state: EncoderState::new(false), }; let mut engine = make_char_engine(); - let out = emit(&mut ir, &mut engine).unwrap(); + let out = emit(&mut ir, &mut engine, None, None).unwrap(); let mut expected = fraction::encode_fraction("1", "2").unwrap(); expected.push(0); @@ -2120,7 +2296,7 @@ mod tests { let mut ir = DocumentIR::parse("", false); ir.state.triple_big_english = true; let mut engine = RuleEngine::new(); - let result = emit(&mut ir, &mut engine).unwrap(); + let result = emit(&mut ir, &mut engine, None, None).unwrap(); assert_eq!( result, vec![32, 4], @@ -2143,6 +2319,42 @@ mod spaced_colon_coverage { }) } + /// 제29항: a spaced `&` joins two Roman words, so it needs a Roman word on + /// each side. Already-encoded output and the end of the stream both prove + /// nothing, so neither side may be read as Roman. + #[rstest::rstest] + #[case::nothing_follows(vec![word("A"), Token::Space(SpaceKind::Regular), word("&")], 2)] + #[case::encoded_output_follows( + vec![ + word("A"), + Token::Space(SpaceKind::Regular), + word("&"), + Token::Space(SpaceKind::Regular), + Token::PreEncoded(vec![1]), + ], + 2 + )] + #[case::nothing_precedes(vec![word("&"), Token::Space(SpaceKind::Regular), word("B")], 0)] + #[case::encoded_output_precedes( + vec![ + Token::PreEncoded(vec![1]), + Token::Space(SpaceKind::Regular), + word("&"), + Token::Space(SpaceKind::Regular), + word("B"), + ], + 2 + )] + fn an_ampersand_without_a_roman_word_on_both_sides_joins_nothing( + #[case] tokens: Vec>, + #[case] ampersand_index: usize, + ) { + assert!(!spaced_ampersand_connects_roman_words( + &tokens, + ampersand_index + )); + } + /// 제29항·제32항·제35항: a standalone colon joins two Roman items only when /// the item after it is proved Roman. #[test] @@ -2260,3 +2472,31 @@ mod roman_chain_resume_coverage { assert!(crate::encode_to_unicode(input).is_ok()); } } + +#[cfg(test)] +mod empty_token_span_tests { + use super::record_token_span; + use crate::rules::trace::{EmitterRule, RuleId, Trace, TraceSink}; + + /// A token can consume input without writing a cell. Attributing it anyway + /// would claim an output position the token never wrote, so the span is + /// dropped rather than recorded as empty. + #[test] + fn a_token_that_wrote_no_cells_records_nothing() { + let mut trace = Trace::default(); + let result = vec![1, 2, 3]; + { + let mut sink = Some(TraceSink::new(&mut trace)); + record_token_span( + &mut sink, + None, + 0, + &result, + result.len(), + RuleId::emitter(EmitterRule::WordSpace), + ); + } + + assert!(trace.events().is_empty(), "{:?}", trace.events()); + } +} diff --git a/libs/braillify/src/rules/engine.rs b/libs/braillify/src/rules/engine.rs index 53567614..0c9d0ecb 100644 --- a/libs/braillify/src/rules/engine.rs +++ b/libs/braillify/src/rules/engine.rs @@ -5,7 +5,9 @@ use std::collections::HashSet; +use super::RuleMeta; use super::context::RuleContext; +use super::trace::{RuleId, RuleOutcome, TraceEvent, TraceSink}; use super::traits::{BrailleRule, Phase, RuleResult}; /// The rule engine — holds all registered rules and applies them. @@ -45,6 +47,12 @@ impl RuleEngine { self.sorted = false; } + /// Metadata of every registered rule, in [`RuleId`] order. + pub(crate) fn registry(&mut self) -> Vec<&'static RuleMeta> { + self.ensure_sorted(); + self.rules.iter().map(|rule| rule.meta()).collect() + } + /// Disable a rule by its section ID (e.g., "11" to disable 제11항). #[cfg(test)] pub fn disable(&mut self, section: &str) { @@ -79,7 +87,7 @@ impl RuleEngine { /// List all registered rule metadata (for introspection/debugging). #[cfg(test)] - pub fn list_rules(&self) -> Vec<&super::RuleMeta> { + pub fn list_rules(&self) -> Vec<&'static RuleMeta> { self.rules.iter().map(|r| r.meta()).collect() } @@ -123,10 +131,16 @@ impl RuleEngine { &mut self, phase: Phase, ctx: &mut RuleContext, + mut trace: Option>, ) -> Result { self.ensure_sorted(); - for rule in &self.rules { + // `rule.apply` is an opaque dyn call that mutates `ctx`, so the reads + // `TraceSpan::open` performs cannot be sunk past it. Gating them on a + // loop-invariant flag keeps the untraced path free of that work. + let tracing = trace.is_some(); + + for (index, rule) in self.rules.iter().enumerate() { if rule.phase() != phase { continue; } @@ -135,7 +149,12 @@ impl RuleEngine { if !rule.matches(ctx) { continue; } - match rule.apply(ctx)? { + let span = tracing.then(|| TraceSpan::open(ctx)); + let outcome = rule.apply(ctx)?; + if let (Some(span), Some(sink)) = (span, trace.as_mut()) { + span.close(RuleId::korean(index), outcome, ctx, sink); + } + match outcome { RuleResult::Consumed => return Ok(RuleResult::Consumed), RuleResult::Continue => {} RuleResult::Skip => {} @@ -146,6 +165,74 @@ impl RuleEngine { } } +struct TraceSpan { + output_start: u32, + char_start: u32, + skip_before: u32, +} + +impl TraceSpan { + fn open(ctx: &RuleContext) -> Self { + Self { + output_start: ctx.result.len() as u32, + char_start: ctx.index as u32, + skip_before: *ctx.skip_count as u32, + } + } + + /// Records by what a rule PRODUCED, not by what it returned. + /// + /// `Skip` normally means the rule declined and explains nothing, so it is + /// dropped — but a few rules emit a mode indicator and still return `Skip` + /// to let the next rule encode the character. Those cells are in the output + /// and something has to account for them. + /// + /// A rule that reported per-article spans (syllable composition) is recorded + /// as those articles instead of as itself, so a syllable names 제3항 for its + /// 받침 rather than one composite entry for the whole character. + fn close( + self, + rule: RuleId, + result: RuleResult, + ctx: &mut RuleContext, + sink: &mut TraceSink<'_>, + ) { + let produced_cells = ctx.result.len() as u32 > self.output_start; + let outcome = match result { + RuleResult::Consumed => RuleOutcome::Consumed, + RuleResult::Continue => RuleOutcome::Continued, + RuleResult::Skip if produced_cells => RuleOutcome::Continued, + RuleResult::Skip => return, + }; + let consumed_extra = (*ctx.skip_count as u32).saturating_sub(self.skip_before); + let word_chars = self.char_start..self.char_start + 1 + consumed_extra; + let end = ctx.result.len() as u32; + + let mut recorded_any = false; + if let Some(spans) = ctx.state.jamo_spans.as_deref_mut() { + for (jamo, span) in spans.drain() { + recorded_any = true; + sink.trace.push(TraceEvent { + rule: RuleId::jamo(jamo), + outcome, + token_index: sink.token_index, + word_chars: word_chars.clone(), + output: self.output_start + span.start..self.output_start + span.end, + }); + } + } + if !recorded_any { + sink.trace.push(TraceEvent { + rule, + outcome, + token_index: sink.token_index, + word_chars, + output: self.output_start..end, + }); + } + } +} + impl Default for RuleEngine { fn default() -> Self { Self::new() @@ -433,6 +520,115 @@ mod tests { assert_eq!(outcome, RuleResult::Skip); } + /// `TraceSpan::close` records by what a rule PRODUCED, not by what it + /// returned. A few rules write a mode indicator and still return `Skip` so + /// the next rule encodes the character — 제29항 로마자표 is the usual one — + /// and those cells are in the output, so something has to account for them. + /// A rule that returned `Skip` without writing anything explains nothing + /// and must stay out of the trace. + #[test] + fn a_skipping_rule_is_recorded_only_when_it_wrote_cells() { + use crate::char_struct::CharType; + use crate::rules::trace::{Trace, TraceSink}; + + static META_INDICATOR: RuleMeta = RuleMeta { + section: "indicator-skip", + subsection: None, + name: "indicator_then_skip", + standard_ref: "", + description: "writes a mode indicator, then defers to the next rule", + }; + static META_SILENT: RuleMeta = RuleMeta { + section: "silent-skip", + subsection: None, + name: "silent_skip", + standard_ref: "", + description: "matches but declines without writing anything", + }; + + struct IndicatorThenSkip; + impl BrailleRule for IndicatorThenSkip { + fn meta(&self) -> &'static RuleMeta { + &META_INDICATOR + } + fn phase(&self) -> Phase { + Phase::CoreEncoding + } + fn matches(&self, _: &RuleContext) -> bool { + true + } + fn apply(&self, ctx: &mut RuleContext) -> Result { + ctx.emit(48); + Ok(RuleResult::Skip) + } + } + + struct SilentSkip; + impl BrailleRule for SilentSkip { + fn meta(&self) -> &'static RuleMeta { + &META_SILENT + } + fn phase(&self) -> Phase { + Phase::CoreEncoding + } + fn matches(&self, _: &RuleContext) -> bool { + true + } + fn apply(&self, _: &mut RuleContext) -> Result { + Ok(RuleResult::Skip) + } + } + + let mut engine = RuleEngine::new(); + engine.register(Box::new(IndicatorThenSkip)); + engine.register(Box::new(SilentSkip)); + + let word_chars = vec!['x']; + let char_type = CharType::English('x'); + let empty: [&str; 0] = []; + let mut skip = 0usize; + let mut state = EncoderState::new(false); + let mut result = Vec::new(); + let mut trace = Trace::default(); + { + let mut ctx = RuleContext { + word_chars: &word_chars, + index: 0, + char_type: &char_type, + prev_word: "", + remaining_words: &empty, + has_korean_char: false, + is_all_uppercase: false, + ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, + skip_count: &mut skip, + state: &mut state, + result: &mut result, + }; + + let outcome = engine + .apply_phase( + Phase::CoreEncoding, + &mut ctx, + Some(TraceSink::new(&mut trace)), + ) + .expect("neither rule fails"); + + assert_eq!(outcome, RuleResult::Skip); + } + + assert_eq!(result, vec![48]); + assert_eq!( + trace.events().len(), + 1, + "only the rule that wrote a cell is recorded: {:?}", + trace.events() + ); + assert_eq!(trace.events()[0].rule, RuleId::korean(0)); + assert_eq!(trace.events()[0].outcome, RuleOutcome::Continued); + assert_eq!(trace.events()[0].output, 0..1); + } + /// engine.rs line 124 - `apply_phase` skip arm for disabled rules. #[test] fn engine_apply_phase_skips_disabled_rules() { @@ -464,7 +660,9 @@ mod tests { }; // TestRule.phase() = CoreEncoding; with disabled section "test", apply_phase // hits the `if !self.is_enabled(meta.section) { continue; }` arm. - let outcome = engine.apply_phase(Phase::CoreEncoding, &mut ctx).unwrap(); + let outcome = engine + .apply_phase(Phase::CoreEncoding, &mut ctx, None) + .unwrap(); assert_eq!(outcome, RuleResult::Skip); } } diff --git a/libs/braillify/src/rules/english_ueb/appendix_3.rs b/libs/braillify/src/rules/english_ueb/appendix_3.rs new file mode 100644 index 00000000..e476d37f --- /dev/null +++ b/libs/braillify/src/rules/english_ueb/appendix_3.rs @@ -0,0 +1,102 @@ +//! Appendix 3 Symbols List. +//! +//! RUEB 2024 Appendix 3 lists every braille symbol with its print symbol. The +//! symbols here are those no section of the rules assigns: mostly the technical +//! signs of the Guidelines for Technical Material, whose part the list cites in +//! square brackets. "When not otherwise indicated, symbols are assumed to take a +//! grade 1 meaning" (the list's usage column), so each entry is the symbol's full +//! grade 1 form and the list is also the reading of a symbol standing on its own. + +use crate::unicode::decode_unicode; + +fn cells(s: &str) -> Vec { + s.chars().map(decode_unicode).collect() +} + +/// The Appendix 3 form of `c`, or `None` if the list gives it nowhere else than a +/// section the other tables already cover. +pub fn encode_symbol(c: char) -> Option> { + let braille = match c { + '∫' => "⠮", // integral sign [11] + '∥' => "⠼⠇", // parallel to [11] + '∞' => "⠼⠿", // infinity sign [11] + '⟂' => "⠼⠤", // perpendicular to [3] + '⦀' => "⠼⠸⠇", // triple vertical bar [3] + '⊾' => "⠼⠸⠪", // measured right angle sign [11] + '∂' => "⠈⠙", // partial derivative [11] + '∅' => "⠈⠚", // null set [10] + '∮' => "⠈⠮", // closed line integral [11] + '¬' => "⠈⠹", // "not" sign [10] + '∨' => "⠈⠖", // or [10] + '∧' => "⠈⠦", // and [10] + '∵' => "⠈⠌", // "since" [11] + '∋' => "⠈⠘⠑", // contains as an element [10] + '⊲' => "⠈⠸⠣", // is a normal subgroup of [10] + '⊣' => "⠈⠸⠒", // reverse assertion [10] + '⊳' => "⠈⠸⠜", // inverse "is normal subgroup" [10] + '∀' => "⠘⠁", // "for all" [11] + '∇' => "⠘⠙", // del, nabla [11] + '∈' => "⠘⠑", // is an element of [10] + '⊂' => "⠘⠣", // is a subset of [10] + '∃' => "⠘⠢", // "there exists" [11] + '≈' => "⠘⠔", // approximately equal to [3] + '⊃' => "⠘⠜", // is a superset of [10] + '⊨' => "⠘⠸⠒", // "is valid" sign [10] + '⇌' => "⠘⠸⠶", // equilibrium arrow [16] + '≏' => "⠘⠐⠶", // difference between [3] + '≡' => "⠸⠿", // equivalent to [3] + '∠' => "⠸⠪", // angle sign [11] + '⊦' => "⠸⠒", // assertion [10] + '±' => "⠸⠖", // plus-or-minus [3] + '≃' => "⠸⠔", // approximately equal to, tilde over line [3] + '∓' => "⠸⠤", // minus-or-plus [3] + '≤' => "⠸⠈⠣", // less than or equal to [3] + '≥' => "⠸⠈⠜", // greater than or equal to [3] + '⊆' => "⠸⠘⠣", // contained in or equal to [10] + '⊇' => "⠸⠘⠜", // contains or equal to [10] + '⊴' => "⠸⠸⠣", // normal subgroup of or equal [10] + '⊵' => "⠸⠸⠜", // inverse "normal subgroup or equal" [10] + '∝' => "⠸⠐⠶", // is proportional to [3, 11] + '√' => "⠐⠩", // radical without vinculum [8] + '∗' => "⠐⠔", // asterisk operator [3] + '∘' => "⠐⠴", // "hollow dot" [11] + '≅' => "⠐⠸⠔", // tilde over equals sign [3] + '`' => "⠨⠡", // grave accent alone + '¦' => "⠨⠳", // broken vertical bar [11] + '∪' => "⠨⠖", // union [10] + '∩' => "⠨⠦", // intersection [10] + '≪' => "⠨⠈⠣", // is much less than [3] + '≫' => "⠨⠈⠜", // is much greater than [3] + '⊊' => "⠨⠘⠣", // proper subset [10] + '⊋' => "⠨⠘⠜", // proper superset [10] + '∡' => "⠨⠸⠪", // measured angle sign [11] + '⫤' => "⠨⠸⠒", // reverse "is valid" sign [10] + '≑' => "⠨⠐⠶", // equals sign dotted above and below [3] + '∴' => "⠠⠡", // "therefore" [11] + _ => return None, + }; + Some(cells(braille)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[rstest::rstest] + #[case::integral('∫', "⠮")] + #[case::perpendicular('⟂', "⠼⠤")] + #[case::element_of('∈', "⠘⠑")] + #[case::less_or_equal('≤', "⠸⠈⠣")] + #[case::radical('√', "⠐⠩")] + #[case::grave_alone('`', "⠨⠡")] + #[case::much_greater('≫', "⠨⠈⠜")] + #[case::therefore('∴', "⠠⠡")] + fn lists_the_grade1_form(#[case] c: char, #[case] expected: &str) { + assert_eq!(encode_symbol(c), Some(cells(expected))); + } + + #[test] + fn symbols_of_other_sections_are_not_listed_here() { + assert_eq!(encode_symbol('+'), None); + } +} diff --git a/libs/braillify/src/rules/english_ueb/contraction.rs b/libs/braillify/src/rules/english_ueb/contraction.rs index a8a914e6..94fb0f59 100644 --- a/libs/braillify/src/rules/english_ueb/contraction.rs +++ b/libs/braillify/src/rules/english_ueb/contraction.rs @@ -26,8 +26,26 @@ pub struct ContractionMatch { pub protect_span: bool, } +/// Placeholder for a contraction rule that has not declared its UEB section yet. +/// Rules keeping this default are reported as unattributed rather than being +/// credited to a section nobody checked against the standard. +pub static UNDECLARED_UEB_RULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "?", + subsection: None, + name: "undeclared_ueb_rule", + standard_ref: "", + description: "", +}; + /// One UEB contraction rule (§10.x). `word` is the lowercased letter slice. pub trait ContractionRule: Send + Sync { + /// The UEB section this rule implements. Defaults to + /// [`UNDECLARED_UEB_RULE`] until someone checks the section against the + /// standard. + fn meta(&self) -> &'static crate::rules::RuleMeta { + &UNDECLARED_UEB_RULE + } + /// Offer a match starting at `pos`, or `None`. fn try_match(&self, word: &[char], pos: usize) -> Option; } @@ -83,6 +101,22 @@ impl ContractionEngine { .collect() } + /// [`Self::matches_at`] with each match paired to its rule's registration + /// index. This is the only place that index is known, and the DP needs it to + /// name the rule behind a match it eventually selects. + pub fn matches_at_indexed(&self, word: &[char], pos: usize) -> Vec<(usize, ContractionMatch)> { + self.rules + .iter() + .enumerate() + .filter_map(|(index, rule)| rule.try_match(word, pos).map(|m| (index, m))) + .collect() + } + + /// Metadata of every registered rule, in registration index order. + pub fn registry(&self) -> Vec<&'static crate::rules::RuleMeta> { + self.rules.iter().map(|rule| rule.meta()).collect() + } + /// Encode a lowercased letter slice to braille cells. /// Returns `None` if a character cannot be encoded as an English letter. #[cfg(test)] @@ -222,6 +256,20 @@ mod tests { assert_eq!(cells, vec![decode_unicode('⠃')]); } + /// §10.4 strong groupsigns and §10.6 lower groupsigns are matched through + /// the §10.11 bridge rule and the §10.6.4/§10.6.8 gated rules rather than + /// registered on their own, so neither has had its section checked against + /// the standard. Reporting the undeclared placeholder keeps their cells out + /// of a section nobody verified instead of crediting a plausible-looking + /// one, which a trace consumer would have no way to distrust. + #[rstest::rstest] + #[case::strong_groupsign(&crate::rules::english_ueb::rule_10_4::StrongGroupsignRule)] + #[case::lower_groupsign(&crate::rules::english_ueb::rule_10_6::LowerGroupsignRule)] + fn an_undeclared_rule_reports_the_placeholder_section(#[case] rule: &dyn ContractionRule) { + assert_eq!(rule.meta().name, "undeclared_ueb_rule"); + assert_eq!(rule.meta().section, "?"); + } + #[test] fn match_longest_accepts_runtime_word_slice() { static MAP: phf::Map<&'static str, u8> = phf::phf_map! { diff --git a/libs/braillify/src/rules/english_ueb/engine.rs b/libs/braillify/src/rules/english_ueb/engine.rs index a9a18b80..29c7bf37 100644 --- a/libs/braillify/src/rules/english_ueb/engine.rs +++ b/libs/braillify/src/rules/english_ueb/engine.rs @@ -61,6 +61,15 @@ pub(super) const SPACE: u8 = 0; type ForeignScope = Option<(super::rule_13::AccentCode, bool)>; type ActiveTypeformPassage = (usize, super::token::Typeform, bool, ForeignScope); +fn settle_inline_technical_attribution(start: Option<(usize, usize)>, out: &[u8]) { + if let Some((start, checkpoint)) = start + && out.len() > start + { + super::rollback_attributions(checkpoint); + super::record_direct(super::UebMoveSource::InlineNemethCode, &out[start..], start); + } +} + /// Capitalisation pattern of a word (§8 subset currently supported). #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Caps { @@ -147,6 +156,18 @@ impl Default for EnglishUebEngine { } impl EnglishUebEngine { + /// §4.2.4: a word with a modified letter, contracted only around that letter. + pub(super) fn encode_modified(&self, chars: &[char]) -> Option> { + let mut out = Vec::new(); + match classify_caps(chars)? { + Caps::None => {} + Caps::Single => out.push(CAPITAL), + Caps::Word => out.extend([CAPITAL, CAPITAL]), + } + encode_modified_word(&self.contractions, chars, true, true, &mut out)?; + Some(out) + } + /// Build the engine with the currently-implemented contraction rules. pub fn new() -> Self { let mut contractions = ContractionEngine::default(); @@ -193,6 +214,11 @@ impl EnglishUebEngine { Self { contractions } } + /// Metadata of every contraction rule, in registration index order. + pub(super) fn contraction_rule_metas(&self) -> Vec<&'static crate::rules::RuleMeta> { + self.contractions.registry() + } + /// Encode one Roman word embedded in Korean text according to Korean rule 37. /// /// At a rule-37 Roman entry, the listed whole-word signs are suppressed @@ -331,7 +357,7 @@ impl EnglishUebEngine { let mut numeric_mode = false; let mut quote_open = false; let mut internal_double_quote_open = false; - let caret_note = contains_caret(tokens); + let caret_note = contains_caret(tokens) && tokens.len() > 1; let transcriber_note = contains_transcriber_note(tokens); // §9: index past a styled run already emitted as a word indicator, so its // member tokens are not re-emitted individually. @@ -544,7 +570,24 @@ impl EnglishUebEngine { 255, ]); } + // Most word branches leave the match with `continue`, so a word cannot be + // checked right after its arm. Carrying the mark to the next iteration + // (and past the loop) reaches every branch without touching any of them. + let mut pending_word: Option<(usize, usize)> = None; + let mut pending_symbol = None; + let mut inline_technical_start = None; + let mut pending_inline_technical = None; + let mut suppress_next_inline_dollar = false; + for i in 0..tokens.len() { + settle_inline_technical_attribution(pending_inline_technical.take(), &out); + super::settle_word_attribution(pending_word.take(), &out); + super::settle_symbol_attribution(pending_symbol.take(), &out); + if matches!(tokens[i], EnglishToken::Technical(_)) { + let start = inline_technical_start.take(); + suppress_next_inline_dollar |= start.is_some(); + settle_inline_technical_attribution(start, &out); + } if let Some((end, form)) = nested_inner_passage && i >= end { @@ -622,6 +665,18 @@ impl EnglishUebEngine { if !spatial_grade1_passage && cap_start[i] { out.extend([CAPITAL, CAPITAL, CAPITAL]); } + if matches!(tokens[i], EnglishToken::Symbol(_)) { + pending_symbol = Some(out.len()); + } + if matches!(tokens[i], EnglishToken::Symbol('$')) { + if suppress_next_inline_dollar { + suppress_next_inline_dollar = false; + } else if let Some(start) = inline_technical_start.take() { + pending_inline_technical = Some(start); + } else { + inline_technical_start = Some((out.len(), super::attribution_checkpoint())); + } + } match &tokens[i] { EnglishToken::Space => { encode_space_arm!(tokens, out, prev_was_number, numeric_mode, skip_to, line_mode_active, preserve_spatial_newlines, flatten_line_layout, spatial_grade1_passage, poem_linear_context, collapse_prose_double_space, skip_flattened_line_indent, numeric_separator_count, i) @@ -629,6 +684,7 @@ impl EnglishUebEngine { EnglishToken::Number(digits) => { skip_flattened_line_indent = false; line_mode_active = false; + let number_start = out.len(); if numeric_mode { // §6.3: already in numeric mode (digit-separator `,`/`.` // bridged us here) — emit digits only, no second `⠼`. @@ -639,16 +695,26 @@ impl EnglishUebEngine { out.extend(super::rule_6::encode_number(digits)?); numeric_separator_count = 0; } + super::record_whole_word(super::UebMoveSource::Numeric, &out[number_start..]); prev_was_number = true; numeric_mode = true; } EnglishToken::Technical(chars) => { skip_flattened_line_indent = false; - out.extend(super::rule_11::encode_technical(chars)?); + let cells = super::rule_11::encode_technical(chars)?; + // Direct records do not change `attempt_count`, so the next + // spelled-out word can still settle to §4.1. They also mask + // any nested word attempts from this complete §14.6.2 span. + super::push_direct( + &mut out, + super::UebMoveSource::InlineNemethCode, + &cells, + ); prev_was_number = false; numeric_mode = false; } EnglishToken::Word(chars) => { + pending_word = Some((out.len(), super::attempt_count())); encode_word_arm!(self, tokens, explicit_english, out, prev_was_number, numeric_mode, skip_to, line_mode_active, grade1_passage, cap_start_grade1, in_passage, escaped_code, regex_listing, spanish_foreign, foreign_passage, scansion_stress_context, early_english, spatial_grade1_passage, skip_flattened_line_indent, i, chars) } EnglishToken::WordDivision { chars, break_at } if poem_linear_context => { @@ -886,7 +952,7 @@ impl EnglishUebEngine { numeric_mode = true; } EnglishToken::Symbol(c) => { - encode_symbol_arm!(self, tokens, out, prev_was_number, numeric_mode, skip_to, line_mode_active, passage, cap_term, in_passage, url_listing, regex_listing, foreign_code, spanish_foreign, foreign_passage, early_english, preserve_spatial_newlines, skip_flattened_line_indent, numeric_separator_count, i, c) + encode_symbol_arm!(self, tokens, out, prev_was_number, numeric_mode, skip_to, line_mode_active, passage, cap_term, in_passage, url_listing, regex_listing, foreign_code, spanish_foreign, foreign_passage, early_english, preserve_spatial_newlines, skip_flattened_line_indent, numeric_separator_count, explicit_english, i, c); } EnglishToken::Styled(_, form) => { encode_styled_arm!(self, tokens, out, prev_was_number, numeric_mode, skip_to, passage, in_passage, foreign_code, spanish_foreign, foreign_passage, drop_styled_typeform_for_code_switch, skip_flattened_line_indent, nested_inner_passage, i, form) @@ -897,6 +963,9 @@ impl EnglishUebEngine { out.extend([CAPITAL, decode_unicode('⠄')]); } } + settle_inline_technical_attribution(pending_inline_technical.take(), &out); + super::settle_word_attribution(pending_word.take(), &out); + super::settle_symbol_attribution(pending_symbol.take(), &out); if let Some(span) = grade1_passage && span.needs_terminator { diff --git a/libs/braillify/src/rules/english_ueb/engine/bibliography.rs b/libs/braillify/src/rules/english_ueb/engine/bibliography.rs index 32098938..96f0aa57 100644 --- a/libs/braillify/src/rules/english_ueb/engine/bibliography.rs +++ b/libs/braillify/src/rules/english_ueb/engine/bibliography.rs @@ -260,7 +260,12 @@ pub(super) fn poem_linear_context(tokens: &[EnglishToken]) -> bool { || (!has_spatial_symbol && tokens .iter() - .filter(|t| matches!(t, EnglishToken::LineBreak)) + .filter(|t| { + matches!( + t, + EnglishToken::LineBreak | EnglishToken::WordDivision { .. } + ) + }) .count() >= 2) } diff --git a/libs/braillify/src/rules/english_ueb/engine/documents.rs b/libs/braillify/src/rules/english_ueb/engine/documents.rs index 33c26e2d..65a77fab 100644 --- a/libs/braillify/src/rules/english_ueb/engine/documents.rs +++ b/libs/braillify/src/rules/english_ueb/engine/documents.rs @@ -351,6 +351,8 @@ mod tests { #[rstest::rstest] #[case::sh_exclamation_spells("Sh!", "⠠⠎⠓⠖")] + #[case::sh_alone_spells("sh", "⠎⠓")] + #[case::st_abbreviation_spells("St Stephen", "⠠⠎⠞⠀⠠⠌⠑⠏⠓⠢")] #[case::th_apostrophe_spells("th'", "⠞⠓⠄")] #[case::th_apostrophe_n_contracts("th'n", "⠹⠄⠝")] fn strong_groupsign_word_ambiguity_10_4_2(#[case] text: &str, #[case] expected: &str) { diff --git a/libs/braillify/src/rules/english_ueb/engine/encode_space.rs b/libs/braillify/src/rules/english_ueb/engine/encode_space.rs index 6ddb32d8..c671aab2 100644 --- a/libs/braillify/src/rules/english_ueb/engine/encode_space.rs +++ b/libs/braillify/src/rules/english_ueb/engine/encode_space.rs @@ -30,12 +30,17 @@ macro_rules! encode_space_arm { } if is_numeric_space($tokens, $i) { $numeric_separator_count += 1; + let numeric_space_start = $out.len(); $skip_to = encode_following_number_as_numeric_space( $tokens, $i, &mut $out, $numeric_separator_count == 6, )?; + super::record_whole_word( + super::UebMoveSource::Numeric, + &$out[numeric_space_start..], + ); $prev_was_number = true; $numeric_mode = true; $line_mode_active = false; @@ -77,9 +82,9 @@ macro_rules! encode_space_arm { // legacy path but land here for Latin-embedded inputs like // `1in는 2.54cm이다.`, where the token stream contains no // multi-space runs and this branch would be a no-op anyway. - if $collapse_prose_double_space + if (($collapse_prose_double_space && styled_prose_double_space($tokens, $i)) + || space_run_inside_parentheses($tokens, $i)) && matches!($tokens.get($i + 1), Some(EnglishToken::Space)) - && styled_prose_double_space($tokens, $i) { $prev_was_number = false; $numeric_mode = false; diff --git a/libs/braillify/src/rules/english_ueb/engine/encode_styled.rs b/libs/braillify/src/rules/english_ueb/engine/encode_styled.rs index db19646b..31b31bcc 100644 --- a/libs/braillify/src/rules/english_ueb/engine/encode_styled.rs +++ b/libs/braillify/src/rules/english_ueb/engine/encode_styled.rs @@ -214,8 +214,10 @@ macro_rules! encode_styled_arm { } } else if chars.len() == 1 && !chars[0].is_ascii_alphabetic() { // §9: a single styled punctuation/symbol mark (`.̲` → `⠸⠆⠲`, - // `%̲` → `⠸⠆⠨⠴`). - $out.extend(super::rule_9::symbol_indicator(*$form)); + // `%̲` → `⠸⠆⠨⠴`). An open passage already covers it (§9.8.1). + if $passage.is_none() { + $out.extend(super::rule_9::symbol_indicator(*$form)); + } encode_styled_nonword_symbol(chars[0], &mut $out)?; } else { // Styled letters: passage / word / symbol level. The word @@ -288,7 +290,9 @@ macro_rules! encode_styled_arm { if matches!($tokens.get(end - 1), Some(EnglishToken::Symbol('.'))) && !styled_passage_introduced_by_colon($tokens, $i) && !bibliography_entry_context($tokens) - && (matches!( + // A period can carry U+0332, so an unmarked one + // after an underlined passage is outside it (§9.1.3). + && ((scope.is_none() && *$form == super::token::Typeform::Underline) || matches!( scope, Some((super::rule_13::AccentCode::Ueb, _)) ) || (styled_word_in_english_title($tokens, $i, *$form) diff --git a/libs/braillify/src/rules/english_ueb/engine/encode_symbol.rs b/libs/braillify/src/rules/english_ueb/engine/encode_symbol.rs index 552f5beb..0a9e5416 100644 --- a/libs/braillify/src/rules/english_ueb/engine/encode_symbol.rs +++ b/libs/braillify/src/rules/english_ueb/engine/encode_symbol.rs @@ -63,10 +63,56 @@ fn ueb_inverted_punctuation_cells(c: char) -> Vec { ] } +/// A symbol that is the whole text: Appendix 3 lists it with its grade 1 form. +/// A sign with several uses takes its primary one there — the §3.11 prime, not a +/// §15.2 stress mark; the §7.2 dash, not the long dash for omitted letters; the +/// §13.5.1 inverted marks, not the foreign-code cells. +fn lone_symbol_cells(c: char) -> Option> { + match c { + '′' => Some(vec![decode_unicode('⠶')]), + '″' => Some(vec![decode_unicode('⠶'), decode_unicode('⠶')]), + '—' => Some(vec![decode_unicode('⠠'), decode_unicode('⠤')]), + '¡' | '¿' => Some(ueb_inverted_punctuation_cells(c)), + _ => super::appendix_3::encode_symbol(c), + } +} + +/// §3.22 with §11.7: a sign printed inside a circle (U+20DD after it) is the +/// circle shape holding that sign — `⠰⠫`, the circle `⠿`, `⠪`, then the sign. +fn circled_sign_cells(c: char) -> Option> { + let sign = super::rule_7::encode_punctuation(c).or_else(|| super::rule_3::encode_symbol(c))?; + let mut cells = vec![ + GRADE1, + decode_unicode('⠫'), + decode_unicode('⠿'), + decode_unicode('⠪'), + ]; + cells.extend(sign); + Some(cells) +} + macro_rules! encode_symbol_arm { - ($engine:expr, $tokens:ident, $out:ident, $prev_was_number:ident, $numeric_mode:ident, $skip_to:ident, $line_mode_active:ident, $passage:ident, $cap_term:ident, $in_passage:ident, $url_listing:ident, $regex_listing:ident, $foreign_code:ident, $spanish_foreign:ident, $foreign_passage:ident, $early_english:ident, $preserve_spatial_newlines:ident, $skip_flattened_line_indent:ident, $numeric_separator_count:ident, $i:ident, $c:ident) => { + ($engine:expr, $tokens:ident, $out:ident, $prev_was_number:ident, $numeric_mode:ident, $skip_to:ident, $line_mode_active:ident, $passage:ident, $cap_term:ident, $in_passage:ident, $url_listing:ident, $regex_listing:ident, $foreign_code:ident, $spanish_foreign:ident, $foreign_passage:ident, $early_english:ident, $preserve_spatial_newlines:ident, $skip_flattened_line_indent:ident, $numeric_separator_count:ident, $explicit_english:ident, $i:ident, $c:ident) => { { $skip_flattened_line_indent = false; + if let Some(cells) = ($tokens.len() == 1) + .then(|| lone_symbol_cells(*$c)) + .flatten() + { + $out.extend(cells); + $prev_was_number = false; + $numeric_mode = false; + continue; + } + if matches!($tokens.get($i + 1), Some(EnglishToken::Symbol('\u{20DD}'))) + && let Some(cells) = circled_sign_cells(*$c) + { + $out.extend(cells); + $skip_to = $i + 2; + $prev_was_number = false; + $numeric_mode = false; + continue; + } if $passage.is_none() && let Some(active_passage) = guillemet_styled_passage( SymbolPassageContext { @@ -667,6 +713,14 @@ macro_rules! encode_symbol_arm { .or_else(|| super::rule_7::encode_punctuation(*$c)) .or_else(|| super::rule_3::encode_symbol(*$c)) .or_else(|| super::rule_6::encode_vulgar_fraction(*$c)) + .or_else(|| super::rule_4::eng_schwa_cells(*$c)) + // Text not declared English leaves these technical signs + // unread here, so it goes on to the math path instead. + .or_else(|| { + $explicit_english + .then(|| super::appendix_3::encode_symbol(*$c)) + .flatten() + }) }?; $out.extend(cells); if solidus_linebreak_space_after($tokens, $i) { diff --git a/libs/braillify/src/rules/english_ueb/engine/encode_word.rs b/libs/braillify/src/rules/english_ueb/engine/encode_word.rs index 6b67ebd3..5b22878a 100644 --- a/libs/braillify/src/rules/english_ueb/engine/encode_word.rs +++ b/libs/braillify/src/rules/english_ueb/engine/encode_word.rs @@ -270,7 +270,11 @@ macro_rules! encode_word_arm { .first() .is_some_and(|c| c.is_ascii_lowercase() && ('a'..='j').contains(c)) { - $out.push(GRADE1); + super::push_indicator( + &mut $out, + super::UebMoveSource::Grade1Indicator, + &[GRADE1], + ); } encode_literal_word($chars, &mut $out)?; } @@ -588,7 +592,11 @@ macro_rules! encode_word_arm { && (matches!(spelled_run, Some((start, _)) if start == $i) || matches!(initialism_run, Some((start, _)) if start == $i)) { - $out.extend([GRADE1, GRADE1]); + super::push_indicator( + &mut $out, + super::UebMoveSource::Grade1Indicator, + &[GRADE1, GRADE1], + ); } let letter_grade1 = !$cap_start_grade1 && spelled_run.is_none() @@ -598,9 +606,20 @@ macro_rules! encode_word_arm { || ($chars.len() == 1 && $chars[0].is_uppercase() && super::rule_5_7::is_wordsign_letter($chars[0]) - && matches!(next, Some(EnglishToken::Symbol('!'))))); + && matches!(next, Some(EnglishToken::Symbol('!')))) + // §2.6.3: a closing transcriber's note indicator after the + // letter's punctuation still leaves it standing alone. + || ($chars.len() == 1 + && super::rule_5_7::is_wordsign_letter($chars[0]) + && closing_transcriber_note_after_transparent_suffix($tokens, $i) + && (matches!(prev, None | Some(EnglishToken::Space)) + || transcriber_note_ends_at($tokens, $i, true)))); if after_number_grade1 || letter_grade1 || apostrophe_wrapped_letter($tokens, $i, $chars) { - $out.push(GRADE1); + super::push_indicator( + &mut $out, + super::UebMoveSource::Grade1Indicator, + &[GRADE1], + ); } if !$foreign_passage && document_all_words($tokens).len() >= 3 @@ -633,8 +652,9 @@ macro_rules! encode_word_arm { $numeric_mode = false; continue; } - let shortform_usable = - standing_alone && !matches!(next, Some(EnglishToken::Symbol('@' | '/'))); + let shortform_usable = (standing_alone + || apostrophe_joined_listed_word($tokens, $i)) + && !matches!(next, Some(EnglishToken::Symbol('@' | '/'))); // §10.5 lower wordsigns need a stricter boundary than §10.1/§10.2. let mut lower_usable = standing_alone && lower_wordsign_usable(prev, next); // §10.5.2: "enough's" keeps the wordsign (its interior apostrophe is diff --git a/libs/braillify/src/rules/english_ueb/engine/foreign.rs b/libs/braillify/src/rules/english_ueb/engine/foreign.rs index c7079835..867c4a49 100644 --- a/libs/braillify/src/rules/english_ueb/engine/foreign.rs +++ b/libs/braillify/src/rules/english_ueb/engine/foreign.rs @@ -47,6 +47,9 @@ pub(super) fn styled_word_is_foreign(chars: &[char]) -> bool { }) { return true; } + if is_capitals_abbreviation(chars) { + return false; + } let word: String = chars.iter().flat_map(|c| c.to_lowercase()).collect(); // §10.12.12: typeform does not block a contraction when the styled letters // themselves form a normal UEB groupsign (`tou𝐜𝐡ed`, `enoug̲h̲`). These short @@ -87,10 +90,19 @@ pub(super) fn styled_word_has_foreign_signal(chars: &[char]) -> bool { /// suppresses contractions inside the styled span. Short digraphs /// (`ch`/`gh`/`sh`/`th`/`wh`) which are themselves UEB groupsigns are /// exempted so a styled emphatic digraph (`tou𝐜𝐡ed`) keeps its contraction. +/// An all-capitals ASCII run (`UEB`) is an abbreviation, not foreign vocabulary, +/// though the pronouncing dictionary does not list it. +fn is_capitals_abbreviation(chars: &[char]) -> bool { + chars.len() >= 2 && chars.iter().all(char::is_ascii_uppercase) +} + pub(super) fn styled_single_word_is_foreign(chars: &[char]) -> bool { if styled_word_has_foreign_signal(chars) { return true; } + if is_capitals_abbreviation(chars) { + return false; + } let word: String = chars.iter().flat_map(|c| c.to_lowercase()).collect(); // A digraph groupsign (`ch`/`gh`/`sh`/`th`/`wh`) is 2 chars, so it is already // rejected by the `< 3` guard above — no separate digraph check is needed. @@ -354,6 +366,27 @@ pub(super) fn styled_word_in_lowercase_phrase_before_word( words.len() >= 2 && styled_words_are_lowercase(&words) && followed_by_word(tokens, k, expected) } +/// A print space run inside parentheses (`DIGEST: August`) is prose spacing, +/// not a column, so it is one braille space (§8.5.6 example). +pub(super) fn space_run_inside_parentheses(tokens: &[EnglishToken], i: usize) -> bool { + let depth = |range: &[EnglishToken]| { + range.iter().fold(0i32, |depth, t| match t { + EnglishToken::Symbol('(') => depth + 1, + EnglishToken::Symbol(')') => depth - 1, + _ => depth, + }) + }; + let line_start = tokens[..i] + .iter() + .rposition(|t| matches!(t, EnglishToken::LineBreak)) + .map_or(0, |p| p + 1); + let line_end = tokens[i..] + .iter() + .position(|t| matches!(t, EnglishToken::LineBreak)) + .map_or(tokens.len(), |p| i + p); + depth(&tokens[line_start..i]) > 0 && depth(&tokens[i..line_end]) < 0 +} + pub(super) fn styled_prose_double_space(tokens: &[EnglishToken], i: usize) -> bool { let has_typeform = tokens.iter().any(|t| matches!(t, EnglishToken::Styled(..))); let prev = i.checked_sub(1).and_then(|p| tokens.get(p)); diff --git a/libs/braillify/src/rules/english_ueb/engine/quotes.rs b/libs/braillify/src/rules/english_ueb/engine/quotes.rs index 7287352e..980be0d1 100644 --- a/libs/braillify/src/rules/english_ueb/engine/quotes.rs +++ b/libs/braillify/src/rules/english_ueb/engine/quotes.rs @@ -272,6 +272,7 @@ pub(super) fn apostrophe_wrapped_letter( tokens.get(index + 1), Some(EnglishToken::Symbol('\'' | '\u{2019}')) ) + && !matches!(tokens.get(index + 2), Some(EnglishToken::Word(_))) } /// §3.27: detect a transcriber's-note marker `[open tn]` / `[close tn]` starting diff --git a/libs/braillify/src/rules/english_ueb/engine/styled_methods.rs b/libs/braillify/src/rules/english_ueb/engine/styled_methods.rs index 580b9ed9..f6e45e5b 100644 --- a/libs/braillify/src/rules/english_ueb/engine/styled_methods.rs +++ b/libs/braillify/src/rules/english_ueb/engine/styled_methods.rs @@ -93,6 +93,11 @@ impl EnglishUebEngine { && !styled_word_in_english_title(ctx.tokens, i, form) && !styled_word_in_lowercase_phrase_before_word(ctx.tokens, i, form, "of") && !domain_component_context(ctx.tokens, i) + // §13.2.3: a foreign place name in English text (`Ždiar, Slovakia`) + // is anglicised and keeps its contractions. + && !(chars.first().is_some_and(|c| c.is_uppercase()) + && chars[1..].iter().all(|c| !c.is_uppercase()) + && matches!(ctx.tokens.get(j), Some(EnglishToken::Symbol(',')))) && styled_single_word_is_foreign(chars) { let doc_letters = document_letters(ctx.tokens); @@ -157,7 +162,13 @@ impl EnglishUebEngine { // anglicised-looking sub-segment (`chai`, `de`) shares the foreign // context. This is a span-level context that the per-segment // `styled_word_is_foreign` check cannot see. - let span_foreign_scope = if ctx.foreign_scope.is_some() { + let is_url_span = ctx.tokens[start..span_end].iter().any(|t| { + matches!( + t, + EnglishToken::Symbol(':' | '/') | EnglishToken::Styled(':' | '/', _) + ) + }); + let span_foreign_scope = if ctx.foreign_scope.is_some() || is_url_span { ctx.foreign_scope } else { let mut any_foreign = false; diff --git a/libs/braillify/src/rules/english_ueb/engine/tokens.rs b/libs/braillify/src/rules/english_ueb/engine/tokens.rs index 41182cb5..7a68eeb4 100644 --- a/libs/braillify/src/rules/english_ueb/engine/tokens.rs +++ b/libs/braillify/src/rules/english_ueb/engine/tokens.rs @@ -162,6 +162,35 @@ pub(super) fn apostrophe_joined_recorded_token_word(tokens: &[EnglishToken], i: super::super::pronunciation::apostrophe_elided_recorded_word_at(&joined, run_start, run_end) } +/// Appendix 1 lists some words written with an apostrophe (`children'swear`); +/// each piece of such a word may use its shortform though it does not stand alone. +pub(super) fn apostrophe_joined_listed_word(tokens: &[EnglishToken], i: usize) -> bool { + let apostrophe_at = |index: usize| { + matches!( + tokens.get(index), + Some(EnglishToken::Symbol('\'' | '\u{2019}')) + ) + }; + let word_at = |index: usize| matches!(tokens.get(index), Some(EnglishToken::Word(_))); + let mut start = i; + while start >= 2 && apostrophe_at(start - 1) && word_at(start - 2) { + start -= 2; + } + if start > 0 && apostrophe_at(start - 1) { + start -= 1; + } + let mut end = i; + while end + 2 < tokens.len() && apostrophe_at(end + 1) && word_at(end + 2) { + end += 2; + } + let joined: String = token_plain_chars(&tokens[start..=end]) + .iter() + .flat_map(|c| c.to_lowercase()) + .map(|c| if c == '\u{2019}' { '\'' } else { c }) + .collect(); + joined.contains('\'') && super::super::rule_10_9_list::is_listed(&joined) +} + pub(super) fn token_plain_chars_preserve_word_division(tokens: &[EnglishToken]) -> Vec { let mut chars = Vec::new(); for token in tokens { diff --git a/libs/braillify/src/rules/english_ueb/engine/word_methods.rs b/libs/braillify/src/rules/english_ueb/engine/word_methods.rs index 2e953a7f..284fa68a 100644 --- a/libs/braillify/src/rules/english_ueb/engine/word_methods.rs +++ b/libs/braillify/src/rules/english_ueb/engine/word_methods.rs @@ -1,4 +1,4 @@ -use super::*; +use super::*; impl EnglishUebEngine { pub(super) fn encode_word( @@ -84,13 +84,21 @@ impl EnglishUebEngine { ); } if shortform_usable && super::super::rule_10_9::is_pure_shortform_abbreviation(&word) { - out.push(GRADE1); + super::super::push_indicator( + out, + super::super::UebMoveSource::Grade1Indicator, + &[GRADE1], + ); } // Inside a §8.4 passage the ⠠⠠⠠ … ⠠⠄ carry capitalisation; `?` still guards // any residual mixed-case word there (→ legacy fallback). if !suppress_caps && !digit_adjacent && chemical_formula_caps(chars) { for &c in chars { - out.push(CAPITAL); + super::super::push_indicator( + out, + super::super::UebMoveSource::CapitalLetterIndicator, + &[CAPITAL], + ); out.push(crate::english::encode_english(c.to_ascii_lowercase()).ok()?); } return Some(()); @@ -98,7 +106,11 @@ impl EnglishUebEngine { match classify_caps(chars)? { _ if suppress_caps => {} Caps::None => {} - Caps::Single => out.push(CAPITAL), + Caps::Single => super::super::push_indicator( + out, + super::super::UebMoveSource::CapitalLetterIndicator, + &[CAPITAL], + ), Caps::Word => { // §8.7 / UEB §5.7.2: a *standing-alone* all-caps acronym whose // lowercase letters form a multi-letter shortform (e.g. `CD` = @@ -113,10 +125,17 @@ impl EnglishUebEngine { && !super::super::rule_10_9::is_pure_shortform_abbreviation(&word) && crate::rules::english_shortform::requires_grade1_indicator(&uppercase_word) { - out.push(GRADE1); + super::super::push_indicator( + out, + super::super::UebMoveSource::Grade1Indicator, + &[GRADE1], + ); } - out.push(CAPITAL); - out.push(CAPITAL); + super::super::push_indicator( + out, + super::super::UebMoveSource::CapitalisedWordIndicator, + &[CAPITAL, CAPITAL], + ); } } // §10.12.1: an all-caps initialism directly abutting a digit (`CH6`, @@ -168,7 +187,20 @@ impl EnglishUebEngine { | ['o', 'u'] | ['s', 't'] ); - if acronym_as_letters || letter_initialism || all_capitals_groupsign_word { + // §10.4.2: a standing-alone word that is exactly one of these strong + // groupsigns would be read as its strong wordsign (⠩ shall, ⠌ still), so + // its letters are brailled individually (`sh` ⠎⠓, `St.` ⠠⠎⠞⠲). + let strong_groupsign_word = standing_alone + && !digit_adjacent + && matches!( + lower.as_slice(), + ['c', 'h'] | ['s', 'h'] | ['t', 'h'] | ['w', 'h'] | ['o', 'u'] | ['s', 't'] + ); + if acronym_as_letters + || letter_initialism + || all_capitals_groupsign_word + || strong_groupsign_word + { for &c in &lower { match super::super::rule_4::accent_cells(c) { Some(cells) => out.extend(cells), @@ -185,20 +217,29 @@ impl EnglishUebEngine { let cell = upper_usable .then(|| { super::super::rule_10_1::wordsign(&word) - .or_else(|| super::super::rule_10_2::wordsign(&word)) + .map(|c| (c, super::super::UebMoveSource::AlphabeticWordsign)) + .or_else(|| { + super::super::rule_10_2::wordsign(&word) + .map(|c| (c, super::super::UebMoveSource::StrongWordsign)) + }) }) .flatten() .or_else(|| { lower_usable - .then(|| super::super::rule_10_5::wordsign(&word)) + .then(|| { + super::super::rule_10_5::wordsign(&word) + .map(|c| (c, super::super::UebMoveSource::LowerWordsign)) + }) .flatten() }); - if let Some(cell) = cell { + if let Some((cell, source)) = cell { + super::super::record_whole_word(source, &[cell]); out.push(cell); return Some(()); } } if shortform_usable && let Some(cells) = super::super::rule_10_9::whole_word_cells(&word) { + super::super::record_whole_word(super::super::UebMoveSource::Shortform, &cells); out.extend(cells); return Some(()); } @@ -365,8 +406,16 @@ impl EnglishUebEngine { // better convey the print meaning than a capitals-word indicator plus // terminator. Plural/suffix acronyms (`CDs`, `OKd`) remain under §8.6.3. for &c in &chars[..2] { - out.push(CAPITAL); - out.push(crate::english::encode_english(c.to_ascii_lowercase()).ok()?); + super::super::push_indicator( + out, + super::super::UebMoveSource::CapitalLetterIndicator, + &[CAPITAL], + ); + super::super::push_direct( + out, + super::super::UebMoveSource::Letter, + &[crate::english::encode_english(c.to_ascii_lowercase()).ok()?], + ); } let suffix: Vec = chars[2..].iter().flat_map(|c| c.to_lowercase()).collect(); out.extend( @@ -509,6 +558,7 @@ impl EnglishUebEngine { } bounds.push(chars.len()); + let attributions_before_buf = super::super::attribution_checkpoint(); let mut buf = Vec::new(); let mut prev_caps_word = false; for w in bounds.windows(2) { @@ -577,12 +627,19 @@ impl EnglishUebEngine { // §8.6.3: a §8.4 caps word (`⠠⠠`) is terminated by `⠠⠄` before lowercase // letters that continue the same word (`ABCs`, `WALKing`, `unSELFish`). if prev_caps_word && matches!(caps, Caps::None) { - buf.push(CAPITAL); - buf.push(decode_unicode('⠄')); + super::super::push_indicator( + &mut buf, + super::super::UebMoveSource::CapitalisedWordIndicator, + &[CAPITAL, decode_unicode('⠄')], + ); } if matches!(caps, Caps::Word) && w[0] > 0 && w[1] < chars.len() && seg.len() <= 2 { for cell in &cells { - buf.push(CAPITAL); + super::super::push_indicator( + &mut buf, + super::super::UebMoveSource::CapitalLetterIndicator, + &[CAPITAL], + ); buf.push(*cell); } prev_caps_word = false; @@ -590,16 +647,22 @@ impl EnglishUebEngine { } else { match caps { Caps::None => {} - Caps::Single => buf.push(CAPITAL), - Caps::Word => { - buf.push(CAPITAL); - buf.push(CAPITAL); - } + Caps::Single => super::super::push_indicator( + &mut buf, + super::super::UebMoveSource::CapitalLetterIndicator, + &[CAPITAL], + ), + Caps::Word => super::super::push_indicator( + &mut buf, + super::super::UebMoveSource::CapitalisedWordIndicator, + &[CAPITAL, CAPITAL], + ), } } buf.extend(&cells); prev_caps_word = matches!(caps, Caps::Word); } + super::super::rebase_attributions(attributions_before_buf, out.len()); out.extend(buf); Some(()) } diff --git a/libs/braillify/src/rules/english_ueb/mod.rs b/libs/braillify/src/rules/english_ueb/mod.rs index e6e2fbc1..210d37e3 100644 --- a/libs/braillify/src/rules/english_ueb/mod.rs +++ b/libs/braillify/src/rules/english_ueb/mod.rs @@ -12,6 +12,7 @@ //! //! Source of truth: `docs/Rules-of-Unified-English-Braille-2024.pdf`. +pub mod appendix_3; pub mod compound; pub mod contraction; pub mod engine; @@ -55,6 +56,532 @@ pub mod token; use engine::EnglishUebEngine; +thread_local! { + /// Word attempts and structural indicators, in emission order. The engine + /// encodes a word under several constraint combinations and keeps one, so + /// [`align_selected`] separates kept attempts from discarded ones while + /// retaining indicators emitted directly into the selected output. + static ATTRIBUTIONS: std::cell::RefCell>> = + const { std::cell::RefCell::new(None) }; +} + +enum AttributionRecord { + Word(WordAttempt), + Indicator(NonWordAttempt), + Direct(NonWordAttempt), +} + +/// The cells one attempt produced, plus where each rule's cells sat inside them. +struct WordAttempt { + cells: Vec, + moves: Vec<(crate::rules::trace::RuleId, u32, u32)>, + /// Where the cells were written, for an attempt that went straight into a + /// buffer rather than being one of several the engine chose between. The + /// literal `` markup of a chemical line repeats the same two-cell run + /// a dozen times, and a search cannot tell those occurrences apart. + offset: Option, +} + +struct NonWordAttempt { + cells: Vec, + rule: crate::rules::trace::RuleId, + /// Where the cells were written, when they went straight into the selected + /// output. A one-cell indicator such as `⠠` recurs all over a capitalised + /// line, so looking for it afterwards finds an earlier occurrence than the + /// one this record wrote. `None` marks a record taken against a buffer that + /// is appended elsewhere, whose final position is not known here. + offset: Option, +} + +/// Accumulates the moves of one word-encoding attempt. +/// +/// Offsets are taken against the attempt's own output as it is built, because +/// the encoder can insert cells between moves (a §10.13 line break), so a move's +/// position is not the running sum of the moves before it. +pub(super) struct AttemptRecorder { + /// `None` when no trace is being collected, so an untraced encode allocates + /// nothing per word. The check costs one thread-local read per attempt + /// rather than one per move. + moves: Option>, +} + +impl AttemptRecorder { + pub(super) fn new() -> Self { + let collecting = ATTRIBUTIONS.with(|slot| slot.borrow().is_some()); + Self { + moves: collecting.then(Vec::new), + } + } + + pub(super) fn push(&mut self, rule: crate::rules::trace::RuleId, offset: usize, len: usize) { + if let Some(moves) = self.moves.as_mut() { + moves.push((rule, offset as u32, len as u32)); + } + } + + pub(super) fn finish(self, cells: &[u8]) { + self.finish_at(cells, None); + } + + pub(super) fn finish_at(self, cells: &[u8], offset: Option) { + let Some(moves) = self.moves else { + return; + }; + ATTRIBUTIONS.with(|slot| { + if let Ok(mut slot) = slot.try_borrow_mut() + && let Some(records) = slot.as_mut() + { + records.push(AttributionRecord::Word(WordAttempt { + cells: cells.to_vec(), + moves, + offset, + })); + } + }); + } +} + +/// [`try_encode`] plus the rule behind each stretch of the output. +/// +/// A word encoder does not know where its cells land in the finished document, +/// so each attempt's ranges are recovered by locating that attempt's cells in +/// the output. Attempts whose cells are absent were discarded by the engine and +/// contribute nothing. A reported range therefore always points at cells its +/// rule actually produced. +pub(crate) fn try_encode_traced(text: &str) -> Option<(Vec, Vec)> { + let encoded = collect_selected(|| try_encode(text)); + encoded.map(|(cells, moves)| { + let spans = align_selected(&cells, &moves); + (cells, spans) + }) +} + +/// [`encode_forced`] plus the rule behind each stretch of the output. +pub(crate) fn encode_forced_traced(text: &str) -> Option<(Vec, Vec)> { + let encoded = collect_selected(|| encode_forced(text)); + encoded.map(|(cells, moves)| { + let spans = align_selected(&cells, &moves); + (cells, spans) + }) +} + +/// One stretch of output and the UEB rule that produced it. +pub(crate) type UebSpan = (crate::rules::trace::RuleId, core::ops::Range); + +fn collect_selected( + encode: impl FnOnce() -> Option>, +) -> Option<(Vec, Vec)> { + ATTRIBUTIONS.with(|slot| *slot.borrow_mut() = Some(Vec::new())); + let encoded = encode(); + let records = ATTRIBUTIONS + .with(|slot| slot.borrow_mut().take()) + .unwrap_or_default(); + encoded.map(|cells| (cells, records)) +} + +fn attribution_checkpoint() -> usize { + ATTRIBUTIONS.with(|slot| slot.borrow().as_ref().map_or(0, Vec::len)) +} + +/// Move records taken since `checkpoint` from a local buffer's coordinates to +/// the output's, once that buffer has been appended at `base`. +/// +/// A word is assembled in its own buffer, so a record made while filling it +/// knows only its place inside that buffer. Rebasing at the append is what +/// turns those into positions the finished output can be indexed by. +fn rebase_attributions(checkpoint: usize, base: usize) { + ATTRIBUTIONS.with(|slot| { + if let Ok(mut slot) = slot.try_borrow_mut() + && let Some(records) = slot.as_mut() + { + for record in records.iter_mut().skip(checkpoint) { + let offset = match record { + AttributionRecord::Word(w) => &mut w.offset, + AttributionRecord::Indicator(a) | AttributionRecord::Direct(a) => &mut a.offset, + }; + if let Some(offset) = offset.as_mut() { + *offset += base; + } + } + } + }); +} + +fn rollback_attributions(checkpoint: usize) { + ATTRIBUTIONS.with(|slot| { + if let Some(records) = slot.borrow_mut().as_mut() { + records.truncate(checkpoint); + } + }); +} + +/// Place each attempt's moves in the finished output, skipping attempts the +/// engine discarded. +/// +/// The scan only moves forward, so an attempt is matched at or after everything +/// already placed. A discarded attempt is recognised by its cells not appearing +/// there — the engine never emitted them. +fn align_selected(cells: &[u8], records: &[AttributionRecord]) -> Vec { + let mut direct_spans = Vec::new(); + let mut direct_cursor = 0usize; + for record in records { + if let AttributionRecord::Direct(direct) = record + && let Some(range) = locate(cells, direct, &mut direct_cursor, &[]) + { + direct_spans.push((direct.rule, range)); + } + } + + let mut indicator_spans = Vec::new(); + let mut indicator_cursor = 0usize; + for record in records { + if let AttributionRecord::Indicator(indicator) = record + && let Some(range) = locate(cells, indicator, &mut indicator_cursor, &direct_spans) + { + push_without_indicators(&mut indicator_spans, (indicator.rule, range), &direct_spans); + } + } + + let mut fixed_spans = direct_spans.clone(); + fixed_spans.extend(indicator_spans.iter().cloned()); + fixed_spans.sort_by_key(|(_, range)| range.start); + + let mut spans = Vec::new(); + let mut cursor = 0usize; + for record in records { + let attempt = match record { + AttributionRecord::Word(attempt) => attempt, + AttributionRecord::Indicator(_) | AttributionRecord::Direct(_) => continue, + }; + let placed = attempt.offset.filter(|offset| { + cells.get(*offset..offset + attempt.cells.len()) == Some(attempt.cells.as_slice()) + }); + let Some(base) = + placed.or_else(|| find_from_outside(cells, &attempt.cells, cursor, &direct_spans)) + else { + continue; + }; + for (rule, offset, len) in &attempt.moves { + let start = base + *offset as usize; + let end = start + *len as usize; + push_without_indicators(&mut spans, (*rule, start as u32..end as u32), &fixed_spans); + } + cursor = base + attempt.cells.len(); + } + spans.extend(indicator_spans); + spans.extend(direct_spans); + // An empty cell between words is the inter-word blank, the same structural + // output the Korean emitter accounts for. It carries no dots, so there is no + // other thing it could be. + let blank = crate::rules::trace::RuleId::emitter(crate::rules::trace::EmitterRule::WordSpace); + spans.extend( + cells + .iter() + .enumerate() + .filter(|(_, cell)| **cell == 0) + .map(|(index, _)| (blank, index as u32..index as u32 + 1)), + ); + spans +} + +/// Where a record's cells sit in the finished output. +/// +/// The position it was written at wins when the output still carries those +/// cells there and nothing already claims them. Searching is the fallback, and +/// the only option for a record taken against a buffer appended elsewhere. +fn locate( + cells: &[u8], + record: &NonWordAttempt, + cursor: &mut usize, + excluded: &[UebSpan], +) -> Option> { + let len = record.cells.len(); + if let Some(offset) = record.offset + && cells.get(offset..offset + len) == Some(record.cells.as_slice()) + && !excluded + .iter() + .any(|(_, taken)| taken.start < (offset + len) as u32 && (offset as u32) < taken.end) + { + *cursor = offset + len; + return Some(offset as u32..(offset + len) as u32); + } + let base = if excluded.is_empty() { + find_from(cells, &record.cells, *cursor)? + } else { + find_from_outside(cells, &record.cells, *cursor, excluded)? + }; + *cursor = base + len; + Some(base as u32..(base + len) as u32) +} + +fn push_without_indicators(spans: &mut Vec, candidate: UebSpan, indicators: &[UebSpan]) { + let (rule, range) = candidate; + let mut start = range.start; + for (_, indicator) in indicators { + if indicator.end <= start { + continue; + } + if indicator.start >= range.end { + break; + } + if start < indicator.start { + spans.push((rule, start..indicator.start)); + } + start = start.max(indicator.end); + } + if start < range.end { + spans.push((rule, start..range.end)); + } +} + +fn find_from(haystack: &[u8], needle: &[u8], from: usize) -> Option { + if needle.is_empty() || from + needle.len() > haystack.len() { + return None; + } + haystack[from..] + .windows(needle.len()) + .position(|window| window == needle) + .map(|offset| offset + from) +} + +fn find_from_outside( + haystack: &[u8], + needle: &[u8], + from: usize, + excluded: &[UebSpan], +) -> Option { + let mut cursor = from; + loop { + let base = find_from(haystack, needle, cursor)?; + let end = base + needle.len(); + let overlap = excluded + .iter() + .find(|(_, range)| range.start < end as u32 && (base as u32) < range.end); + let Some((_, range)) = overlap else { + return Some(base); + }; + cursor = range.end as usize; + } +} + +/// Sources of a selected contraction move that are not [`ContractionRule`] +/// objects. They occupy the first slots of the UEB id space so a contraction +/// rule's id stays a fixed offset from its registration index. +/// +/// [`ContractionRule`]: contraction::ContractionRule +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum UebMoveSource { + Shortform = 0, + Anglicised = 1, + Letter = 2, + AlphabeticWordsign = 3, + StrongWordsign = 4, + LowerWordsign = 5, + Numeric = 6, + Symbol = 7, + Grade1Indicator = 8, + CapitalLetterIndicator = 9, + CapitalisedWordIndicator = 10, + InlineNemethCode = 11, +} + +/// Number of non-rule slots reserved before the contraction rules. +pub(crate) const UEB_RESERVED_SLOTS: usize = 12; + +/// Record a whole word that a lookup table resolved in one step, bypassing the +/// contraction search. Without this a wordsign or shortform would leave its +/// cells unexplained even though its section is known exactly. +/// How many attempts have been recorded so far, so a caller can tell whether the +/// encoder it just ran attributed its own output. +pub(super) fn attempt_count() -> usize { + ATTRIBUTIONS.with(|slot| { + slot.borrow().as_ref().map_or(0, |records| { + records + .iter() + .filter(|record| matches!(record, AttributionRecord::Word(_))) + .count() + }) + }) +} + +/// A word whose attribution has not been settled yet: where its cells start in +/// the output, and how many attempts existed before it ran. +pub(super) type PendingWord = (usize, usize); + +/// Attribute a finished word that nothing else claimed. +/// +/// A word normally names itself through the contraction search or a wordsign +/// lookup. The branches that simply spell it out — letters after a digit, an +/// acronym abutting one — reach neither, and this leaves their cells explained +/// as §4.1 letters without double-counting the words that did claim themselves. +pub(super) fn settle_word_attribution(pending: Option, out: &[u8]) { + let Some((start, attempts_before)) = pending else { + return; + }; + if attempt_count() == attempts_before && out.len() > start { + record_whole_word_at(UebMoveSource::Letter, &out[start..], Some(start)); + } +} + +pub(super) fn settle_symbol_attribution(start: Option, out: &[u8]) { + if let Some(start) = start + && out.len() > start + { + record_whole_word_at(UebMoveSource::Symbol, &out[start..], Some(start)); + } +} + +pub(super) fn record_whole_word(source: UebMoveSource, cells: &[u8]) { + record_whole_word_at(source, cells, None); +} + +pub(super) fn record_whole_word_at(source: UebMoveSource, cells: &[u8], offset: Option) { + let mut attempt = AttemptRecorder::new(); + attempt.push( + crate::rules::trace::RuleId::ueb(source as usize), + 0, + cells.len(), + ); + attempt.finish_at(cells, offset); +} + +pub(super) fn push_indicator(out: &mut Vec, source: UebMoveSource, cells: &[u8]) { + let offset = Some(out.len()); + out.extend_from_slice(cells); + push_record(source, cells, offset, AttributionRecord::Indicator); +} + +pub(super) fn push_direct(out: &mut Vec, source: UebMoveSource, cells: &[u8]) { + let offset = Some(out.len()); + out.extend_from_slice(cells); + push_record(source, cells, offset, AttributionRecord::Direct); +} + +/// [`push_direct`] for a buffer that is appended into the output later, where +/// the position here would not be the position the cells end up at. +pub(super) fn push_direct_unplaced(out: &mut Vec, source: UebMoveSource, cells: &[u8]) { + out.extend_from_slice(cells); + push_record(source, cells, None, AttributionRecord::Direct); +} + +pub(super) fn record_direct(source: UebMoveSource, cells: &[u8], offset: usize) { + push_record(source, cells, Some(offset), AttributionRecord::Direct); +} + +fn push_record( + source: UebMoveSource, + cells: &[u8], + offset: Option, + wrap: fn(NonWordAttempt) -> AttributionRecord, +) { + ATTRIBUTIONS.with(|slot| { + if let Ok(mut slot) = slot.try_borrow_mut() + && let Some(records) = slot.as_mut() + { + records.push(wrap(NonWordAttempt { + cells: cells.to_vec(), + rule: crate::rules::trace::RuleId::ueb(source as usize), + offset, + })); + } + }); +} + +static UEB_NON_RULE_METAS: [crate::rules::RuleMeta; UEB_RESERVED_SLOTS] = [ + crate::rules::RuleMeta { + section: "10.9", + subsection: None, + name: "ueb_shortform", + standard_ref: "UEB 2024 §10.9", + description: "Shortform standing for a longer word", + }, + crate::rules::RuleMeta { + section: "13.2", + subsection: Some("3"), + name: "ueb_anglicised_contraction", + standard_ref: "UEB 2024 §13.2.3", + description: "Contraction in an anglicised or borrowed word", + }, + crate::rules::RuleMeta { + section: "4.1", + subsection: None, + name: "ueb_letter", + standard_ref: "UEB 2024 §4.1 / §4.2", + description: "Uncontracted letter, with an accent indicator where needed", + }, + crate::rules::RuleMeta { + section: "10.1", + subsection: None, + name: "ueb_alphabetic_wordsign", + standard_ref: "UEB 2024 §10.1", + description: "Single letter standing for a whole word", + }, + crate::rules::RuleMeta { + section: "10.2", + subsection: None, + name: "ueb_strong_wordsign", + standard_ref: "UEB 2024 §10.2", + description: "Strong groupsign cell standing for a whole word", + }, + crate::rules::RuleMeta { + section: "10.5", + subsection: None, + name: "ueb_lower_wordsign", + standard_ref: "UEB 2024 §10.5", + description: "Lower-cell sign standing for a whole word", + }, + crate::rules::RuleMeta { + section: "6", + subsection: None, + name: "ueb_numeric", + standard_ref: "UEB 2024 §6", + description: "Numeric indicator and the digits that follow it", + }, + crate::rules::RuleMeta { + section: "3", + subsection: None, + name: "ueb_symbol", + standard_ref: "UEB 2024 §3", + description: "General symbol such as percent, ampersand or asterisk", + }, + crate::rules::RuleMeta { + section: "5", + subsection: None, + name: "ueb_grade1_indicator", + standard_ref: "RUEB 2024 §5", + description: "Grade-1 indicator establishing grade-1 mode", + }, + crate::rules::RuleMeta { + section: "8.3", + subsection: None, + name: "ueb_capital_letter_indicator", + standard_ref: "RUEB 2024 §8.3", + description: "Capital indicator applying to the following letter", + }, + crate::rules::RuleMeta { + section: "8.4", + subsection: None, + name: "ueb_capitalised_word_indicator", + standard_ref: "RUEB 2024 §8.4", + description: "Capital indicators applying to the following word", + }, + crate::rules::RuleMeta { + section: "14.6.2", + subsection: None, + name: "ueb_inline_nemeth_code", + standard_ref: "RUEB 2024 §14.6.2", + description: "Nemeth Code within UEB text", + }, +]; + +/// Metadata of every UEB move source, in [`crate::rules::trace::RuleId`] order: +/// the reserved non-rule slots first, then the contraction rules. +pub(crate) fn ueb_rule_registry() -> Vec<&'static crate::rules::RuleMeta> { + let mut metas: Vec<&'static crate::rules::RuleMeta> = UEB_NON_RULE_METAS.iter().collect(); + metas.extend(EnglishUebEngine::new().contraction_rule_metas()); + metas +} + /// Attempt to encode `text` as standalone UEB Grade-2. Returns `None` if the /// input is empty or contains a construct the engine does not yet support, so /// the caller can fall back to the legacy encoding path. @@ -106,6 +633,7 @@ fn encode_english(text: &str, explicit_english: bool) -> Option> { // §14.3.1/14.3.2: non-UEB (Arabic/Greek/IPA/music) runs inside English prose // take the non-UEB word/passage indicators, with the surrounding English // encoded by the closure. Returns None when no code-switch span is present. + let attribution_checkpoint = attribution_checkpoint(); if let Some(cells) = rule_14::encode_with_code_switches(&composed, |segment| { let tokens = parser::parse_english(segment); if tokens.is_empty() { @@ -116,6 +644,7 @@ fn encode_english(text: &str, explicit_english: bool) -> Option> { }) { return Some(cells); } + rollback_attributions(attribution_checkpoint); let tokens = parser::parse_english(&composed); if tokens.is_empty() { return None; @@ -136,6 +665,14 @@ fn encode_struck_ligature_text(text: &str) -> Option> { return None; } let chars: Vec = text.chars().collect(); + // §9.5: a run of three or more struck letters shows strikeout is a typeform + // in this text, so struck letters are not §4.3 ligatures. + if chars + .windows(6) + .any(|w| w[1] == '\u{0336}' && w[3] == '\u{0336}' && w[5] == '\u{0336}') + { + return None; + } let mut out = Vec::new(); let mut i = 0; while i < chars.len() { @@ -177,11 +714,8 @@ fn encode_single_caron_word(text: &str, _explicit_english: bool) -> Option = text.chars().collect(); + EnglishUebEngine::new().encode_modified(&chars) } fn is_ligature_letter(c: char) -> bool { @@ -731,6 +1265,77 @@ mod is_ueb_eligible_tests { mod encode_pipeline_tests { use super::encode_forced; + #[rstest::rstest] + #[case::two_assignments("$P_{D}$=1,000kN, $P_{L}$=600kN", 12)] + #[case::preceded_by_prose("abc $P_{D}$", 6)] + #[case::followed_by_prose("$I_{7}H^{T}$(mod 2)", 12)] + #[case::ethylene_equation("$C_{2}H_{4}$(g)+$H_{2}O$(g)→$C_{2}H_{5}OH$(g)", 36)] + #[case::carbon_monoxide_equation("$CO$(g)+$H_{2}O$(g)→$CO_{2}$(g)+$H_{2}$(g)", 27)] + #[case::capitalised_spelled_word_after_span("abc $P_{D}$ Cat", 6)] + fn inline_technical_cells_are_claimed_by_14_6_2( + #[case] input: &str, + #[case] expected_technical_cells: usize, + ) { + let (cells, trace) = + crate::encode_with_trace(input).expect("inline technical input must encode"); + let untraced = crate::encode(input).expect("inline technical input must encode untraced"); + let technical_cells: usize = trace + .events() + .iter() + .filter(|event| { + event + .rule + .meta() + .is_some_and(|meta| meta.section == "14.6.2") + }) + .map(|event| event.output.len()) + .sum(); + let mut claims = vec![0u8; cells.len()]; + for event in trace.events() { + for index in event.output.clone() { + claims[index as usize] += 1; + } + } + + assert_eq!(cells, untraced, "trace collection must not change output"); + assert!( + claims.iter().all(|count| *count == 1), + "claims={claims:?}, events={:?}", + trace.events() + ); + assert_eq!( + technical_cells, + expected_technical_cells, + "events={:?}", + trace.events() + ); + } + + /// §8.8.2 gives a two-letter chemical symbol its capitals one at a time + /// (`CCl`, `HCl`). Those indicators and letters are written straight into + /// the output, so each must claim the cell it wrote. + #[rstest::rstest] + #[case::two_letter_symbols("SO2, CCl4, HCl, $SF_{6}$")] + #[case::camel_subunit_word("aMgO")] + #[case::camel_caps_word("dCO")] + #[case::balanced_equation("aMgO(s)$+$bC(s)→cMg(s)$+$dCO(g)$+$eCO2(g)")] + #[case::repeated_subscript_markup("CO2, SO2, CO2")] + #[case::word_repeated_later_in_the_line("CO$+$H2O↔CO2$+$H2")] + #[case::lone_capitals_between_inline_spans("A $1s^{2}2s^{2}2p^{5}$, B $1s^{2}2s^{2}2p^{2}$")] + fn every_cell_of_a_chemical_line_names_a_rule(#[case] input: &str) { + let (cells, trace) = crate::encode_with_trace(input).expect("input must encode"); + let untraced = crate::encode(input).expect("input must encode untraced"); + + assert_eq!(cells, untraced, "trace collection must not change output"); + assert_eq!( + trace.unattributed_cells(), + 0, + "{} of {} cells name no rule", + trace.unattributed_cells(), + cells.len() + ); + } + /// An input that parses to zero tokens — the empty string, reached through /// the eligibility-free `encode_forced` entry — yields None rather than an /// empty cell vector. @@ -739,3 +1344,28 @@ mod encode_pipeline_tests { assert_eq!(encode_forced(""), None); } } + +#[cfg(test)] +mod indicator_clipping_tests { + use super::push_without_indicators; + use crate::rules::trace::{EmitterRule, RuleId}; + + /// An indicator can land inside the cells a rule produced. The cells before + /// it still belong to that rule, so they are recorded as their own span + /// instead of being surrendered along with the indicator. + #[test] + fn a_span_interrupted_by_an_indicator_keeps_the_part_before_it() { + let rule = RuleId::emitter(EmitterRule::WordSpace); + let mut spans = Vec::new(); + + push_without_indicators(&mut spans, (rule, 0..6), &[(rule, 2..4)]); + + assert_eq!( + spans + .iter() + .map(|(_, range)| range.clone()) + .collect::>(), + vec![0..2, 4..6] + ); + } +} diff --git a/libs/braillify/src/rules/english_ueb/parser.rs b/libs/braillify/src/rules/english_ueb/parser.rs index 7cf6d4f5..32f99f67 100644 --- a/libs/braillify/src/rules/english_ueb/parser.rs +++ b/libs/braillify/src/rules/english_ueb/parser.rs @@ -117,10 +117,89 @@ fn dollar_span_is_technical(span: &[char]) -> bool { }) || span.iter().any(|c| c.is_ascii_digit()) } +/// §9.5: a word whose every letter carries a dot below (`ṃục̣ḥ`) is a +/// transcriber-defined typeform, not accented letters (Yoruba `tọrọ` marks only +/// some vowels). NFC composes such letters, so spell the dot out again. +fn dot_below_words(chars: Vec) -> Vec { + use unicode_normalization::UnicodeNormalization; + let base_of = |c: char| -> Option { + let mut parts = std::iter::once(c).nfd(); + let base = parts.next()?; + (parts.next() == Some('\u{0323}') && parts.next().is_none() && base.is_ascii_alphabetic()) + .then_some(base) + }; + let mut out = Vec::with_capacity(chars.len()); + let mut i = 0; + while i < chars.len() { + let mut j = i; + let mut letters = Vec::new(); + while j < chars.len() { + if let Some(base) = base_of(chars[j]) { + letters.push(base); + j += 1; + } else if chars[j].is_ascii_alphabetic() && chars.get(j + 1) == Some(&'\u{0323}') { + letters.push(chars[j]); + j += 2; + } else { + break; + } + } + let whole_word = letters.len() >= 2 && !chars.get(j).is_some_and(|c| c.is_alphabetic()); + if whole_word { + for base in letters { + out.extend([base, '\u{0323}']); + } + i = j; + } else { + out.push(chars[i]); + i += 1; + } + } + out +} + +/// §9.1.3: small capitals that open a paragraph (`Oɴ Tᴜᴇꜱᴅᴀʏ, …`) are a print +/// convention, so they are read as ordinary lowercase letters. +fn plain_paragraph_opening(mut chars: Vec) -> Vec { + let mut line_start = 0; + while line_start < chars.len() { + let mut k = line_start; + let mut saw_small_cap = false; + while k < chars.len() && chars[k] != '\n' { + let c = chars[k]; + if super::rule_9::decode_small_cap(c).is_some() { + saw_small_cap = true; + } else if !(c.is_ascii_uppercase() || c == ' ') { + break; + } + k += 1; + } + let line_end = chars[k..] + .iter() + .position(|c| *c == '\n') + .map_or(chars.len(), |p| k + p); + let opens_prose = chars.get(line_start).is_some_and(char::is_ascii_uppercase) + && chars.get(k).is_some_and(|c| c.is_ascii_punctuation()) + && chars[k..line_end].iter().any(char::is_ascii_lowercase); + if saw_small_cap && opens_prose { + for c in &mut chars[line_start..k] { + if let Some(cap) = super::rule_9::decode_small_cap(*c) { + *c = cap.to_ascii_lowercase(); + } + } + } + line_start = chars[line_start..] + .iter() + .position(|c| *c == '\n') + .map_or(chars.len(), |p| line_start + p + 1); + } + chars +} + /// Tokenize `text`: runs of word letters become `Word`, runs of ASCII digits /// become `Number`, a single space becomes `Space`, anything else `Symbol`. pub fn parse_english(text: &str) -> Vec { - let chars: Vec = text.chars().collect(); + let chars = plain_paragraph_opening(dot_below_words(text.chars().collect())); let mut tokens = Vec::new(); let mut i = 0; while i < chars.len() { diff --git a/libs/braillify/src/rules/english_ueb/rule_10_11.rs b/libs/braillify/src/rules/english_ueb/rule_10_11.rs index 32d5565e..297bede8 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_11.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_11.rs @@ -60,7 +60,19 @@ fn is_bridging_digraph(a: char, b: char) -> bool { /// splits its two letters) is left to spell out. pub struct BridgeAwareStrongGroupsignRule; +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.11", + subsection: None, + name: "ueb_bridge_aware_strong_groupsign", + standard_ref: "UEB 2024 §10.11", + description: "Strong groupsign that must not bridge a compound boundary", +}; + impl ContractionRule for BridgeAwareStrongGroupsignRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { let m = StrongGroupsignRule.try_match(word, pos)?; if m.consumed == 2 diff --git a/libs/braillify/src/rules/english_ueb/rule_10_3.rs b/libs/braillify/src/rules/english_ueb/rule_10_3.rs index 796364d6..b0080a51 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_3.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_3.rs @@ -25,7 +25,19 @@ pub fn is_strong_contraction_word(word: &str) -> bool { /// §10.3 strong contraction rule. pub struct StrongContractionRule; +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.3", + subsection: None, + name: "ueb_strong_contraction", + standard_ref: "UEB 2024 §10.3", + description: "Strong contractions: and, for, of, the, with", +}; + impl ContractionRule for StrongContractionRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { match_longest(word, pos, &STRONG, 50) } diff --git a/libs/braillify/src/rules/english_ueb/rule_10_6_8.rs b/libs/braillify/src/rules/english_ueb/rule_10_6_8.rs index 67e88521..1f8b24f8 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_6_8.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_6_8.rs @@ -29,21 +29,48 @@ impl EnInBeforeNessRule { /// Whether the shared `n` onsets a `ness` syllable — the whole word is /// pronounced with a final `N S`, every recorded variant agreeing. - /// An unknown word yields `false` (keep `en`/`in`). - fn n_onsets_ness(&self, word: &[char]) -> bool { + /// An unknown word takes the `ness` suffix when the letters before it form a + /// word, directly or with a final `i` for `y` (`immediate·ness`, + /// `musti·ness` ← musty); `captai·ness` has no such stem and keeps `in`. + fn n_onsets_ness(&self, word: &[char], pos: usize) -> bool { let spelling: String = word.iter().collect(); let prons = self.provider.pronunciations(&spelling); - !prons.is_empty() && prons.iter().all(|p| ends_n_vowel_s(p)) + if prons.is_empty() { + return self.is_ness_stem(&word[..=pos]); + } + prons.iter().all(|p| ends_n_vowel_s(p)) + } + + fn is_ness_stem(&self, stem: &[char]) -> bool { + if stem.len() < 3 { + return false; + } + let known = |letters: String| !self.provider.pronunciations(&letters).is_empty(); + known(stem.iter().collect()) + || (stem.last() == Some(&'i') + && known(stem[..stem.len() - 1].iter().chain(['y'].iter()).collect())) } } +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.6", + subsection: Some("8"), + name: "ueb_en_in_before_ness", + standard_ref: "UEB 2024 §10.6.8", + description: "en/in kept or dropped where they overlap a final ness", +}; + impl ContractionRule for EnInBeforeNessRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { let mut m = LowerGroupsignRule.try_match(word, pos)?; // §10.6.8: where `en`/`in` overlaps a following `ness` at the shared `n`, // pronunciation decides which is kept. if ness_overlaps(word, pos) { - if self.n_onsets_ness(word) { + if self.n_onsets_ness(word, pos) { // The `n` onsets the `ness` syllable (`busi·ness`, `fi·ness·e`) → // suppress `en`/`in` so the cheaper `ness` final groupsign wins. return None; @@ -86,6 +113,8 @@ mod tests { #[case::finesse("finesse", 1)] #[case::happiness("happiness", 4)] #[case::friendliness("friendliness", 7)] + #[case::unknown_mustiness("mustiness", 4)] + #[case::unknown_immediateness("immediateness", 8)] fn suppresses_en_in_before_onset_ness(#[case] word: &str, #[case] pos: usize) { let chars: Vec = word.chars().collect(); assert!(rule().try_match(&chars, pos).is_none()); @@ -93,7 +122,7 @@ mod tests { /// `en`/`in` is kept (→ `Some`) when the `n` is a coda — the base ends in /n/ /// (`citi·zen·ess`, `captain·ess`); these base+`ess` words are absent from - /// CMUdict, so KEEP by default. + /// CMUdict, and their base up to the `n` is a word. #[rstest::rstest] #[case::citizeness("citizeness", 5)] #[case::captainess("captainess", 5)] diff --git a/libs/braillify/src/rules/english_ueb/rule_10_6_middle.rs b/libs/braillify/src/rules/english_ueb/rule_10_6_middle.rs index c59f1627..6b331aff 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_6_middle.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_6_middle.rs @@ -134,7 +134,25 @@ impl MiddleLowerGroupsignRule { if !EA_BRIDGING_PREFIXES.contains(&prefix.as_str()) { return false; } - right.len() >= 4 && self.is_word(right) + right.len() >= 4 && self.is_word(right) && self.prefix_is_a_syllable(word, right) + } + + /// A prefix is its own syllable: the word sounds one more vowel than the root + /// it would be split from (`re·action` 3 against `action` 2). A root that is + /// spelled with `ea` sounds no extra vowel (`reader` 2 against `ader` 2). An + /// unrecorded word keeps the prefix reading. + fn prefix_is_a_syllable(&self, word: &[char], right: &[char]) -> bool { + let fewest_vowels = |chars: &[char]| { + self.provider + .pronunciations(&collect(chars)) + .iter() + .map(|pron| pron.iter().filter(|phoneme| phoneme.is_vowel()).count()) + .min() + }; + match (fewest_vowels(word), fewest_vowels(right)) { + (Some(whole), Some(root)) => whole > root, + _ => true, + } } /// Additional solid-compound seams recover obvious free-word compounds not in @@ -291,7 +309,19 @@ fn bridges_compound_seam(word: &[char], pos: usize, consumed: usize) -> bool { .any(|&seam| pos < seam && seam < pos + consumed) } +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.6", + subsection: Some("5"), + name: "ueb_middle_lower_groupsign", + standard_ref: "UEB 2024 §10.6.5", + description: "Middle lower groupsigns ea bb cc ff gg, morpheme-gated", +}; + impl ContractionRule for MiddleLowerGroupsignRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { let m = middle_lower_groupsign(word, pos)?; // `middle_lower_groupsign` only matches `ea` or a doubled letter diff --git a/libs/braillify/src/rules/english_ueb/rule_10_6_restricted.rs b/libs/braillify/src/rules/english_ueb/rule_10_6_restricted.rs index 4132b6dc..74877afe 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_6_restricted.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_6_restricted.rs @@ -24,7 +24,19 @@ impl RestrictedLowerGroupsignRule { } } +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.6", + subsection: Some("restricted"), + name: "ueb_restricted_lower_groupsign", + standard_ref: "UEB 2024 §10.6", + description: "Restricted lower groupsigns be, con, dis", +}; + impl ContractionRule for RestrictedLowerGroupsignRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { // Restricted groupsigns are word-initial only (§10.6.2). if pos != 0 { @@ -39,7 +51,22 @@ impl ContractionRule for RestrictedLowerGroupsignRule { } else { return None; }; - match classify(word, prefix, self.provider.as_ref()) { + let decision = match classify(word, prefix, self.provider.as_ref()) { + // A derivative missing from the dictionary shares its stem's first + // syllable (`belittle·ment` like `belittle`). + Decision::Unknown => suffix_stem(word) + .filter(|stem| { + !self + .provider + .pronunciations(&stem.iter().collect::()) + .is_empty() + }) + .map_or(Decision::Unknown, |stem| { + classify(stem, prefix, self.provider.as_ref()) + }), + decision => decision, + }; + match decision { Decision::Use => Some(ContractionMatch { cells: vec![cell], consumed, @@ -55,6 +82,13 @@ impl ContractionRule for RestrictedLowerGroupsignRule { } } +fn suffix_stem(word: &[char]) -> Option<&[char]> { + ["ment", "ness", "ly"].iter().find_map(|suffix| { + let cut = word.len().checked_sub(suffix.len())?; + (cut >= 5 && word[cut..].iter().copied().eq(suffix.chars())).then(|| &word[..cut]) + }) +} + #[cfg(test)] mod tests { use super::super::pronunciation::cmudict::CmuDictProvider; @@ -76,6 +110,8 @@ mod tests { #[case::concept("concept", Some((decode_unicode('⠒'), 3)))] #[case::dislike_rest_word("dislike", Some((decode_unicode('⠲'), 3)))] #[case::dishonest_rest_word("dishonest", Some((decode_unicode('⠲'), 3)))] + #[case::unknown_derivative_of_a_known_stem("belittlement", Some((decode_unicode('⠆'), 2)))] + #[case::unknown_name_without_a_word_stem("beagans", None)] #[case::beckon("beckon", None)] #[case::cone("cone", None)] #[case::dispirited("dispirited", None)] diff --git a/libs/braillify/src/rules/english_ueb/rule_10_7.rs b/libs/braillify/src/rules/english_ueb/rule_10_7.rs index db77e3ae..4a964580 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_7.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_7.rs @@ -88,7 +88,19 @@ pub fn is_initial_letter_contraction_word(word: &str) -> bool { /// §10.7 initial-letter contraction rule. pub struct InitialContractionRule; +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.7", + subsection: None, + name: "ueb_initial_contraction", + standard_ref: "UEB 2024 §10.7", + description: "Initial-letter contractions standing for whole words", +}; + impl ContractionRule for InitialContractionRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { let mut best: Option<(usize, [u8; 2])> = None; for (key, &cells) in INITIAL_CONTRACTIONS.entries() { diff --git a/libs/braillify/src/rules/english_ueb/rule_10_7_pron.rs b/libs/braillify/src/rules/english_ueb/rule_10_7_pron.rs index 68829a28..8a24750c 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_7_pron.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_7_pron.rs @@ -264,7 +264,19 @@ impl InitialContractionPronunciationRule { } } +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.7", + subsection: Some("pronunciation"), + name: "ueb_initial_contraction_pronunciation", + standard_ref: "UEB 2024 §10.7", + description: "Initial-letter contractions gated by pronunciation", +}; + impl ContractionRule for InitialContractionPronunciationRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { let full: String = word.iter().collect(); let mut best: Option<(usize, [u8; 2])> = None; @@ -286,10 +298,14 @@ impl ContractionRule for InitialContractionPronunciationRule { { continue; } + // Appendix 1 lists compounds of a shortform word and another word + // (`between·time`, `herein·after`), so their components are whole words. + let listed_compound = super::rule_10_9_list::is_listed(&full); if *key == "time" && pos > 0 && word[pos - 1] == 'n' && !matches!(word.get(..pos), Some(['u', 'n'])) + && !listed_compound { continue; } @@ -329,6 +345,7 @@ impl ContractionRule for InitialContractionPronunciationRule { let danger = key.ends_with('e') && word.get(end).is_some_and(|c| matches!(c, 'r' | 'd')); let accept = if (*key == "had" && pos == 0 && !matches!(word.get(3), Some('e' | 'r'))) + || (listed_compound && pos == 0 && matches!(*key, "here" | "there" | "where")) || (*key == "day" && (end == word.len() || word diff --git a/libs/braillify/src/rules/english_ueb/rule_10_7_struct.rs b/libs/braillify/src/rules/english_ueb/rule_10_7_struct.rs index d6bdf90f..4efc7046 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_7_struct.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_7_struct.rs @@ -93,7 +93,19 @@ impl StructuralInitialContractionRule { } } +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.7", + subsection: Some("structure"), + name: "ueb_initial_contraction_structural", + standard_ref: "UEB 2024 §10.7", + description: "Initial-letter contractions gated by morpheme structure", +}; + impl ContractionRule for StructuralInitialContractionRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { for (key, &cells) in STRUCT_CONTRACTIONS.entries() { let klen = key.chars().count(); diff --git a/libs/braillify/src/rules/english_ueb/rule_10_8.rs b/libs/braillify/src/rules/english_ueb/rule_10_8.rs index 1c523c74..0ce69dad 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_8.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_8.rs @@ -35,12 +35,15 @@ pub(crate) fn final_groupsign_cells(cluster: &str) -> Option<[u8; 2]> { } /// RUEB 2024 §10.8.3 printed exceptions for the `ity` final-letter groupsign. +/// `hoity-toity` reaches here as its two hyphen-separated halves. fn ity_exception(word: &[char]) -> bool { matches!( word, ['b', 'i', 's', 'c', 'u', 'i', 't', 'y'] | ['d', 'a', 'c', 'o', 'i', 't', 'y'] | ['f', 'r', 'u', 'i', 't', 'y'] + | ['h', 'o', 'i', 't', 'y'] + | ['t', 'o', 'i', 't', 'y'] | ['p', 'i', 't', 'y', 'a', 'r', 'd'] | ['r', 'a', 'b', 'b', 'i', 't', 'y'] ) @@ -53,7 +56,19 @@ fn ness_exception(word: &[char]) -> bool { /// §10.8 final-letter groupsign rule. pub struct FinalGroupsignRule; +static META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "10.8", + subsection: None, + name: "ueb_final_groupsign", + standard_ref: "UEB 2024 §10.8", + description: "Final-letter groupsigns for word-final letter clusters", +}; + impl ContractionRule for FinalGroupsignRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META + } + fn try_match(&self, word: &[char], pos: usize) -> Option { // §10.8: never used at the start of a word. if pos == 0 { @@ -93,6 +108,8 @@ mod tests { #[case::ount_mid("amount", 2, Some((vec![decode_unicode('⠨'), decode_unicode('⠞')], 4)))] #[case::ness_final("baroness", 4, Some((vec![decode_unicode('⠰'), decode_unicode('⠎')], 4)))] #[case::ity_final("circuity", 5, Some((vec![decode_unicode('⠰'), decode_unicode('⠽')], 3)))] + #[case::hoity_exception("hoity", 2, None)] + #[case::toity_exception("toity", 2, None)] #[case::no_match_at_start("tion", 0, None)] #[case::no_cluster("cat", 1, None)] fn matches_final_groupsigns( diff --git a/libs/braillify/src/rules/english_ueb/rule_10_9.rs b/libs/braillify/src/rules/english_ueb/rule_10_9.rs index 214aa22f..ae9f29c0 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_9.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_9.rs @@ -7,9 +7,11 @@ use phf::phf_map; +use super::UebMoveSource; use super::contraction::{ContractionEngine, ContractionMatch}; use super::rule_10_13::WordDivision; use crate::english::encode_english; +use crate::rules::trace::RuleId; use crate::unicode::decode_unicode; static SHORTFORMS: phf::Map<&'static str, &'static str> = phf_map! { @@ -137,7 +139,10 @@ fn rule_10_9_3_reading_exists(shortform: &str, suffix: &[char]) -> bool { /// and would consequently miss cell-equivalent sequences such as `fst` (`f` + /// the `st` groupsign) and `shd` (the `sh` groupsign + `d`). fn korean_letter_sequence_cells(letters: &[char]) -> Vec { - super::span::encode_korean_word( + // A collision test only compares cells. Left recorded, its attempts are + // matched against a later identical word in the output (`CO … CO`). + let checkpoint = super::attribution_checkpoint(); + let cells = super::span::encode_korean_word( letters, true, // capitalization indicators are compared separately false, // do not recursively prepend grade 1 false, // rule 37 suppresses whole-word signs on Roman entry @@ -148,7 +153,9 @@ fn korean_letter_sequence_cells(letters: &[char]) -> Vec { false, // not split by an apostrophe false, // lowercase, so never a §10.12.1 initialism ) - .expect("a lowercase ASCII letters-sequence must be encodable") + .expect("a lowercase ASCII letters-sequence must be encodable"); + super::rollback_attributions(checkpoint); + cells } /// Encode a word as the §10.10.2 cell-minimising contraction sequence. @@ -355,12 +362,12 @@ fn encode_with_constraints( // lower-preference groupsign that overlaps its start here (`en`, 70): // `re·name·d`, not `r·en·amed`; `mis·time·d`, not `mis·st·imed`. let mut path_priority = vec![u16::MAX; n + 1]; - let mut back: Vec, usize)>> = vec![None; n + 1]; + let mut back: Vec, usize, RuleId)>> = vec![None; n + 1]; cost[n] = 0; for pos in (0..n).rev() { - // Best candidate so far: (total cells, path priority, consumed, cells). - let mut best: Option<(usize, u16, usize, Vec)> = None; - for (cells, consumed, priority) in candidate_moves( + // Best candidate so far: (total cells, path priority, consumed, cells, source). + let mut best: Option<(usize, u16, usize, Vec, RuleId)> = None; + for (cells, consumed, priority, source) in candidate_moves( word, pos, contractions, @@ -379,31 +386,36 @@ fn encode_with_constraints( // The preference of the whole remaining path: the best contraction in // this move or anything the tail already chose. let this_priority = priority.min(path_priority[next]); - let better = best.as_ref().is_none_or(|(bt, bp, bc, _)| { + let better = best.as_ref().is_none_or(|(bt, bp, bc, _, _)| { total < *bt || (total == *bt && this_priority < *bp) || (total == *bt && this_priority == *bp && consumed > *bc) }); if better { - best = Some((total, this_priority, consumed, cells)); + best = Some((total, this_priority, consumed, cells, source)); } } - let (total, pp, consumed, cells) = best?; + let (total, pp, consumed, cells, source) = best?; cost[pos] = total; path_priority[pos] = pp; - back[pos] = Some((cells, consumed)); + back[pos] = Some((cells, consumed, source)); } - // Reconstruct the chosen sequence from the start. + // Reconstruct the chosen sequence from the start. Only the moves on this + // path produced output; the candidates the DP rejected did not, so this walk + // is the only place a contraction may be credited for a cell. let mut out = Vec::with_capacity(cost[0]); + let mut attempt = super::AttemptRecorder::new(); let mut pos = 0; while pos < n { - let (cells, consumed) = back[pos].as_ref()?; + let (cells, consumed, source) = back[pos].as_ref()?; + attempt.push(*source, out.len(), cells.len()); out.extend(cells.iter().copied()); pos += consumed; if division.is_some_and(|d| pos == d.index) { super::rule_10_13::append_break(&mut out, true); } } + attempt.finish(&out); Some(out) } @@ -428,7 +440,7 @@ fn candidate_moves( allow_longer_shortforms: bool, relax_shortforms: bool, suppress_whole_word_wordsign: bool, -) -> Vec<(Vec, usize, u16)> { +) -> Vec<(Vec, usize, u16, RuleId)> { let mut moves = Vec::new(); // §10.9 longer-word shortform placement (preferred on a cost tie → priority 0). if allow_longer_shortforms { @@ -442,14 +454,14 @@ fn candidate_moves( if let Some((len, cells)) = longer && division.is_none_or(|d| !d.blocks_span(pos, len)) { - moves.push((cells, len, 0)); + moves.push((cells, len, 0, source_id(UebMoveSource::Shortform))); } } if relax_shortforms && let Some((cells, len)) = anglicised_initial_contraction(word, pos) { - moves.push((cells, len, 55)); + moves.push((cells, len, 55, source_id(UebMoveSource::Anglicised))); } let protected_here = inside_protected[pos]; - for m in contractions.matches_at(word, pos) { + for (rule_index, m) in contractions.matches_at_indexed(word, pos) { // Korean rule 37: immediately after the Roman indicator, a lower // wordsign is written with alphabet/multi-letter groupsigns instead. // Reject only a contraction consuming the complete wordsign; inner @@ -565,19 +577,27 @@ fn candidate_moves( }) { continue; } - moves.push((m.cells, m.consumed, m.priority)); + moves.push((m.cells, m.consumed, m.priority, rule_id(rule_index))); } // §4.2 accent / §4.1 single letter — always available so the DP never stalls. if let Some(cells) = super::rule_12::early_letter(word[pos]) { - moves.push((cells, 1, u16::MAX)); + moves.push((cells, 1, u16::MAX, source_id(UebMoveSource::Letter))); } else if let Some(cells) = super::rule_4::accent_cells(word[pos]) { - moves.push((cells, 1, u16::MAX)); + moves.push((cells, 1, u16::MAX, source_id(UebMoveSource::Letter))); } else if let Ok(cell) = encode_english(word[pos]) { - moves.push((vec![cell], 1, u16::MAX)); + moves.push((vec![cell], 1, u16::MAX, source_id(UebMoveSource::Letter))); } moves } +fn source_id(source: super::UebMoveSource) -> RuleId { + RuleId::ueb(source as usize) +} + +fn rule_id(rule_index: usize) -> RuleId { + RuleId::ueb(super::UEB_RESERVED_SLOTS + rule_index) +} + /// §13.2.3 anglicised words may use ordinary UEB contractions even when CMUdict /// has no entry for the borrowed/proper word. Initial-letter contractions whose /// English phonology gate cannot fire for an unrecorded word are safe when the @@ -1001,7 +1021,7 @@ mod tests { false, false, ); - assert!(moves.iter().all(|(cells, consumed, _)| { + assert!(moves.iter().all(|(cells, consumed, _, _)| { *consumed != pattern.len() || cells != &vec![decode_unicode('⠆')] })); } @@ -1131,7 +1151,7 @@ mod tests { false, ); - assert!(moves.iter().any(|(cells, consumed, priority)| { + assert!(moves.iter().any(|(cells, consumed, priority, _)| { *cells == vec![decode_unicode('⠵')] && *consumed == 1 && *priority == u16::MAX })); } @@ -1324,7 +1344,7 @@ mod tests { false, false, ); - assert!(moves.iter().any(|(cells_, consumed, priority)| { + assert!(moves.iter().any(|(cells_, consumed, priority, _)| { *cells_ == cells("⠼⠮") && *consumed == 1 && *priority == u16::MAX })); } diff --git a/libs/braillify/src/rules/english_ueb/rule_10_9_list.rs b/libs/braillify/src/rules/english_ueb/rule_10_9_list.rs index d9daaa64..0a8d1744 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_9_list.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_9_list.rs @@ -76,8 +76,21 @@ static APPENDIX_MIXED_CASE_LONGER_WORDS: Set<&'static str> = phf_set! { "deafblind", }; +/// A listed word written with an apostrophe or hyphen (`couldn't`, `'twould`, +/// `do-it-yourselfer`) reaches the shortform rules one piece at a time. +fn is_listed_piece(word: &str) -> bool { + APPENDIX_LONGER_WORDS.iter().any(|listed| { + listed.split(['\'', '-']).count() > 1 + && listed.split(['\'', '-']).any(|piece| piece == word) + }) +} + +pub fn is_listed(word: &str) -> bool { + APPENDIX_LONGER_WORDS.contains(word) +} + pub fn listed_or_added_s(word: &str) -> bool { - if APPENDIX_LONGER_WORDS.contains(word) { + if APPENDIX_LONGER_WORDS.contains(word) || is_listed_piece(word) { return true; } if matches!(word, "abouts" | "almosts" | "hims") { @@ -101,6 +114,9 @@ mod tests { #[case::listed_base("afterburn", true)] #[case::listed_plural("afterburns", true)] #[case::listed_longer_variant("afterburner", true)] + #[case::piece_before_apostrophe("couldn", true)] + #[case::piece_after_apostrophe("twould", true)] + #[case::piece_after_hyphen("yourselfer", true)] #[case::blocked_abouts("abouts", false)] #[case::unlisted_plural("zzzzs", false)] #[case::unlisted("zzzz", false)] diff --git a/libs/braillify/src/rules/english_ueb/rule_11.rs b/libs/braillify/src/rules/english_ueb/rule_11.rs index df1df478..dbbd129e 100644 --- a/libs/braillify/src/rules/english_ueb/rule_11.rs +++ b/libs/braillify/src/rules/english_ueb/rule_11.rs @@ -440,8 +440,24 @@ pub fn encode_technical(chars: &[char]) -> Option> { return Some(cells); } if chars.windows(5).any(|w| w == ['\\', 's', 'q', 'r', 't']) { - let mut out = vec![GRADE1, GRADE1]; - encode_expr_with_options(chars, &mut out, false, true)?; + // §11.5.1: over a number the radical needs only the grade 1 symbol + // indicator before ⠩ — the numeric indicator carries grade 1 mode through + // ⠬ (`√9 = 3` → ⠰⠩⠼⠊⠬ ⠐⠶ ⠼⠉). A letter, or an index whose superscript + // indicator is a second grade 1 symbol (§11.5.2 `∛8` → ⠰⠰⠩⠔⠼⠉⠼⠓⠬), + // takes the grade 1 word indicator (§5.3). + let text: String = chars.iter().collect(); + let mut out = Vec::new(); + if text.contains("\\sqrt[") + || text + .replace("\\sqrt", "") + .chars() + .any(|c| c.is_ascii_alphabetic()) + { + out.extend([GRADE1, GRADE1]); + encode_expr_with_options(chars, &mut out, false, true)?; + } else { + encode_expr_with_options(chars, &mut out, false, false)?; + } return Some(out); } if chars.contains(&' ') { diff --git a/libs/braillify/src/rules/english_ueb/rule_12.rs b/libs/braillify/src/rules/english_ueb/rule_12.rs index f1f56f94..30d9244b 100644 --- a/libs/braillify/src/rules/english_ueb/rule_12.rs +++ b/libs/braillify/src/rules/english_ueb/rule_12.rs @@ -129,8 +129,11 @@ pub fn encode_uncontracted_word(chars: &[char]) -> Option> { out.extend(cells("⠠⠠")); } for &c in chars { - let is_early_capital = - c.is_uppercase() && early_letter(c.to_lowercase().next().unwrap_or(c)).is_some(); + // A lone capital is the §12.2 letter itself (`Ȝ` → ⠠⠼⠽); only inside a + // word does the Wyclif example drop the indicator. + let is_early_capital = chars.len() > 1 + && c.is_uppercase() + && early_letter(c.to_lowercase().next().unwrap_or(c)).is_some(); if !word_caps && c.is_uppercase() && !is_early_capital { out.push(decode_unicode('⠠')); } @@ -189,6 +192,7 @@ mod tests { #[case::breve_cyrillic_y(&['ў'], Some("⠽"))] #[case::all_caps_ascii_word(&['A', 'L'], Some("⠠⠠⠁⠇"))] #[case::early_capital_omits_cap_indicator(&['Ȝ', 'e', 'e'], Some("⠼⠽⠑⠑"))] + #[case::lone_early_capital_keeps_cap_indicator(&['Ȝ'], Some("⠠⠼⠽"))] #[case::unknown_letter(&['🙂'], None)] fn uncontracted_word_paths(#[case] chars: &[char], #[case] expected: Option<&str>) { let expected_cells = expected.map(|s| s.chars().map(decode_unicode).collect()); diff --git a/libs/braillify/src/rules/english_ueb/rule_13.rs b/libs/braillify/src/rules/english_ueb/rule_13.rs index 7db5dcb5..c63cf756 100644 --- a/libs/braillify/src/rules/english_ueb/rule_13.rs +++ b/libs/braillify/src/rules/english_ueb/rule_13.rs @@ -314,7 +314,8 @@ pub fn encode_uncontracted_word( ) -> Option> { let mut out = Vec::new(); for &c in chars { - if c.is_uppercase() && !is_foreign_letter(c) { + let capital_pushed = c.is_uppercase() && !is_foreign_letter(c); + if capital_pushed { out.push(decode_unicode('⠠')); } let lower = c.to_lowercase().next()?; @@ -330,7 +331,9 @@ pub fn encode_uncontracted_word( } AccentCode::Ueb => { if let Some(cells) = rule_4::accent_cells(c) { - out.extend(cells); + let skip = + usize::from(capital_pushed && cells.first() == Some(&decode_unicode('⠠'))); + out.extend(&cells[skip..]); } else if let Some(cells) = rule_12::early_letter(c) { out.extend(cells); } else { diff --git a/libs/braillify/src/rules/english_ueb/rule_14.rs b/libs/braillify/src/rules/english_ueb/rule_14.rs index a60fd7f4..860f1856 100644 --- a/libs/braillify/src/rules/english_ueb/rule_14.rs +++ b/libs/braillify/src/rules/english_ueb/rule_14.rs @@ -229,6 +229,27 @@ fn has_nemeth_span(input: &str) -> bool { false } +/// The switch indicators and the maths they wrap are all §14.6.2 output, so +/// they are recorded as they are appended rather than left for a later pass to +/// guess at. +fn push_nemeth(out: &mut Vec, cells: &[u8]) { + super::push_direct_unplaced(out, super::UebMoveSource::InlineNemethCode, cells); +} + +/// Append prose that was encoded into its own buffer, moving the records it +/// made from that buffer's coordinates onto this one's. +fn extend_prose( + out: &mut Vec, + encode_ueb: &mut impl FnMut(&str) -> Option>, + text: &str, +) -> Option<()> { + let checkpoint = super::attribution_checkpoint(); + let cells = encode_ueb(text)?; + super::rebase_attributions(checkpoint, out.len()); + out.extend(cells); + Some(()) +} + fn encode_nemeth_spans( input: &str, encode_ueb: &mut impl FnMut(&str) -> Option>, @@ -238,37 +259,33 @@ fn encode_nemeth_spans( let mut continued = false; while let Some(start) = rest.find('$') { if continued { - out.extend(encode_ueb(&rest[..start])?); + extend_prose(&mut out, encode_ueb, &rest[..start])?; } else if rest[..start].ends_with('"') { let prefix = &rest[..start - '"'.len_utf8()]; - out.extend(encode_ueb(prefix)?); + extend_prose(&mut out, encode_ueb, prefix)?; out.push(decode_unicode('⠦')); } else { - out.extend(encode_ueb(&rest[..start])?); + extend_prose(&mut out, encode_ueb, &rest[..start])?; } let after = &rest[start + '$'.len_utf8()..]; let end = after.find('$')?; if !continued { - out.extend(cells("⠸⠩⠀")); + push_nemeth(&mut out, &cells("⠸⠩⠀")); } - out.extend(encode_nemeth_math(&after[..end])?); + push_nemeth(&mut out, &encode_nemeth_math(&after[..end])?); let tail = &after[end + '$'.len_utf8()..]; if tail.starts_with(", $") { - out.extend(cells("⠠⠀")); + push_nemeth(&mut out, &cells("⠠⠀")); rest = &tail[", ".len()..]; continued = true; } else { - out.extend(cells("⠀⠸⠱")); + push_nemeth(&mut out, &cells("⠀⠸⠱")); rest = tail; continued = false; } } - if !rest.is_empty() { - if let Some(cells) = encode_ueb(rest) { - out.extend(cells); - } else { - out.extend(encode_simple_ueb_symbols(rest)?); - } + if !rest.is_empty() && extend_prose(&mut out, encode_ueb, rest).is_none() { + out.extend(encode_simple_ueb_symbols(rest)?); } Some(out) } @@ -1135,15 +1152,25 @@ fn is_omicron_ypsilon(word: &[char], i: usize) -> bool { matches!(word.get(i), Some('ο' | 'ὀ')) && matches!(word.get(i + 1), Some('υ')) } -const fn is_ipa_char(c: char) -> bool { +/// IPA has no capitals; a capital in print (`SCHWA /Ə/` in an all-capitals +/// heading) is ornamentation of the lowercase symbol (§2.3.2). +fn ipa_symbol(c: char) -> char { + if c.is_ascii() { + c + } else { + c.to_lowercase().next().unwrap_or(c) + } +} + +fn is_ipa_char(c: char) -> bool { matches!( - c, + ipa_symbol(c), 'ː' | 'ə' | 'ɔ' | 'ˈ' | 'ˌ' | 'ɹ' | 'θ' | 'ɪ' | 'ð' | 'ɾ' | 'ŋ' | 'ʃ' | 'č' ) } fn ipa_cell(c: char) -> Option> { - let cells = match c { + let cells = match ipa_symbol(c) { 'ː' => "⠒", 'ə' => "⠢", 'ɔ' => "⠣", diff --git a/libs/braillify/src/rules/english_ueb/rule_15.rs b/libs/braillify/src/rules/english_ueb/rule_15.rs index 6c12b5e7..01e76cbd 100644 --- a/libs/braillify/src/rules/english_ueb/rule_15.rs +++ b/libs/braillify/src/rules/english_ueb/rule_15.rs @@ -23,6 +23,10 @@ pub fn encode_symbol(c: char) -> Option> { 'ˌ' => cells("⠘⠨⠆"), '″' => cells("⠘⠨⠃"), 'ə' => cells("⠸⠢"), + // §15.3 level tones: the high, mid and low tone letters. + '˦' => cells("⠘⠨⠉"), + '˧' => cells("⠘⠨⠒"), + '˨' => cells("⠘⠨⠤"), '➘' => cells("⠘⠨⠴"), // §15.3.2 example uses `↗` for low rising in prose (⠘⠨⠔). `ˊ` (modifier // acute) is the high-rising tone letter (⠘⠨⠊) — the two arrows share the @@ -50,6 +54,9 @@ mod tests { #[case::line('|', "⠸⠳")] #[case::double_line('‖', "⠸⠳⠸⠳")] #[case::scansion_solidus('/', "⠸⠌")] + #[case::tone_high('˦', "⠘⠨⠉")] + #[case::tone_mid('˧', "⠘⠨⠒")] + #[case::tone_low('˨', "⠘⠨⠤")] #[case::tone_down('↓', "⠘⠨⠮")] #[case::tone_fall('➘', "⠘⠨⠴")] #[case::tone_low_rising('↗', "⠘⠨⠔")] diff --git a/libs/braillify/src/rules/english_ueb/rule_3.rs b/libs/braillify/src/rules/english_ueb/rule_3.rs index 3e16cee8..7febed7f 100644 --- a/libs/braillify/src/rules/english_ueb/rule_3.rs +++ b/libs/braillify/src/rules/english_ueb/rule_3.rs @@ -29,6 +29,8 @@ pub fn encode_symbol(c: char) -> Option> { '\u{2212}' => vec![decode_unicode('⠐'), decode_unicode('⠤')], // − minus sign '<' => vec![decode_unicode('⠈'), decode_unicode('⠣')], '>' => vec![decode_unicode('⠈'), decode_unicode('⠜')], + '\u{2236}' => vec![decode_unicode('⠒')], // ∶ ratio + '\u{2237}' => vec![decode_unicode('⠒'), decode_unicode('⠒')], // ∷ proportion '⟨' | '〈' => vec![decode_unicode('⠈'), decode_unicode('⠣')], // §3.17 angle bracket less-than shape '⟩' | '〉' => vec![decode_unicode('⠈'), decode_unicode('⠜')], // §3.17 angle bracket greater-than shape '\u{00F7}' => vec![decode_unicode('⠐'), decode_unicode('⠌')], // ÷ division @@ -176,6 +178,8 @@ mod tests { #[case::minus('\u{2212}', vec![decode_unicode('⠐'), decode_unicode('⠤')])] #[case::less_than('<', vec![decode_unicode('⠈'), decode_unicode('⠣')])] #[case::greater_than('>', vec![decode_unicode('⠈'), decode_unicode('⠜')])] + #[case::ratio('\u{2236}', vec![decode_unicode('⠒')])] + #[case::proportion('\u{2237}', vec![decode_unicode('⠒'), decode_unicode('⠒')])] #[case::division('\u{00F7}', vec![decode_unicode('⠐'), decode_unicode('⠌')])] #[case::multiplication('\u{00D7}', vec![decode_unicode('⠐'), decode_unicode('⠦')])] #[case::tilde('~', vec![decode_unicode('⠈'), decode_unicode('⠔')])] diff --git a/libs/braillify/src/rules/english_ueb/rule_4.rs b/libs/braillify/src/rules/english_ueb/rule_4.rs index 6124262b..fc9dd027 100644 --- a/libs/braillify/src/rules/english_ueb/rule_4.rs +++ b/libs/braillify/src/rules/english_ueb/rule_4.rs @@ -120,6 +120,22 @@ fn eszett_cells(c: char) -> Option> { }) } +/// §4.4 eng and schwa: `ŋ` → ⠘⠝ and `ə` → ⠸⠢, the capital forms carrying the §8 +/// capital indicator (`Ŋ` → ⠠⠘⠝, `Ə` → ⠠⠸⠢). +pub fn eng_schwa_cells(c: char) -> Option> { + let (prefix, letter) = match c.to_lowercase().next()? { + 'ŋ' => ('⠘', '⠝'), + 'ə' => ('⠸', '⠢'), + _ => return None, + }; + let mut cells = Vec::with_capacity(3); + if c.is_uppercase() { + cells.push(decode_unicode('⠠')); + } + cells.extend([decode_unicode(prefix), decode_unicode(letter)]); + Some(cells) +} + /// Whether `c` is a supported accented or ligatured letter (so the parser keeps /// it in a word). pub fn is_accented(c: char) -> bool { @@ -213,6 +229,17 @@ mod tests { assert_eq!(accent_cells(c), Some(want)); } + #[rstest::rstest] + #[case::eng('ŋ', Some("⠘⠝"))] + #[case::eng_upper('Ŋ', Some("⠠⠘⠝"))] + #[case::schwa('ə', Some("⠸⠢"))] + #[case::schwa_upper('Ə', Some("⠠⠸⠢"))] + #[case::plain_letter('n', None)] + fn eng_schwa_cells_follow_section_4_4(#[case] c: char, #[case] expected: Option<&str>) { + let want = expected.map(|s| s.chars().map(decode_unicode).collect::>()); + assert_eq!(eng_schwa_cells(c), want); + } + #[test] fn plain_letter_is_not_accented() { assert!(!is_accented('e')); diff --git a/libs/braillify/src/rules/korean/rule_41.rs b/libs/braillify/src/rules/korean/rule_41.rs index aceeb27a..85bf7777 100644 --- a/libs/braillify/src/rules/korean/rule_41.rs +++ b/libs/braillify/src/rules/korean/rule_41.rs @@ -59,7 +59,11 @@ impl BrailleRule for Rule41 { // Comma between numbers, or between ASCII and alphanumeric ((ctx.state.is_number || has_numeric_prefix) && next_is_digit) - || (has_ascii_prefix && next_is_alphanumeric) + || (has_ascii_prefix + && next_is_alphanumeric + && !crate::english_logic::opens_korean_number( + ctx.word_chars[ctx.index + 1..].iter().copied(), + )) } fn apply(&self, ctx: &mut RuleContext) -> Result { @@ -166,6 +170,16 @@ mod tests { assert_eq!(comma_cell, Some(expected_comma)); } + /// 제33항 — 로마자와 한글이 붙은 수 사이의 쉼표는 한글 쉼표다. 로마자 뒤 맨 수는 + /// 제41항대로 로마자 쉼표를 쓴다. + #[rstest::rstest] + #[case::korean_counter_follows("최희섭(KIA,27개)과의", "⠅⠊⠁⠐⠼⠃⠛")] + #[case::bare_number_follows("A,1", "⠁⠂⠼⠁")] + fn a_comma_before_a_korean_number_is_korean(#[case] input: &str, #[case] cells: &str) { + let actual = crate::encode_to_unicode(input).expect("comma must encode"); + assert!(actual.contains(cells), "{actual}"); + } + /// rule_41 line 75 — `j -= 1;` when prev char is a space (continues backward scan). #[test] fn scan_prefix_skips_space_then_finds_digit() { diff --git a/libs/braillify/src/rules/korean/rule_60.rs b/libs/braillify/src/rules/korean/rule_60.rs index 4eac1de0..f3406de8 100644 --- a/libs/braillify/src/rules/korean/rule_60.rs +++ b/libs/braillify/src/rules/korean/rule_60.rs @@ -1,10 +1,5 @@ -//! 제60항 — 별표(*)는 앞뒤를 한 칸씩 띄어 쓴다. -//! -//! Asterisks require surrounding spaces. When the asterisk is a standalone word, -//! spaces are added before and after. The inter-word spacing mechanism handles -//! most cases, but explicit spacing is needed at word boundaries. -//! -//! Reference: 2024 Korean Braille Standard, Chapter 6, Section 13, Article 60 +//! 제60항 — 별표(*)는 앞뒤를 한 칸씩 띄어 쓴다. 홀로 선 별표 앞뒤의 한 칸은 +//! 묵자의 어절 사이 빈칸이다. use crate::char_struct::CharType; use crate::rules::RuleMeta; @@ -20,12 +15,6 @@ pub static META: RuleMeta = RuleMeta { description: "Asterisk (*) requires surrounding spaces", }; -/// Plugin struct for the rule engine. -/// -/// Handles asterisk encoding with spacing. -/// When the asterisk is the first and only character in a word, and there's -/// a previous word, insert a space before it. The asterisk symbol encoding -/// is then emitted normally. pub struct Rule60; impl BrailleRule for Rule60 { @@ -46,10 +35,6 @@ impl BrailleRule for Rule60 { } fn apply(&self, ctx: &mut RuleContext) -> Result { - // 제60항: asterisk as standalone word with previous word → prepend space - if ctx.index == 0 && ctx.word_len() == 1 && !ctx.prev_word.is_empty() { - ctx.emit(0); // Space before asterisk - } let encoded = symbol_shortcut::encode_char_symbol_shortcut('*')?; ctx.emit_slice(encoded); Ok(RuleResult::Consumed) @@ -74,16 +59,13 @@ mod tests { // Just exercise apply() for coverage } - /// 제60항 — 별표(*)가 단독 어절로 직전 단어가 있을 때 앞에 공백 0을 emit - /// (line 50-52). - #[test] - fn rule60_apply_standalone_asterisk_after_word_prepends_space() { - let mut owned = crate::test_helpers::CtxOwned::for_text("*", false).with_prev_word("가"); - let mut ctx = owned.ctx_at(0); - let outcome = Rule60.apply(&mut ctx).unwrap(); - assert!(matches!(outcome, RuleResult::Consumed)); - // Space (0) is the first emitted byte - assert_eq!(owned.result[0], 0); - assert!(owned.result.len() > 1); + /// 제60항 — 앞뒤 한 칸씩. 묵자의 빈칸에 한 칸을 더 얹지 않는다. + #[rstest::rstest] + #[case::between_words("가나 * 다라", "⠫⠉⠀⠐⠔⠀⠊⠐⠣")] + #[case::after_a_number("1만7천원 * 0.2", "⠒⠀⠐⠔⠀⠼⠚")] + #[case::after_an_exclamation("들! * 가", "⠖⠀⠐⠔⠀⠫")] + fn a_standalone_asterisk_takes_one_blank_each_side(#[case] input: &str, #[case] cells: &str) { + let encoded = crate::encode_to_unicode(input).unwrap(); + assert!(encoded.contains(cells), "{encoded}"); } } diff --git a/libs/braillify/src/rules/korean/rule_64.rs b/libs/braillify/src/rules/korean/rule_64.rs index d33b1aa4..ae0f9c9e 100644 --- a/libs/braillify/src/rules/korean/rule_64.rs +++ b/libs/braillify/src/rules/korean/rule_64.rs @@ -37,6 +37,7 @@ pub static META_SQUARE: RuleMeta = RuleMeta { const CIRCLE: u8 = 54; // ⠶ const LETTER_MARKER: u8 = 52; // ⠴ +const CAPITAL_MARKER: u8 = 32; // ⠠ const NUMBER_MARKER: u8 = 60; // ⠼ /// Open marker for square enclosing: ⠸⠦ (cells 56, 38) @@ -81,7 +82,7 @@ const CIRCLED_JAMO: &[(char, char)] = &[ ]; pub fn is_enclosed_symbol(c: char) -> bool { - matches!(c, '①'..='⑳' | 'ⓐ'..='ⓩ') + matches!(c, '①'..='⑳' | 'ⓐ'..='ⓩ' | 'Ⓐ'..='Ⓩ') || CIRCLED_SYLLABLES.iter().any(|(enclosed, _)| *enclosed == c) || CIRCLED_JAMO.iter().any(|(enclosed, _)| *enclosed == c) } @@ -174,6 +175,16 @@ pub fn encode_enclosed_symbol(c: char) -> Result, String> { ])); } + if ('Ⓐ'..='Ⓩ').contains(&c) { + let letter = char::from_u32((c as u32) - ('Ⓐ' as u32) + ('a' as u32)) + .ok_or_else(|| "Invalid enclosed latin letter".to_string())?; + return Ok(wrap_circle(vec![ + LETTER_MARKER, + CAPITAL_MARKER, + english::encode_english(letter)?, + ])); + } + if let Some((_, syllable)) = CIRCLED_SYLLABLES .iter() .find(|(enclosed, _)| *enclosed == c) @@ -294,6 +305,21 @@ mod tests { assert_eq!(to_unicode(&encode_enclosed_symbol('ⓐ').unwrap()), "⠶⠴⠁⠶"); } + /// 제64항 shows only the lowercase ⓐ as 70a7, leaving the order of the two + /// indicators for a capital undetermined. 국립국어원 settled it on + /// 2026-09-21: the roman sign comes first, then the capital sign, then the + /// letter — 7 0 , a 7. + #[rstest::rstest] + #[case::first('Ⓐ', "⠶⠴⠠⠁⠶")] + #[case::last('Ⓩ', "⠶⠴⠠⠵⠶")] + fn encodes_circled_capital(#[case] symbol: char, #[case] expected: &str) { + assert!(is_enclosed_symbol(symbol)); + assert_eq!( + to_unicode(&encode_enclosed_symbol(symbol).unwrap()), + expected + ); + } + #[test] fn encodes_circled_syllable() { assert_eq!(to_unicode(&encode_enclosed_symbol('㉮').unwrap()), "⠶⠫⠶"); diff --git a/libs/braillify/src/rules/korean/rule_68.rs b/libs/braillify/src/rules/korean/rule_68.rs index fe64769a..9544a28c 100644 --- a/libs/braillify/src/rules/korean/rule_68.rs +++ b/libs/braillify/src/rules/korean/rule_68.rs @@ -64,6 +64,19 @@ fn is_superscript_symbol(c: char) -> bool { matches!(c, '⁺' | '⁻') } +pub(crate) fn is_superscript_digit(c: char) -> bool { + matches!(c, '⁰' | '¹' | '²' | '³' | '⁴'..='⁹') +} + +/// 위 첨자가 이어진 자리인가. 제68항은 위 첨자 기호 ⠘ 뒤에 첨자의 내용을 적으므로 +/// 이어진 첨자(`⁻¹`, `²³`)는 ⠘ 하나 뒤에 적는다 — 수학 제18항 `x⁻¹` = ⠭⠘⠔⠼⠁. +pub(crate) fn continues_superscript(word: &[char], index: usize) -> bool { + index + .checked_sub(1) + .and_then(|previous| word.get(previous)) + .is_some_and(|c| is_superscript_symbol(*c) || is_superscript_digit(*c)) +} + fn is_subscript_digit(c: char) -> bool { matches!(c, '₀'..='₉') } @@ -230,6 +243,13 @@ impl BrailleRule for Rule68 { return Ok(RuleResult::Skip); }; let is_roman_unit = matches!(ctx.current_char(), '㎡' | '㏊'); + // 제69항 [붙임 3] — 빗금으로 이어진 로마자 단위(`kgf/㎡`)는 한 로마자 구간이다. + if is_roman_unit + && super::rule_69::roman_unit_chain_continues_before(ctx) + && encoded.first() == Some(&ROMAN_INDICATOR) + { + encoded.remove(0); + } let continues = is_roman_unit && super::rule_69::adjust_roman_unit_boundary(ctx, ctx.index + 1, &mut encoded); ctx.emit_slice(&encoded); @@ -243,8 +263,9 @@ impl BrailleRule for Rule68 { } } -/// PDF — `1++등급` 같은 digit + 연속 `+` 등급 표기 패턴인지 검사. -/// 직전이 digit이고 현재가 `+`이며 이후에 한글 등급 키워드(등급)가 나오면 true. +/// 제68항 [붙임 2] — `1++등급` 의 `+` 는 등급을 나타내는 위 첨자다. 수 뒤에 붙은 +/// `+`·`-` 는 그 뒤에 `등급` 이 이어질 때만 첨자이고, 그 밖의 한글 앞(`50+캠퍼스`, +/// `4.1+실업률`)에서는 덧셈표다. fn is_digit_grade_plus_notation(word: &[char], index: usize) -> bool { if index == 0 { return false; @@ -261,9 +282,7 @@ fn is_digit_grade_plus_notation(word: &[char], index: usize) -> bool { break; } } - // 직후에 한글이 와야 grade context로 본다 (`1++등급` 등). - word.get(cursor) - .is_some_and(|c| crate::utils::is_korean_char(*c)) + word[cursor..].starts_with(&['등', '급']) } #[cfg(test)] @@ -294,6 +313,13 @@ mod tests { assert!(!is_rule_68_symbol('1')); } + #[rstest::rstest] + #[case::negative_exponents("cm³g⁻¹sec⁻²", "⠴⠉⠍⠘⠼⠉⠛⠘⠔⠼⠁⠎⠑⠉⠘⠔⠼⠃")] + #[case::two_digit_exponent("cm³g⁻¹⁰", "⠴⠉⠍⠘⠼⠉⠛⠘⠔⠼⠁⠚")] + fn writes_a_superscript_run_after_one_sign(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + #[test] fn is_superscript_symbol_plus_minus_only() { assert!(is_superscript_symbol('⁺')); @@ -410,22 +436,21 @@ mod tests { assert!(result.is_none()); } - #[test] - fn is_digit_grade_plus_notation_paths() { - // "1+등급" - let word: Vec = "1+등급".chars().collect(); - assert!(is_digit_grade_plus_notation(&word, 1)); - // "1++등급" - let word: Vec = "1++등급".chars().collect(); - assert!(is_digit_grade_plus_notation(&word, 1)); - // Index 0 - not preceded by digit - assert!(!is_digit_grade_plus_notation(&word, 0)); - // Without Korean following - let word: Vec = "1++x".chars().collect(); - assert!(!is_digit_grade_plus_notation(&word, 1)); - // Not preceded by digit - let word: Vec = "a+등급".chars().collect(); - assert!(!is_digit_grade_plus_notation(&word, 1)); + #[rstest::rstest] + #[case::one_plus_grade("1+등급", 1, true)] + #[case::double_plus_grade("1++등급", 1, true)] + #[case::at_the_start("1++등급", 0, false)] + #[case::before_a_letter("1++x", 1, false)] + #[case::after_a_letter("a+등급", 1, false)] + #[case::fifty_plus_campus("50+캠퍼스", 2, false)] + #[case::sum_of_rates("4.1+실업률", 3, false)] + fn a_plus_after_a_number_is_a_script_only_before_a_grade( + #[case] text: &str, + #[case] index: usize, + #[case] expected: bool, + ) { + let word: Vec = text.chars().collect(); + assert_eq!(is_digit_grade_plus_notation(&word, index), expected); } #[test] @@ -548,28 +573,19 @@ mod tests { assert!(should_insert_separator_after_symbol(&ctx)); } - /// rule_68:198 — digit-grade chain with `-` triggers the `'-' => ⠔` arm. - /// Input MUST have a Korean char after the +/- chain to satisfy - /// is_digit_grade_plus_notation (per the function source). #[test] - fn rule68_digit_grade_with_minus_in_chain() { - // 1+-가 — digit, then +, -, then Korean → satisfies notation predicate. - let _ = crate::encode("1+-가"); - let _ = crate::encode("5-+나"); - let _ = crate::encode("3--다"); + fn a_grade_script_writes_plus_and_minus_in_order() { + assert_eq!(crate::encode_to_unicode("1+-등급").unwrap(), "⠼⠁⠘⠢⠔⠀⠊⠪⠶⠈⠪⠃"); } - /// rule_68:108 — direct call to `encode_compact_ascii_notation` with a base - /// letter followed by a single ⁺/⁻ then a non-super char. The inner loop - /// breaks at line 108 when next char is neither ⁺ nor ⁻. - #[test] - fn rule68_superscript_block_breaks_on_non_super_direct() { - // "A⁺x" — uppercase A + ⁺ (consumed) + x (breaks loop) - let word: Vec = "A\u{207A}x".chars().collect(); - let result = encode_compact_ascii_notation(&word, 0, false).unwrap(); - assert!(result.is_some()); - let (_, consumed) = result.unwrap(); - // Only A and ⁺ are consumed; x triggers the break at line 108. + #[rstest::rstest] + #[case::superscript_then_letter("A\u{207A}x")] + #[case::subscript_then_letter("H\u{2082}O")] + fn compact_notation_stops_at_the_first_non_script(#[case] text: &str) { + let word: Vec = text.chars().collect(); + let (_, consumed) = encode_compact_ascii_notation(&word, 0, false) + .unwrap() + .expect("a capital with a script is compact notation"); assert_eq!(consumed, 2); } diff --git a/libs/braillify/src/rules/korean/rule_69.rs b/libs/braillify/src/rules/korean/rule_69.rs index 01e7a6bd..c3caa87f 100644 --- a/libs/braillify/src/rules/korean/rule_69.rs +++ b/libs/braillify/src/rules/korean/rule_69.rs @@ -52,7 +52,7 @@ const PDF_ASCII_UNIT_SYMBOLS: &[&str] = &[ /// SI prefixes are case-sensitive. This set is used only as the grammar for a /// complete measured-unit suffix; it never reclassifies a separated Roman word. -const SI_PREFIXES: &[&str] = &[ +pub(crate) const SI_PREFIXES: &[&str] = &[ "q", "r", "y", "z", "a", "f", "p", "n", "u", "m", "c", "d", "da", "h", "k", "M", "G", "T", "P", "E", "Z", "Y", "R", "Q", ]; @@ -187,10 +187,10 @@ fn encode_compatibility_unit( } fn is_roman_unit_component(ch: char) -> bool { - ch.is_ascii_alphabetic() || ch == 'μ' || compatibility_unit_decomposition(ch).is_some() + ch.is_ascii_alphabetic() || ch == 'μ' || is_compatibility_unit_presentation(ch) } -fn roman_unit_chain_continues_before(ctx: &RuleContext) -> bool { +pub(crate) fn roman_unit_chain_continues_before(ctx: &RuleContext) -> bool { ctx.index >= 2 && ctx.word_chars.get(ctx.index - 1) == Some(&'/') && ctx @@ -593,12 +593,22 @@ pub(crate) fn parse_numeric_ascii_unit_expression(word: &[char]) -> Option) { /// Rule 35 keeps a Roman unit directly following a number in the already-open /// Roman section. The number temporarily places the emitter in /// `roman_number_chain`; in that state the unit's self-contained Rule-69 entry -/// marker would be a duplicate. +/// marker would be a duplicate. UEB 6.5.2: a lowercase a–j right after the +/// number would read as a digit, so it takes the grade-1 sign ⠰ instead +/// (`GS 450h`). fn omit_unit_entry_in_open_roman_number_chain(encoded: &mut Vec, state: &EncoderState) { if state.roman_number_chain && encoded.first() == Some(&ROMAN_INDICATOR) { - encoded.remove(0); + let reads_as_digit = encoded + .get(1) + .is_some_and(|cell| matches!(*cell, 1 | 3 | 9 | 25 | 17 | 11 | 27 | 19 | 10 | 26)); + if reads_as_digit { + encoded[0] = ENGLISH_CONTINUATION; + } else { + encoded.remove(0); + } } } @@ -792,7 +811,11 @@ impl BrailleRule for Rule69 { encode_complete_numeric_ascii_unit(ctx.word_chars, ctx.index) { let continues = adjust_roman_unit_boundary(ctx, ctx.index + consumed, &mut encoded); - if roman_unit_chain_continues_before(ctx) && encoded.first() == Some(&ROMAN_INDICATOR) { + // 제35항 `MP4 Player`: 앞 어절에서 이어진 로마자 구간(`FM 98.1 MHz`)의 단위는 + // 새 로마자표 없이 잇는다. + let continues_section = roman_unit_chain_continues_before(ctx) + || (ctx.index == 0 && ctx.roman_section_continues_from_previous_word); + if continues_section && encoded.first() == Some(&ROMAN_INDICATOR) { encoded.remove(0); } trim_recent_english_indicator(ctx.result); @@ -1125,6 +1148,11 @@ mod tests { #[case::middle_dot_missing_right_unit("3kg·4", 0)] #[case::middle_dot_numeric_list("54·55·56", 0)] #[case::unknown_ascii_suffix("3.5~8.5models", 0)] + #[case::hyphen_range("100-130mm", 9)] + #[case::hyphen_after_unit("3kg-4", 3)] + #[case::hyphen_before_letter("100-mm", 0)] + #[case::middle_dot_product_unit("36.0kgf·m", 9)] + #[case::middle_dot_before_non_unit("5N·x", 0)] fn recognizes_only_complete_numeric_unit_expressions( #[case] input: &str, #[case] expected_consumed: usize, @@ -1139,6 +1167,10 @@ mod tests { #[rstest::rstest] #[case::ascii_range("범위는 3.5~8.5m이다", "⠼⠉⠲⠑⠈⠔⠼⠓⠲⠑⠴⠍⠲")] #[case::unicode_range("범위는 3∼5kg이다", "⠼⠉⠈⠔⠼⠑⠴⠅⠛⠲")] + #[case::unit_continuing_a_roman_section("라디오(FM 98.1 MHz) 김", "⠼⠊⠓⠲⠁⠀⠠⠍⠠⠓⠵")] + #[case::unit_after_a_korean_number("가 98.1 MHz 나", "⠼⠊⠓⠲⠁⠀⠴⠠⠍")] + #[case::digit_like_unit_after_a_number("가 GS 450h 나", "⠼⠙⠑⠚⠰⠓")] + #[case::digit_like_unit_before_a_bracket("가 ES 300h(하이브리드)를", "⠼⠉⠚⠚⠰⠓⠦⠄")] fn numeric_unit_ranges_stay_on_korean_number_and_unit_rules( #[case] input: &str, #[case] expected_segment: &str, @@ -1406,6 +1438,7 @@ mod tests { #[rstest::rstest] #[case::milligram_per_decilitre("160㎎/㎗", "⠼⠁⠋⠚⠴⠍⠛⠸⠌⠙⠇⠲")] #[case::calorie_per_square_centimetre_per_minute("cal/㎠/min", "⠴⠉⠁⠇⠸⠌⠉⠍⠘⠼⠃⠸⠌⠍⠔⠲")] + #[case::kilogram_force_per_square_metre("kgf/㎡이", "⠴⠅⠛⠋⠸⠌⠍⠘⠼⠃⠕")] #[case::megahertz("96.7 ㎒", "⠼⠊⠋⠲⠛⠀⠴⠠⠍⠠⠓⠵⠲")] #[case::kilometres_per_hour("80 ㎞/시", "⠼⠓⠚⠀⠴⠅⠍⠲⠸⠌⠠⠕")] fn preserves_pdf_unit_examples(#[case] input: &str, #[case] expected: &str) { diff --git a/libs/braillify/src/rules/korean/rule_72.rs b/libs/braillify/src/rules/korean/rule_72.rs index 96fd8aec..21280e66 100644 --- a/libs/braillify/src/rules/korean/rule_72.rs +++ b/libs/braillify/src/rules/korean/rule_72.rs @@ -116,6 +116,10 @@ fn owned_word(text: String) -> Token<'static> { pub struct Rule72AttachedMarkerTokenRule; impl TokenRule for Rule72AttachedMarkerTokenRule { + fn meta(&self) -> &'static RuleMeta { + &META + } + fn phase(&self) -> TokenPhase { TokenPhase::Normalization } @@ -203,11 +207,14 @@ impl BrailleRule for Rule72 { // 실제로 이어지는지만 판정한다. let tight_before_content = matches!(current, '△' | '▲' | '▴') && ctx.next_char().is_some_and(|c| !c.is_whitespace()); + // 칠한 세모는 제49항 사물부호가 아니므로 어절 끝에 붙어도(`씨▲ 도의상`) 같은 표지다. + let closes_word = matches!(current, '▲' | '▴') && ctx.next_char().is_none(); let contextual_marker = ctx.word_len() == 1 || ctx .next_char() .is_some_and(|c| c.is_whitespace() || matches!(c, '(' | '\'' | '"')) || tight_before_content + || closes_word || matches!(current, '◎' | '▣'); if !contextual_marker { return Ok(RuleResult::Skip); @@ -277,6 +284,13 @@ mod tests { assert_ne!(cell_after_triangle_marker(input), Some('\u{2800}')); } + #[rstest::rstest] + #[case::filled("씨▲ 도의상")] + #[case::small_filled("씨▴ 도의상")] + fn a_filled_triangle_closing_a_word_is_the_same_marker(#[case] input: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), "⠠⠠⠕⠸⠬⠀⠊⠥⠺⠇⠶"); + } + #[rstest::rstest] #[case::outline("△ 문화")] #[case::roman_item("목록은 △ R&D이다")] diff --git a/libs/braillify/src/rules/korean/rule_english_symbol.rs b/libs/braillify/src/rules/korean/rule_english_symbol.rs index beb7115d..ef027af0 100644 --- a/libs/braillify/src/rules/korean/rule_english_symbol.rs +++ b/libs/braillify/src/rules/korean/rule_english_symbol.rs @@ -14,11 +14,16 @@ use crate::rules::traits::{BrailleRule, Phase, RuleResult}; use crate::symbol_shortcut; use crate::utils; +/// Three articles decide together here, and 국립국어원 answered on 2026-09-21 +/// that such a cell should name them all rather than pick one. 제33항 keeps a +/// comma between Roman and Korean in the Korean shape, 제34항 drops the Roman +/// terminator when brackets or quotes enclose the Roman text, and 제49항 gives +/// the punctuation its cells. pub static META: RuleMeta = RuleMeta { - section: "49", - subsection: Some("eng"), + section: "33, 34, 49", + subsection: None, name: "english_symbol_context", - standard_ref: "2024 Korean Braille Standard, Ch.4 Sec.10 + Ch.6 Sec.13", + standard_ref: "2024 Korean Braille Standard, 한글 제33항·제34항·제49항", description: "English-context punctuation rendering with parenthesis tracking", }; diff --git a/libs/braillify/src/rules/korean/rule_fraction.rs b/libs/braillify/src/rules/korean/rule_fraction.rs index aafc4010..82d2b022 100644 --- a/libs/braillify/src/rules/korean/rule_fraction.rs +++ b/libs/braillify/src/rules/korean/rule_fraction.rs @@ -7,10 +7,10 @@ use crate::rules::context::RuleContext; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; pub static META: RuleMeta = RuleMeta { - section: "fraction", + section: "47", subsection: None, name: "unicode_fraction_encoding", - standard_ref: "2024 Korean Braille Standard (fractions)", + standard_ref: "2024 Korean Braille Standard, 제47항", description: "Unicode fraction characters (½, ⅓, ¼, etc.)", }; @@ -81,7 +81,7 @@ mod tests { fn rule_metadata_reports_phase() { let rule = RuleFraction; - assert_eq!(rule.meta().section, "fraction"); + assert_eq!(rule.meta().section, "47"); assert!(matches!(rule.phase(), Phase::CoreEncoding)); } } diff --git a/libs/braillify/src/rules/korean/rule_korean.rs b/libs/braillify/src/rules/korean/rule_korean.rs index 9c11281f..916d1ac5 100644 --- a/libs/braillify/src/rules/korean/rule_korean.rs +++ b/libs/braillify/src/rules/korean/rule_korean.rs @@ -8,8 +8,8 @@ //! and 13 (single-char abbreviation), serving as the general-purpose fallback //! for Korean syllables that weren't caught by those specialized rules. -use crate::char_struct::CharType; -use crate::korean_char::encode_korean_char; +use crate::char_struct::{CharType, KoreanChar}; +use crate::korean_char::{JamoSpans, encode_korean_char, encode_korean_char_with_spans}; use crate::rules::RuleMeta; use crate::rules::context::RuleContext; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; @@ -50,7 +50,21 @@ impl BrailleRule for RuleKorean { let Some(korean) = ctx.as_korean() else { return Ok(RuleResult::Skip); }; - let encoded = encode_korean_char(korean)?; + if ctx.state.jamo_spans.is_none() { + let encoded = encode_korean_char(korean)?; + ctx.emit_slice(&encoded); + return Ok(RuleResult::Consumed); + } + let korean = KoreanChar { + cho: korean.cho, + jung: korean.jung, + jong: korean.jong, + }; + let mut spans = JamoSpans::default(); + let encoded = encode_korean_char_with_spans(&korean, &mut spans)?; + if let Some(slot) = ctx.state.jamo_spans.as_deref_mut() { + *slot = spans; + } ctx.emit_slice(&encoded); Ok(RuleResult::Consumed) } diff --git a/libs/braillify/src/rules/korean/rule_math.rs b/libs/braillify/src/rules/korean/rule_math.rs index 450aec84..1b6c7fcc 100644 --- a/libs/braillify/src/rules/korean/rule_math.rs +++ b/libs/braillify/src/rules/korean/rule_math.rs @@ -1,4 +1,4 @@ -//! Math symbol encoding with Korean spacing rules. +//! 제46항: 연산 기호와 비교 기호가 한글 사이에 나올 때에는 기호의 앞뒤를 한 칸씩 띄어 쓴다. //! //! Math symbols (+, −, ×, ÷, etc.) need spacing around them when //! adjacent to Korean text, unless the Korean is a grammatical particle (josa). @@ -11,10 +11,10 @@ use crate::rules::traits::{BrailleRule, Phase, RuleResult}; use crate::utils; pub static META: RuleMeta = RuleMeta { - section: "math", + section: "46", subsection: None, name: "math_symbol_encoding", - standard_ref: "2024 Korean Braille Standard (math symbols)", + standard_ref: "2024 Korean Braille Standard, 제46항", description: "Math symbols with Korean spacing rules", }; @@ -188,48 +188,7 @@ fn is_semantic_ascii_minus(ctx: &RuleContext) -> bool { if ctx.current_char() != '-' { return false; } - - let next_starts_number = ctx.next_char().is_some_and(|next| { - next.is_ascii_digit() - || (next == '.' - && ctx - .word_chars - .get(ctx.index + 2) - .is_some_and(char::is_ascii_digit)) - }); - let unary_boundary = ctx.prev_char().is_none_or(|prev| { - matches!( - prev, - '(' | '[' - | '{' - | '〈' - | '《' - | '「' - | '『' - | '【' - | '〔' - | '〖' - | '〘' - | '〚' - | '‘' - | '“' - | '\'' - | '"' - | ',' - | ':' - | ';' - | '=' - | '+' - | '×' - | '÷' - | '<' - | '>' - | '≤' - | '≥' - | '≠' - ) - }); - if next_starts_number && unary_boundary { + if starts_signed_number(ctx) { return true; } @@ -275,7 +234,63 @@ fn is_semantic_ascii_minus(ctx: &RuleContext) -> bool { || matches!(prev, ')' | ']' | '}' | '〉' | '》' | '」' | '』' | '】') }); - prev_ends_operand && next_starts_number && has_other_math_operator + prev_ends_operand && next_number(ctx) && has_other_math_operator +} + +/// Print often sets a minus sign as the horizontal bar `―` (`수익률이 ―43.22%로`). +/// At the start of a number it is the 제45항 뺄셈표, while a bar joining two +/// parts (`3―1로`, `?”―28`) stays the 제49항 줄표. +fn is_dash_minus(ctx: &RuleContext) -> bool { + ctx.current_char() == '\u{2015}' && starts_signed_number(ctx) +} + +fn next_number(ctx: &RuleContext) -> bool { + ctx.next_char().is_some_and(|next| { + next.is_ascii_digit() + || (next == '.' + && ctx + .word_chars + .get(ctx.index + 2) + .is_some_and(char::is_ascii_digit)) + }) +} + +/// A sign at the start of a number: at the start of the word or after an +/// opening mark, separator or operator. +fn starts_signed_number(ctx: &RuleContext) -> bool { + let unary_boundary = ctx.prev_char().is_none_or(|prev| { + matches!( + prev, + '(' | '[' + | '{' + | '〈' + | '《' + | '「' + | '『' + | '【' + | '〔' + | '〖' + | '〘' + | '〚' + | '‘' + | '“' + | '\'' + | '"' + | ',' + | ':' + | ';' + | '=' + | '+' + | '×' + | '÷' + | '<' + | '>' + | '≤' + | '≥' + | '≠' + ) + }); + unary_boundary && next_number(ctx) } fn is_roman_grade_minus(ctx: &RuleContext) -> bool { @@ -300,6 +315,7 @@ impl BrailleRule for RuleMath { matches!(ctx.char_type, CharType::MathSymbol(_)) || (matches!(ctx.char_type, CharType::Symbol('-')) && (is_semantic_ascii_minus(ctx) || is_roman_grade_minus(ctx))) + || (matches!(ctx.char_type, CharType::Symbol('\u{2015}')) && is_dash_minus(ctx)) } fn apply(&self, ctx: &mut RuleContext) -> Result { @@ -313,6 +329,7 @@ impl BrailleRule for RuleMath { let c = match ctx.char_type { CharType::MathSymbol(c) => *c, CharType::Symbol('-') if is_semantic_ascii_minus(ctx) => '\u{2212}', + CharType::Symbol('\u{2015}') if is_dash_minus(ctx) => '\u{2212}', _ => return Ok(RuleResult::Skip), }; @@ -350,7 +367,14 @@ impl BrailleRule for RuleMath { ctx.emit(0); } - let encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(c)?; + let mut encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(c)?; + if super::rule_68::is_superscript_digit(c) + && super::rule_68::continues_superscript(ctx.word_chars, ctx.index) + { + let number_continues = + super::rule_68::is_superscript_digit(ctx.word_chars[ctx.index - 1]); + encoded = &encoded[if number_continues { 2 } else { 1 }..]; + } ctx.emit_slice(encoded); if pad_after { @@ -529,6 +553,31 @@ mod tests { ); } + #[rstest::rstest] + #[case::percentage("수익률이 ―43.22%로")] + #[case::after_opening_quote("‘―5도’")] + #[case::decimal("수익률은 ―1.20%였다")] + fn a_bar_opening_a_number_is_the_minus_sign(#[case] bar: &str) { + assert_eq!( + crate::encode_to_unicode(bar).expect("bar must encode"), + crate::encode_to_unicode(&bar.replace('―', "−")).expect("minus must encode"), + "input={bar}" + ); + } + + #[rstest::rstest] + #[case::score_between_numbers("3―1로")] + #[case::count_after_parenthesis("2(볼)―1(스트라이크)")] + #[case::attribution_after_quote("”―28")] + fn a_bar_joining_two_parts_stays_the_dash(#[case] bar: &str) { + assert!( + crate::encode_to_unicode(bar) + .expect("bar must encode") + .contains("⠤⠤"), + "input={bar}" + ); + } + #[rstest::rstest] #[case::pdf_phone_number("02-799-1000")] #[case::identifier_suffix("A-3")] @@ -541,6 +590,13 @@ mod tests { "input={input}" ); } + + /// 제46항 "연산 기호와 비교 기호가 한글 사이에 나올 때에는 기호의 앞뒤를 한 칸씩 띄어 쓴다" + /// 이 규칙의 메타데이터는 제46항을 명시해야 한다. + #[test] + fn meta_section_is_article_46() { + assert_eq!(META.section, "46", "META.section must be article 46"); + } } #[cfg(test)] @@ -585,6 +641,13 @@ mod roman_grade_minus_coverage { fn a_credit_grade_minus_encodes(#[case] input: &str) { assert!(crate::encode_to_unicode(input).is_ok()); } + + /// 제34항: 등급 뒤 한글 주석이 붙어도 붙임표는 등급의 뺄셈 기호다. + #[test] + fn a_grade_before_a_korean_annotation_keeps_its_minus() { + let encoded = crate::encode_to_unicode("등급을 AA-(안정적)에서").unwrap(); + assert!(encoded.contains("⠁⠁⠐⠤⠦⠄"), "{encoded}"); + } } #[cfg(test)] diff --git a/libs/braillify/src/rules/korean/rule_space.rs b/libs/braillify/src/rules/korean/rule_space.rs index 60f3bf01..ae3dcdf4 100644 --- a/libs/braillify/src/rules/korean/rule_space.rs +++ b/libs/braillify/src/rules/korean/rule_space.rs @@ -1,4 +1,4 @@ -//! Space character encoding. +//! 빈칸 자체는 어떤 규정 항목에도 속하지 않음. //! //! Spaces → 0, newlines → 255. @@ -8,10 +8,10 @@ use crate::rules::context::RuleContext; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; pub static META: RuleMeta = RuleMeta { - section: "space", + section: "-", subsection: None, name: "space_encoding", - standard_ref: "N/A", + standard_ref: "빈칸 자체", description: "Encode space (0) and newline (255)", }; @@ -57,4 +57,14 @@ mod tests { let ctx = owned.ctx_at(0); let _ = RuleSpace.matches(&ctx); } + + /// 빈칸 자체는 어떤 규정 항목에도 속하지 않으므로, 정직한 표시로 "-"를 사용한다. + /// trace.rs의 WORD_SPACE_META 선례를 따른다. + #[test] + fn meta_section_is_dash_for_non_article() { + assert_eq!( + META.section, "-", + "META.section must be dash for non-article blank cell" + ); + } } diff --git a/libs/braillify/src/rules/math/encoder.rs b/libs/braillify/src/rules/math/encoder.rs index ce19324d..9b931b64 100644 --- a/libs/braillify/src/rules/math/encoder.rs +++ b/libs/braillify/src/rules/math/encoder.rs @@ -14,6 +14,36 @@ use super::{ rule_54, rule_57, }; use crate::math_symbol_shortcut; +use crate::rules::RuleMeta; + +static DIGIT_SEPARATOR_META: RuleMeta = RuleMeta { + section: "41", + subsection: None, + name: "math_digit_separator", + standard_ref: "2024 Korean Braille Standard, 수학 제41항", + description: "Comma and grouping point between digits", +}; + +static SPACE_META: RuleMeta = RuleMeta { + section: "11", + subsection: None, + name: "math_expression_spacing", + standard_ref: "2024 Korean Braille Standard, 수학 제11항", + description: "Spacing around mathematical expressions", +}; + +static KOREAN_WORD_META: RuleMeta = RuleMeta { + section: "6", + subsection: None, + name: "math_korean_word", + standard_ref: "2024 Korean Braille Standard, 수학 제6항", + description: "Korean text and grouping brackets inside mathematics", +}; + +static RAW_TOKEN_VARIANT_METAS: &[&RuleMeta] = &[ + &math_symbol_shortcut::META_KOREAN_51, + &math_symbol_shortcut::META_KOREAN_59, +]; struct DigitSeparatorRule; @@ -21,13 +51,17 @@ pub(super) fn encode_generic_math_symbol( c: char, _is_direct_shortcut_symbol: bool, result: &mut Vec, -) -> Result<(), String> { - let encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(c)?; - result.extend_from_slice(encoded); - Ok(()) +) -> Result<&'static crate::rules::RuleMeta, String> { + let shortcut = math_symbol_shortcut::math_symbol_shortcut(c)?; + result.extend_from_slice(shortcut.cells); + Ok(shortcut.fallback_meta) } impl MathTokenRule for DigitSeparatorRule { + fn meta(&self) -> &'static RuleMeta { + &DIGIT_SEPARATOR_META + } + fn name(&self) -> &'static str { "DigitSeparatorRule" } @@ -113,6 +147,10 @@ fn should_suppress_space(tokens: &[MathToken], index: usize) -> bool { } impl MathTokenRule for SpaceRule { + fn meta(&self) -> &'static RuleMeta { + &SPACE_META + } + fn name(&self) -> &'static str { "SpaceRule" } @@ -259,6 +297,10 @@ fn should_suppress_after_operator(tokens: &[MathToken], index: usize) -> bool { } impl MathTokenRule for KoreanWordRule { + fn meta(&self) -> &'static RuleMeta { + &KOREAN_WORD_META + } + fn name(&self) -> &'static str { "KoreanWordRule" } @@ -302,6 +344,14 @@ use symbol_rule::MathSymbolRule; struct RawTokenRule; impl MathTokenRule for RawTokenRule { + fn meta(&self) -> &'static RuleMeta { + &math_symbol_shortcut::META_KOREAN_49 + } + + fn variant_metas(&self) -> &'static [&'static RuleMeta] { + RAW_TOKEN_VARIANT_METAS + } + fn name(&self) -> &'static str { "RawTokenRule" } @@ -325,15 +375,18 @@ impl MathTokenRule for RawTokenRule { let Some(MathToken::Raw(c)) = tokens.get(index) else { return Ok(MathTokenResult::Skip); }; - // PDF — 수학 컨텍스트 내 일반 구두점 중 PDF 65항 등에서 정의된 것만 처리한다. + // PDF 한글 제49·51·59항 — 수학 입력 안의 물음표·느낌표·쌍점·쌍반점. // 무차별 fallback은 다른 컨텍스트(예: 인용 부호)와 충돌하므로 명시적 매핑으로 한정. - if matches!(*c, ':' | ';' | '?' | '!') - && let Ok(encoded) = crate::symbol_shortcut::encode_char_symbol_shortcut(*c) - { - result.extend_from_slice(encoded); - return Ok(MathTokenResult::Consumed(1)); - } - Err(format!("Unrecognized math character: '{}'", c)) + let meta = match *c { + '?' | '!' => &math_symbol_shortcut::META_KOREAN_49, + ':' => &math_symbol_shortcut::META_KOREAN_51, + ';' => &math_symbol_shortcut::META_KOREAN_59, + _ => return Err(format!("Unrecognized math character: '{}'", c)), + }; + let encoded = crate::symbol_shortcut::encode_char_symbol_shortcut(*c) + .map_err(|_| format!("Unrecognized math character: '{}'", c))?; + result.extend_from_slice(encoded); + Ok(MathTokenResult::ConsumedWithMeta { tokens: 1, meta }) } } @@ -361,6 +414,13 @@ static MATRIX_MATH_MODE_ENGINE: LazyLock = LazyLock::new(|| { }) }); +/// Metadata of every math rule, in [`crate::rules::trace::RuleId`] order. All +/// four context engines register the same rules in the same order, so one +/// engine's ordering describes them all. +pub(crate) fn math_rule_registry() -> Vec<&'static crate::rules::RuleMeta> { + DEFAULT_MATH_ENGINE.registry() +} + pub(super) fn math_engine_for_context(context: MathContext) -> &'static MathTokenEngine { match (context.matrix_context_active, context.math_mode_active) { (false, false) => &DEFAULT_MATH_ENGINE, @@ -407,12 +467,29 @@ fn build_math_engine(context: MathContext) -> MathTokenEngine { /// Encode a full math expression string into braille bytes. pub fn encode_math_expression(input: &str) -> Result, String> { + encode_math_expression_traced(input, MathContext::default(), None) +} + +pub(crate) fn encode_math_expression_traced( + input: &str, + context: MathContext, + trace: Option<&mut crate::rules::trace::TraceSink<'_>>, +) -> Result, String> { if rule_14::is_roman_numeral_expression(input) { return rule_14::encode_roman_numeral_expression(input); } - let tokens = super::parser::parse_math_expression(input)?; - encode_math_tokens_with_context(&tokens, MathContext::default()) + // A non-default context parses with the math-mode parser; the two parsers + // disagree on tokenisation, so this branch decides the output. + let tokens = if context == MathContext::default() { + super::parser::parse_math_expression(input)? + } else { + super::parser::parse_math_expression_with_math_mode(input, context.math_mode_active)? + }; + let engine = math_engine_for_context(context); + let mut result = Vec::new(); + engine.encode_tokens_traced(&tokens, &mut result, trace)?; + Ok(result) } /// Encode a full math expression string with encoder-scoped context flags. @@ -423,24 +500,7 @@ pub fn encode_math_expression_with_context( if context == MathContext::default() { return encode_math_expression(input); } - - if rule_14::is_roman_numeral_expression(input) { - return rule_14::encode_roman_numeral_expression(input); - } - - let tokens = - super::parser::parse_math_expression_with_math_mode(input, context.math_mode_active)?; - encode_math_tokens_with_context(&tokens, context) -} - -fn encode_math_tokens_with_context( - tokens: &[MathToken], - context: MathContext, -) -> Result, String> { - let engine = math_engine_for_context(context); - let mut result = Vec::new(); - engine.encode_tokens(tokens, &mut result)?; - Ok(result) + encode_math_expression_traced(input, context, None) } #[cfg(test)] @@ -455,6 +515,30 @@ mod tests { assert!(result.is_ok(), "Should encode ax+b=0: {:?}", result); } + /// The raw rule carries only the four punctuation marks the Korean articles + /// name. A mark outside that list has no article behind it, so it is refused + /// rather than borrowed from another context. + #[test] + fn a_raw_character_outside_the_named_punctuation_is_refused() { + let context = MathContext::default(); + let tokens = [MathToken::Raw('@')]; + let mut result = Vec::new(); + let mut state = MathEncodeState::with_context(false, context); + + let outcome = RawTokenRule.apply( + &tokens, + 0, + &mut result, + &mut state, + math_engine_for_context(context), + ); + + let Err(err) = outcome else { + panic!("an unmapped raw character must not encode"); + }; + assert!(err.contains("Unrecognized math character"), "{err}"); + } + #[test] fn test_number_encoding() { // Pure number should get # prefix @@ -742,6 +826,7 @@ mod tests { let rule = SpaceRule; assert_eq!(rule.name(), "SpaceRule"); assert_eq!(rule.priority(), 50); + assert_eq!(rule.meta().section, "11"); } /// DigitSeparatorRule metadata must remain stable. @@ -750,6 +835,7 @@ mod tests { let rule = DigitSeparatorRule; assert_eq!(rule.name(), "DigitSeparatorRule"); assert_eq!(rule.priority(), 50); + assert_eq!(rule.meta().section, "41"); let state = MathEncodeState::with_context(false, MathContext::default()); // matches returns true ONLY for DigitSeparator. let yes = vec![MathToken::DigitSeparator]; @@ -1030,6 +1116,7 @@ mod tests { let rule = KoreanWordRule; assert_eq!(rule.name(), "KoreanWordRule"); assert_eq!(rule.priority(), 50); + assert_eq!(rule.meta().section, "6"); let state = MathEncodeState::with_context(false, MathContext::default()); let yes = vec![kw("원")]; assert!(rule.matches(&yes, 0, &state)); @@ -1044,6 +1131,14 @@ mod tests { let rule = RawTokenRule; assert_eq!(rule.name(), "RawTokenRule"); assert_eq!(rule.priority(), 500); + assert_eq!(rule.meta().section, "49"); + assert_eq!( + rule.variant_metas() + .iter() + .map(|meta| meta.section) + .collect::>(), + vec!["51", "59"] + ); let state = MathEncodeState::with_context(false, MathContext::default()); let yes = vec![MathToken::Raw('?')]; assert!(rule.matches(&yes, 0, &state)); @@ -1198,6 +1293,46 @@ mod tests { }); } + #[rstest::rstest] + #[case::matrix(MathContext { + matrix_context_active: true, + math_mode_active: false, + })] + #[case::math_mode(MathContext { + matrix_context_active: false, + math_mode_active: true, + })] + #[case::matrix_math_mode(MathContext { + matrix_context_active: true, + math_mode_active: true, + })] + fn every_context_engine_exposes_the_same_flattened_registry(#[case] context: MathContext) { + let default_registry = math_engine_for_context(MathContext::default()).registry(); + let context_registry = math_engine_for_context(context).registry(); + + assert_eq!(context_registry.len(), default_registry.len()); + assert!( + context_registry + .iter() + .zip(default_registry) + .all(|(actual, expected)| std::ptr::eq(*actual, expected)) + ); + } + + /// `MathTokenRule::meta` has no default, so a rule without a checked article + /// cannot be written at all. What remains to assert is that every article + /// the engine reports is a real one. + #[test] + fn every_math_rule_reports_a_real_article() { + let unnamed: Vec<&str> = math_rule_registry() + .into_iter() + .map(|meta| meta.section) + .filter(|section| section.is_empty() || *section == "?") + .collect(); + + assert_eq!(unnamed, Vec::<&str>::new()); + } + /// `KoreanWordRule.apply` defensive Skip when token is not KoreanWord. /// `matches()` guarantees correctness; the Skip arm is type-safety only. #[test] @@ -1278,6 +1413,32 @@ mod tests { assert!(result.is_empty()); } + #[rstest::rstest] + #[case::question('?', "49")] + #[case::exclamation('!', "49")] + #[case::colon(':', "51")] + #[case::semicolon(';', "59")] + fn raw_token_rule_reports_korean_punctuation_article( + #[case] symbol: char, + #[case] expected_section: &str, + ) { + let context = MathContext::default(); + let engine = MathTokenEngine::with_context(context); + let tokens = [MathToken::Raw(symbol)]; + let mut state = MathEncodeState::with_context(false, context); + let mut output = Vec::new(); + + let outcome = RawTokenRule + .apply(&tokens, 0, &mut output, &mut state, &engine) + .expect("supported punctuation should encode"); + + let MathTokenResult::ConsumedWithMeta { tokens, meta } = outcome else { + panic!("raw punctuation did not report selected metadata"); + }; + assert_eq!(tokens, 1); + assert_eq!(meta.section, expected_section); + } + /// encoder.rs line 348 — `encode_math_expression_with_context` Roman numeral fast-path /// when context is NON-default (forces the second is_roman_numeral check). #[test] diff --git a/libs/braillify/src/rules/math/encoder/symbol_rule.rs b/libs/braillify/src/rules/math/encoder/symbol_rule.rs index 15a738a6..8c67173a 100644 --- a/libs/braillify/src/rules/math/encoder/symbol_rule.rs +++ b/libs/braillify/src/rules/math/encoder/symbol_rule.rs @@ -6,10 +6,10 @@ use super::super::math_token_rule::{ use super::super::parser::{BracketKind, MathToken}; use super::super::{ rule_1, rule_2, rule_3, rule_4, rule_5, rule_6, rule_9, rule_10, rule_11, rule_12, rule_13, - rule_15, rule_16, rule_17, rule_21, rule_22, rule_23, rule_24, rule_25, rule_26, rule_27, - rule_28, rule_30, rule_31, rule_32, rule_33, rule_36, rule_37, rule_38, rule_39, rule_40, - rule_41, rule_42, rule_43, rule_44, rule_50, rule_54, rule_55, rule_56, rule_58, rule_59, - rule_60, rule_61, rule_64, rule_65, + rule_15, rule_16, rule_17, rule_21, rule_22, rule_23, rule_25, rule_26, rule_27, rule_28, + rule_30, rule_31, rule_32, rule_33, rule_36, rule_37, rule_38, rule_39, rule_40, rule_41, + rule_42, rule_43, rule_44, rule_50, rule_54, rule_55, rule_56, rule_58, rule_59, rule_60, + rule_61, rule_64, rule_65, }; use super::encode_generic_math_symbol; use crate::math_symbol_shortcut; @@ -26,6 +26,15 @@ impl MathSymbolRule { } None } + + /// True iff the arrow at `index` is drawn over a pair of points, which is + /// how 제37항 and 제38항 write a line and a ray: `3o,,AB`, the arrow ahead of + /// both capitals with nothing before it. An arrow with an operand on its + /// left is standing between two things and belongs to 제10항 instead. + fn names_two_points(tokens: &[MathToken], index: usize) -> bool { + matches!(tokens.get(index + 1), Some(MathToken::UpperVariable(_))) + && rule_12::prev_non_space(tokens, index).is_none() + } } /// True iff tokens at `index+1..=index+5` form the `( N , N )` math-paren @@ -53,6 +62,14 @@ impl MathTokenRule for MathSymbolRule { "MathSymbolRule" } + fn meta(&self) -> &'static crate::rules::RuleMeta { + &math_symbol_shortcut::META_3 + } + + fn variant_metas(&self) -> &'static [&'static crate::rules::RuleMeta] { + math_symbol_shortcut::MATH_SYMBOL_VARIANT_METAS + } + fn priority(&self) -> u16 { 100 } @@ -96,7 +113,10 @@ impl MathTokenRule for MathSymbolRule { } result.push(52); state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(i - index)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: i - index, + meta: &math_symbol_shortcut::META_65, + }); } // PDF 수학 제65항 1 — `#(UpperVar)` 패턴: 기수 표기. @@ -138,7 +158,10 @@ impl MathTokenRule for MathSymbolRule { result.push(52); // ⠴ (MathParen close) state.prev_was_number = false; let consumed = i + 1 - index; - return Ok(MathTokenResult::Consumed(consumed)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: consumed, + meta: &math_symbol_shortcut::META_65, + }); } } } @@ -176,7 +199,10 @@ impl MathTokenRule for MathSymbolRule { } result.push(0); // PDF 제61항 ∀x/∃x 다음 한 칸 띄움 state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(2)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: 2, + meta: &math_symbol_shortcut::META_61, + }); } } @@ -218,7 +244,10 @@ impl MathTokenRule for MathSymbolRule { } state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(close_idx + 1 - index)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: close_idx + 1 - index, + meta: &math_symbol_shortcut::META_25, + }); } if *c == '\u{03A0}' && is_capital_pi_numeric_pair(tokens, index) { @@ -234,20 +263,25 @@ impl MathTokenRule for MathSymbolRule { } result.push(62); state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(6)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: 6, + meta: &math_symbol_shortcut::META_13, + }); } // In derivative/product formulas (제53항), middle dot is used as // multiplication sign when the same expression also contains // arithmetic composition (= or +). - if *c == '\u{00B7}' - && tokens - .iter() - .any(|t| matches!(t, MathToken::Operator('=' | '+'))) - { + let composes_arithmetically = tokens + .iter() + .any(|t| matches!(t, MathToken::Operator('=' | '+'))); + if *c == '\u{00B7}' && composes_arithmetically { rule_2::encode_operator('\u{00D7}', tokens, index, result)?; state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(1)); + return Ok(MathTokenResult::ConsumedWithMeta { + tokens: 1, + meta: &math_symbol_shortcut::META_53, + }); } let next_for_padding = Self::next_non_space(tokens, index + 1); @@ -295,24 +329,34 @@ impl MathTokenRule for MathSymbolRule { } } - if rule_3::is_equality_symbol(*c) { + let selected_meta: &'static crate::rules::RuleMeta = if rule_3::is_equality_symbol(*c) { rule_3::encode_equality_symbol(*c, result)?; + &math_symbol_shortcut::META_3 } else if rule_4::is_comparison_symbol(*c) { rule_4::encode_comparison_symbol(*c, result)?; + &math_symbol_shortcut::META_4 } else if rule_5::is_proportion_symbol(*c) { rule_5::encode_proportion_symbol(*c, result)?; - } else if rule_37::is_double_arrow_line_symbol(*c) { + &math_symbol_shortcut::META_SCIENCE_29 + } else if rule_37::is_double_arrow_line_symbol(*c) && Self::names_two_points(tokens, index) + { rule_37::encode_double_arrow_line_symbol(*c, result)?; - } else if rule_38::is_right_arrow_ray_symbol(*c) { + &math_symbol_shortcut::META_37 + } else if rule_38::is_right_arrow_ray_symbol(*c) && Self::names_two_points(tokens, index) { rule_38::encode_right_arrow_ray_symbol(*c, result)?; + &math_symbol_shortcut::META_38 } else if rule_10::is_arrow_symbol(*c) { rule_10::encode_arrow_symbol(*c, result)?; + &math_symbol_shortcut::META_10 } else if rule_13::is_greek_symbol(*c) { rule_13::encode_greek_symbol(*c, result)?; + &math_symbol_shortcut::META_13 } else if rule_15::is_custom_binary_operator(*c) { rule_15::encode_custom_binary_operator(*c, result)?; + &math_symbol_shortcut::META_15 } else if rule_17::is_prime_mark(*c) { rule_17::encode_prime(*c, result)?; + &math_symbol_shortcut::META_17 // rule_20 (U+2252 ≒) and rule_29 (U+2248 ≈) dispatch arms were removed: // both chars are claimed by `rule_3::is_equality_symbol` earlier in the // chain, making rule_20/rule_29 arms structurally unreachable. @@ -327,15 +371,16 @@ impl MathTokenRule for MathSymbolRule { } else { rule_21::encode_absolute_value_close(result)?; } + &math_symbol_shortcut::META_21 } else if rule_23::is_overline_mark(*c) { rule_23::encode_overline(result)?; - } else if rule_24::is_sequence_brace(*c) { - rule_24::encode_sequence_brace(*c, result)?; + &math_symbol_shortcut::META_23 } else if rule_27::is_divisibility_symbol(*c) { // `|` is always handled by rule_21::is_absolute_value_bar above; only // U+2224 (∤) reaches this arm. Probe-verified 2026-05-23. let encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(*c)?; result.extend_from_slice(encoded); + &math_symbol_shortcut::META_27 } else if rule_28::is_norm_symbol(*c) { if index == 0 { rule_28::encode_norm_open(result)?; @@ -344,30 +389,43 @@ impl MathTokenRule for MathSymbolRule { } else { rule_28::encode_norm_symbol(*c, result)?; } + &math_symbol_shortcut::META_28 } else if rule_30::is_dot_congruence(*c) { rule_30::encode_dot_congruence(*c, result)?; + &math_symbol_shortcut::META_30 } else if rule_31::is_asymptotic_equal(*c) { rule_31::encode_asymptotic_equal(*c, result)?; + &math_symbol_shortcut::META_31 } else if rule_32::is_congruence_symbol(*c) { rule_32::encode_congruence_symbol(*c, result)?; + &math_symbol_shortcut::META_32 } else if rule_33::is_geometric_operator(*c) { rule_33::encode_geometric_operator(*c, result)?; + &math_symbol_shortcut::META_33 } else if rule_36::is_arc_symbol(*c) { rule_36::encode_arc(*c, result)?; + &math_symbol_shortcut::META_36 } else if rule_39::is_angle_symbol(*c) { rule_39::encode_angle_symbol(*c, result)?; + &math_symbol_shortcut::META_39 } else if rule_40::is_geometric_shape(*c) { rule_40::encode_geometric_shape(*c, result)?; + &math_symbol_shortcut::META_40 } else if rule_41::is_perpendicular_symbol(*c) { rule_41::encode_perpendicular(*c, result)?; + &math_symbol_shortcut::META_41 } else if rule_42::is_similarity_symbol(*c) { rule_42::encode_similarity_symbol(*c, result)?; + &math_symbol_shortcut::META_42 } else if rule_43::is_identity_symbol(*c) { rule_43::encode_identity_symbol(*c, result)?; + &math_symbol_shortcut::META_43 } else if rule_44::is_parallel_symbol(*c) { rule_44::encode_parallel_symbol(*c, result)?; + &math_symbol_shortcut::META_44 } else if rule_50::is_special_constant(*c) { rule_50::encode_special_constant(*c, result)?; + &math_symbol_shortcut::META_50 } // 제52항 (Δ, U+0394) is captured by `rule_13::is_greek_symbol` earlier in // this dispatch chain, so an explicit rule_52 arm would be unreachable. @@ -375,16 +433,22 @@ impl MathTokenRule for MathSymbolRule { // callers that want delta encoding without going through MathSymbolRule. else if rule_54::is_partial_derivative(*c) { rule_54::encode_partial_derivative(*c, result)?; + &math_symbol_shortcut::META_54 } else if rule_55::is_nabla_symbol(*c) { rule_55::encode_nabla_symbol(*c, result)?; + &math_symbol_shortcut::META_55 } else if rule_56::is_integral_symbol(*c) { rule_56::encode_integral_symbol(*c, result)?; + &math_symbol_shortcut::META_56 } else if *c == '\u{222C}' { rule_58::encode_double_integral(*c, result)?; + &math_symbol_shortcut::META_58 } else if rule_59::is_contour_integral(*c) { rule_59::encode_contour_integral(*c, result)?; + &math_symbol_shortcut::META_59 } else if rule_65::is_therefore_because(*c) { rule_65::encode_therefore_because(*c, result)?; + &math_symbol_shortcut::META_65 } else if *c == '\u{0307}' && matches!( rule_12::prev_non_space(tokens, index), @@ -394,6 +458,7 @@ impl MathTokenRule for MathSymbolRule { // PDF 수학 제65항 5 — 문자 뒤 결합 윗 한 점 (ȧ 등). 숫자 뒤 순환소수와 구분. result.push(crate::unicode::decode_unicode('⠈')); result.push(crate::unicode::decode_unicode('⠲')); + &math_symbol_shortcut::META_65 } else { let is_direct_shortcut_symbol = rule_11::is_math_sentence_delimiter(*c) || rule_16::is_base_notation_subscript(*c) @@ -401,8 +466,8 @@ impl MathTokenRule for MathSymbolRule { || rule_60::is_set_symbol(*c) || rule_61::is_logic_symbol(*c) || rule_64::is_hat_notation(*c); - encode_generic_math_symbol(*c, is_direct_shortcut_symbol, result)?; - } + encode_generic_math_symbol(*c, is_direct_shortcut_symbol, result)? + }; if matches!(*c, '\u{2234}' | '\u{2235}') { let next_is_space = matches!(tokens.get(index + 1), Some(MathToken::Space)); @@ -434,7 +499,10 @@ impl MathTokenRule for MathSymbolRule { } state.prev_was_number = rule_9::is_repeating_decimal_mark(*c); - Ok(MathTokenResult::Consumed(1)) + Ok(MathTokenResult::ConsumedWithMeta { + tokens: 1, + meta: selected_meta, + }) } } @@ -454,7 +522,8 @@ impl MathTokenRule for MathSymbolRule { // ============================================================ #[cfg(test)] mod tests { - use super::super::super::math_token_rule::MathContext; + use super::super::super::math_token_rule::{MathContext, MathTokenResult}; + use super::super::super::parser::MathToken; use super::super::encode_math_expression; use super::super::encode_math_expression_with_context; @@ -466,6 +535,141 @@ mod tests { encode_math_expression_with_context(s, ctx).expect("math encode should succeed") } + #[rstest::rstest] + #[case::equality('=', "3")] + #[case::proportion('∝', "29")] + #[case::greek('α', "13")] + #[case::root('√', "22")] + #[case::set_membership('∈', "60")] + #[case::negation_overlay('\u{0338}', "34")] + fn reports_the_selected_symbol_article(#[case] symbol: char, #[case] expected_section: &str) { + use super::super::super::encoder::math_engine_for_context; + use super::super::super::math_token_rule::{MathEncodeState, MathTokenRule}; + + let context = MathContext::default(); + let engine = math_engine_for_context(context); + let tokens = [MathToken::MathSymbol(symbol)]; + let mut output = Vec::new(); + let mut state = MathEncodeState::with_context(false, context); + + let outcome = super::MathSymbolRule + .apply(&tokens, 0, &mut output, &mut state, engine) + .expect("math symbol should encode"); + + let MathTokenResult::ConsumedWithMeta { tokens, meta } = outcome else { + panic!("math symbol did not report selected metadata"); + }; + assert_eq!(tokens, 1); + assert_eq!(meta.section, expected_section); + } + + fn apply_symbol(tokens: &[MathToken]) -> (Vec, MathTokenResult) { + use super::super::super::encoder::math_engine_for_context; + use super::super::super::math_token_rule::{MathEncodeState, MathTokenRule}; + + let context = MathContext::default(); + let mut output = Vec::new(); + let mut state = MathEncodeState::with_context(false, context); + let outcome = super::MathSymbolRule + .apply( + tokens, + 0, + &mut output, + &mut state, + math_engine_for_context(context), + ) + .expect("math symbol should encode"); + (output, outcome) + } + + /// 제25항 writes the summation's bounds in a group and then leaves a blank + /// before the body. A summation that already has a space after it, or that + /// ends the expression, must not gain a second blank. + #[rstest::rstest] + #[case::runs_into_the_body(Some(MathToken::Variable('x')), true)] + #[case::already_spaced(Some(MathToken::Space), false)] + #[case::ends_the_expression(None, false)] + fn a_summation_separates_itself_from_the_body( + #[case] trailing: Option, + #[case] expects_blank: bool, + ) { + use super::super::super::parser::BracketKind; + + let mut tokens = vec![ + MathToken::MathSymbol('\u{03A3}'), + MathToken::OpenParen(BracketKind::MathParen), + MathToken::Variable('n'), + MathToken::Operator('='), + MathToken::Number("1".to_string()), + MathToken::CloseParen(BracketKind::MathParen), + ]; + tokens.extend(trailing); + + let (output, _) = apply_symbol(&tokens); + + assert_eq!(output.last() == Some(&0), expects_blank); + } + + /// 제37항 and 제38항 draw a line and a ray over a pair of points, writing + /// the arrow ahead of both capitals as `3o,,AB`. 제10항 covers the arrow + /// standing between two things, which is how a reaction equation and an + /// ordinary mapping are written. The cells are the same either way, so only + /// the article distinguishes them. + #[rstest::rstest] + #[case::ray_over_points('\u{2192}', 0, "38")] + #[case::line_over_points('\u{2194}', 0, "37")] + #[case::ray_between_operands('\u{2192}', 1, "10")] + #[case::line_between_operands('\u{2194}', 1, "10")] + fn an_arrow_over_points_is_not_an_arrow_between_them( + #[case] arrow: char, + #[case] index: usize, + #[case] section: &str, + ) { + use super::super::super::encoder::math_engine_for_context; + use super::super::super::math_token_rule::{MathEncodeState, MathTokenRule}; + + let mut tokens = vec![MathToken::MathSymbol(arrow), MathToken::UpperVariable('A')]; + if index == 1 { + tokens.insert(0, MathToken::UpperVariable('B')); + } else { + tokens.push(MathToken::UpperVariable('B')); + } + + let context = MathContext::default(); + let mut output = Vec::new(); + let mut state = MathEncodeState::with_context(false, context); + let outcome = super::MathSymbolRule + .apply( + &tokens, + index, + &mut output, + &mut state, + math_engine_for_context(context), + ) + .expect("an arrow should encode"); + + let MathTokenResult::ConsumedWithMeta { meta, .. } = outcome else { + panic!("the arrow did not report selected metadata"); + }; + assert_eq!(meta.section, section); + } + + /// 제53항 reads a middle dot as the multiplication sign when the same + /// expression also composes arithmetically, which is how derivative and + /// product formulas are written. + #[test] + fn a_middle_dot_multiplies_inside_an_equation() { + let (output, outcome) = + apply_symbol(&[MathToken::MathSymbol('\u{00B7}'), MathToken::Operator('=')]); + + let MathTokenResult::ConsumedWithMeta { tokens, meta } = outcome else { + panic!("the middle dot did not report selected metadata"); + }; + assert_eq!(tokens, 1); + assert_eq!(meta.section, "53"); + assert!(!output.is_empty()); + } + // ---------------- Specialised prefix arms ---------------- /// Math rule 61: a negation sign keeps its complete two-cell mapping @@ -715,13 +919,10 @@ mod tests { assert!(!result.is_empty(), "a̅ must encode"); } - /// `{a,b,c}` — sequence brace (U+007B/U+007D) → rule_24 arm at lines - /// 341-342. (Note: parser routes `{` to OpenParen, but a bare math - /// symbol `{` outside grouping context can hit this arm.) + /// The parser routes `{`/`}` to OpenParen/CloseParen, so a brace + /// expression encodes through the bracket path rather than as a symbol. #[test] - fn sequence_brace_dispatch() { - // Use a curly-brace expression — the inner `{`/`}` are parsed as - // OpenParen/CloseParen, but rule_24 still detects them. + fn brace_expression_encodes_through_the_bracket_path() { let result = enc("{a,b}"); assert!(!result.is_empty(), "{{a,b}} must encode"); } @@ -1104,7 +1305,10 @@ mod tests { .apply(&tokens, 1, &mut result, &mut state, engine) .expect("operator should encode"); - assert!(matches!(action, MathTokenResult::Consumed(1))); + assert!(matches!( + action, + MathTokenResult::ConsumedWithMeta { tokens: 1, meta } if meta.section == "61" + )); assert_eq!(result.first().copied(), Some(0)); assert!(!state.prev_was_number); } diff --git a/libs/braillify/src/rules/math/math_token_rule.rs b/libs/braillify/src/rules/math/math_token_rule.rs index f74727e6..8967c036 100644 --- a/libs/braillify/src/rules/math/math_token_rule.rs +++ b/libs/braillify/src/rules/math/math_token_rule.rs @@ -35,15 +35,32 @@ impl MathEncodeState { pub enum MathTokenResult { /// Rule consumed N tokens (advance index by N). Consumed(usize), + /// Rule consumed tokens under one of its declared metadata variants. + ConsumedWithMeta { + tokens: usize, + meta: &'static crate::rules::RuleMeta, + }, /// Rule did not apply. Try next rule. Skip, } +use crate::rules::trace::{RuleId, TraceSink}; + /// Plugin interface for math token encoding rules. pub trait MathTokenRule: Send + Sync { /// Rule name for debugging. fn name(&self) -> &'static str; + /// The article of the standard this rule implements. Required rather than + /// defaulted: a rule that has not been checked against the standard should + /// fail to compile, not quietly report an article nobody verified. + fn meta(&self) -> &'static crate::rules::RuleMeta; + + /// Additional articles this rule can select while dispatching variants. + fn variant_metas(&self) -> &'static [&'static crate::rules::RuleMeta] { + &[] + } + /// Priority (lower runs first). Default: 100. fn priority(&self) -> u16 { 100 @@ -86,24 +103,83 @@ impl MathTokenEngine { self.rules.sort_by_key(|r| r.priority()); } + /// Metadata of every registered math rule, in [`RuleId`] order. + pub(crate) fn registry(&self) -> Vec<&'static crate::rules::RuleMeta> { + self.rules + .iter() + .flat_map(|rule| { + std::iter::once(rule.meta()).chain(rule.variant_metas().iter().copied()) + }) + .collect() + } + /// Encode a sequence of math tokens into braille bytes. pub fn encode_tokens(&self, tokens: &[MathToken], result: &mut Vec) -> Result<(), String> { + self.encode_tokens_traced(tokens, result, None) + } + + /// [`Self::encode_tokens`], recording which rule produced each stretch. + /// + /// The spans go to a collector rather than to a sink parameter because a math + /// expression usually reaches here from a token rule, which emits the cells + /// much later without knowing where they land. [`crate::rules::emit`] pairs + /// the collected spans back to those cells. + pub(crate) fn encode_tokens_traced( + &self, + tokens: &[MathToken], + result: &mut Vec, + mut trace: Option<&mut TraceSink<'_>>, + ) -> Result<(), String> { + let mut attempt = super::MathAttempt::new(); + let attempt_base = result.len(); let logic_context = Self::has_logic_symbol(tokens); let mut state = MathEncodeState::with_context(logic_context, self.context); let mut i = 0usize; while i < tokens.len() { let mut handled = false; + let mut registry_base = 0usize; for rule in &self.rules { let _ = rule.name(); - if rule.matches(tokens, i, &state) - && let MathTokenResult::Consumed(n) = - rule.apply(tokens, i, result, &mut state, self)? - { - i += n; - handled = true; - break; + let variant_metas = rule.variant_metas(); + let registry_len = 1 + variant_metas.len(); + if !rule.matches(tokens, i, &state) { + registry_base += registry_len; + continue; } + + let start = result.len(); + let (consumed, local_offset) = + match rule.apply(tokens, i, result, &mut state, self)? { + MathTokenResult::Consumed(tokens) => (tokens, 0), + MathTokenResult::ConsumedWithMeta { tokens, meta } => { + let local_offset = std::iter::once(rule.meta()) + .chain(variant_metas.iter().copied()) + .position(|declared| std::ptr::eq(declared, meta)) + .ok_or_else(|| { + format!( + "{} reported undeclared metadata: {} {}", + rule.name(), + meta.section, + meta.name + ) + })?; + (tokens, local_offset) + } + MathTokenResult::Skip => { + registry_base += registry_len; + continue; + } + }; + let rule_id = RuleId::math(registry_base + local_offset); + attempt.push(rule_id, start - attempt_base, result.len() - start); + if let Some(sink) = trace.as_deref_mut() { + let token_index = sink.token_index() as usize; + sink.record_span(rule_id, token_index, start..result.len()); + } + i += consumed; + handled = true; + break; } if !handled { return Err(format!( @@ -112,6 +188,7 @@ impl MathTokenEngine { )); } } + attempt.finish(&result[attempt_base..]); Ok(()) } @@ -142,6 +219,150 @@ impl MathTokenEngine { #[cfg(test)] mod tests { use super::*; + use crate::rules::RuleMeta; + use crate::rules::trace::Trace; + + static PRIMARY_META: RuleMeta = RuleMeta { + section: "101", + subsection: None, + name: "test_primary", + standard_ref: "test primary", + description: "test primary", + }; + static SECONDARY_META: RuleMeta = RuleMeta { + section: "102", + subsection: None, + name: "test_secondary", + standard_ref: "test secondary", + description: "test secondary", + }; + static FOLLOWING_META: RuleMeta = RuleMeta { + section: "103", + subsection: None, + name: "test_following", + standard_ref: "test following", + description: "test following", + }; + static FOREIGN_META: RuleMeta = RuleMeta { + section: "104", + subsection: None, + name: "test_foreign", + standard_ref: "test foreign", + description: "test foreign", + }; + static VARIANT_METAS: [&RuleMeta; 1] = [&SECONDARY_META]; + + struct VariantRule; + + impl MathTokenRule for VariantRule { + fn name(&self) -> &'static str { + "VariantRule" + } + + fn meta(&self) -> &'static RuleMeta { + &PRIMARY_META + } + + fn variant_metas(&self) -> &'static [&'static RuleMeta] { + &VARIANT_METAS + } + + fn priority(&self) -> u16 { + 10 + } + + fn matches(&self, tokens: &[MathToken], index: usize, _state: &MathEncodeState) -> bool { + matches!(tokens.get(index), Some(MathToken::Variable(_))) + } + + fn apply( + &self, + _tokens: &[MathToken], + _index: usize, + result: &mut Vec, + _state: &mut MathEncodeState, + _engine: &MathTokenEngine, + ) -> Result { + result.push(1); + Ok(MathTokenResult::ConsumedWithMeta { + tokens: 1, + meta: &SECONDARY_META, + }) + } + } + + struct FollowingRule; + + impl MathTokenRule for FollowingRule { + fn name(&self) -> &'static str { + "FollowingRule" + } + + fn meta(&self) -> &'static RuleMeta { + &FOLLOWING_META + } + + fn priority(&self) -> u16 { + 20 + } + + fn matches(&self, tokens: &[MathToken], index: usize, _state: &MathEncodeState) -> bool { + matches!(tokens.get(index), Some(MathToken::Number(_))) + } + + fn apply( + &self, + _tokens: &[MathToken], + _index: usize, + result: &mut Vec, + _state: &mut MathEncodeState, + _engine: &MathTokenEngine, + ) -> Result { + result.push(2); + Ok(MathTokenResult::Consumed(1)) + } + } + + struct UndeclaredMetaRule; + + impl MathTokenRule for UndeclaredMetaRule { + fn name(&self) -> &'static str { + "UndeclaredMetaRule" + } + + fn meta(&self) -> &'static RuleMeta { + &PRIMARY_META + } + + fn matches(&self, tokens: &[MathToken], index: usize, _state: &MathEncodeState) -> bool { + matches!(tokens.get(index), Some(MathToken::Variable(_))) + } + + fn apply( + &self, + _tokens: &[MathToken], + _index: usize, + result: &mut Vec, + _state: &mut MathEncodeState, + _engine: &MathTokenEngine, + ) -> Result { + result.push(1); + Ok(MathTokenResult::ConsumedWithMeta { + tokens: 1, + meta: &FOREIGN_META, + }) + } + } + + /// Stand-in article for the dummy rules below. They exercise dispatch and + /// never reach the registry, so the number only has to be well formed. + static TEST_META: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "1", + subsection: None, + name: "test_rule", + standard_ref: "", + description: "", + }; /// `MathTokenRule::priority()` default implementation returns 100. /// Exercised by a dummy rule that doesn't override `priority()`. @@ -150,6 +371,9 @@ mod tests { fn priority_default_impl_returns_100() { struct DummyRule; impl MathTokenRule for DummyRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &TEST_META + } fn name(&self) -> &'static str { "DummyRule" } @@ -174,6 +398,73 @@ mod tests { } let r = DummyRule; assert_eq!(r.priority(), 100); + assert!(r.variant_metas().is_empty()); + } + + #[test] + fn registry_flattens_primary_then_variant_metadata_per_rule() { + let mut engine = MathTokenEngine::with_context(MathContext::default()); + engine.register(Box::new(FollowingRule)); + engine.register(Box::new(VariantRule)); + engine.finalize(); + + let registry = engine.registry(); + + assert_eq!(registry.len(), 3); + assert!(std::ptr::eq(registry[0], &PRIMARY_META)); + assert!(std::ptr::eq(registry[1], &SECONDARY_META)); + assert!(std::ptr::eq(registry[2], &FOLLOWING_META)); + } + + #[test] + fn traced_variant_uses_its_secondary_registry_slot() { + let mut engine = MathTokenEngine::with_context(MathContext::default()); + engine.register(Box::new(VariantRule)); + engine.finalize(); + let mut output = Vec::new(); + let mut trace = Trace::default(); + let mut sink = TraceSink::new(&mut trace); + + engine + .encode_tokens_traced(&[MathToken::Variable('x')], &mut output, Some(&mut sink)) + .unwrap(); + + assert_eq!(trace.events()[0].rule, RuleId::math(1)); + } + + #[test] + fn traced_rule_after_variant_rule_uses_flattened_registry_base() { + let mut engine = MathTokenEngine::with_context(MathContext::default()); + engine.register(Box::new(FollowingRule)); + engine.register(Box::new(VariantRule)); + engine.finalize(); + let mut output = Vec::new(); + let mut trace = Trace::default(); + let mut sink = TraceSink::new(&mut trace); + + engine + .encode_tokens_traced( + &[MathToken::Number("1".to_string())], + &mut output, + Some(&mut sink), + ) + .unwrap(); + + assert_eq!(trace.events()[0].rule, RuleId::math(2)); + } + + #[test] + fn encoded_variant_rejects_metadata_the_rule_never_declared() { + let mut engine = MathTokenEngine::with_context(MathContext::default()); + engine.register(Box::new(UndeclaredMetaRule)); + engine.finalize(); + let mut output = Vec::new(); + + let error = engine + .encode_tokens(&[MathToken::Variable('x')], &mut output) + .unwrap_err(); + + assert!(error.contains("UndeclaredMetaRule reported undeclared metadata")); } /// math_token_rule.rs line 97 - `MathTokenEngine.encode_tokens` returns Err @@ -195,6 +486,9 @@ mod tests { fn encode_tokens_continues_after_matching_rule_skips() { struct SkippingRule; impl MathTokenRule for SkippingRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &TEST_META + } fn name(&self) -> &'static str { "SkippingRule" } @@ -223,6 +517,9 @@ mod tests { struct ConsumingRule; impl MathTokenRule for ConsumingRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &TEST_META + } fn name(&self) -> &'static str { "ConsumingRule" } diff --git a/libs/braillify/src/rules/math/mod.rs b/libs/braillify/src/rules/math/mod.rs index 9dce6f99..00a17da2 100644 --- a/libs/braillify/src/rules/math/mod.rs +++ b/libs/braillify/src/rules/math/mod.rs @@ -12,6 +12,73 @@ pub mod function; pub mod math_token_rule; pub mod parser; +thread_local! { + /// Rule spans of each math expression encoded during one traced encode. + /// + /// A math expression is normally encoded by a token rule, which emits the + /// cells into the document much later. Holding the spans here lets + /// [`super::emit`] pair them back to those cells instead of reporting the + /// whole expression as one token-rule span. + static MATH_ATTEMPTS: std::cell::RefCell>> = + const { std::cell::RefCell::new(None) }; +} + +type MathSpans = (Vec, Vec<(super::trace::RuleId, u32, u32)>); + +/// Collects the rule spans of one math expression. +pub(crate) struct MathAttempt { + moves: Option>, +} + +impl MathAttempt { + pub(crate) fn new() -> Self { + let collecting = MATH_ATTEMPTS.with(|slot| slot.borrow().is_some()); + Self { + moves: collecting.then(Vec::new), + } + } + + pub(crate) fn push(&mut self, rule: super::trace::RuleId, offset: usize, len: usize) { + if let Some(moves) = self.moves.as_mut() { + moves.push((rule, offset as u32, len as u32)); + } + } + + pub(crate) fn finish(self, cells: &[u8]) { + let Some(moves) = self.moves else { + return; + }; + MATH_ATTEMPTS.with(|slot| { + if let Ok(mut slot) = slot.try_borrow_mut() + && let Some(attempts) = slot.as_mut() + { + attempts.push((cells.to_vec(), moves)); + } + }); + } +} + +pub(crate) fn begin_collection() { + MATH_ATTEMPTS.with(|slot| *slot.borrow_mut() = Some(Vec::new())); +} + +pub(crate) fn end_collection() { + MATH_ATTEMPTS.with(|slot| *slot.borrow_mut() = None); +} + +/// Spans of the math expression whose output is exactly `cells`, if one was +/// encoded during this trace. +pub(crate) fn spans_for(cells: &[u8]) -> Option> { + MATH_ATTEMPTS.with(|slot| { + let slot = slot.try_borrow().ok()?; + let attempts = slot.as_ref()?; + attempts + .iter() + .find(|(produced, _)| produced == cells) + .map(|(_, moves)| moves.clone()) + }) +} + // ── 제1항–제10항: 숫자, 연산, 등식, 비교, 괄호, 분수, 소수, 비 ── pub mod rule_1; pub mod rule_10; @@ -40,7 +107,6 @@ pub mod rule_20; pub mod rule_21; pub mod rule_22; pub mod rule_23; -pub mod rule_24; pub mod rule_25; pub mod rule_26; pub mod rule_27; diff --git a/libs/braillify/src/rules/math/parser/parse.rs b/libs/braillify/src/rules/math/parser/parse.rs index 9c684283..772e2643 100644 --- a/libs/braillify/src/rules/math/parser/parse.rs +++ b/libs/braillify/src/rules/math/parser/parse.rs @@ -681,17 +681,6 @@ pub(crate) fn parse_math_expression_with_math_mode( c, '+' | '=' | '>' | '<' | '/' | '-' | '!' | '×' | '÷' | '\u{2212}' ) { - // In chained inequalities like -5 < x < -2, the second minus is omitted. - if c == '-' - && i > 0 - && chars[i - 1] == '<' - && i + 1 < chars.len() - && chars[i + 1].is_ascii_digit() - { - i += 1; - continue; - } - let op = normalize_operator_char(c); if matches!(op, '+' | '×' | '/') { for group in &mut bracket_stack { diff --git a/libs/braillify/src/rules/math/rule_1.rs b/libs/braillify/src/rules/math/rule_1.rs index 1cef83b5..4ad8d726 100644 --- a/libs/braillify/src/rules/math/rule_1.rs +++ b/libs/braillify/src/rules/math/rule_1.rs @@ -33,7 +33,19 @@ pub fn encode_number_literal(digits: &str, result: &mut Vec) { pub struct NumberRule; +static META_NUMBERRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "1", + subsection: None, + name: "math_number", + standard_ref: "2024 Korean Braille Standard, 수학 제1항", + description: "수 표기", +}; + impl MathTokenRule for NumberRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_NUMBERRULE + } + fn name(&self) -> &'static str { "NumberRule" } diff --git a/libs/braillify/src/rules/math/rule_12.rs b/libs/braillify/src/rules/math/rule_12.rs index d82bd7d7..55f36fe4 100644 --- a/libs/braillify/src/rules/math/rule_12.rs +++ b/libs/braillify/src/rules/math/rule_12.rs @@ -246,6 +246,22 @@ pub fn encode_upper_variable( } } + /// 과학 제4항 — 화학식의 원소 기호는 모두 1급 점자로 적는다. 수학 제12항의 + /// 대문자 이어쓰기(`⠠⠠`)가 아니라 글자마다 대문자표를 붙인다. + /// + /// 아래 첨자가 있는 식으로 한정한다. 한 글자짜리 원소 기호는 수학 변수와 + /// 글자가 겹치므로(`P`, `V`, `B`, `C`), 첨자라는 화학식 신호가 없으면 + /// 행렬 아닌 대문자 변수까지 갈라놓게 된다. + /// + /// 프라임이 낀 대문자열(`O′H`)은 수학 변수이지 원소 기호의 나열이 아니다. + fn names_element_symbols(tokens: &[MathToken], start: usize, end: usize) -> bool { + use crate::rules::science::elements::is_single_letter_element; + let is_element = |token: &MathToken| matches!(token, MathToken::UpperVariable(letter) if is_single_letter_element(*letter)); + tokens.iter().any(|t| matches!(t, MathToken::Subscript(_))) + && end - start >= 2 + && tokens[start..end].iter().all(is_element) + } + let mut seq_end = *i; let mut uppercase_count = 0usize; while let Some(MathToken::UpperVariable(_)) = tokens.get(seq_end) { @@ -259,7 +275,8 @@ pub fn encode_upper_variable( // PDF 제12항 붙임 1 — 행렬 컨텍스트면 2-cap 행렬명(`AB`)을 ⠠+letter 개별 표기. // The seq_end loop above guarantees tokens[*i..seq_end] contains only // UpperVariable and Prime tokens (no other arms reachable). - if uppercase_count == 2 && matrix_context_active { + if (uppercase_count == 2 && matrix_context_active) || names_element_symbols(tokens, *i, seq_end) + { for token in &tokens[*i..seq_end] { if let MathToken::UpperVariable(upper) = token { result.push(32); @@ -362,7 +379,19 @@ pub fn encode_upper_variable( pub struct CombinatoricsRule; +static META_COMBINATORICSRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "12", + subsection: None, + name: "math_combinatorics", + standard_ref: "2024 Korean Braille Standard, 수학 제12항", + description: "순열·조합", +}; + impl MathTokenRule for CombinatoricsRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_COMBINATORICSRULE + } + fn name(&self) -> &'static str { "CombinatoricsRule" } @@ -415,7 +444,19 @@ impl MathTokenRule for CombinatoricsRule { pub struct VariableRule; +static META_VARIABLERULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "12", + subsection: None, + name: "math_variable", + standard_ref: "2024 Korean Braille Standard, 수학 제12항", + description: "소문자 변수", +}; + impl MathTokenRule for VariableRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_VARIABLERULE + } + fn name(&self) -> &'static str { "VariableRule" } @@ -457,7 +498,19 @@ impl MathTokenRule for VariableRule { pub struct UpperVariableRule; +static META_UPPERVARIABLERULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "12", + subsection: None, + name: "math_upper_variable", + standard_ref: "2024 Korean Braille Standard, 수학 제12항", + description: "대문자 변수", +}; + impl MathTokenRule for UpperVariableRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_UPPERVARIABLERULE + } + fn name(&self) -> &'static str { "UpperVariableRule" } diff --git a/libs/braillify/src/rules/math/rule_18.rs b/libs/braillify/src/rules/math/rule_18.rs index 1d6c19ea..d2412063 100644 --- a/libs/braillify/src/rules/math/rule_18.rs +++ b/libs/braillify/src/rules/math/rule_18.rs @@ -31,6 +31,68 @@ fn next_non_space(tokens: &[MathToken], mut idx: usize) -> Option<&MathToken> { /// PDF 수학 제18항 2 — 좌상첨자: 위첨자가 변수 앞에 단독 위치할 때. /// 앞에 피첨자(변수/숫자/괄호닫기)가 없고 뒤에 변수가 이어지면 좌상첨자다. /// 단, 합/적분/극한 등 한정자 뒤의 첨자(예: ∑_{k=0}^{∞} 의 ^∞)는 좌상첨자가 아니다. +/// 원소 기호를 대문자표와 함께 emit하고 소비한 토큰 수를 돌려준다. +/// +/// 원소 기호가 아니면 `None`을 돌려주고 아무것도 쓰지 않는다. 한 글자짜리 +/// 원소 기호는 수학 변수와 글자가 겹치므로, 실제 원소 기호 목록에 있는 것만 +/// 통과시켜 좌상첨자가 붙은 일반 변수(`ⁿx`)를 건드리지 않는다. +pub(super) fn emit_element_symbol( + tokens: &[MathToken], + index: usize, + result: &mut Vec, +) -> Result, String> { + use crate::rules::science::elements::is_element; + let Some(MathToken::UpperVariable(upper)) = tokens.get(index) else { + return Ok(None); + }; + let lower = match tokens.get(index + 1) { + Some(MathToken::Variable(letter)) if letter.is_ascii_lowercase() => Some(*letter), + _ => None, + }; + let two: Option = lower.map(|letter| format!("{upper}{letter}")); + let (symbol, consumed) = match two { + Some(ref name) if is_element(name) => (name.as_str(), 2), + _ if is_element(&upper.to_string()) => return single(*upper, result), + _ => return Ok(None), + }; + result.push(32); + for letter in symbol.chars() { + result.push(crate::english::encode_english(letter.to_ascii_lowercase())?); + } + Ok(Some(consumed)) +} + +fn single(upper: char, result: &mut Vec) -> Result, String> { + result.push(32); + result.push(crate::english::encode_english(upper.to_ascii_lowercase())?); + Ok(Some(1)) +} + +/// 과학 제3항 — 동위원소 표기의 앞 첨자인가. 원자 번호와 질량수는 수이고, 첨자는 +/// 식의 처음이나 연산자·여는 괄호 뒤에 선다. `\mu_{0}I` 의 `₀` 는 μ 의 첨자이고 +/// `1/^{\circ}C` 의 `°` 는 수가 아니므로 원소 앞으로 옮기지 않는다. +pub(super) fn is_isotope_prescript( + tokens: &[MathToken], + index: usize, + content: &[MathToken], +) -> bool { + // LaTeX 는 앞 첨자를 빈 묶음 뒤에 적는다(`{}^{235}_{92}U`). + let after_empty_group = index >= 2 + && matches!(tokens[index - 1], MathToken::CloseParen(_)) + && matches!(tokens[index - 2], MathToken::OpenParen(_)); + !content.is_empty() + && content + .iter() + .all(|token| matches!(token, MathToken::Number(_))) + && (after_empty_group + || matches!( + prev_non_space(tokens, index), + None | Some( + MathToken::Operator(_) | MathToken::OpenParen(_) | MathToken::KoreanWord(_) + ) + )) +} + fn is_left_superscript_position(tokens: &[MathToken], index: usize) -> bool { let prev_blocks = matches!( prev_non_space(tokens, index), @@ -250,6 +312,29 @@ pub fn encode_superscript( // 좌상첨자는 단일 토큰이라도 그룹 괄호로 묶는다. let is_left_superscript = is_left_superscript_position(tokens, *i); + // 과학 제3항 — 원소 기호를 먼저 적고 원자 번호·질량수를 아래·위 첨자로 적는다 + // (⁷Li → ,li~#g, ²³⁵₉₂U → ,u;#ib~#bce). 수학 제18항 2의 좌상첨자는 제자리에 + // 괄호로 묶이지만 동위원소는 원소가 앞선다. + let atomic_number = match tokens.get(*i + 1) { + Some(MathToken::Subscript(sub)) if is_isotope_prescript(tokens, *i, sub) => Some(sub), + _ => None, + }; + if (is_left_superscript || atomic_number.is_some()) + && is_isotope_prescript(tokens, *i, sup_content) + { + let base = *i + 1 + usize::from(atomic_number.is_some()); + if let Some(consumed) = emit_element_symbol(tokens, base, result)? { + if let Some(sub) = atomic_number { + result.push(48); + engine.encode_tokens(sub, result)?; + } + result.push(24); + engine.encode_tokens(sup_content, result)?; + *i = base + consumed; + return Ok(false); + } + } + result.push(24); if wrapped_simple_index { // 본문 그대로 emit하여 ⠦⠴(MathParen) 보존. @@ -267,7 +352,19 @@ pub fn encode_superscript( pub struct SuperscriptRule; +static META_SUPERSCRIPTRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "18", + subsection: None, + name: "math_superscript", + standard_ref: "2024 Korean Braille Standard, 수학 제18항", + description: "위첨자", +}; + impl MathTokenRule for SuperscriptRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_SUPERSCRIPTRULE + } + fn name(&self) -> &'static str { "SuperscriptRule" } @@ -306,6 +403,32 @@ mod tests { crate::encode(input).unwrap_or_default() } + /// 과학 제3항 — 식 첫머리의 수 첨자만 원소 뒤로 옮긴다. + #[rstest::rstest] + #[case::mass_number("⁷Li", "⠠⠇⠊⠘⠼⠛")] + #[case::atomic_and_mass_number_in_a_sentence( + "우라늄 ${}^{235}_{92}U$가 있다", + "⠍⠐⠣⠉⠩⠢⠀⠀⠠⠥⠰⠼⠊⠃⠘⠼⠃⠉⠑⠀⠀⠫⠀⠕⠌⠊" + )] + #[case::greek_base_keeps_its_script("$\\mu_{0}I$", "⠨⠍⠰⠷⠼⠚⠾⠠⠊")] + #[case::degree_is_not_a_mass_number("$1/^{\\circ}C$", "⠼⠁⠌⠘⠷⠸⠴⠾⠠⠉")] + fn moves_only_isotope_numbers_behind_the_element(#[case] input: &str, #[case] expected: &str) { + let braille: String = enc(input) + .iter() + .map(|cell| crate::unicode::encode_unicode(*cell)) + .collect(); + assert_eq!(braille, expected); + } + + #[rstest::rstest] + #[case::capital_that_is_no_element(vec![MathToken::UpperVariable('Q')])] + #[case::pair_that_is_no_element(vec![MathToken::UpperVariable('Q'), MathToken::Variable('x')])] + fn leaves_capitals_that_are_no_element_symbol(#[case] tokens: Vec) { + let mut result = Vec::new(); + assert_eq!(emit_element_symbol(&tokens, 0, &mut result), Ok(None)); + assert!(result.is_empty()); + } + #[test] fn is_simple_signed_number_paths() { let with_ascii_minus = vec![MathToken::Operator('-'), MathToken::Number("1".into())]; diff --git a/libs/braillify/src/rules/math/rule_19.rs b/libs/braillify/src/rules/math/rule_19.rs index b95c715a..46932958 100644 --- a/libs/braillify/src/rules/math/rule_19.rs +++ b/libs/braillify/src/rules/math/rule_19.rs @@ -169,6 +169,18 @@ pub fn encode_subscript( return Ok(false); } + // 과학 제3항 — 원소 기호를 먼저 적고 원자 번호를 아래 첨자로 적는다(₈O → ,o;#h). + // 수학 제19항 2의 좌하첨자는 제자리에 묶이지만 원자 번호는 원소가 앞선다. + if is_left_subscript_position(tokens, *i) + && super::rule_18::is_isotope_prescript(tokens, *i, content) + && let Some(consumed) = super::rule_18::emit_element_symbol(tokens, *i + 1, result)? + { + result.push(48); + engine.encode_tokens(content, result)?; + *i += 1 + consumed; + return Ok(false); + } + result.push(48); // 적분/합/곱(∫ ∑ ∏ 등) 한정자 뒤 첨자는 묶음 없이 본문 그대로 출력한다. // PDF 제51항 [붙임] — `\substack`로 펼쳐진 두 번째 이상 첨자도 동일한 한정자 @@ -270,7 +282,19 @@ fn needs_quantifier_trailing_space(tokens: &[MathToken], idx: usize) -> bool { pub struct SubscriptRule; +static META_SUBSCRIPTRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "19", + subsection: None, + name: "math_subscript", + standard_ref: "2024 Korean Braille Standard, 수학 제19항", + description: "아래첨자", +}; + impl MathTokenRule for SubscriptRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_SUBSCRIPTRULE + } + fn name(&self) -> &'static str { "SubscriptRule" } diff --git a/libs/braillify/src/rules/math/rule_2.rs b/libs/braillify/src/rules/math/rule_2.rs index 2bef2018..01caaf25 100644 --- a/libs/braillify/src/rules/math/rule_2.rs +++ b/libs/braillify/src/rules/math/rule_2.rs @@ -367,7 +367,19 @@ mod tests { pub struct OperatorRule; +static META_OPERATORRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "2", + subsection: None, + name: "math_operator", + standard_ref: "2024 Korean Braille Standard, 수학 제2항", + description: "연산 기호", +}; + impl MathTokenRule for OperatorRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_OPERATORRULE + } + fn name(&self) -> &'static str { "OperatorRule" } diff --git a/libs/braillify/src/rules/math/rule_24.rs b/libs/braillify/src/rules/math/rule_24.rs deleted file mode 100644 index 91c0700d..00000000 --- a/libs/braillify/src/rules/math/rule_24.rs +++ /dev/null @@ -1,38 +0,0 @@ -//! 수학 제24항 — 수열 표기 `{aₙ}`. -//! -//! 수열은 중괄호로 구간을 감싸고 항 기호(예: `aₙ`)를 내부에 배치한다. -//! 인코딩 파이프라인에서는 중괄호 경계와 첨자 정보를 분리해 후속 규칙에 전달한다. - -use crate::math_symbol_shortcut; - -pub fn is_sequence_brace(c: char) -> bool { - matches!(c, '{' | '}') -} - -pub fn encode_sequence_brace(c: char, result: &mut Vec) -> Result<(), String> { - math_symbol_shortcut::encode_char_math_symbol_shortcut(c) - .map(|encoded| result.extend_from_slice(encoded)) -} - -#[cfg(test)] -mod tests { - use super::*; - - fn is_sequence_notation_char(c: char) -> bool { - is_sequence_brace(c) || c == '\u{2099}' - } - - #[test] - fn detects_sequence_braces() { - assert!(is_sequence_brace('{')); - assert!(is_sequence_brace('}')); - } - - #[test] - fn detects_sequence_notation_chars() { - assert!(is_sequence_notation_char('{')); - assert!(is_sequence_notation_char('}')); - assert!(is_sequence_notation_char('\u{2099}')); // subscript n - assert!(!is_sequence_notation_char('a')); - } -} diff --git a/libs/braillify/src/rules/math/rule_47.rs b/libs/braillify/src/rules/math/rule_47.rs index dabc0b5f..a05d1d9b 100644 --- a/libs/braillify/src/rules/math/rule_47.rs +++ b/libs/braillify/src/rules/math/rule_47.rs @@ -276,7 +276,19 @@ fn next_is_lim_body(tokens: &[MathToken], idx: usize) -> bool { pub struct FunctionNameRule; +static META_FUNCTIONNAMERULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "47", + subsection: None, + name: "math_function_name", + standard_ref: "2024 Korean Braille Standard, 수학 제47항", + description: "함수 이름", +}; + impl MathTokenRule for FunctionNameRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_FUNCTIONNAMERULE + } + fn name(&self) -> &'static str { "FunctionNameRule" } diff --git a/libs/braillify/src/rules/math/rule_5.rs b/libs/braillify/src/rules/math/rule_5.rs index b8c66ec5..855f97b4 100644 --- a/libs/braillify/src/rules/math/rule_5.rs +++ b/libs/braillify/src/rules/math/rule_5.rs @@ -1,6 +1,7 @@ -//! 수학 제5항 — 비례식 기호. +//! 과학 제29항 — 비례 기호. //! -//! 비례식에서 사용하는 ∝(U+221D) 기호를 단축표 인코딩으로 준비한다. +//! ∝(U+221D)는 과학 제29항이 `+3`으로 정한 기호다. 수학 제5항의 비 기호 +//! ∶(U+2236)와는 다른 조문이므로 단축표에서도 따로 귀속한다. use crate::math_symbol_shortcut; diff --git a/libs/braillify/src/rules/math/rule_53.rs b/libs/braillify/src/rules/math/rule_53.rs index 8e746fdd..9846eba2 100644 --- a/libs/braillify/src/rules/math/rule_53.rs +++ b/libs/braillify/src/rules/math/rule_53.rs @@ -11,7 +11,19 @@ pub fn encode_prime(result: &mut Vec) { pub struct PrimeRule; +static META_PRIMERULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "53", + subsection: None, + name: "math_prime", + standard_ref: "2024 Korean Braille Standard, 수학 제53항", + description: "프라임 기호", +}; + impl MathTokenRule for PrimeRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_PRIMERULE + } + fn name(&self) -> &'static str { "PrimeRule" } diff --git a/libs/braillify/src/rules/math/rule_54.rs b/libs/braillify/src/rules/math/rule_54.rs index 9fb9276d..3ed57b87 100644 --- a/libs/braillify/src/rules/math/rule_54.rs +++ b/libs/braillify/src/rules/math/rule_54.rs @@ -43,7 +43,19 @@ fn is_slash_operator(tok: Option<&MathToken>) -> bool { pub struct PartialDerivativeFractionRule; +static META_PARTIALDERIVATIVEFRACTIONRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "54", + subsection: None, + name: "math_partial_derivative", + standard_ref: "2024 Korean Braille Standard, 수학 제54항", + description: "편미분 분수", +}; + impl MathTokenRule for PartialDerivativeFractionRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_PARTIALDERIVATIVEFRACTIONRULE + } + fn name(&self) -> &'static str { "PartialDerivativeFractionRule" } diff --git a/libs/braillify/src/rules/math/rule_57.rs b/libs/braillify/src/rules/math/rule_57.rs index 2625de6d..cbf75b16 100644 --- a/libs/braillify/src/rules/math/rule_57.rs +++ b/libs/braillify/src/rules/math/rule_57.rs @@ -40,7 +40,19 @@ fn split_definite_integral_bounds( pub struct DefiniteIntegralRule; +static META_DEFINITEINTEGRALRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "57", + subsection: None, + name: "math_definite_integral", + standard_ref: "2024 Korean Braille Standard, 수학 제57항", + description: "정적분", +}; + impl MathTokenRule for DefiniteIntegralRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_DEFINITEINTEGRALRULE + } + fn name(&self) -> &'static str { "DefiniteIntegralRule" } diff --git a/libs/braillify/src/rules/math/rule_6.rs b/libs/braillify/src/rules/math/rule_6.rs index 6b5107e6..2f5d5cea 100644 --- a/libs/braillify/src/rules/math/rule_6.rs +++ b/libs/braillify/src/rules/math/rule_6.rs @@ -62,7 +62,19 @@ pub fn find_matching_paren(tokens: &[MathToken], start: usize) -> Option pub struct BracketRule; +static META_BRACKETRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "6", + subsection: None, + name: "math_bracket", + standard_ref: "2024 Korean Braille Standard, 수학 제6항", + description: "괄호", +}; + impl MathTokenRule for BracketRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_BRACKETRULE + } + fn name(&self) -> &'static str { "BracketRule" } diff --git a/libs/braillify/src/rules/math/rule_7.rs b/libs/braillify/src/rules/math/rule_7.rs index c88d8d81..6345be96 100644 --- a/libs/braillify/src/rules/math/rule_7.rs +++ b/libs/braillify/src/rules/math/rule_7.rs @@ -47,7 +47,19 @@ pub struct FractionReversalRule; pub struct GroupedFractionReversalRule; +static META_GROUPEDFRACTIONREVERSALRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "7", + subsection: None, + name: "math_grouped_fraction", + standard_ref: "2024 Korean Braille Standard, 수학 제7항", + description: "묶음 분수 - 분모 먼저", +}; + impl MathTokenRule for GroupedFractionReversalRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_GROUPEDFRACTIONREVERSALRULE + } + fn name(&self) -> &'static str { "GroupedFractionReversalRule" } @@ -152,7 +164,19 @@ fn find_simple_right_end(tokens: &[MathToken], start: usize) -> usize { i } +static META_FRACTIONREVERSALRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "7", + subsection: None, + name: "math_fraction", + standard_ref: "2024 Korean Braille Standard, 수학 제7항", + description: "분수 - 분모 먼저", +}; + impl MathTokenRule for FractionReversalRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_FRACTIONREVERSALRULE + } + fn name(&self) -> &'static str { "FractionReversalRule" } @@ -194,7 +218,19 @@ impl MathTokenRule for FractionReversalRule { /// `f/x` → `x/f` (분모 먼저). 안전을 위해 prev가 OpenParen 또는 comma일 때만 발동. pub struct VariableFractionInListRule; +static META_VARIABLEFRACTIONINLISTRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "7", + subsection: None, + name: "math_variable_fraction", + standard_ref: "2024 Korean Braille Standard, 수학 제7항", + description: "나열 속 변수 분수", +}; + impl MathTokenRule for VariableFractionInListRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_VARIABLEFRACTIONINLISTRULE + } + fn name(&self) -> &'static str { "VariableFractionInListRule" } @@ -255,7 +291,19 @@ impl MathTokenRule for VariableFractionInListRule { pub struct ConditionalProbFractionRule; +static META_CONDITIONALPROBFRACTIONRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "7", + subsection: None, + name: "math_conditional_fraction", + standard_ref: "2024 Korean Braille Standard, 수학 제7항", + description: "조건부 확률 분수", +}; + impl MathTokenRule for ConditionalProbFractionRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_CONDITIONALPROBFRACTIONRULE + } + fn name(&self) -> &'static str { "ConditionalProbFractionRule" } diff --git a/libs/braillify/src/rules/math/rule_8.rs b/libs/braillify/src/rules/math/rule_8.rs index f7e1b453..9944ccee 100644 --- a/libs/braillify/src/rules/math/rule_8.rs +++ b/libs/braillify/src/rules/math/rule_8.rs @@ -66,7 +66,19 @@ pub fn encode_decimal_point( pub struct DecimalPointRule; +static META_DECIMALPOINTRULE: crate::rules::RuleMeta = crate::rules::RuleMeta { + section: "8", + subsection: None, + name: "math_decimal_point", + standard_ref: "2024 Korean Braille Standard, 수학 제8항", + description: "소수점", +}; + impl MathTokenRule for DecimalPointRule { + fn meta(&self) -> &'static crate::rules::RuleMeta { + &META_DECIMALPOINTRULE + } + fn name(&self) -> &'static str { "DecimalPointRule" } diff --git a/libs/braillify/src/rules/mod.rs b/libs/braillify/src/rules/mod.rs index 82ab12c3..851702d1 100644 --- a/libs/braillify/src/rules/mod.rs +++ b/libs/braillify/src/rules/mod.rs @@ -29,12 +29,14 @@ pub mod token; pub mod token_engine; pub mod token_rule; pub mod token_rules; +pub mod trace; pub mod traits; // ── Rule domains ──────────────────────────────────────── pub mod english_ueb; // 통일영어점자 규정 (Unified English Braille) pub mod korean; // 한글 점자 규정 (Korean Braille rules) pub mod math; // 수학 점자 규정 (Math Braille rules) +pub mod science; // 과학 점자 규정 (Science Braille rules) /// Metadata identifying a braille rule and its source in the standard. #[derive(Debug, Clone, Copy, PartialEq, Eq)] diff --git a/libs/braillify/src/rules/science/bond_lines.rs b/libs/braillify/src/rules/science/bond_lines.rs new file mode 100644 index 00000000..cbfe1265 --- /dev/null +++ b/libs/braillify/src/rules/science/bond_lines.rs @@ -0,0 +1,72 @@ +//! 과학 제13항 3·4 — 결합선만으로 그린 환핵 그림의 공간 표기 형식. +//! +//! 묵자 그림을 한 칸씩 옮긴다. 오른쪽 위로 가는 사선은 ⠜, 왼쪽 위로 가는 사선은 +//! ⠣, 세로선은 ⠸ 이고 빈칸은 그대로 둔다. + +use crate::unicode::decode_unicode; + +const LINE_BREAK: u8 = 255; + +fn line_cell(c: char) -> Option { + match c { + '/' => Some(decode_unicode('⠜')), + '\\' => Some(decode_unicode('⠣')), + '|' => Some(decode_unicode('⠸')), + ' ' => Some(0), + _ => None, + } +} + +/// 결합선만으로 된 여러 줄 그림이면 그 점자. 아니면 `None`. +pub(crate) fn encode(text: &str) -> Option> { + if !text.contains('\n') || text.trim().is_empty() { + return None; + } + let mut out = Vec::new(); + for (at, line) in text.split('\n').enumerate() { + if at > 0 { + out.push(LINE_BREAK); + } + for c in line.trim_end().chars() { + out.push(line_cell(c)?); + } + } + Some(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn braille(text: &str) -> Option { + encode(text).map(|cells| { + cells + .into_iter() + .map(|c| { + if c == LINE_BREAK { + '\n' + } else { + crate::unicode::encode_unicode(c) + } + }) + .collect() + }) + } + + #[rstest::rstest] + #[case::slopes(" /\\\n/ \\", " ⠜⠣\n⠜⠀⠀⠣")] + #[case::double_vertical("| ||\n| ||", "⠸⠀⠀⠸⠸\n⠸⠀⠀⠸⠸")] + fn copies_the_drawing_cell_by_cell(#[case] text: &str, #[case] expected: &str) { + assert_eq!( + braille(text).as_deref(), + Some(expected.replace(' ', "⠀").as_str()) + ); + } + + #[rstest::rstest] + #[case::one_line("/\\")] + #[case::with_atoms(" H\n/ \\")] + fn leaves_other_text_alone(#[case] text: &str) { + assert_eq!(braille(text), None); + } +} diff --git a/libs/braillify/src/rules/science/circuit.rs b/libs/braillify/src/rules/science/circuit.rs new file mode 100644 index 00000000..23d1f22a --- /dev/null +++ b/libs/braillify/src/rules/science/circuit.rs @@ -0,0 +1,63 @@ +//! 과학 [부록] 9 — 전기·전자 회로 및 소자 기호. +//! +//! 묵자가 그림이라 `[그림: 명칭]` 으로 받는다. 점형은 모두 ⠯ 과 ⠽ 사이에 소자의 +//! 약호를 적은 것이다. + +use crate::unicode::decode_unicode; + +/// 글 전체가 회로 소자 그림 하나이면 그 점형. +pub(crate) fn encode(text: &str) -> Option> { + let abbreviation = match super::picture_name(text)? { + "전지(직류 전원)" => "⠢⠠⠃⠔", + "교류 전원" => "⠠⠁⠉", + "전구" => "⠠⠇⠏", + "열린 스위치" => "⠕⠠⠎", + "닫힌 스위치" => "⠉⠠⠎", + "전류계" => "⠠⠁", + "전압계" => "⠠⠧", + "직류 전류계" => "⠙⠠⠁", + "직류 전압계" => "⠙⠠⠧", + "교류 전류계" => "⠁⠠⠁", + "교류 전압계" => "⠁⠠⠧", + "검류계" => "⠠⠛⠁", + "변압기" => "⠠⠞", + "콘덴서" => "⠠⠉", + "가변 콘덴서" => "⠧⠠⠉", + "저항" => "⠠⠗", + "가변 저항" => "⠧⠠⠗", + "코일" => "⠠⠇", + "가변 코일" => "⠧⠠⠇", + "다이오드" => "⠠⠙", + "발광 다이오드" => "⠠⠇⠑⠙", + "N-P-N형 트랜지스터" => "⠠⠝⠏⠝", + "P-N-P형 트랜지스터" => "⠠⠏⠝⠏", + _ => return None, + }; + let mut out = vec![decode_unicode('⠯')]; + out.extend(abbreviation.chars().map(decode_unicode)); + out.push(decode_unicode('⠽')); + Some(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[rstest::rstest] + #[case::battery("[그림: 전지(직류 전원)]", "⠯⠢⠠⠃⠔⠽")] + #[case::open_switch("[그림: 열린 스위치]", "⠯⠕⠠⠎⠽")] + #[case::direct_current_ammeter("[그림: 직류 전류계]", "⠯⠙⠠⠁⠽")] + #[case::light_emitting_diode("[그림: 발광 다이오드]", "⠯⠠⠇⠑⠙⠽")] + fn writes_the_element_sign(#[case] text: &str, #[case] expected: &str) { + let want: Vec = expected.chars().map(decode_unicode).collect(); + assert_eq!(encode(text), Some(want)); + } + + #[rstest::rstest] + #[case::name_alone("전구")] + #[case::unknown_element("[그림: 퓨즈]")] + #[case::unclosed("[그림: 전구")] + fn leaves_other_text_alone(#[case] text: &str) { + assert_eq!(encode(text), None); + } +} diff --git a/libs/braillify/src/rules/science/conditions.rs b/libs/braillify/src/rules/science/conditions.rs new file mode 100644 index 00000000..c344a04b --- /dev/null +++ b/libs/braillify/src/rules/science/conditions.rs @@ -0,0 +1,127 @@ +//! 과학 제18항 5 — 화살표 위아래에 적은 반응 조건. 위의 내용은 화살표 앞에, +//! 아래의 내용은 화살표 뒤에 ⠸⠷ ⠸⠾ 으로 묶어 붙여 적는다. +//! +//! 입력은 조건을 화살표와 같은 칸에 맞춰 식의 위아래 줄에 적은 글이다. + +use crate::unicode::decode_unicode; + +const ARROWS: [char; 4] = ['→', '←', '⇄', '⇌']; + +/// 조건 줄의 글자가 화살표 칸에서 시작하면 그 조건. +fn condition_over(line: &str, arrow_column: usize) -> Option<&str> { + let start = line.chars().take_while(|c| *c == ' ').count(); + let text = line.trim(); + (!text.is_empty() && start.abs_diff(arrow_column) <= 1).then_some(text) +} + +fn wrap(condition: &[u8]) -> Vec { + let mut out = vec![decode_unicode('⠸'), decode_unicode('⠷')]; + out.extend_from_slice(condition); + out.extend([decode_unicode('⠸'), decode_unicode('⠾')]); + out +} + +/// 식 한 줄과 그 위·아래 조건 줄로 된 글을 적는다. 그런 모양이 아니면 `None`. +pub(crate) fn encode( + text: &str, + encode_formula: impl Fn(&str) -> Option>, + encode_condition: impl Fn(&str) -> Option>, +) -> Option> { + let lines: Vec<&str> = text.split('\n').collect(); + let formula_at = lines + .iter() + .position(|line| line.chars().filter(|c| ARROWS.contains(c)).count() == 1)?; + if lines.len() > 3 || formula_at > 1 { + return None; + } + let formula = lines[formula_at]; + let arrow_column = formula.chars().position(|c| ARROWS.contains(&c))?; + let above = formula_at + .checked_sub(1) + .map(|at| condition_over(lines[at], arrow_column)); + let below = lines + .get(formula_at + 1) + .map(|line| condition_over(line, arrow_column)); + if above.flatten().is_none() && below.flatten().is_none() + || above.is_some_and(|c| c.is_none()) + || below.is_some_and(|c| c.is_none()) + { + return None; + } + + // 화살표의 점형은 화살표만 바꾼 식과 견주어 찾는다. + let arrow = formula.chars().nth(arrow_column)?; + let other = if arrow == '→' { '←' } else { '→' }; + let cells = encode_formula(formula)?; + let swapped = encode_formula(&formula.replace(arrow, &other.to_string()))?; + let start = cells + .iter() + .zip(&swapped) + .take_while(|(a, b)| a == b) + .count(); + let end = cells[start..] + .iter() + .position(|cell| *cell == 0) + .map_or(cells.len(), |p| start + p); + if start == cells.len() || start == end { + return None; + } + + let mut out = cells[..start].to_vec(); + if let Some(condition) = above.flatten() { + out.extend(wrap(&encode_condition(condition)?)); + } + out.extend_from_slice(&cells[start..end]); + if let Some(condition) = below.flatten() { + out.extend(wrap(&encode_condition(condition)?)); + } + out.extend_from_slice(&cells[end..]); + Some(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn formula(text: &str) -> Option> { + Some( + text.chars() + .map(|c| match c { + ' ' => 0, + '→' => decode_unicode('⠒'), + '⇄' => decode_unicode('⠶'), + _ => decode_unicode('⠁'), + }) + .collect(), + ) + } + + fn condition(_: &str) -> Option> { + Some(vec![decode_unicode('⠃')]) + } + + fn braille(cells: Vec) -> String { + cells + .into_iter() + .map(crate::unicode::encode_unicode) + .collect() + } + + #[rstest::rstest] + #[case::both(" x\na ⇄ a\n y", "⠁⠀⠸⠷⠃⠸⠾⠶⠸⠷⠃⠸⠾⠀⠁")] + #[case::above_only(" x\na → a", "⠁⠀⠸⠷⠃⠸⠾⠒⠀⠁")] + fn places_conditions_around_the_arrow(#[case] text: &str, #[case] expected: &str) { + assert_eq!( + encode(text, formula, condition).map(braille).as_deref(), + Some(expected) + ); + } + + #[rstest::rstest] + #[case::one_line("a → a")] + #[case::not_over_the_arrow("x\na → a")] + #[case::no_arrow("x\na a")] + fn leaves_other_text_alone(#[case] text: &str) { + assert_eq!(encode(text, formula, condition), None); + } +} diff --git a/libs/braillify/src/rules/science/diagram.rs b/libs/braillify/src/rules/science/diagram.rs new file mode 100644 index 00000000..e76b1a6d --- /dev/null +++ b/libs/braillify/src/rules/science/diagram.rs @@ -0,0 +1,847 @@ +//! 과학 제9·10·13·15·16·25항 — 묵자에서 2차원으로 그린 구조식·전자 점식·가계도. +//! +//! 입력은 묵자 배치를 줄마다 옮긴 여러 줄 글이다. 구조식과 전자 점식은 기호 표기 +//! 형식(제10항 3, 제16항 3·4)으로 풀어 한 줄에 적고, 가계도는 세대마다 한 줄씩 +//! 적는다(제25항). + +use super::formula::{self, Item}; +use super::genotype::gene_cells; +use crate::unicode::decode_unicode; + +const LINE_BREAK: u8 = 255; + +/// 두 원소를 잇는 것 — 결합선(제10항 2)이나 전자(제16항 1). +#[derive(Clone, Copy)] +enum Link { + Bond(u8), + /// 전자 기호 하나에 든 전자 수와 그 기호의 개수(`::` 은 둘씩 둘). + Electrons(usize, usize), +} + +impl Link { + fn item(self) -> Item { + match self { + Link::Bond(order) => Item::Bond(order), + Link::Electrons(count, marks) => Item::Electrons(count * marks), + } + } +} + +fn electrons(mark: char) -> Option { + match mark { + ':' | '‥' => Some(2), + '⋮' => Some(3), + '∷' => Some(4), + _ => None, + } +} + +fn horizontal_link(gap: &[char]) -> Option { + let marks: Vec = gap.iter().copied().filter(|c| *c != ' ').collect(); + match marks.as_slice() { + [] => None, + ['-' | '–'] => Some(Link::Bond(1)), + ['='] => Some(Link::Bond(2)), + ['≡'] => Some(Link::Bond(3)), + dots => { + let counts: Vec = dots + .iter() + .map(|mark| electrons(*mark)) + .collect::>()?; + counts + .iter() + .all(|count| *count == counts[0]) + .then(|| Link::Electrons(counts[0], counts.len())) + } + } +} + +fn vertical_link(mark: char) -> Option { + match mark { + '|' | '│' => Some(Link::Bond(1)), + '‖' => Some(Link::Bond(2)), + '⦀' => Some(Link::Bond(3)), + _ => electrons(mark).map(|count| Link::Electrons(count, 1)), + } +} + +/// 원소 앞뒤에 홀로 붙은 전자. 비어 있으면 `Some(None)`, 결합선이면 `None`. +fn dangling(gap: &[char]) -> Option> { + match horizontal_link(gap) { + None if gap.iter().all(|c| *c == ' ') => Some(None), + Some(link @ Link::Electrons(..)) => Some(Some(link)), + _ => None, + } +} + +fn is_atom_char(ch: char) -> bool { + ch.is_ascii_alphabetic() || ('₀'..='₉').contains(&ch) +} + +/// 원소 기호와 그 아래 첨자로 된 묶음(`C`, `OH`, `CH₃`) 하나. +struct Atom { + row: usize, + columns: std::ops::Range, + items: Vec, +} + +/// 한 줄에 가로로 이어진 원소들. +struct Chain { + lead: Option, + atoms: Vec, + links: Vec, + trail: Option, +} + +fn atom_row(row: usize, chars: &[char], atoms: &mut Vec) -> Option { + let mut gaps: Vec> = vec![Vec::new()]; + let mut members = Vec::new(); + let mut at = 0; + while at < chars.len() { + if !is_atom_char(chars[at]) { + gaps.last_mut()?.push(chars[at]); + at += 1; + continue; + } + let start = at; + while chars.get(at).copied().is_some_and(is_atom_char) { + at += 1; + } + let items = formula::parse(&chars[start..at].iter().collect::())?; + if !items + .iter() + .all(|item| matches!(item, Item::Capital { element: true, .. } | Item::Sub(_))) + { + return None; + } + members.push(atoms.len()); + atoms.push(Atom { + row, + columns: start..at, + items, + }); + gaps.push(Vec::new()); + } + let links = gaps[1..gaps.len() - 1] + .iter() + .map(|gap| horizontal_link(gap)) + .collect::>>()?; + Some(Chain { + lead: dangling(&gaps[0])?, + atoms: members, + links, + trail: dangling(&gaps[gaps.len() - 1])?, + }) +} + +const UP: usize = 0; +const DOWN: usize = 1; + +/// 원소 하나에 달린 것 — 위·아래의 결합과 그 끝 원소, 같은 줄의 왼쪽·오른쪽. +#[derive(Default)] +struct Neighbours { + vertical: [Option<(Link, Option)>; 2], + left: Option<(Link, usize)>, + right: Option<(Link, usize)>, +} + +struct Walk<'a> { + atoms: &'a [Atom], + neighbours: &'a [Neighbours], + seen: Vec, + vertical_links: usize, + items: Vec, +} + +impl Walk<'_> { + fn visit(&mut self, atom: usize) -> Option<()> { + if std::mem::replace(&mut self.seen[atom], true) { + return None; + } + self.items.extend(self.atoms[atom].items.iter().cloned()); + Some(()) + } + + fn link(&mut self, link: Link, atom: usize) -> Option<()> { + self.items.push(link.item()); + self.visit(atom) + } + + /// 제10항 3 가·나, 제16항 3 — 중심 원소의 위(⠬)와 아래(⠩)를 적는다. 적었으면 + /// `true` — 그 뒤 오른쪽 원소나 전자 앞에 ⠤ 을 적는다(제10항 3 라). + fn branches(&mut self, atom: usize) -> Option { + let mut branched = false; + for (side, mark) in [(UP, '⠬'), (DOWN, '⠩')] { + let Some((link, target)) = self.neighbours[atom].vertical[side] else { + continue; + }; + branched = true; + self.vertical_links += 1; + self.items.push(Item::Branch(mark)); + self.items.push(link.item()); + if let Some(target) = target { + self.visit(target)?; + self.side_chain(target, side)?; + } + } + Some(branched) + } + + /// 제10항 3 다, 제16항 4 — 측쇄에 또 달린 측쇄는 왼쪽 ⠣, 오른쪽 ⠜ 을 먼저 적는다. + fn side_chain(&mut self, atom: usize, side: usize) -> Option<()> { + if self.neighbours[atom].vertical[side].is_some() { + return None; + } + for (mark, rightward) in [('⠣', false), ('⠜', true)] { + let step = |n: &Neighbours| if rightward { n.right } else { n.left }; + let mut next = step(&self.neighbours[atom]); + if next.is_some() { + self.items.push(Item::Branch(mark)); + } + while let Some((link, target)) = next { + self.link(link, target)?; + next = step(&self.neighbours[target]); + } + } + Some(()) + } + + fn main_chain(&mut self, chain: &Chain) -> Option<()> { + self.items.extend(chain.lead.map(Link::item)); + let mut branched = false; + for (at, &atom) in chain.atoms.iter().enumerate() { + if at > 0 { + self.right_of(branched, chain.links[at - 1]); + } + self.visit(atom)?; + branched = self.branches(atom)?; + } + if let Some(trail) = chain.trail { + self.right_of(branched, trail); + } + Some(()) + } + + fn right_of(&mut self, branched: bool, link: Link) { + if branched { + self.items.push(Item::Branch('⠤')); + } + self.items.push(link.item()); + } +} + +struct Mark { + row: usize, + column: usize, + link: Link, +} + +struct Diagram { + atoms: Vec, + chains: Vec, + marks: Vec, + neighbours: Vec, + /// 도식 위에 따로 적은 화학식(제11항의 `CH₄`) — 첫 줄에서 어디에도 잇지 않은 원소. + caption: Option, + main: usize, +} + +fn parse(text: &str) -> Option { + let mut atoms = Vec::new(); + let mut chains = Vec::new(); + let mut marks = Vec::new(); + for (row, line) in text.lines().enumerate() { + let chars: Vec = line.chars().collect(); + if chars.iter().copied().any(is_atom_char) { + chains.push(atom_row(row, &chars, &mut atoms)?); + continue; + } + let row_marks: Vec<(usize, char)> = chars + .iter() + .copied() + .enumerate() + .filter(|(_, ch)| *ch != ' ') + .collect(); + if row_marks.is_empty() { + return None; + } + for (column, mark) in row_marks { + marks.push(Mark { + row, + column, + link: vertical_link(mark)?, + }); + } + } + if marks.is_empty() { + return None; + } + + let mut neighbours: Vec = atoms.iter().map(|_| Neighbours::default()).collect(); + for chain in &chains { + for (at, link) in chain.links.iter().enumerate() { + let (left, right) = (chain.atoms[at], chain.atoms[at + 1]); + neighbours[left].right = Some((*link, right)); + neighbours[right].left = Some((*link, left)); + } + } + for mark in &marks { + let above = atom_at(&atoms, mark.row.checked_sub(1), mark.column); + let below = atom_at(&atoms, Some(mark.row + 1), mark.column); + match (above, below, mark.link) { + (None, None, _) | (None, _, Link::Bond(_)) | (_, None, Link::Bond(_)) => return None, + _ => {} + } + for (end, side, other) in [(above, DOWN, below), (below, UP, above)] { + if let Some(end) = end { + let slot = &mut neighbours[end].vertical[side]; + if slot.replace((mark.link, other)).is_some() { + return None; + } + } + } + } + + let isolated = |atom: usize| { + let n = &neighbours[atom]; + n.left.is_none() && n.right.is_none() && n.vertical.iter().all(Option::is_none) + }; + let caption = chains + .first() + .filter(|chain| { + chains.len() > 1 + && chain.atoms.len() == 1 + && chain.lead.is_none() + && chain.trail.is_none() + && atoms[chain.atoms[0]].row == 0 + && isolated(chain.atoms[0]) + }) + .map(|chain| chain.atoms[0]); + let main = (0..chains.len()) + .filter(|at| caption != Some(chains[*at].atoms[0])) + .min_by_key(|at| { + let first = &atoms[chains[*at].atoms[0]]; + (first.columns.start, first.row) + })?; + Some(Diagram { + atoms, + chains, + marks, + neighbours, + caption, + main, + }) +} + +fn atom_at(atoms: &[Atom], row: Option, column: usize) -> Option { + let row = row?; + atoms + .iter() + .position(|atom| atom.row == row && atom.columns.contains(&column)) +} + +/// 과학 제9항 1·제10항 3·제15항 1·제16항 — 2차원 구조식과 전자 점식을 기호 표기 +/// 형식으로 적는다. 왼쪽에서 오른쪽으로, 위쪽에서 아래쪽으로 풀어 적는다. 도식 +/// 위의 화학식은 적지 않는다(제10항의 예). +fn symbol_form(diagram: &Diagram) -> Option> { + let side_electrons = + diagram.chains.iter().enumerate().any(|(at, chain)| { + at != diagram.main && (chain.lead.is_some() || chain.trail.is_some()) + }); + if side_electrons { + return None; + } + let mut walk = Walk { + atoms: &diagram.atoms, + neighbours: &diagram.neighbours, + seen: vec![false; diagram.atoms.len()], + vertical_links: 0, + items: Vec::new(), + }; + if let Some(caption) = diagram.caption { + walk.seen[caption] = true; + } + walk.main_chain(&diagram.chains[diagram.main])?; + let complete = walk.seen.iter().all(|seen| *seen) && walk.vertical_links == diagram.marks.len(); + complete + .then(|| formula::encode(&walk.items).ok()) + .flatten() +} + +/// 전자 기호마다 전자 수만큼 ⠔ 을 적고 기호 사이를 한 칸 띄운다(제17항 2·3). +fn electron_cells(count: usize, marks: usize) -> Vec { + let mut out = Vec::new(); + for mark in 0..marks { + if mark > 0 { + out.push(0); + } + out.extend(std::iter::repeat_n(decode_unicode('⠔'), count)); + } + out +} + +fn bond_count(order: u8) -> u8 { + decode_unicode(match order { + 1 => '⠂', + 2 => '⠆', + _ => '⠒', + }) +} + +/// 공간 표기 형식의 결합선과 전자 — 가로 결합선 ⠠⠤, 세로 결합선 ⠸ 뒤에 결합 수를 +/// 적는다(제11항 4). +fn link_cells(link: Link, vertical: bool) -> Vec { + match link { + Link::Bond(order) if vertical => vec![decode_unicode('⠸'), bond_count(order)], + Link::Bond(order) => vec![decode_unicode('⠠'), decode_unicode('⠤'), bond_count(order)], + Link::Electrons(count, marks) => electron_cells(count, marks), + } +} + +struct Row { + cells: Vec, + starts: Vec<(usize, usize)>, +} + +type Encoder = fn(&[Item]) -> Result, String>; + +/// 과학 제13항 6 — 세로 결합선은 원소 묶음에서 수소가 아닌 첫 원소(`H₂C` 의 `C`)의 첫 +/// 칸에 닿는다. 그 칸이 묶음의 몇째 칸인지. +fn bonding_cell(items: &[Item], encode: Encoder) -> usize { + let Some(element) = items + .iter() + .position(|item| matches!(item, Item::Capital { symbol, .. } if symbol != "H")) + else { + return 0; + }; + let through = encode(&items[..=element]).map_or(0, |cells| cells.len()); + let alone = encode(&items[element..=element]).map_or(0, |cells| cells.len()); + through.saturating_sub(alone) +} + +fn lay_out_chain(chain: &Chain, atom_cells: &[Vec], dots: bool) -> Row { + let mut row = Row { + cells: Vec::new(), + starts: Vec::new(), + }; + let mut segments: Vec<(Vec, Option)> = Vec::new(); + segments.extend(chain.lead.map(|link| (link_cells(link, false), None))); + for (at, &atom) in chain.atoms.iter().enumerate() { + if at > 0 { + segments.push((link_cells(chain.links[at - 1], false), None)); + } + segments.push((atom_cells[atom].clone(), Some(atom))); + } + segments.extend(chain.trail.map(|link| (link_cells(link, false), None))); + for (at, (cells, atom)) in segments.into_iter().enumerate() { + if dots && at > 0 { + row.cells.push(0); + } + if let Some(atom) = atom { + row.starts.push((atom, row.cells.len())); + } + row.cells.extend(cells); + } + row +} + +/// 과학 제9항 2·제11항·제13항 6·제15항 2·제17항 — 구조식과 전자 점식을 모양대로 여러 +/// 줄에 적는다. 세로로 잇는 결합선·전자와 위아래 원소는 이어지는 원소의 첫 칸(대문자 +/// 기호표가 있으면 그 칸)에 맞추고, 도식 위의 화학식은 첫 줄에 적는다. +fn spatial_form(diagram: &Diagram) -> Option> { + let links = diagram + .chains + .iter() + .flat_map(|chain| chain.links.iter().chain(&chain.lead).chain(&chain.trail)) + .chain(diagram.marks.iter().map(|mark| &mark.link)); + let dots = links + .clone() + .any(|link| matches!(link, Link::Electrons(..))); + // 제11항 2 — 한 글자 원소가 붙어 나오는 묶음(`OH`, `CH₃`)이 있으면 대문자 구절표로 + // 감싸고 원소 기호마다의 대문자 기호표는 적지 않는다. + let passage = !dots + && diagram.atoms.iter().enumerate().any(|(at, atom)| { + Some(at) != diagram.caption + && atom + .items + .iter() + .filter(|item| matches!(item, Item::Capital { symbol, element: true, .. } if symbol.len() == 1)) + .count() + >= 2 + }); + let encoder = |at: usize| -> Encoder { + if passage && Some(at) != diagram.caption { + formula::encode_in_phrase + } else { + formula::encode + } + }; + let atom_cells = diagram + .atoms + .iter() + .enumerate() + .map(|(at, atom)| encoder(at)(&atom.items)) + .collect::, _>>() + .ok()?; + let rows: Vec = diagram + .chains + .iter() + .map(|chain| lay_out_chain(chain, &atom_cells, dots)) + .collect(); + let mut chain_of = vec![0; diagram.atoms.len()]; + let mut anchor_of = vec![0; diagram.atoms.len()]; + for (at, row) in rows.iter().enumerate() { + for &(atom, start) in &row.starts { + chain_of[atom] = at; + anchor_of[atom] = start + bonding_cell(&diagram.atoms[atom].items, encoder(atom)); + } + } + + let mut offsets: Vec> = vec![None; rows.len()]; + offsets[diagram.main] = Some(0); + let column = |atom: usize, offsets: &[Option]| { + offsets[chain_of[atom]].map(|offset| offset + anchor_of[atom] as isize) + }; + loop { + let mut placed = false; + for mark in &diagram.marks { + let above = atom_at(&diagram.atoms, mark.row.checked_sub(1), mark.column); + let below = atom_at(&diagram.atoms, Some(mark.row + 1), mark.column); + let (Some(above), Some(below)) = (above, below) else { + continue; + }; + match (column(above, &offsets), column(below, &offsets)) { + (Some(top), None) => { + offsets[chain_of[below]] = Some(top - anchor_of[below] as isize); + placed = true; + } + (None, Some(bottom)) => { + offsets[chain_of[above]] = Some(bottom - anchor_of[above] as isize); + placed = true; + } + (Some(top), Some(bottom)) if top != bottom => return None, + _ => {} + } + } + if !placed { + break; + } + } + let unplaced = offsets.iter().enumerate().any(|(at, offset)| { + offset.is_none() && Some(diagram.chains[at].atoms[0]) != diagram.caption + }); + if unplaced { + return None; + } + let margin = offsets.iter().flatten().copied().min().unwrap_or(0); + + let mut lines: Vec<(usize, Vec)> = Vec::new(); + for (at, row) in rows.iter().enumerate() { + let line_row = diagram.atoms[diagram.chains[at].atoms[0]].row; + match offsets[at] { + Some(offset) => { + let mut line = vec![0; (offset - margin) as usize]; + line.extend(&row.cells); + lines.push((line_row, line)); + } + None => lines.push((line_row, row.cells.clone())), + } + } + let mut mark_lines: std::collections::BTreeMap)>> = + std::collections::BTreeMap::new(); + for mark in &diagram.marks { + let anchor = atom_at(&diagram.atoms, mark.row.checked_sub(1), mark.column) + .or_else(|| atom_at(&diagram.atoms, Some(mark.row + 1), mark.column))?; + let at = (column(anchor, &offsets)? - margin) as usize; + mark_lines + .entry(mark.row) + .or_default() + .push((at, link_cells(mark.link, true))); + } + for (row, mut placed) in mark_lines { + placed.sort_by_key(|(at, _)| *at); + let mut line = Vec::new(); + for (at, cells) in placed { + (line.len() <= at).then_some(())?; + line.resize(at, 0); + line.extend(cells); + } + lines.push((row, line)); + } + lines.sort_by_key(|(row, _)| *row); + + let caption_row = diagram.caption.map(|atom| diagram.atoms[atom].row); + let (caption, body): (Vec<_>, Vec<_>) = lines + .into_iter() + .partition(|(row, _)| Some(*row) == caption_row); + let mut out_lines: Vec> = caption.into_iter().map(|(_, line)| line).collect(); + if passage { + out_lines.push(cells_of("⠐⠐⠿⠠⠠⠠")); + } + out_lines.extend(body.into_iter().map(|(_, line)| line)); + if passage { + out_lines.push(cells_of("⠐⠐⠿⠠⠄")); + } + Some(out_lines.join(&LINE_BREAK)) +} + +fn cells_of(pattern: &str) -> Vec { + pattern.chars().map(decode_unicode).collect() +} + +fn is_connector_row(line: &str) -> bool { + !line.trim().is_empty() + && line.chars().all(|ch| { + matches!( + ch, + ' ' | '|' | '│' | '┌' | '┐' | '┬' | '┴' | '─' | '└' | '┘' | '├' | '┤' + ) + }) +} + +/// 세대를 나타내는 문자 — 어버이 `P`, 자손 `F₁`·`F₂`. 홀로 선 `P` 는 통일영어점자 +/// §5.7.1 에 따라 1급 점자 기호표를 앞세운다. +fn generation_label(word: &str) -> Option> { + let items = formula::parse(word)?; + let cells = formula::encode(&items).ok()?; + match items.as_slice() { + [Item::Capital { symbol, .. }] if symbol == "P" => { + Some([vec![decode_unicode('⠰')], cells].concat()) + } + [Item::Capital { symbol, .. }, Item::Sub(number)] + if symbol == "F" && number.chars().all(|c| c.is_ascii_digit()) => + { + Some(cells) + } + _ => None, + } +} + +fn genes(word: &str) -> Option> { + let chars: Vec = word.chars().collect(); + let pairs = !chars.is_empty() + && chars.len().is_multiple_of(2) + && chars.iter().all(char::is_ascii_alphabetic) + && chars + .chunks(2) + .all(|pair| pair[0].eq_ignore_ascii_case(&pair[1])); + pairs.then(|| gene_cells(&chars)).flatten() +} + +/// 제25항 2·4 — 곱하기 기호 앞뒤는 한 칸씩 띄고, 한 세대의 여럿은 묶음 괄호로 묶는다. +fn generation(words: &[&str]) -> Option> { + match words { + [] => None, + [one] => genes(one), + [mother, "×", father] => Some( + [ + genes(mother)?, + vec![0, decode_unicode('⠡'), 0], + genes(father)?, + ] + .concat(), + ), + siblings => { + let mut out = vec![decode_unicode('⠷')]; + for (at, word) in siblings.iter().enumerate() { + if at > 0 { + out.push(0); + } + out.extend(genes(word)?); + } + out.push(decode_unicode('⠾')); + Some(out) + } + } +} + +/// 과학 제25항 — 가계도. 세대마다 한 줄에 세대 문자와 유전자를 두 칸 띄어 적고, +/// 다음 세대가 있으면 끝에 오른쪽 화살표를 적는다. +fn pedigree(text: &str) -> Option> { + let mut generations = Vec::new(); + for line in text.lines().filter(|line| !is_connector_row(line)) { + let words: Vec<&str> = line.split_whitespace().collect(); + let (label, content) = words.split_first()?; + generations.push((generation_label(label)?, generation(content)?)); + } + if generations.len() < 2 { + return None; + } + let last = generations.len() - 1; + let mut out = Vec::new(); + for (at, (label, content)) in generations.into_iter().enumerate() { + if at > 0 { + out.push(LINE_BREAK); + } + out.extend(label); + out.extend([0, 0]); + out.extend(content); + if at < last { + out.extend([0, decode_unicode('⠒'), decode_unicode('⠕')]); + } + } + Some(out) +} + +/// 여러 줄로 그린 과학 도식을 점자로 적는다. 도식이 아니면 `None`. `spatial` 이면 +/// 구조식과 전자 점식을 공간 표기 형식으로 적는다(제9항·제15항은 두 형식을 다 둔다). +pub(crate) fn encode(text: &str, spatial: bool) -> Option> { + if !text.contains('\n') { + return None; + } + parse(text) + .and_then(|diagram| { + if spatial { + spatial_form(&diagram) + } else { + symbol_form(&diagram) + } + }) + .or_else(|| pedigree(text)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn braille(cells: &[u8]) -> String { + cells + .iter() + .map(|cell| crate::unicode::encode_unicode(*cell)) + .collect() + } + + #[rstest::rstest] + #[case::ammonia(" H\n |\nH-N\n |\n H", "⠠⠠⠠⠓⠰⠂⠝⠬⠰⠂⠓⠩⠰⠂⠓⠠⠄")] + #[case::ethane_carbon(" H\n |\nH-C-H\n |\n H", "⠠⠠⠠⠓⠰⠂⠉⠬⠰⠂⠓⠩⠰⠂⠓⠤⠰⠂⠓⠠⠄")] + #[case::side_chain_left(" O\n ‖\nH-C\n |\nH-O", "⠠⠠⠠⠓⠰⠂⠉⠬⠰⠆⠕⠩⠰⠂⠕⠣⠰⠂⠓⠠⠄")] + #[case::grouped_atoms("CH₃ - CH₂\n |\n OH", "⠠⠠⠠⠉⠓⠰⠼⠉⠰⠂⠉⠓⠰⠼⠃⠩⠰⠂⠕⠓⠠⠄")] + #[case::lone_pairs(" ‥\n:F:\n ‥", "⠔⠔⠠⠋⠬⠔⠔⠩⠔⠔⠤⠔⠔")] + #[case::triple_bonds("C≡N\n⦀\nN", "⠠⠠⠠⠉⠩⠰⠒⠝⠤⠰⠒⠝⠠⠄")] + #[case::three_and_four_electrons("H⋮N\n ∷\n H", "⠠⠠⠠⠓⠔⠔⠔⠝⠩⠔⠔⠔⠔⠓⠠⠄")] + #[case::caption_is_left_out("OH₂\n H\n |\nH-O", "⠠⠠⠠⠓⠰⠂⠕⠬⠰⠂⠓⠠⠄")] + fn writes_diagrams_in_symbol_form(#[case] text: &str, #[case] expected: &str) { + assert_eq!( + encode(text, false).map(|cells| braille(&cells)).as_deref(), + Some(expected) + ); + } + + #[rstest::rstest] + #[case::ammonia( + " H\n |\nH-N\n |\n H", + " ,h\n _1\n,h,-1,n\n _1\n ,h" + )] + #[case::caption_first( + "NH₃\n H\n |\nH-N\n |\n H", + ",n,h;#c\n ,h\n _1\n,h,-1,n\n _1\n ,h" + )] + #[case::grouped_atoms_in_a_passage("CH₃-OH\n|\nH", "\"\"=,,,\nch;#c,-1oh\n_1\nh\n\"\"=,'")] + #[case::caption_outside_the_passage( + "CH₄O\nCH₃-OH\n|\nH", + ",,,ch;#do,'\n\"\"=,,,\nch;#c,-1oh\n_1\nh\n\"\"=,'" + )] + #[case::row_above_aligns_to_its_atom(" H-O\n |\nH-C", " ,h,-1,o\n _1\n,h,-1,c")] + #[case::electron_pairs(" ‥\n:F:F:", " 99\n99 ,f 99 ,f 99")] + #[case::shared_pair_between_rows( + " :O:\n ‥\n:O:P:O:", + " 99 ,o 99\n 99\n99 ,o 99 ,p 99 ,o 99" + )] + #[case::triple_vertical("N\n⦀\nN", ",n\n_3\n,n")] + #[case::three_electrons("N⋮N\n‥", ",n 999 ,n\n99")] + #[case::square_ring( + "H₂C - CH₂\n | |\nH₂C - CH₂", + "\"\"=,,,\nh;#b\"c,-1ch;#b\n _1 _1\nh;#b\"c,-1ch;#b\n\"\"=,'" + )] + #[case::bond_under_the_carbon( + "H₃C-OH\n |\n H", + "\"\"=,,,\nh;#c\"c,-1oh\n _1\n h\n\"\"=,'" + )] + fn writes_diagrams_in_spatial_form(#[case] text: &str, #[case] internal: &str) { + let expected: String = internal + .chars() + .map(|ch| match ch { + '\n' => '\n', + ' ' => '⠀', + ',' => '⠠', + '-' => '⠤', + '_' => '⠸', + '"' => '⠐', + '=' => '⠿', + '\'' => '⠄', + ';' => '⠰', + '#' => '⠼', + '1' => '⠂', + '2' => '⠆', + '3' => '⠒', + '9' => '⠔', + 'a' => '⠁', + 'b' => '⠃', + 'c' => '⠉', + 'd' => '⠙', + 'f' => '⠋', + 'h' => '⠓', + 'n' => '⠝', + 'o' => '⠕', + 'p' => '⠏', + _ => unreachable!("unmapped internal cell {ch}"), + }) + .collect(); + assert_eq!( + encode(text, true).map(|cells| braille(&cells)).as_deref(), + Some(expected.as_str()) + ); + } + + #[rstest::rstest] + #[case::unconnected_row("H\n|\nH\nO-H")] + #[case::rows_disagree("O:O\n‥ ‥\nO⋮O")] + #[case::pedigree_is_not_spatial("P BB × bb\n |\nF₁ Bb")] + fn leaves_other_text_alone_in_spatial_form(#[case] text: &str) { + assert_eq!(parse(text).and_then(|diagram| spatial_form(&diagram)), None); + } + + #[rstest::rstest] + #[case::one_line("H-O-H")] + #[case::no_vertical_link("H-O\nO-H")] + #[case::blank_row("H\n\n|\nH")] + #[case::unknown_mark("H\n*\nH")] + #[case::dangling_bond("H\n|")] + #[case::bond_into_nothing("H-N\n |")] + #[case::double_vertical("H H\n| |\n H")] + #[case::not_an_element("A\n|\nH")] + #[case::english_words("hello\n|\nworld")] + #[case::missing_link("H N\n|\nH")] + #[case::bond_before_atom("-H\n |\n H")] + #[case::side_row_electrons("H\n|\nH:")] + #[case::branch_beyond_branch("H\n|\nO\n|\nH")] + #[case::ring("H-O\n| |\nH-H")] + #[case::two_links_one_slot("OH\n||\nOH")] + #[case::unreached_atom("H\n|\nH\nO")] + #[case::mixed_marks("H-:O\n|\nH")] + fn leaves_other_text_alone(#[case] text: &str) { + assert_eq!(encode(text, false), None); + } + + #[rstest::rstest] + #[case::cross_then_offspring("P BB × bb\n |\nF₁ Bb", "⠰⠠⠏⠀⠀⠠⠠⠃⠃⠀⠡⠀⠃⠃⠀⠒⠕\n⠠⠋⠰⠼⠁⠀⠀⠠⠃⠃")] + #[case::siblings("F₁ Bb\n ┌──┐\nF₂ BB bb", "⠠⠋⠰⠼⠁⠀⠀⠠⠃⠃⠀⠒⠕\n⠠⠋⠰⠼⠃⠀⠀⠷⠠⠠⠃⠃⠀⠃⠃⠾")] + fn writes_pedigrees(#[case] text: &str, #[case] expected: &str) { + assert_eq!( + encode(text, false).map(|cells| braille(&cells)).as_deref(), + Some(expected) + ); + } + + #[rstest::rstest] + #[case::single_generation("P BB\n |")] + #[case::unknown_label("Q BB\nF₁ Bb")] + #[case::label_only("P\nF₁ Bb")] + #[case::not_genes("P BBB\nF₁ Bb")] + #[case::sibling_not_genes("P BB\nF₁ Bb xyz")] + #[case::letter_subscript("P BB\nFₐ Bb")] + #[case::empty_line("P BB\n\nF₁ Bb")] + fn leaves_other_pedigrees_alone(#[case] text: &str) { + assert_eq!(pedigree(text), None); + } +} diff --git a/libs/braillify/src/rules/science/elements.rs b/libs/braillify/src/rules/science/elements.rs new file mode 100644 index 00000000..ff352073 --- /dev/null +++ b/libs/braillify/src/rules/science/elements.rs @@ -0,0 +1,75 @@ +//! 원소 기호 — 과학 점자의 모든 규칙이 함께 쓰는 단일 목록. +//! +//! 한 글자짜리 원소 기호는 수학 변수와 글자가 겹치므로(`P`, `V`, `B`, `C`), 어떤 +//! 대문자가 원소인지는 이 목록으로만 판정한다. 화학식이라는 신호(첨자, 반응 +//! 화살표 등)를 따로 보는 것은 각 규칙의 몫이다. + +use phf::{Set, phf_set}; + +/// 주기율표의 118개 원소 기호. +static ELEMENTS: Set<&'static str> = phf_set! { + "H", "He", "Li", "Be", "B", "C", "N", "O", "F", "Ne", "Na", "Mg", "Al", "Si", "P", "S", "Cl", + "Ar", "K", "Ca", "Sc", "Ti", "V", "Cr", "Mn", "Fe", "Co", "Ni", "Cu", "Zn", "Ga", "Ge", "As", + "Se", "Br", "Kr", "Rb", "Sr", "Y", "Zr", "Nb", "Mo", "Tc", "Ru", "Rh", "Pd", "Ag", "Cd", "In", + "Sn", "Sb", "Te", "I", "Xe", "Cs", "Ba", "La", "Ce", "Pr", "Nd", "Pm", "Sm", "Eu", "Gd", "Tb", + "Dy", "Ho", "Er", "Tm", "Yb", "Lu", "Hf", "Ta", "W", "Re", "Os", "Ir", "Pt", "Au", "Hg", "Tl", + "Pb", "Bi", "Po", "At", "Rn", "Fr", "Ra", "Ac", "Th", "Pa", "U", "Np", "Pu", "Am", "Cm", "Bk", + "Cf", "Es", "Fm", "Md", "No", "Lr", "Rf", "Db", "Sg", "Bh", "Hs", "Mt", "Ds", "Rg", "Cn", "Nh", + "Fl", "Mc", "Lv", "Ts", "Og", +}; + +/// `symbol` 이 원소 기호인가. +pub fn is_element(symbol: &str) -> bool { + ELEMENTS.contains(symbol) +} + +/// 대문자 하나가 그대로 원소 기호인가. 과학 제4항 대문자 구절표의 대상이다. +pub fn is_single_letter_element(letter: char) -> bool { + let mut buf = [0u8; 4]; + ELEMENTS.contains(letter.encode_utf8(&mut buf)) +} + +/// 대문자와 소문자가 이어진 두 글자 원소 기호인가(`Na`, `Cl`). +pub fn is_two_letter_element(upper: char, lower: char) -> bool { + upper.is_ascii_uppercase() + && lower.is_ascii_lowercase() + && ELEMENTS.contains(format!("{upper}{lower}").as_str()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[rstest::rstest] + #[case::hydrogen("H", true)] + #[case::sodium("Na", true)] + #[case::oganesson("Og", true)] + #[case::not_an_element("A", false)] + #[case::lowercase("na", false)] + fn knows_the_periodic_table(#[case] symbol: &str, #[case] expected: bool) { + assert_eq!(is_element(symbol), expected); + } + + #[rstest::rstest] + #[case::hydrogen('H', true)] + #[case::uranium('U', true)] + #[case::rest_group('R', false)] + #[case::lowercase('h', false)] + fn knows_single_letter_elements(#[case] letter: char, #[case] expected: bool) { + assert_eq!(is_single_letter_element(letter), expected); + } + + #[rstest::rstest] + #[case::chlorine('C', 'l', true)] + #[case::cobalt('C', 'o', true)] + #[case::not_an_element('C', 'x', false)] + #[case::wrong_case('c', 'l', false)] + fn knows_two_letter_elements(#[case] upper: char, #[case] lower: char, #[case] expected: bool) { + assert_eq!(is_two_letter_element(upper, lower), expected); + } + + #[test] + fn holds_every_element_once() { + assert_eq!(ELEMENTS.len(), 118); + } +} diff --git a/libs/braillify/src/rules/science/formula.rs b/libs/braillify/src/rules/science/formula.rs new file mode 100644 index 00000000..63b0a91c --- /dev/null +++ b/libs/braillify/src/rules/science/formula.rs @@ -0,0 +1,1438 @@ +//! 과학 제1~8·18항 — 화학식과 화학 반응식. +//! +//! 원소 기호는 모두 1급 점자로 적고(제4항 1) 원소 기호마다 대문자표를 붙인다 +//! (제7항 1). 로마자 하나로 된 원소 기호가 셋 이상 이어지면 대문자 구절표 +//! ⠠⠠⠠ 로 묶고, 마지막 한 글자 원소 기호와 그 첨자 뒤에 대문자 종료표 ⠠⠄ 를 +//! 적는다(제4항, [붙임 1]). 구절 안에서 숫자 뒤에 붙는 H·B·C·F·I 앞에는 ⠐ 을 +//! 적는다(제5항). + +use crate::english::encode_english; +use crate::number::encode_number; +use crate::rules::english_ueb::rule_9::decode_styled; +use crate::rules::english_ueb::token::Typeform; +use crate::unicode::decode_unicode; + +use super::elements::{is_single_letter_element, is_two_letter_element}; + +/// 상태 기호. 과학 제18항 3 — 한글 소괄호로 묶는다. +const STATES: &[&str] = &["s", "l", "g", "aq"]; + +/// 과학 제7항 5 — 화학식 안에서 강조된 문자는 통일영어점자 §9 에 따른다. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum Style { + Italic, + Bold, + Underline, +} + +impl Style { + fn lead(self) -> u8 { + decode_unicode(match self { + Style::Italic => '⠨', + Style::Bold => '⠘', + Style::Underline => '⠸', + }) + } +} + +/// 식을 이루는 낱낱. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) enum Item { + /// 원소 기호, 또는 원소가 아닌 로마자 대문자(`R`, `NAD` 의 `A`·`D`). + Capital { + symbol: String, + element: bool, + style: Option