From bbc7951250a56afebe91dddf386fc51bd317a251 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Thu, 1 Oct 2026 13:56:54 -0700 Subject: [PATCH 1/5] fix(#559): ask the particle chain's titles-ahead test once per chain `chain()` asked `all(is_leading_title(...) for x in range(k))` at every ambiguous-particle chain site, so leading titles x chain sites `is_leading_title` calls: `"Dr. " * n + "Jan " + "van Berg " * n` cost 12.2x the time for 4x the input at 2.2.0 and 2.3.0 (12.5x at 2.0.0). merge(k, j) changes only indices from k on and k only grows, so a piece behind k is final: a forward-only cursor over the leading-title run answers the same question with one walk per chain. Full parse output (repr, ambiguities, initials) is byte-identical to master over the differential corpora plus 120,000 fuzzed names under three configurations. Co-Authored-By: Claude Opus 5.5 --- nameparser/_pipeline/_group.py | 33 ++++++++++++++++++++------------- 1 file changed, 20 insertions(+), 13 deletions(-) diff --git a/nameparser/_pipeline/_group.py b/nameparser/_pipeline/_group.py index b515bc02..5669405f 100644 --- a/nameparser/_pipeline/_group.py +++ b/nameparser/_pipeline/_group.py @@ -1724,6 +1724,12 @@ def merge(lo: int, hi: int, add: Set[str] = frozenset(), tail = len(pieces) - trailing_start(name_start, pieces, ptags, tokens, one_case=one_case) def chain(tail: int) -> None: + # `pieces[:titled]` are known to be leading titles. merge(k, + # j) changes only indices from k on, and k only grows, so a + # piece behind k is final and its answer can be kept: the + # cursor makes the all-titles-ahead test below one walk per + # chain rather than one per chain site (#559). + titled = 0 k = 0 while k < len(pieces): if k == leading or not prefix(k): @@ -1743,8 +1749,8 @@ def chain(tail: int) -> None: # A fork whose two sides are decided in different stages # needs an emitter in each. # - # Narrow, and #367 is why. `all(is_leading_title(...))` - # says every piece ahead of this one is a title, and the + # Narrow, and #367 is why. `titled == k` says every + # piece ahead of this one is a title, and the # loop skipped k == leading, so `leading` is STRICTLY # before k -- and being before k it is one of those titles, # while being `leading` it satisfies `not title or prefix`. @@ -1784,17 +1790,18 @@ def chain(tail: int) -> None: # an ambiguous particle, while title() is a call per piece.) if (j > k + 1 and "vocab:particle-ambiguous" - in tokens[pieces[k][0]].tags - and all(is_leading_title(pieces[x], ptags[x], - tokens) - for x in range(k))): - i = pieces[k][0] - ambiguities.append(PendingAmbiguity( - AmbiguityKind.PARTICLE_OR_GIVEN, - f"{tokens[i].text!r} was chained onto the following " - f"name piece; it is also a given name in other " - f"names", - (i,))) + in tokens[pieces[k][0]].tags): + while titled < k and is_leading_title( + pieces[titled], ptags[titled], tokens): + titled += 1 + if titled == k: + i = pieces[k][0] + ambiguities.append(PendingAmbiguity( + AmbiguityKind.PARTICLE_OR_GIVEN, + f"{tokens[i].text!r} was chained onto the " + f"following name piece; it is also a given " + f"name in other names", + (i,))) # rules.md#S2: "A BARE ambiguous acronym is consumed # only when the name has words to spare — as the second # of two words it stays the family name — and at the From 708f8d3fc2248af8cb0818611027e24c5ec21eed Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Thu, 1 Oct 2026 13:56:54 -0700 Subject: [PATCH 2/5] fix(#558): resume tail_reading's peel instead of re-peeling every pass The S2 peel / H5 chain fixed point re-ran `peel_trailing` over the whole walk once per title the chain took, re-reading every suffix the passes before had peeled: `"John Smith " + "MA Prof. " * n` cost 12.9x for 4x the input since 2.3.0, and the maiden clause, which runs the fixed point three times, 14.8x on master. A pass now resumes the walk at the splice, over `rest` as written. The pieces in front of the splice are the ones a fresh walk meets, and `credential_anchors` is a front-to-back pass, so its carried values hold for them. Two of the walk's tests read position COUNTS of pieces already peeled, which the splice lowers, so a pass resumes only where neither can move: two or more pieces in front of the splice (the acronym fork's words to spare) and two or more peeled pieces behind it (the numeral's last-pair test). Otherwise it walks afresh. A pick the stopping title made is dropped, a fresh walk never meeting it. Checked against the re-peeling loop on every tail_reading call over the corpora plus 150,000 fuzzed names under four configurations, including a lexicon putting title words in the ambiguous acronym class: 698,009 calls, 108,453 through the resume path, all identical. Weakening each of the three conditions fails that check. Co-Authored-By: Claude Opus 5.5 --- nameparser/_pipeline/_pieces.py | 110 ++++++++++++++++++++++++++------ 1 file changed, 91 insertions(+), 19 deletions(-) diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 89a3fa59..0b3a2421 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -590,11 +590,16 @@ class Peel(NamedTuple): one token long: `numeral` is the piece the roman-numeral fork took (None when it did not fire; always the walk's last piece), `picks` the bare ambiguous acronyms the peel had to resolve, in - peel order, either way (the last may sit at rest[names - 1]).""" + peel order, either way (the last may sit at rest[names - 1]). + `anchors` is the walk's `credential_anchors` pass over `rest`, or + None where no member needed it. It is `tail_reading`'s alone, + handed to the next pass of its fixed point (`peel_trailing`'s + `start`); no caller reads it.""" names: int numeral: tuple[int, ...] | None picks: tuple[tuple[int, ...], ...] + anchors: list[bool] | None # rules.md#S2: "a trailing word of the suffix vocabulary reads as a @@ -742,7 +747,9 @@ def credential_at_the_given_slot( def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], tokens: Sequence[WorkToken], - one_case: bool | None) -> Peel: + one_case: bool | None, + start: int | None = None, + anchors: list[bool] | None = None) -> Peel: """The S2 trailing peel over `rest`, a peel_walk list. In the piece layer rather than in assign because group's bound-given reserve asks the same question of the view the join would leave @@ -754,11 +761,18 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], nobody asked, which is every caller that has no state to ask with, and reads as "no lean" -- rules.md#S2's count alone, the behavior of every release before this one. + + `start` and `anchors` resume a walk rather than begin one: the + walk picks up at `rest[start - 1]`, as though the pieces from + `start` on were already peeled, with `anchors` the pass an earlier + walk built (or None). Only `tail_reading` passes them, and its + docstring states when that answers what a fresh walk would. The + returned picks are the resumed walk's own; the numeral is never + read, `start` standing short of the walk's last piece. """ picks: list[tuple[int, ...]] = [] numeral: tuple[int, ...] | None = None - anchors: list[bool] | None = None - k = len(rest) + k = len(rest) if start is None else start while k > 0: piece = pieces[rest[k - 1]] if is_suffix_piece(piece, ptags[rest[k - 1]], tokens): @@ -848,7 +862,7 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], k -= 1 continue break - return Peel(k, numeral, tuple(picks)) + return Peel(k, numeral, tuple(picks), anchors) # rules.md#H5: "only a word the vocabulary knows as a title is one, @@ -863,7 +877,8 @@ def peel_trailing(rest: Sequence[int], pieces: Sequence[Sequence[int]], def trailing_titles(rest: Sequence[int], pieces: Sequence[Sequence[int]], ptags: Sequence[Set[str]], tokens: Sequence[WorkToken], - floor: int = 1) -> int: + floor: int = 1, + end: int | None = None) -> int: """How many pieces of `rest` the trailing title chain LEAVES standing: `rest[:kept]` are the name pieces and `rest[kept:]` the period-marked title words the chain took, in piece order. Counted @@ -872,7 +887,8 @@ def trailing_titles(rest: Sequence[int], pieces: Sequence[Sequence[int]], on the no-comma path what the S2 peel left, after a family comma the segment's pieces that the segment's own suffix reading does not claim, and in `tail_reading` the leftovers of whichever peel - is current. + is current. `end`, where given, reads `rest[:end]` without the + copy, for `tail_reading`'s passes. Floor: `floor` leading positions of `rest` are never taken -- 1 by default, so one name piece stands and a name is never all title; the maiden walk passes the position just past the marker's @@ -900,7 +916,7 @@ def trailing_titles(rest: Sequence[int], pieces: Sequence[Sequence[int]], period-marked word, so the ordinary parse pays the one match and stops (decisions.md#parse-cost). """ - k = len(rest) + k = len(rest) if end is None else end while k > floor: idx = rest[k - 1] piece = pieces[idx] @@ -991,18 +1007,74 @@ def tail_reading(rest: list[int], pieces: Sequence[Sequence[int]], 'Jane Doe nee King. ba' lost the suffix 'Jane Doe nee Smith ba' keeps. The splice only ever removes positions at or past the floor, so the floor names the same pieces every pass. + + Linear in the walk (#558). Re-peeling the whole walk every pass + re-read each suffix the passes before had already peeled, so a + name ending 'MA Prof. MA Prof. ...' cost the square of its tail. + A pass instead RESUMES the walk where the splice left it, over + `rest` as written: the pieces in front of the splice are the ones + a fresh walk would meet, and `credential_anchors` reads each + position off the pieces in front of it, so the carried pass + answers for them too. What a fresh walk reads differently is the + position COUNT of the pieces it already peeled, which the splice + lowers, and two tests of the walk read it. The acronym fork's + words to spare: a member whose count falls below three can + decline where it was taken -- 'John Prof. MA Prof.' is exactly + that -- so a pass resumes only with two or more pieces in front + of the splice, every peeled piece then counting three or more. + And the numeral fork, which reads the walk's last piece against + the one before it: a pass resumes only with two or more peeled + pieces behind the splice, so that pair is the one the first walk + read. Elsewhere -- a splice near the front of the walk, or before + two pieces have been peeled -- the pass walks afresh over the + spliced pieces. The fuzz that checked this against the + re-peeling loop, the three conditions each shown load-bearing by + a weakened copy failing it, is recorded on the PR closing #558. + The splice itself is never materialized in between: the runs are + collected back to front and joined once. """ - titled: list[int] = [] - while True: - peeled = peel_trailing(rest, pieces, ptags, tokens, one_case) - kept = trailing_titles(rest[:peeled.names], pieces, ptags, - tokens, floor) - if kept == peeled.names: - return rest, tuple(titled), peeled - # the chain's pieces reach this list back to front, so each - # run goes in FRONT of what the pass before it took - titled[:0] = rest[kept:peeled.names] - rest = rest[:kept] + rest[peeled.names:] + peeled = peel_trailing(rest, pieces, ptags, tokens, one_case) + kept = trailing_titles(rest, pieces, ptags, tokens, floor, + peeled.names) + if kept == peeled.names: + return rest, (), peeled + # `rest[:hi]` stands as written; `behind` holds the peeled runs + # the splices left after it, and `titled` the chained ones, each + # back to front -- a pass's run goes in FRONT of what the pass + # before it took + hi = len(rest) + behind: list[list[int]] = [] + behind_count = 0 + titled: list[list[int]] = [] + picks = list(peeled.picks) + numeral = peeled.numeral + while kept < peeled.names: + names = peeled.names + titled.append(rest[kept:names]) + behind.append(rest[names:hi]) + behind_count += hi - names + # the walk stopped AT the chain's last title, and a fresh walk + # will not meet that piece: drop the pick it made there + if picks and picks[-1] == tuple(pieces[rest[names - 1]]): + picks.pop() + hi = kept + if kept >= 2 and behind_count >= 2: + peeled = peel_trailing(rest, pieces, ptags, tokens, one_case, + start=hi, anchors=peeled.anchors) + picks.extend(peeled.picks) + else: + rest = rest[:hi] + [j for run in reversed(behind) for j in run] + hi = len(rest) + behind, behind_count = [], 0 + peeled = peel_trailing(rest, pieces, ptags, tokens, one_case) + picks = list(peeled.picks) + numeral = peeled.numeral + kept = trailing_titles(rest, pieces, ptags, tokens, floor, + peeled.names) + if behind: + rest = rest[:hi] + [j for run in reversed(behind) for j in run] + return (rest, tuple(j for run in reversed(titled) for j in run), + Peel(peeled.names, numeral, tuple(picks), None)) # rules.md#H5: "the title is TRANSPARENT to the suffix reading: where From 40318e18610d6a517a19df48232b983cef2c0406 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Thu, 1 Oct 2026 13:56:54 -0700 Subject: [PATCH 3/5] test(#558, #559): frame-count guards for the re-reading walks `_REREAD_SHAPES` counts the one function each defect re-ran, at 8 units against 32: under 3.8x on this tree, 11.0-14.7x at fc682e36. Each row fails against a copy reverting only its own half of the fix and passes against one reverting only the other. Frame-counted rather than a `_PREFIXED_SHAPES` clock row because the cost is Python-level, which keeps the guard deterministic and running under a line tracer. Release-log bullets for both fixes, and AGENTS.md's perf gotcha names the new table as the home for a prefixed Python-level shape. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- docs/release_log.rst | 4 ++ tests/v2/test_benchmark.py | 78 +++++++++++++++++++++++++++++++++++++- 3 files changed, 82 insertions(+), 2 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 55b7d728..0793c9bf 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -383,7 +383,7 @@ Add a dedicated `copy.deepcopy()` round-trip test for it too (see `test_regexes_ **`_normalize` must reach a fixed point** — storage and match-time share the one fold, and `Lexicon.__setstate__` re-validates, so a value that changes on re-normalization changes under its owner. `strip().strip(".")` alone is not idempotent (`'. a .'` → `' a '` → `'a'`). The loop is the fix; keep any new stripping inside it. **Anything built on `_normalize` must converge too** — `_fold_words` runs `_normalize` per word and DROPS the words that fold away (`_title_key` is that list space-joined, and `_run_addresses_by_given` reads the list itself, so its last-word arm is the last word of the FOLDED key by construction); keeping the empty slot stored `'lt .'` as `'lt '`, a key match-time can never rebuild (so the entry is silently inert) and `__setstate__` rejects on the next round-trip as "not written by this version". -**Perf regressions are caught by the scaling test, not the absolute-time ones** — `tests/v2/test_benchmark.py::test_parse_cost_grows_no_worse_than_linearly` times a repeated unit at n vs 4n over a table of shapes (one per pipeline inner loop) and bounds the ratio; the `_thousand_names` tests use constant-size, delimiter-free input and are structurally blind to a complexity regression. Two rules when touching it: calibrate `_MAX_RATIO` against the WEAKEST quadratic's signal (a mixed quadratic surfaces far below the textbook 16×, so the operating point `_BASE` matters more than the bound), and confirm a planted regression fails it across REPEATED runs — one failure is a coin-flip on a timing test. The shapes cover different dimensions (segment count only via `commas`, intra-piece accumulation only via `particles`/`conjunctions`, non-ASCII input only via `honorifics` — every other unit is pure ASCII, so `script_segment` returns at its bail and the CJK stages go unmeasured, M2's clause view only via `maiden_clause`, whose unit has to END on a class member: `MA nee ` holds the same two words, the peel stops at the trailing marker, and the shape reaches nothing — and connective RUN LENGTH only via `link_run`, whose unit has to be MIXED CASE and hold a connective of the generational class: `i und ` reads the letter as an initial and reaches nothing where `i Und ` reaches everything, and S2's anchor pass only via `credential_run`, whose unit has to hold a Title-case member behind an unambiguous credential: `PhD MA ` reads the same fields and never asks the pass, the capitals deciding each member first); measure before pruning one. **A shape whose input needs a PREFIX cannot be a `_SHAPES` row at all**, since that table repeats a unit and nothing else — a maiden clause needs a name word and a marker before the run it is about, and `"nee i Und " * n` reaches the clause rule not at all, measuring the identical ratio on a broken tree and a fixed one. That one is `test_a_clause_link_run_does_not_cost_quadratically`, which builds its own input and counts FRAMES. A prefixed shape whose cost is C-level work — a list scanned by `in`, a tuple re-hashed as a dict key — emits no frame for that count to see, so it goes in `_PREFIXED_SHAPES` instead: (prefix, unit, reachability probe) rows timed on the clock like `_SHAPES`, at a base of their own recorded beside the table (#553, whose run shape read 5.97× once at base 800 on the pre-fix tree, under the bound; py3.11, 2026-09-28). **That table skips under a line tracer** (`sys.settrace`, coverage's core below py3.14), which slows every Python line and not the C-level scan, diluting the quadratic below the bound (5.46–5.52× broken at base 1600 under `coverage run` on py3.11, same date). The `sys.monitoring` core coverage uses from 3.14 does not dilute it (8.87–10.91× broken), and CI collects coverage in its 3.14 build job alone, so every CI job runs the table; `ja-extra` also sets `NAMEPARSER_REQUIRE_CLOCK_GUARDS` so that a line tracer there fails the rows instead of skipping them. **Coverage is collected on ONE interpreter, the one that uploads it** (`COVERAGE_ON` in `.github/workflows/python-package.yml`): the package has no version-conditional code, so one report stands for all of them, and under a line tracer the property grids in `tests/v2/test_properties.py` had cost the 3.12 job most of its ~12 minutes (measured 2026-09-29: the three #397 grids take 139s under coverage's settrace core on 3.12, 24s under its sys.monitoring core, 17s with none). **A shape the CLOCK cannot reach needs a FRAME-count guard instead**, which is the second scaling test in that file (`test_a_trailing_credential_run_does_not_cost_exponentially`, #531): where the defect is an exponential rather than a quadratic, the input length that separates the curves on a timing test does not finish, so the guard counts frames over 8 units against 16 and bounds THAT ratio. One pair does not see every curve, and the fix round for #531 measured why: at 2× the input the per-member LINEAR work swamps a quadratic (2.08× for a genuine one against 1.73× clean), so that pair guards the exponential alone and a second, longer pair — 16 against 64, where the same quadratic reads 7.42× against 3.53× clean — is what can see one. Assert them in that order: an exponential never returns from the longer run, so the cheap pair has to have failed first. Frame counts do not move under load, so this shape needs no repeated-run calibration — but it does need the same reachability assertion `_POLICY_SHAPES` rows carry, since the walk under measurement runs only while every unit still reads as a credential. A stage gated on an opt-in `Policy` field needs a `_POLICY_SHAPES` entry instead, since bare `parse()` never enters it — and that table's rows carry a **reachability probe** run before the measurement, because a precedence change can quietly stop the shape reaching the stage and leave a green test measuring a no-op (`_POLICY_SHAPES` is also asserted non-empty: an empty `parametrize` is a skip, not a failure, so deleting its last row would retire the guard silently). +**Perf regressions are caught by the scaling test, not the absolute-time ones** — `tests/v2/test_benchmark.py::test_parse_cost_grows_no_worse_than_linearly` times a repeated unit at n vs 4n over a table of shapes (one per pipeline inner loop) and bounds the ratio; the `_thousand_names` tests use constant-size, delimiter-free input and are structurally blind to a complexity regression. Two rules when touching it: calibrate `_MAX_RATIO` against the WEAKEST quadratic's signal (a mixed quadratic surfaces far below the textbook 16×, so the operating point `_BASE` matters more than the bound), and confirm a planted regression fails it across REPEATED runs — one failure is a coin-flip on a timing test. The shapes cover different dimensions (segment count only via `commas`, intra-piece accumulation only via `particles`/`conjunctions`, non-ASCII input only via `honorifics` — every other unit is pure ASCII, so `script_segment` returns at its bail and the CJK stages go unmeasured, M2's clause view only via `maiden_clause`, whose unit has to END on a class member: `MA nee ` holds the same two words, the peel stops at the trailing marker, and the shape reaches nothing — and connective RUN LENGTH only via `link_run`, whose unit has to be MIXED CASE and hold a connective of the generational class: `i und ` reads the letter as an initial and reaches nothing where `i Und ` reaches everything, and S2's anchor pass only via `credential_run`, whose unit has to hold a Title-case member behind an unambiguous credential: `PhD MA ` reads the same fields and never asks the pass, the capitals deciding each member first); measure before pruning one. **A shape whose input needs a PREFIX cannot be a `_SHAPES` row at all**, since that table repeats a unit and nothing else — a maiden clause needs a name word and a marker before the run it is about, and `"nee i Und " * n` reaches the clause rule not at all, measuring the identical ratio on a broken tree and a fixed one. That one is `test_a_clause_link_run_does_not_cost_quadratically`, which builds its own input and counts FRAMES. A prefixed shape whose cost is Python-level joins `_REREAD_SHAPES` beside it — (input builder, the one function the defect re-runs, reachability probe) rows counted in that function's frames at 8 units against 32 — which holds #558's trailing fixed point, its maiden-clause twin, and #559's particle chain, a shape that grows at both ends and so has no unit to repeat. A prefixed shape whose cost is C-level work — a list scanned by `in`, a tuple re-hashed as a dict key — emits no frame for that count to see, so it goes in `_PREFIXED_SHAPES` instead: (prefix, unit, reachability probe) rows timed on the clock like `_SHAPES`, at a base of their own recorded beside the table (#553, whose run shape read 5.97× once at base 800 on the pre-fix tree, under the bound; py3.11, 2026-09-28). **That table skips under a line tracer** (`sys.settrace`, coverage's core below py3.14), which slows every Python line and not the C-level scan, diluting the quadratic below the bound (5.46–5.52× broken at base 1600 under `coverage run` on py3.11, same date). The `sys.monitoring` core coverage uses from 3.14 does not dilute it (8.87–10.91× broken), and CI collects coverage in its 3.14 build job alone, so every CI job runs the table; `ja-extra` also sets `NAMEPARSER_REQUIRE_CLOCK_GUARDS` so that a line tracer there fails the rows instead of skipping them. **Coverage is collected on ONE interpreter, the one that uploads it** (`COVERAGE_ON` in `.github/workflows/python-package.yml`): the package has no version-conditional code, so one report stands for all of them, and under a line tracer the property grids in `tests/v2/test_properties.py` had cost the 3.12 job most of its ~12 minutes (measured 2026-09-29: the three #397 grids take 139s under coverage's settrace core on 3.12, 24s under its sys.monitoring core, 17s with none). **A shape the CLOCK cannot reach needs a FRAME-count guard instead**, which is the second scaling test in that file (`test_a_trailing_credential_run_does_not_cost_exponentially`, #531): where the defect is an exponential rather than a quadratic, the input length that separates the curves on a timing test does not finish, so the guard counts frames over 8 units against 16 and bounds THAT ratio. One pair does not see every curve, and the fix round for #531 measured why: at 2× the input the per-member LINEAR work swamps a quadratic (2.08× for a genuine one against 1.73× clean), so that pair guards the exponential alone and a second, longer pair — 16 against 64, where the same quadratic reads 7.42× against 3.53× clean — is what can see one. Assert them in that order: an exponential never returns from the longer run, so the cheap pair has to have failed first. Frame counts do not move under load, so this shape needs no repeated-run calibration — but it does need the same reachability assertion `_POLICY_SHAPES` rows carry, since the walk under measurement runs only while every unit still reads as a credential. A stage gated on an opt-in `Policy` field needs a `_POLICY_SHAPES` entry instead, since bare `parse()` never enters it — and that table's rows carry a **reachability probe** run before the measurement, because a precedence change can quietly stop the shape reaching the stage and leave a green test measuring a no-op (`_POLICY_SHAPES` is also asserted non-empty: an empty `parametrize` is a skip, not a failure, so deleting its last row would retire the guard silently). **Expected-failure tests use `@pytest.mark.xfail`** — the conftest parametrized fixture breaks `@unittest.expectedFailure`; always use `@pytest.mark.xfail` instead. diff --git a/docs/release_log.rst b/docs/release_log.rst index f5b20a0d..9631c654 100644 --- a/docs/release_log.rst +++ b/docs/release_log.rst @@ -50,6 +50,10 @@ Release Log - **Fix a long given part after a family comma costing quadratic time.** Since 2.3.0, parsing ``"Doe, Jane " + "Smith " * n`` took time growing with the square of the part's length: going from 1,600 to 6,400 words cost 8.4x the time, where 2.2.0 and the comma-less form cost 4x. A run of trailing titles in the same place (``Doe, Jane Smith Prof. Prof. ...``) cost 10x. Both cost 4x again (Python 3.11, measured 2026-09-28). No field moves (closes #553) + - **Fix a name ending in alternating credentials and titles costing quadratic time.** Since 2.3.0, parsing ``"John Smith " + "MA Prof. " * n`` took time growing with the square of ``n``: going from 400 to 1,600 pairs cost 12.9x the time, where 2.2.0 cost 4x. It costs 4x again (Python 3.11, ``HumanName``, measured 2026-10-01). No field moves (closes #558) + + - **Fix many leading titles plus many surname particles costing quadratic time.** Parsing ``"Dr. " * n + "Jan " + "van Berg " * n`` took time growing with the product of the two counts in every 2.x release: going from 400 to 1,600 of each cost 12.2x the time at 2.2.0 and 2.3.0, and 12.5x at 2.0.0. It costs 4x now (Python 3.11, ``HumanName``, measured 2026-10-01). No field moves (closes #559) + **Additions** - **Add Lexicon.conjunctions_ambiguous, the one-letter connectives that read as initials.** A subset of ``conjunctions`` holding ``e`` and ``i`` by default; it is the knob for the change above rather than a switch. Portuguese data, where ``e`` links surnames the way ``y`` does in Spanish, takes it out: ``Lexicon.default().remove(conjunctions_ambiguous={"e"})`` restores the joining reading. Dutch data, where a bare single letter is an initial and never a connective, adds the other one: ``Lexicon.default().add(conjunctions_ambiguous={"y"})``. A v1 ``Constants`` has no manager of its own for it -- deleting the word from ``conjunctions`` is what turns the marking off, the same rule the glued-honorific tails follow. See ``docs/customize.rst`` (#383, #479) diff --git a/tests/v2/test_benchmark.py b/tests/v2/test_benchmark.py index 0fd38a25..d463a7a7 100644 --- a/tests/v2/test_benchmark.py +++ b/tests/v2/test_benchmark.py @@ -30,7 +30,7 @@ import pytest -from nameparser import Parser, parse +from nameparser import ParsedName, Parser, parse from nameparser._policy import Policy @@ -497,6 +497,7 @@ def test_shape_tables_are_not_empty() -> None: assert _SHAPES assert _POLICY_SHAPES assert _PREFIXED_SHAPES + assert _REREAD_SHAPES @pytest.mark.parametrize("unit,parser,reaches", _POLICY_SHAPES.values(), @@ -732,6 +733,81 @@ def test_the_paired_initials_title_scan_does_not_cost_quadratically() -> None: f"is running once per pair again (#563)") +# Fixed points and scans that re-read what they had already read, each +# found by the `_pipeline/` sweep for #553. A `_SHAPES` row cannot +# express any of them: the tail ones need a name in front of the run, +# and the chain one grows at BOTH ends. The cost is Python-level, so it +# is counted in frames of the one function each re-ran, which isolates +# it from the rest of the parse and holds under a line tracer. +# +# tail #558: `tail_reading` re-peeled the whole walk once per +# title the H5 chain took, asking `listed_lean` of every +# member it had already peeled. +# clause #558 again, through the maiden walk, which runs that fixed +# point three times (the clause's take, its release check, +# and `trailing_start_past_titles`). +# chain #559: the particle chain asked "is every piece ahead of +# this one a title?" afresh at every chain site, so leading +# titles x particle sites `is_leading_title` calls. +# +# Measured 2026-10-01 on py3.11 through `_frames_for(..., only=...)`, k +# units, ratio for 4x the units: +# +# k=8 -> k=32, fixed k=8 -> k=32, at fc682e36 (broken) +# tail 9 -> 33 3.67x 36 -> 528 14.67x +# clause 27 -> 99 3.67x 108 -> 1,584 14.67x +# chain 52 -> 196 3.77x 108 -> 1,188 11.00x +# +# 6.0 sits between the populations with room on both sides; counts are +# deterministic, so the margin is for future shape changes, not noise. +# Each probe pins the reading the shape needs, so a change that stops +# the shape reaching the walk fails here instead of leaving the row +# counting nothing. +_REREAD_SMALL = 8 +_REREAD_LARGE = 32 +_REREAD_MAX_RATIO = 6.0 +_REREAD_SHAPES: dict[str, tuple[Callable[[int], str], str, + Callable[[ParsedName, int], bool]]] = { + "tail": ( + lambda k: "John Smith " + "MA Prof. " * k, "listed_lean", + lambda n, k: (n.title.split() == ["Prof."] * k + and n.suffix.split() == ["MA"] * k + and n.family == "Smith"), + ), + "clause": ( + lambda k: "Jane Doe nee Smith " + "MA Prof. " * k, "listed_lean", + lambda n, k: (n.title.split() == ["Prof."] * k + and n.suffix.split() == ["MA"] * k + and n.maiden == "Smith"), + ), + "chain": ( + lambda k: "Dr. " * k + "Jan " + "van Berg " * k, "is_leading_title", + lambda n, k: (n.title.split() == ["Dr."] * k + and n.given == "Jan" and n.family == "van Berg"), + ), +} + + +@pytest.mark.parametrize("text,only,reaches", _REREAD_SHAPES.values(), + ids=list(_REREAD_SHAPES)) +def test_a_fixed_point_does_not_reread_what_it_has_read( + text: Callable[[int], str], only: str, + reaches: Callable[[ParsedName, int], bool]) -> None: + if sys.getprofile() is not None: + pytest.skip("a profile hook is already installed; this test owns it") + for k in (_REREAD_SMALL, _REREAD_LARGE): + assert reaches(parse(text(k)), k), ( + f"shape no longer reaches the walk at k={k}") + small = _frames_for(text(_REREAD_SMALL), only=only) + large = _frames_for(text(_REREAD_LARGE), only=only) + ratio = large / small + assert ratio < _REREAD_MAX_RATIO, ( + f"{_REREAD_SMALL} units cost {small} {only} calls and " + f"{_REREAD_LARGE} cost {large} -- {ratio:.1f}x for 4x the input, " + f"where this tree measured under 3.8x and the re-reading walk " + f"11.0-14.7x (#558, #559)") + + # THE ABSOLUTE COST OF A LINK, which the ratio above cannot see: a # change costing ONE MORE FRAME PER LINK moves both ends of the pair # and leaves the ratio where it was. Re-splitting the #397 follow-up's From 74301f40d9eb951a519e75fa320832bf4ebdf6d2 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Thu, 1 Oct 2026 14:04:52 -0700 Subject: [PATCH 4/5] docs(#558): correct tail_reading's count examples 'John Prof. MA Prof.' never reaches the resume path's words-to-spare condition: one peeled piece behind the splice already forces a fresh walk, and since #289 the capitals take a mixed-case MA whatever the count. 'john prof. ma ma prof.' is the input the condition decides (a copy resuming at one piece in front reads family 'john', suffix 'ma ma'). The transparency paragraph's 'John Prof. MA' pair went stale the same way with #289 and now names the one-case spelling, which still reads family 'ma'. Peel's `anchors` note no longer reads as a promise about every returned Peel. Co-Authored-By: Claude Opus 5.5 --- nameparser/_pipeline/_pieces.py | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/nameparser/_pipeline/_pieces.py b/nameparser/_pipeline/_pieces.py index 0b3a2421..dd5d2b4b 100644 --- a/nameparser/_pipeline/_pieces.py +++ b/nameparser/_pipeline/_pieces.py @@ -591,10 +591,11 @@ class Peel(NamedTuple): took (None when it did not fire; always the walk's last piece), `picks` the bare ambiguous acronyms the peel had to resolve, in peel order, either way (the last may sit at rest[names - 1]). - `anchors` is the walk's `credential_anchors` pass over `rest`, or - None where no member needed it. It is `tail_reading`'s alone, - handed to the next pass of its fixed point (`peel_trailing`'s - `start`); no caller reads it.""" + `anchors` is `peel_trailing`'s working state for `tail_reading` + alone -- the walk's `credential_anchors` pass over `rest` where a + member needed one, handed to the next pass of the fixed point + (`peel_trailing`'s `start`). No caller reads it, and the Peel + `tail_reading` returns carries None.""" names: int numeral: tuple[int, ...] | None @@ -984,7 +985,10 @@ def tail_reading(rest: list[int], pieces: Sequence[Sequence[int]], written and wherever the peel then stops. Iterating ONCE reads a second title only half way -- 'John Prof. MA Prof.' un-peeled the acronym and re-exposed the first title, reading family 'Prof.' - with suffix 'MA' where 'John Prof. MA' reads family 'MA'. + with suffix 'MA' where 'John Prof. MA' read family 'MA' (written + in one case today: since #289 capitals in a mixed-case name take + the acronym whatever the count, so 'john prof. ma prof.' and + 'john prof. ma' are the pair that still reads family 'ma'). One function for assign's placement, group's bound-given reserve (P5), which must count the name words assign will leave, and the @@ -1019,8 +1023,9 @@ def tail_reading(rest: list[int], pieces: Sequence[Sequence[int]], position COUNT of the pieces it already peeled, which the splice lowers, and two tests of the walk read it. The acronym fork's words to spare: a member whose count falls below three can - decline where it was taken -- 'John Prof. MA Prof.' is exactly - that -- so a pass resumes only with two or more pieces in front + decline where it was taken -- 'john prof. ma ma prof.' is exactly + that, keeping family 'ma' only because the second pass walks + afresh -- so a pass resumes only with two or more pieces in front of the splice, every peeled piece then counting three or more. And the numeral fork, which reads the walk's last piece against the one before it: a pass resumes only with two or more peeled From 7a4fa3879d2ad10248f59ed6252f7c9678d20262 Mon Sep 17 00:00:00 2001 From: Derek Gulbranson Date: Thu, 1 Oct 2026 14:28:23 -0700 Subject: [PATCH 5/5] test(#558): pin the stopping title's dropped peel pick The one line of the #558 diff no test reached: a resumed pass drops the pick the previous walk made at the title the chain then took. No default title is an ambiguous acronym, so the row configures one ('ma'). Without the drop, 'John Smith Ma. PhD Jr.' keeps its fields and reports 'suffix-or-name' on the title 'Ma.'. Co-Authored-By: Claude Opus 5.5 --- tests/v2/pipeline/test_pieces.py | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index 6c1f2211..306447e9 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -11,7 +11,7 @@ import pytest -from nameparser import parse +from nameparser import Parser, parse from nameparser._lexicon import Lexicon, _normalize from nameparser._pipeline import STAGES from nameparser._pipeline._assign import assign @@ -822,3 +822,19 @@ def test_anchor_in_reach_never_hides_an_anchor() -> None: failures = _reach_failures(sorted(texts)) assert not failures, (f"{len(failures)} anchored member(s) the " f"reach test hides:\n" + "\n".join(failures[:15])) + + +def test_a_title_the_fixed_point_splices_out_keeps_no_peel_pick() -> None: + # tail_reading's resumed pass (#558) carries the picks of the walk + # before it, and that walk STOPPED at the title the chain then took: + # where the title is also an ambiguous acronym the stop was a pick, + # and a fresh walk over the spliced pieces never meets the word. + # No default title is an ambiguous acronym, so a caller's lexicon is + # the only way here. Recorded negative control (2026-10-01): with + # the drop removed, this reads the same fields and reports + # 'suffix-or-name' on 'Ma.', a word the parse took as a title. + parser = Parser(lexicon=Lexicon.default().add(titles={"ma"})) + name = parser.parse("John Smith Ma. PhD Jr.") + assert (name.title, name.given, name.family, name.suffix) == ( + "Ma.", "John", "Smith", "PhD Jr.") + assert name.ambiguities == ()