diff --git a/AGENTS.md b/AGENTS.md index 2f44f85a..d7e8ff4f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -28,7 +28,7 @@ Three committed contributor docs carry the parser's normative rules and their re **Counting claims.** A bare count in prose is either an assertion or a liability, keyed by who observes its staleness: asserted counts (a test holds the number) fail CI at change time — the useful kind; dated snapshots ("51 sites at spec time") cannot go stale; standing present-tense prose counts are the forbidden class — promote to an assertion, add a date, or state the invariant and let a test count. After changing how many times something runs, sweep for counts, not for the thing's name. -**Release-log claims.** Quantified or universal behavior claims in release bullets must come from the differential gate's classified summary, be verified against rules.md examples, or -- for a view the gate cannot see -- carry a recompute recipe stored with the design entry the bullet cites; never write one from memory. The classified summary covers the CONTRACT tier plus whatever radar diffs a rule classifies; a radar corpus's unmatched diffs are listed under UNCLASSIFIED (radar) and are not in it, so a claim quantified from the summary alone is silent about them. The gate compares the seven role fields, `_ambiguities` and -- since #484 -- `initials()`, under the `_initials` pseudo-field; so an initials-only change DOES show in the classified summary. But that pseudo-field sees only the names whose initials moved WHILE EVERY ROLE AND EVERY AMBIGUITY KIND STAYED PUT -- main()'s roles-identical guard keeps it out of any diff a role or a report is already in -- so a count taken from it is a FLOOR on initials movement, not the population: measured 2026-09-02 at 2.1.0 → tree on the v2 surface, 83 of the 1120 compared ENTRIES (1116 distinct names; seven entries carry a declared order rather than the default, and three strings are compared under more than one) changed their `initials()` string and only 28 were visible under `_initials`, the other 55 having moved a role as well. A bullet about how many names' initials changed still needs the recompute recipe. `capitalized()` and any other render view stay invisible to the gate (decisions.md#R4, #R3), and for those the first two sources still cannot reach a claim: a gate run is byte-identical across the change, and an example line witnesses an output without counting anything. A recipe names the corpus files, the policy sweep, and -- the part that is easy to omit and fatal -- THE COMPARATOR, which must be something the shipped tree is not: #408's first recipe said to compare `initials()` against a folded-first partition, which is what `initials()` now IS, so it reproduced 0 where the bullet claimed 660 and was the only stated provenance for the number. Run the recipe as written before shipping the bullet. Cross-version numbers (a released wheel, the pre-change tree) are dated snapshots under Counting claims, since nothing in the repository re-runs them. Per-rule ledger toml comments asserting PARSER behavior cite rule IDs under the excerpt discipline; free prose is for ledger mechanics only (owned by tools/differential/README.md). +**Release-log claims.** Quantified or universal behavior claims in release bullets must come from the differential gate's classified summary, be verified against rules.md examples, or -- for a view the gate cannot see -- carry a recompute recipe stored with the design entry the bullet cites; never write one from memory. The classified summary covers the CONTRACT tier plus whatever radar diffs a rule classifies; a radar corpus's unmatched diffs are listed under UNCLASSIFIED (radar) and are not in it, so a claim quantified from the summary alone is silent about them. The gate compares the seven role fields, `_ambiguities` and -- since #484 -- `initials()`, under the `_initials` pseudo-field; so an initials-only change DOES show in the classified summary. But that pseudo-field sees only the names whose initials moved WHILE EVERY ROLE AND EVERY AMBIGUITY KIND STAYED PUT -- main()'s roles-identical guard keeps it out of any diff a role or a report is already in -- so a count taken from it is a FLOOR on initials movement, not the population: measured 2026-09-02 at 2.1.0 → tree on the v2 surface, 83 of the 1120 compared ENTRIES (1116 distinct names; seven entries carry a declared order rather than the default, and three strings are compared under more than one) changed their `initials()` string and only 28 were visible under `_initials`, the other 55 having moved a role as well. A bullet about how many names' initials changed still needs the recompute recipe. `capitalized()` and any other render view stay invisible to the gate (decisions.md#R4, #R3), and for those the first two sources still cannot reach a claim: a gate run is byte-identical across the change, and an example line witnesses an output without counting anything. A recipe names the corpus files, the policy sweep, and -- the part that is easy to omit and fatal -- THE COMPARATOR, which must be something the shipped tree is not: #408's first recipe said to compare `initials()` against a folded-first partition, which is what `initials()` now IS, so it reproduced 0 where the bullet claimed 660 and was the only stated provenance for the number. Run the recipe as written before shipping the bullet. Cross-version numbers (a released wheel, the pre-change tree) are dated snapshots under Counting claims, since nothing in the repository re-runs them. Per-rule ledger toml comments asserting PARSER behavior cite rule IDs under the excerpt discipline, and since 2026-10-10 tests/v2/test_doc_citations.py checks them as it checks code. A ledger comment is copied from ledger to ledger and reworded by nothing: measured 2026-10-10 by running the check as it now stands over the tree before it (8e6e524a, master after #631), 27 quotes were stale and 5 colon citations quoted nothing, beyond the five #631's review found by hand -- most in a shape the colon-only check could not read (`rules.md#X -- "..."`, `rules.md#X's Accepted clause ("...")`, a second quote chained by "and"). A rule's statement and its `Accepted:` clauses are quotable, its example lines are not; `rules.md#H` names the H section, whose Background is quotable; a decisions.md citation may paraphrase, but what it quotes must be verbatim; and a reference with no quote must still name something that exists. The check's reach is a quote within four words of the ID or chained to one that is: a quote further off reads exactly like a name written in double quotes, so it is NOT checked, and the second review of #632 found 14 such quotes stale by hand. Keep a quote beside its ID. Quotes of AGENTS.md itself are not checked at all. Free prose is for ledger mechanics only (owned by tools/differential/README.md). **Writing the user docs (docs/*.rst) has its own AGENTS.md too.** `docs/AGENTS.md` carries the style the docs are written in — tables as indexes, subheadings per task, measured claims, recipe doctests — distilled from PR #588. It loads the same way as the docs/design/ one below; if your tool does not do nested discovery, read it before editing a `.rst` file under docs/. diff --git a/nameparser/_pipeline/_script_segment.py b/nameparser/_pipeline/_script_segment.py index 0c07e578..c7a5ce6f 100644 --- a/nameparser/_pipeline/_script_segment.py +++ b/nameparser/_pipeline/_script_segment.py @@ -715,9 +715,9 @@ def _split_surname_site(state: ParseState) -> ParseState: for j in state.segments[0]): return state # No try/except around the call: rules.md#A1's Accepted clause - # ("a user-supplied segmenter's own error propagates"). The two - # checks below are that same doctrine, curated, - # and they are where the line this module draws is easiest to state: + # ("a user-supplied segmenter's own error, which propagates"). The + # two checks below are that same doctrine, curated, and they are + # where the line this module draws is easiest to state: # a PROTOCOL VIOLATION BY THE SEGMENTER AUTHOR RAISES, while an # ADAPTER'S DEFENSE AGAINST ITS LIBRARY DECLINES. Both checks here # are the first kind -- a wrong answer TYPE and an answer indexing diff --git a/tests/v2/pipeline/test_pieces.py b/tests/v2/pipeline/test_pieces.py index 4f045239..1f6998aa 100644 --- a/tests/v2/pipeline/test_pieces.py +++ b/tests/v2/pipeline/test_pieces.py @@ -426,7 +426,7 @@ def test_the_walks_own_leading_piece_never_anchors_what_follows_it( def test_a_reserve_kept_leading_piece_beside_a_genuine_family_loss( ) -> None: - # decisions.md#S2's Accepted boundary ("an unambiguous suffix is + # rules.md#S2's Accepted clause ("an unambiguous suffix is # consumed even when that leaves no family name at all", 'Smith # Jr.' -> family='') applies just the same when the LEADING piece # is itself listed suffix vocabulary rather than an ordinary name: diff --git a/tests/v2/test_doc_citations.py b/tests/v2/test_doc_citations.py index 7153c45d..d18fd05b 100644 --- a/tests/v2/test_doc_citations.py +++ b/tests/v2/test_doc_citations.py @@ -1,7 +1,17 @@ """Referential integrity for docs/design/ citations. -Checks: cited rule/mechanism IDs exist; citation sentences are -whitespace-normalized verbatim excerpts of their statements; +Checks: every reference to a design doc, in code and in the +differential ledgers' comments, names a rule, section, mechanism or +decisions entry that exists; every quote a citation carries -- in any +of the shapes _LEAD_RE accepts, chained quotes included -- is a +normalized verbatim excerpt of what it cites ("[...]" eliding, in +order): a rule's statement and Accepted: clauses, a section's +Background, a mechanism's Contract statement, a decisions entry's +text. Normalized means whitespace, case, "--" for the em dash and ' +for a nested ", so emphasis in capitals is not checked. A quote +standing more than four words from its ID and chained to nothing is +prose to this module and is NOT checked: the reader cannot tell it +from a name written in double quotes, which is the common case; ``implemented:`` lists match the set of modules actually citing the rule; ``interacts:`` IDs exist (existence only -- the field is advisory). The legacy-pattern check (armed) keeps gitignored-spec @@ -11,8 +21,10 @@ import re from pathlib import Path +from typing import NamedTuple -from tests.v2.rules_doc import RULES_DOC, parse_rules_doc +from tests.v2.rules_doc import ( + _POINTER_RE, _RULE_RE, RULES_DOC, parse_rules_doc) REPO = Path(__file__).resolve().parents[2] MECH_DOC = REPO / "docs" / "design" / "mechanisms.md" @@ -22,29 +34,99 @@ re.compile(r"plan[\s#]+deviation"), re.compile(r"spec\s+[S§]?\d")) -_CITE_RE = re.compile( - r"(?:rules|mechanisms|decisions)\.md#" - r"(?P[A-Z]\d+|[A-Z][A-Z0-9_]*(?:-[A-Z0-9_]+)*)" - r":\s*(?P.*)") -# The excerpt is the FIRST double-quoted span after the ID, wrapped -# over continuation comment lines; text outside the quotes (v1 names, -# history pointers, code-local notes) is free. -_EXCERPT_RE = re.compile(r'"(.*?)"') +# A reference to a design doc: the doc and the ID or entry key. A +# decisions.md key may be a lowercase slug ('suffix-acronym-collisions'). +_MENTION_RE = re.compile( + r"(?Prules|mechanisms|decisions)\.md#" + r"(?P[A-Za-z][A-Za-z0-9_]*(?:-[A-Za-z0-9_]+)*)") +# A reference is a CITATION when a quote follows it in one of the +# shapes the tree writes: the colon form (`rules.md#S2: "..."`, where +# prose may stand before the quote), or the ID, an optional "'s", up +# to four words and optionally a dash, colon, comma or parenthesis, +# then the +# quote (`rules.md#P5 -- "..."`, `rules.md#M2's Accepted clause +# ("...")`, `rules.md#S2 consumes "..."`, `rules.md#M2 -- the +# numeral is taken "..."`). Any other reference is a +# pointer, which must still name something that exists. +_COLON_RE = re.compile(r"\s*:") +_LEAD_RE = re.compile( + r"(?:'s)?(?:\s*(?:--|—))?(?:\s+[A-Za-z]+){0,4}?" + r"\s*(?:--|—|[:,(])?\s*(?=\")") +# A quote, and the quotes chained to it by "and", a comma or a dash: +# every one of them is part of the citation and must be verbatim. +_QUOTE_RE = re.compile(r'"([^"]*)"') +_CHAIN_RE = re.compile(r'\s*,?\s*(?:and\s+|--\s*|—\s*)?(?=")') +# An excerpt may elide with "[...]": each fragment must then be +# verbatim, non-empty, and in order. +_ELISION = "[...]" +DEC_DOC = REPO / "docs" / "design" / "decisions.md" +# Ledger comments cite rules under the same discipline as code +# (AGENTS.md, "Release-log claims"); they are swept with the code but +# are not "citing modules" -- rules.md's implemented: names parser +# modules, never ledgers. +_LEDGERS = REPO / "tools" / "differential" + + +class Citation(NamedTuple): + path: Path + line: int + doc: str + cid: str + excerpts: tuple[str, ...] # normalized; () for a pointer + colon: bool # written in the colon form + lead: str # the text between the ID and its quote def _norm(s: str) -> str: + # Case folds because a quote routinely lowercases the first letter + # of the sentence it lifts; "--" is how an ASCII comment spells the + # em dash the docs use, and a quote nested in a quoted excerpt can + # only be written single + s = s.replace(" -- ", " — ").replace('"', "'") return " ".join(s.split()).lower() def _statements() -> dict[str, str]: - out: dict[str, str] = {} - text = RULES_DOC.read_text(encoding="utf-8") - for block in re.split(r"^(?=[A-Z]\d+\.\s)", text, flags=re.M): - m = re.match(r"([A-Z]\d+)\.\s(.*)", block, flags=re.S) + """What a citation may quote: for a rule, its prose -- the + statement and its Accepted: clauses, everything in the block but + example, pointer and marker lines; for a section letter + (`rules.md#H Background`), the section's Background paragraph; for + a mechanism, its Contract statement.""" + out: dict[str, list[str]] = {} + current: list[str] | None = None + background: list[str] | None = None + section = None + marker = False # inside a wrapped no-boundary: marker + for line in RULES_DOC.read_text(encoding="utf-8").splitlines(): + m = _RULE_RE.match(line) + hm = re.match(r"## .*\(([A-Z])\)\s*$", line) + if background is not None: + if line.strip(): + background.append(line) + continue + background = None + if hm: + section = hm.group(1) + elif section and line.startswith("Background:"): + background = out[section] = [line] + section = None + continue + if re.match(r"\s*(no-boundary|tolerated):", line): + marker = True + continue + if marker and (_POINTER_RE.match(line) or not line.strip() + or line.startswith(" ") or m): + marker = False if m: - # stop at the first example-like line, any indent/quote kind - body = re.split(r'\n\s+["“„\[]', m.group(2))[0] - out[m.group(1)] = _norm(body) + current = out.setdefault(m.group(1) + m.group(2), []) + current.append(line[m.end():]) + elif line.startswith("#"): + current = None + elif current is not None and line.startswith(" ") and not ( + marker or line.startswith(" ") + or _POINTER_RE.match(line)): + current.append(line) + stmts = {rid: _norm(" ".join(body)) for rid, body in out.items()} mech = MECH_DOC.read_text(encoding="utf-8") # split into sections first so a missing Contract line cannot # bleed into the next section's statement @@ -55,51 +137,190 @@ def _statements() -> dict[str, str]: cm = re.search(r"Contract statement[.:*]*\s*(?P.+?)(?=\n\n|\Z)", sec, flags=re.S) if cm: - out[hm.group("slug")] = _norm(cm.group("stmt")) + stmts[hm.group("slug")] = _norm(cm.group("stmt")) + return stmts + + +def _decision_entries() -> dict[str, str]: + """decisions.md entry bodies keyed by their ### key. An entry is a + dated record rather than a statement, so a citation of one need + not quote it; but what it does quote must be verbatim. An arc + spread over several headings ('### differential-ledger, the ... + arc') answers to its key with all of them.""" + out: dict[str, str] = {} + text = DEC_DOC.read_text(encoding="utf-8") + for sec in re.split(r"^(?=#{2,3} )", text, flags=re.M): + hm = re.match(r"### ([^\s—,]+)", sec) + if hm: + out[hm.group(1)] = out.get(hm.group(1), "") + " " + _norm(sec) return out -def _citations() -> list[tuple[Path, int, str, str]]: +def _is_excerpt(excerpt: str, body: str) -> bool: + pos = 0 + for frag in excerpt.split(_ELISION): + frag = frag.strip() + if not frag: + return False + found = body.find(frag, pos) + if found < 0: + return False + pos = found + len(frag) + return True + + +def _swept_files() -> list[Path]: + # this module's own comments show citation shapes, citing nothing + files = [p for d in SWEEP_DIRS for p in sorted((REPO / d).rglob("*.py")) + if p != Path(__file__).resolve()] + return files + sorted(_LEDGERS.glob("*.toml")) + + +def _quotes(text: str, at: int) -> tuple[str, ...]: + """The quote opening at `at` and every quote chained to it.""" + out = [] + while (qm := _QUOTE_RE.match(text, at)): + out.append(_norm(qm.group(1))) + cm = _CHAIN_RE.match(text, qm.end()) + if not cm: + break + at = cm.end() + return tuple(out) + + +def _citations() -> list[Citation]: + """Every reference to a design doc, with what it quotes. A + reference's text runs to the next reference, over the comment + lines that continue it; on a line that is not a comment (a + docstring, a string) it ends with the line, since a quote there + may be the string's own delimiter.""" found = [] - for d in SWEEP_DIRS: - for path in sorted((REPO / d).rglob("*.py")): - lines = path.read_text(encoding="utf-8").splitlines() - for i, line in enumerate(lines): - m = _CITE_RE.search(line) - if not m: - continue - block = [m.group("first")] - for cont in lines[i + 1:]: - cs = cont.strip() - if cs.startswith("#") and not _CITE_RE.search(cont): - block.append(cs.lstrip("# ")) - else: - break - qm = _EXCERPT_RE.search(" ".join(block)) - found.append((path, i + 1, m.group("cid"), - _norm(qm.group(1)) if qm else "")) + for path in _swept_files(): + lines = path.read_text(encoding="utf-8").splitlines() + for i, line in enumerate(lines): + for m in _MENTION_RE.finditer(line): + cid, text = m.group("cid"), line[m.end():] + if re.fullmatch(r"-[\"']?", text) and i + 1 < len(lines): + # a key wrapped at its hyphen ('ONE-PREDICATE-PER-' + # then 'QUESTION' on the next line) + wm = re.match(r"[\s#\"']*([A-Za-z0-9_]+" + r"(?:-[A-Za-z0-9_]+)*)", lines[i + 1]) + if wm: + cid += "-" + wm.group(1) + if "#" in line[:m.start()]: + for cont in lines[i + 1:]: + cs = cont.strip() + if not cs.startswith("#"): + break + text += " " + re.sub(r"^#+:?\s*", "", cs) + nxt = _MENTION_RE.search(text) + if nxt: + text = text[:nxt.start()] + colon = bool(_COLON_RE.match(text)) + doc = m.group("doc") + if colon and doc != "decisions": + # the colon form's quote may follow a paraphrase + q = text.find('"') + else: + lm = _LEAD_RE.match(text) + q = lm.end() if lm else -1 + excerpts = _quotes(text, q) if q >= 0 else () + found.append(Citation(path, i + 1, doc, cid, excerpts, + colon, text[:q] if excerpts else "")) return found def test_citations_are_verbatim_excerpts() -> None: statements = _statements() + entries = _decision_entries() problems = [] - for path, lineno, cid, excerpt in _citations(): - if cid not in statements: - problems.append(f"{path}:{lineno}: cites unknown ID {cid}") - elif not excerpt: - problems.append( - f"{path}:{lineno}: citation of {cid} has no quoted excerpt") - elif excerpt not in statements[cid]: + for c in _citations(): + where = f"{c.path.relative_to(REPO)}:{c.line}" + known = entries if c.doc == "decisions" else statements + if c.cid not in known: + problems.append(f"{where}: {c.doc}.md#{c.cid} names nothing") + continue + if c.colon and c.doc != "decisions" and not c.excerpts: problems.append( - f"{path}:{lineno}: not a verbatim excerpt of {cid}") + f"{where}: citation of {c.cid} has no quoted excerpt") + for excerpt in c.excerpts: + if not _is_excerpt(excerpt, known[c.cid]): + problems.append( + f"{where}: not a verbatim excerpt of " + f"{c.doc}.md#{c.cid}: {excerpt[:60]!r}") assert not problems, "\n".join(problems) +def test_the_sweep_reaches_every_citation_shape() -> None: + # Each half of the sweep decides its own scope -- a glob, a lead + # grammar, an elision, the decisions routing -- and one matching + # nothing would let the excerpt test pass on what is left. + cites = _citations() + ledgers = {c.path.name for c in cites if c.excerpts + and c.path.suffix == ".toml"} + assert {"expected_since_1.4.0.toml", + "expected_since_2.3.0.toml"} <= ledgers + leads = [c.lead for c in cites if c.excerpts and not c.colon] + assert any(ld.startswith("'s") for ld in leads) + assert any("(" in ld for ld in leads) + assert any(ld.strip().startswith("--") for ld in leads) + assert any(re.fullmatch(r"(?:\s+[A-Za-z]+)+\s*", ld) for ld in leads) + assert any(c.excerpts and c.doc == "decisions" for c in cites) + assert any(len(c.excerpts) > 1 for c in cites) + assert any(_ELISION in e for c in cites for e in c.excerpts) + + +def test_an_elided_excerpt_must_keep_its_order() -> None: + assert _is_excerpt("a b [...] d", "a b c d") + assert not _is_excerpt("d [...] a b", "a b c d") + assert not _is_excerpt("[...]", "a b c d") + assert not _is_excerpt("a [...]", "a b c d") + + +def test_the_quotable_text_is_the_prose_and_only_the_prose() -> None: + stmts = _statements() + # an Accepted: clause, and a Background's wrapped lines, are quotable + assert _is_excerpt(_norm("an unambiguous suffix is consumed even " + "when that leaves no family name at all"), + stmts["S2"]) + assert _is_excerpt(_norm("another's opener"), stmts["N"]) + # an example line, and a wrapped no-boundary: marker, are not + assert not _is_excerpt(_norm("Smith Jr."), stmts["S2"]) + assert not _is_excerpt(_norm("its boundaries are the other rules"), + stmts["O4"]) + + +# The recorded negative control: excerpts that stood stale in the tree +# until #632, each still false against the text the check reads. They +# guard the QUOTABLE TEXT -- an Accepted clause or a fold loosening the +# match far enough to let one back in. The scanner's reach is the reach +# test's to guard. +_RETIRED_EXCERPTS = ( + ("P5", "a trailing roman numeral that assign reads as the " + "suffix (S2) is no word to spare, and is not joined"), + ("M2", "a bare acronym the reading declines is maiden text " + "all the same"), + ("S2", "A suffix never BEGINS a name: position outranks " + "the vocabulary match"), + ("C1", "The part is read as its words stand, before any " + "join"), + ("A1", "a kind is worth adding only if a reader would " + "hesitate too"), +) + + +def test_retired_excerpts_stay_rejected() -> None: + statements = _statements() + for cid, old in _RETIRED_EXCERPTS: + assert not _is_excerpt(_norm(old), statements[cid]), (cid, old) + + def test_implemented_matches_citing_modules() -> None: citing: dict[str, set[str]] = {} - for path, _lineno, cid, _x in _citations(): - citing.setdefault(cid, set()).add(str(path.relative_to(REPO))) + for c in _citations(): + if not c.colon or c.doc == "decisions" or c.path.suffix != ".py": + continue + citing.setdefault(c.cid, set()).add(str(c.path.relative_to(REPO))) problems = [] for rule in parse_rules_doc(RULES_DOC.read_text(encoding="utf-8")): actual = citing.get(rule.rule_id, set()) diff --git a/tools/differential/expected_since_1.4.0.toml b/tools/differential/expected_since_1.4.0.toml index 81d81793..011547ae 100644 --- a/tools/differential/expected_since_1.4.0.toml +++ b/tools/differential/expected_since_1.4.0.toml @@ -2002,11 +2002,11 @@ fields = ["given", "family"] [[change]] issue = "fix(#401) the bound-given reserve counts the trailing numeral assign reads as the suffix" -# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "a trailing -# roman numeral that assign reads as the suffix (S2) is no word to -# spare, and is not joined". first 'abdul Smith', last 'V' -> given -# 'abdul', family 'Smith', suffix 'V'; at this baseline the V was not -# read as a suffix, so all three fields move here where the 2.x +# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "A trailing roman +# numeral, or a bare acronym the run takes, or a trailing title word the +# run takes, is no word to spare". first 'abdul Smith', last 'V' -> +# given 'abdul', family 'Smith', suffix 'V'; at this baseline the V was +# not read as a suffix, so all three fields move here where the 2.x # ledgers list two. Both are rules.md examples; no differential corpus # name has the shape. The 'abdul Smith V Ph. D.' spelling stays out # of the corpus: a trailing split credential is this ledger's @@ -2071,21 +2071,24 @@ fields = ["given", "family"] [[change]] issue = "fix(#424) accepted: the maiden walk keeps the numeral an initial before the marker vetoes" -# 'J. née Jones Smith V': rules.md#M2 -- the numeral is taken "read as -# the take would leave the name, the word before the numeral being -# then the word before the marker". 1.4.0 has no maiden support and -# read middle 'née Jones', last 'Smith', suffix 'V'; 2.x reads maiden -# 'Jones Smith V' with the J. before the marker vetoing the fork, as -# 'J. V' reads the V as a name. The fix(#274) rule's fields omit -# the suffix v1 read, so this accepted reading names its own. +# 'J. née Jones Smith V': rules.md#M2 -- "It takes them up to the +# trailing run of post-nominals and titles that the end of the name +# reads as if the clause were not written". 1.4.0 has no maiden support +# and read middle 'née Jones', last 'Smith', suffix 'V'; 2.x reads +# maiden 'Jones Smith V' with the J. before the marker vetoing the fork, +# as 'J. V' reads the V as a name. The fix(#274) rule's fields omit the +# suffix v1 read, so this accepted reading names its own. name_regex = "(?i)^j\\.\\s+n[ée]e\\s+jones\\s+smith\\s+v$" fields = ["middle", "family", "suffix", "maiden"] [[change]] issue = "fix(#424) accepted: the chain keeps an acronym assign will not peel behind a title-and-particle word" -# 'Freiherr von Berg MA': rules.md#P2 -- the trailing suffix is read -# "over the pieces the chain leaves". #367 stops the leading-particle -# scan at a title-and-particle word, so the chain takes 'von Berg', +# 'Freiherr von Berg MA': P2 at #424 read the trailing suffix over the +# pieces the chain leaves; since #614 it is read first -- rules.md#P2: +# "a trailing suffix begins -- the trailing run S2 reads, once, before +# the chain is made" -- and the 2026-09-18 note below is the reading +# that ships. #367 stops the leading-particle scan at a +# title-and-particle word, so the chain takes 'von Berg', # and the acronym the fork counted with three pieces meets assign # with two; the chain asks the peel again and takes it. first 'von # Berg', last 'MA' -> family 'von Berg MA' (master's reading since @@ -2142,14 +2145,14 @@ fields = ["middle", "family", "suffix"] [[change]] issue = "fix(#424) the particle chain stops before the trailing numeral" -# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- -# read as assign will read it (S2): a trailing roman numeral, or a -# bare acronym with words to spare, ends the chain as a suffix word -# does". last 'van der Berg V' -> family 'van der Berg', suffix 'V', -# as 'John Smith V' reads. Shipped since 1.x: the chain's stop asked -# with the suffix-piece test, whose initial veto does not see a bare -# V. A rules.md example; no differential corpus name has the shape. -# Without this rule the name landed on the fields-only +# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- the +# trailing run S2 reads, once, before the chain is made [...] a trailing +# roman numeral, or a bare acronym with words to spare, ends the chain +# as a suffix word does". last 'van der Berg V' -> family 'van der +# Berg', suffix 'V', as 'John Smith V' reads. Shipped since 1.x: the +# chain's stop asked with the suffix-piece test, whose initial veto does +# not see a bare V. A rules.md example; no differential corpus name has +# the shape. Without this rule the name landed on the fields-only # fix(suffix-routing) catch-all -- the count was the tell. #451 # deleted that catch-all, so this ledger has no fields-only rule left # to absorb anything. @@ -2158,15 +2161,16 @@ fields = ["family", "suffix"] [[change]] issue = "fix(#424/#445) accepted: the maiden walk keeps a bare acronym, and the lone name word is the family" -# 'John née Jones Smith Ma': rules.md#M2 -- "a bare acronym the -# reading declines is maiden text all the same". middle -# 'née Jones', family 'Smith', suffix 'Ma' -> maiden 'Jones Smith Ma': -# v1 had no maiden support; 2.0 has read the name so since #274, and -# since #533 the walk asks the acronym fork rather than leaving it: -# what declines this word is the WRITING, Title case saying name -# where capitals would say credential, and the count such a reading -# needs is taken over the name the take would leave rather than over -# the words as they stand. The fix(#274) rule +# 'John née Jones Smith Ma': rules.md#M2 -- "It takes them up to the +# trailing run of post-nominals and titles that the end of the name +# reads as if the clause were not written", and 'John Ma' reads the Ma +# as the family. middle 'née Jones', family 'Smith', suffix 'Ma' -> +# maiden 'Jones Smith Ma': v1 had no maiden support; 2.0 has read the +# name so since #274, and since #533 the walk asks the acronym fork +# rather than leaving it: what declines this word is the WRITING, Title +# case saying name where capitals would say credential, and the count +# such a reading needs is taken over the name the take would leave +# rather than over the words as they stand. The fix(#274) rule # cannot carry it: its fields omit the suffix v1 read. A rules.md # Accepted example, first witnessed here. # @@ -2734,10 +2738,10 @@ fields = ["nickname", "maiden"] [[change]] issue = "fix(#335) a marker-led clause leaves the one name word its bare reading" # 'Smith (née Jones)', which rules.md#N3 carries as an Accepted line. -# N3 reads a name that is only a nickname plus one name word as "that -# word is the family name", and a marker-led clause is not a nickname -# clause, so N3 no longer reaches this shape: the word keeps the -# reading the bare "Smith née Jones" gives it. +# rules.md#N3: "A name that is only a nickname and one name word reads +# that word as the family name", and a marker-led clause is not a +# nickname clause, so N3 no longer reaches this shape: the word keeps +# the reading the bare "Smith née Jones" gives it. # # Its own rule rather than a sixth alternative in the fix(#335) rule # above, for the reason that rule states: one rule holding both would @@ -2994,8 +2998,8 @@ issue = "fix(suffix-routing) a two-token name ending in a credential acronym kee # positional reading and are not tracked as deviations." 'Donald mc' # has no comma, so P6 never reaches it. # -# What stands instead is rules.md#S2: "an unambiguous suffix is -# consumed even when that leaves no family name at all". That is the +# What stands instead is rules.md#S2: "an unambiguous suffix is consumed +# even when that leaves no family name at all". That is the # authority here, and NOT P6's scope note, which does not support it: # P6 promises the comma-less shapes keep their POSITIONAL reading, and # for the words that are both particle and suffix vocabulary it does @@ -3006,7 +3010,7 @@ issue = "fix(suffix-routing) a two-token name ending in a credential acronym kee # 2026-09-07: P6's Accepted clause now names vd and mc as the two words # the positional reading does not hold for, which is what this rule has # been classifying all along. This rule is unchanged by that repair and -# still rests on S2's statement. +# still rests on S2, its Accepted clause as quoted above. # # The two neighbouring shapes both already have rules above, and the # contrast is the point: 'Mc Donald' has the particle LEADING and folds @@ -4050,25 +4054,25 @@ issue = "fix(#563) paired initials after a comma read as the given name unless a # two words before the comma may be one surname ('García Márquez, # G.J.'), so two dotted single letters alone after it stay the given # name and report the fork, where #516 had read them as a credential. -# "Only a credential in front of them, or another word the class -# admits by its dotted shape standing in the same part, makes them the -# credential run": 'John Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' -# move to the run, while 'De La Cruz, M.J. PhD' keeps its initials, -# the degree standing behind them, and so does 'García Márquez, Ms -# G.J.', a title in front ("a suffix word in front of them that is -# not also title vocabulary"), which moves nothing at any baseline. -# The second period is optional ('García Márquez, G.J'). Two pairs -# speaking only for each other make the run and report it behind two -# name words ('John Smith, X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, -# M.J. K.L.' is no longer that case -- 'De La Cruz' is one name word, -# so the count no longer reaches it and it reads as 'Cruz, M.J. K.L.' -# does, given 'M.J.', suffix 'K.L.', reporting both; a listed member -# in front speaks for nothing -# ('García Márquez, Ed G.J.' keeps the family comma, and the fix(#531) -# rule claims what its given part's trailing slot then reads). Against a 2.x baseline the names -# whose roles match the baseline's move only the report, which a -# 1.4.0 comparison cannot see; that ledger lists the three role movers -# alone. +# "Only an unambiguous suffix word in front of them that is not also +# title vocabulary, or another word the class admits by its dotted shape +# standing in the same part, makes them the credential run": 'John +# Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' move to the run, while +# 'De La Cruz, M.J. PhD' keeps its initials, the degree standing behind +# them, and so does 'García Márquez, Ms G.J.', a title in front ("an +# unambiguous suffix word in front of them that is not also title +# vocabulary"), which moves nothing at any baseline. The second period +# is optional ('García Márquez, G.J'). Two pairs speaking only for each +# other make the run and report it behind two name words ('John Smith, +# X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, M.J. K.L.' is no longer +# that case -- 'De La Cruz' is one name word, so the count no longer +# reaches it and it reads as 'Cruz, M.J. K.L.' does, given 'M.J.', +# suffix 'K.L.', reporting both; a listed member in front speaks for +# nothing ('García Márquez, Ed G.J.' keeps the family comma, and the +# fix(#531) rule claims what its given part's trailing slot then reads). +# Against a 2.x baseline the names whose roles match the baseline's move +# only the report, which a 1.4.0 comparison cannot see; that ledger +# lists the three role movers alone. name_regex = "^(?:De La Cruz, M\\.J\\. K\\.L\\.|García Márquez, PhD G\\.J\\.|John Smith, PhD X\\.Y\\.|John Smith, X\\.Y\\. P\\.Q\\.)$" fields = ["family", "given", "middle", "suffix", "title"] orders = ["DEFAULT"] @@ -4272,21 +4276,22 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 read as a post-nominal" -# Six names at landing (seven since #544, below) whose reading this -# change does not touch: each reads on -# this tree exactly as it read at 2f57ff21, measured name by name, -# and the whole of the 1.4.0 diff is v1 having no maiden field. The -# marker takes the words after it, the trailing member among them -- -# rules.md#M2's Accepted reading, which has shipped since #274 and -# which #533 leaves standing wherever the WRITING declines the word -# (a Title-cased member in a mixed-case name: 'Jane Doe nee Smith -# Ma', 'Jane Doe nee Yo-Yo Ma', 'Doe, Jane nee Smith Ma'), wherever -# the member is the only word the marker would leave ('Jane Doe nee -# MA'), and wherever P6's attachment claims the word instead ('Doe, -# Jane nee Smith do', 'Doe, Jane nee Smith MA do'). What #533 adds on -# these names is a REPORT, and 1.4.0 has no surface to compare one -# against -- so nothing of this change is visible from here, and the -# rule is named for the change that is. +# Six names at landing (seven from #544; three since #601, which moved +# the three comma names to fix(#601) and the Jr. name to fix(#601/#602), +# below) whose reading this change does not touch: each reads on this +# tree exactly as it read at 2f57ff21, measured name by name, and the +# whole of the 1.4.0 diff is v1 having no maiden field. The marker takes +# the words after it, the trailing member among them -- rules.md#M2's +# Accepted reading, which has shipped since #274 and which #533 leaves +# standing wherever the WRITING declines the word (a Title-cased member +# in a mixed-case name: 'Jane Doe nee Smith Ma', 'Jane Doe nee Yo-Yo +# Ma', 'Doe, Jane nee Smith Ma'), wherever the member is the only word +# the marker would leave ('Jane Doe nee MA'), and wherever P6's +# attachment claims the word instead ('Doe, Jane nee Smith do', 'Doe, +# Jane nee Smith MA do'). What #533 adds on these names is a REPORT, and +# 1.4.0 has no surface to compare one against -- so nothing of this +# change is visible from here, and the rule is named for the change that +# is. # # Its own rule rather than a widening of fix(#274) above, and the # reason is that rule's `fields`: it stops at maiden/middle/family @@ -4302,12 +4307,16 @@ issue = "fix(#274/#424) accepted: a maiden clause keeps a trailing credential v1 # # 2026-09-27, #544: a seventh joins, 'Jane Doe Jr. nee Smith Ma' -- # maiden 'Smith Ma', suffix 'Jr.', where v1 read middle 'Doe Jr. nee', -# last 'Smith', suffix 'Ma'. It reads so at 2.0.0 through 2.3.0, at the -# tree before #544 and at the tree, the report aside: rules.md#M2's -# Accepted boundary, "a credential written in front of the marker speaks -# for no word of the clause". #544 decided that boundary and moves no role -# on the name, so the 1.4.0 diff is this rule's, the Title-case member's -# writing declining it as it does for 'Jane Doe nee Smith Ma'. +# last 'Smith', suffix 'Ma'. It read so at 2.0.0 through 2.3.0, at the +# tree before #544 and at the tree, the report aside, by M2's Accepted +# boundary as it then stood: a credential in front of the marker spoke +# for no word of the clause. #544 decided that boundary and moved no +# role on the name, so the 1.4.0 diff was this rule's, the Title-case +# member's writing declining it as it does for 'Jane Doe nee Smith Ma'. +# #601 (2026-10-04) moved it on: it reads suffix 'Jr. nee Smith +# Ma' -- rules.md#M2: "a marker behind a credential is an ordinary word, +# and the credential's run (S2) takes it in with everything after it" -- +# and fix(#601/#602) below claims it. name_regex = "^(?:Jane Doe nee MA|Jane Doe nee Smith Ma|Jane Doe nee Yo-Yo Ma)$" fields = ["family", "maiden", "middle", "suffix"] @@ -4907,10 +4916,11 @@ fields = ["title", "given"] [[change]] # rules.md#S2: "The run at a name's trailing slot is read ONCE, once # the connectives have joined (P3) and before the particle chain (P2) -# and the bound given-name join (P5) are made" (#614, 2026-10-07), and -# rules.md#H5: "a particle chain (P2) stops in front of a trailing -# title rather than taking it into the family name". This baseline -# made the chain first and the title was a word of the family name. +# and the bound given-name join (P5) are made" and "neither join +# reaches a word the run took" (#614, 2026-10-07), and rules.md#H5: "a +# particle chain (P2) stops in front of a trailing title rather than +# taking it into the family name". This baseline made the chain first +# and the title was a word of the family name. issue = "fix(#614) a trailing title is read before the particle chain" name_regex = "^John van der Berg Prof\\.$" fields = ["family", "title"] diff --git a/tools/differential/expected_since_2.0.0.toml b/tools/differential/expected_since_2.0.0.toml index f638ef33..e213596e 100644 --- a/tools/differential/expected_since_2.0.0.toml +++ b/tools/differential/expected_since_2.0.0.toml @@ -142,10 +142,15 @@ fields = ["_ambiguities"] [[change]] issue = "feat(#449) an interpunct transcription declines the script order, so the convention decides it" -# '王·Smith'. rules.md#T3: the 間隔号 marks a transcription and -# suppresses the script_orders lookup whole, so name_order governs and -# the one remaining name word is O5's. Literal-anchored, one corpus -# name. +# '王·Smith'. rules.md#T3: "The interpunct divides a name only between +# two characters of a classified East Asian script; anywhere else it is +# part of the word" -- 'Smith' is Latin, so the name is one word, and +# '王Smith' reads the same. rules.md#W4: "A name written wholly in one +# East Asian script, or in the kana-licensed Japanese repertoire, reads +# family-first whatever order the caller declared" -- a word mixing Han +# and Latin is neither, so no script order applies, name_order governs +# and the one name word is O5's. The issue names the interpunct, which +# is incidental to this name. Literal-anchored, one corpus name. name_regex = "^王·Smith$" fields = ["_ambiguities"] @@ -769,10 +774,10 @@ fields = ["given", "family"] [[change]] issue = "fix(#401) the bound-given reserve counts the trailing numeral assign reads as the suffix" -# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "a trailing -# roman numeral that assign reads as the suffix (S2) is no word to -# spare, and is not joined". given 'abdul Smith', family '' -> given -# 'abdul', family 'Smith'; the suffix was already read at this +# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "A trailing roman +# numeral, or a bare acronym the run takes, or a trailing title word the +# run takes, is no word to spare". given 'abdul Smith', family '' -> +# given 'abdul', family 'Smith'; the suffix was already read at this # baseline, which is why the 1.4.0 rule lists three fields and this # one two. Both are rules.md examples; no differential corpus name # has the shape, and the 'abdul Smith V Ph. D.' spelling stays out of @@ -821,15 +826,18 @@ fields = ["given", "middle", "suffix"] [[change]] issue = "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view" -# 'abdul Smith Jr Ma', 'abdul Smith Ma': rules.md#P5 -- "the join is -# tried on the pieces as it would leave them, assign's trailing peel -# (S2) is read over that, and the name words it leaves are the words -# to spare" and "a word the peel reads as a suffix unjoined must read -# so joined, or the join declines". given 'abdul Smith', family '' -> -# given 'abdul', family 'Smith' (the acronym and the suffix peel as -# they do for 'John Smith Jr Ma'); given 'abdul Smith', family 'Ma' -# -> given 'abdul', family 'Smith', suffix 'Ma' (the join would have -# turned a credential into the family). Both 1.4.0 parity, so the +# 'abdul Smith Jr Ma', 'abdul Smith Ma': P5 as it landed tried the join +# on the pieces as it would leave them and read the trailing peel over +# that, a word the peel read as a suffix unjoined having to read so +# joined. Since #614 rules.md#P5 reads the run first -- "the trailing +# suffix run (S2) and the trailing title run (H5), each read over what +# the other leaves until neither takes anything more, are read once, +# before the join is tried" -- and the #289 note below records how +# 'abdul Smith Ma' has read since 2026-09-18. given 'abdul Smith', +# family '' -> given 'abdul', family 'Smith' (the acronym and the suffix +# peel as they do for 'John Smith Jr Ma'); given 'abdul Smith', family +# 'Ma' -> given 'abdul', family 'Smith', suffix 'Ma' (the join would +# have turned a credential into the family). Both 1.4.0 parity, so the # 1.4.0 ledger has no twin. Rules.md examples; no differential corpus # name has either shape. # @@ -1127,14 +1135,14 @@ fields = ["family", "suffix", "_ambiguities"] [[change]] issue = "fix(#424) the particle chain stops before the trailing numeral" -# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- -# read as assign will read it (S2): a trailing roman numeral, or a -# bare acronym with words to spare, ends the chain as a suffix word -# does". last 'van der Berg V' -> family 'van der Berg', suffix 'V', -# as 'John Smith V' reads. Shipped since 1.x: the chain's stop asked -# with the suffix-piece test, whose initial veto does not see a bare -# V. A rules.md example; no differential corpus name has the shape. -# The fork's report arrives with the reading. +# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- the +# trailing run S2 reads, once, before the chain is made [...] a trailing +# roman numeral, or a bare acronym with words to spare, ends the chain +# as a suffix word does". last 'van der Berg V' -> family 'van der +# Berg', suffix 'V', as 'John Smith V' reads. Shipped since 1.x: the +# chain's stop asked with the suffix-piece test, whose initial veto does +# not see a bare V. A rules.md example; no differential corpus name has +# the shape. The fork's report arrives with the reading. name_regex = "(?i)^john\\s+van\\s+der\\s+berg\\s+v$" fields = ["family", "suffix", "_ambiguities"] @@ -1163,13 +1171,14 @@ fields = ["family", "suffix", "_ambiguities"] [[change]] issue = "fix(#424/#445) the maiden walk stops before the trailing numeral, and the lone name word is the family" -# 'John née Jones Smith V': rules.md#M2 -- "takes the words after it -# -- up to any suffix word, or the trailing roman numeral assign -# reads as the suffix (S2) -- as the maiden name". maiden 'Jones -# Smith V' -> maiden 'Jones Smith', suffix 'V'. The walk's stop asked -# with the suffix-piece test too. No 1.4.0 twin: v1 had no maiden -# support, and the fix(#274) rule carries the name there. A rules.md -# example; no differential corpus name has the shape. +# 'John née Jones Smith V': rules.md#M2 -- "It takes them up to the +# trailing run of post-nominals and titles that the end of the name +# reads as if the clause were not written", and 'John V' reads the V as +# the suffix. maiden 'Jones Smith V' -> maiden 'Jones Smith', suffix +# 'V'. The walk's stop asked with the suffix-piece test too. No 1.4.0 +# twin: v1 had no maiden support, and the fix(#274) rule carries the +# name there. A rules.md example; no differential corpus name has the +# shape. # # Renamed for #445 (2026-08-27) rather than widened in place, because # the diff has two causes now and one rule has to explain the whole @@ -1642,10 +1651,10 @@ fields = ["given", "family", "nickname", "maiden"] [[change]] issue = "fix(#335) a marker-led clause leaves the one name word its bare reading" # 'Smith (née Jones)', which rules.md#N3 carries as an Accepted line. -# N3 reads a name that is only a nickname plus one name word as "that -# word is the family name", and a marker-led clause is not a nickname -# clause, so N3 no longer reaches this shape: the word keeps the -# reading the bare "Smith née Jones" gives it -- `given` when this +# rules.md#N3: "A name that is only a nickname and one name word reads +# that word as the family name", and a marker-led clause is not a +# nickname clause, so N3 no longer reaches this shape: the word keeps +# the reading the bare "Smith née Jones" gives it -- `given` when this # rule was written, and `family` since #445 gave the bare spelling # the same reading from the other side. # @@ -1741,7 +1750,7 @@ fields = ["given", "middle", "family", "maiden"] [[change]] issue = "fix(#371) a suffix never begins a name: the Ph./D. merge declines at the head" # 'Ph. D. Van Johnson', 'Ph. D. John Smith', 'Ph. D.', 'Ph. D., Jr.': -# rules.md#S2 -- "A suffix never BEGINS a name: position outranks the +# rules.md#S2 -- "A suffix never OPENS THE STRING: position outranks the # vocabulary match". The v1 fix_phd merge joined the pair wherever it # stood, which is what made a leading credential possible at all, and # it emptied the family doing it: 'Ph. D. Van Johnson' read given @@ -2598,8 +2607,10 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#562) a comma credential run holding a particle chain reads by the name-word count" -# rules.md#C1: "The part is read as its words stand, before any join" -# (#613, 2026-10-06): no particle chain reaches it, so each member's +# rules.md#C1: "A part read as postnominal [...] is read as its words +# stand, before any join, and that reading is final: [...] no particle +# run, connective join or bound given name reaches into it" (#613, +# 2026-10-06): no particle chain reaches it, so each member's # capitals settle it and the run reads whole as the credential run -- # silently since #613, where #562 (2026-10-01) had read the pair by the # count and reported the flip. The family-comma path had read the part as name text: @@ -2622,25 +2633,25 @@ issue = "fix(#563) paired initials after a comma read as the given name unless a # two words before the comma may be one surname ('García Márquez, # G.J.'), so two dotted single letters alone after it stay the given # name and report the fork, where #516 had read them as a credential. -# "Only a credential in front of them, or another word the class -# admits by its dotted shape standing in the same part, makes them the -# credential run": 'John Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' -# move to the run, while 'De La Cruz, M.J. PhD' keeps its initials, -# the degree standing behind them, and so does 'García Márquez, Ms -# G.J.', a title in front ("a suffix word in front of them that is -# not also title vocabulary"), which moves nothing at any baseline. -# The second period is optional ('García Márquez, G.J'). Two pairs -# speaking only for each other make the run and report it behind two -# name words ('John Smith, X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, -# M.J. K.L.' is no longer that case -- 'De La Cruz' is one name word, -# so the count no longer reaches it and it reads as 'Cruz, M.J. K.L.' -# does, given 'M.J.', suffix 'K.L.', reporting both; a listed member -# in front speaks for nothing -# ('García Márquez, Ed G.J.' keeps the family comma, and the fix(#531) -# rule claims what its given part's trailing slot then reads). Against a 2.x baseline the names -# whose roles match the baseline's move only the report, which a -# 1.4.0 comparison cannot see; that ledger lists the three role movers -# alone. +# "Only an unambiguous suffix word in front of them that is not also +# title vocabulary, or another word the class admits by its dotted shape +# standing in the same part, makes them the credential run": 'John +# Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' move to the run, while +# 'De La Cruz, M.J. PhD' keeps its initials, the degree standing behind +# them, and so does 'García Márquez, Ms G.J.', a title in front ("an +# unambiguous suffix word in front of them that is not also title +# vocabulary"), which moves nothing at any baseline. The second period +# is optional ('García Márquez, G.J'). Two pairs speaking only for each +# other make the run and report it behind two name words ('John Smith, +# X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, M.J. K.L.' is no longer +# that case -- 'De La Cruz' is one name word, so the count no longer +# reaches it and it reads as 'Cruz, M.J. K.L.' does, given 'M.J.', +# suffix 'K.L.', reporting both; a listed member in front speaks for +# nothing ('García Márquez, Ed G.J.' keeps the family comma, and the +# fix(#531) rule claims what its given part's trailing slot then reads). +# Against a 2.x baseline the names whose roles match the baseline's move +# only the report, which a 1.4.0 comparison cannot see; that ledger +# lists the three role movers alone. name_regex = "^(?:De La Cruz, M\\.J\\. K\\.L\\.|De La Cruz, M\\.J\\. PhD|García Márquez, G\\.J|García Márquez, G\\.J\\.|García Márquez, PhD G\\.J\\.|John Smith, A\\.B\\.|John Smith, A\\.B\\. Ph\\.D\\.|John Smith, PhD X\\.Y\\.|John Smith, X\\.Y\\. P\\.Q\\.)$" fields = ["family", "given", "middle", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -2794,8 +2805,8 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # text whole. # # Reporting a DECLINED fork is #530's stated rule -- the report tracks -# the fork consulted, not the lean -- and rules.md#A1's "a kind is -# worth adding only if a reader would hesitate too" is the standing +# the fork consulted, not the lean -- and AGENTS.md's "A kind is worth +# adding only if a reader would hesitate too" is the standing # objection it answers. It is also the only part of #531 that adds a # report without moving a field, which is why it is a rule of its own # rather than a widening of the one above: `fields` is @@ -2818,12 +2829,13 @@ issue = "fix(#531) capitals take the do collision from the family-comma particle # 'Doe, John DO', alone, because the word is alone in the class: `do` # is the one ambiguous credential that is also particle vocabulary, so # this slot and P6's attachment (rules.md#P6) want the same word. -# Derek's decision, recorded at decisions.md#S2: CAPITALS DECIDE, AND -# THE PARTICLE RULE KEEPS EVERY OTHER SPELLING. An all-caps member in -# a name written in more than one case carries a positive credential -# lean, so the credential reading wins and P6 stands down; every other -# spelling attaches exactly as it did, with P6's own -# `particle-or-given` and no second report. +# Derek's decision, recorded at decisions.md#S2: "CAPITALS DECIDE FOR +# `do`, AND THE PARTICLE RULE KEEPS EVERY OTHER SPELLING". An all-caps +# member in a name written in more than one case carries a positive +# credential lean, so the credential reading wins and P6 stands down +# (and since #544 so does an unambiguous credential in front of it, +# 'doe, jane v phd do'); every other spelling attaches exactly as it +# did, with P6's own `particle-or-given` and no second report. # # THE PAIRING IS THE ARGUMENT and the accepted cost is half of it. In # ONE CASE the rule cannot tell 'NASCIMENTO, EDSON ARANTES DO' from @@ -4019,10 +4031,11 @@ fields = ["_ambiguities", "suffix", "title"] [[change]] # rules.md#S2: "The run at a name's trailing slot is read ONCE, once # the connectives have joined (P3) and before the particle chain (P2) -# and the bound given-name join (P5) are made" (#614, 2026-10-07), and -# rules.md#H5: "a particle chain (P2) stops in front of a trailing -# title rather than taking it into the family name". This baseline -# made the chain first and the title was a word of the family name. +# and the bound given-name join (P5) are made" and "neither join +# reaches a word the run took" (#614, 2026-10-07), and rules.md#H5: "a +# particle chain (P2) stops in front of a trailing title rather than +# taking it into the family name". This baseline made the chain first +# and the title was a word of the family name. issue = "fix(#614) a trailing title is read before the particle chain" name_regex = "^John van der Berg Prof\\.$" fields = ["family", "title"] diff --git a/tools/differential/expected_since_2.1.0.toml b/tools/differential/expected_since_2.1.0.toml index 4c314d91..8a5268d9 100644 --- a/tools/differential/expected_since_2.1.0.toml +++ b/tools/differential/expected_since_2.1.0.toml @@ -166,10 +166,15 @@ fields = ["_ambiguities"] [[change]] issue = "feat(#449) an interpunct transcription declines the script order, so the convention decides it" -# '王·Smith'. rules.md#T3: the 間隔号 marks a transcription and -# suppresses the script_orders lookup whole, so name_order governs and -# the one remaining name word is O5's. Literal-anchored, one corpus -# name. +# '王·Smith'. rules.md#T3: "The interpunct divides a name only between +# two characters of a classified East Asian script; anywhere else it is +# part of the word" -- 'Smith' is Latin, so the name is one word, and +# '王Smith' reads the same. rules.md#W4: "A name written wholly in one +# East Asian script, or in the kana-licensed Japanese repertoire, reads +# family-first whatever order the caller declared" -- a word mixing Han +# and Latin is neither, so no script order applies, name_order governs +# and the one name word is O5's. The issue names the interpunct, which +# is incidental to this name. Literal-anchored, one corpus name. name_regex = "^王·Smith$" fields = ["_ambiguities"] @@ -425,10 +430,10 @@ fields = ["given", "family"] [[change]] issue = "fix(#401) the bound-given reserve counts the trailing numeral assign reads as the suffix" -# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "a trailing -# roman numeral that assign reads as the suffix (S2) is no word to -# spare, and is not joined". given 'abdul Smith', family '' -> given -# 'abdul', family 'Smith'; the suffix was already read at this +# 'abdul Smith V', 'abdul Smith Jr V': rules.md#P5 -- "A trailing roman +# numeral, or a bare acronym the run takes, or a trailing title word the +# run takes, is no word to spare". given 'abdul Smith', family '' -> +# given 'abdul', family 'Smith'; the suffix was already read at this # baseline, which is why the 1.4.0 rule lists three fields and this # one two. Both are rules.md examples; no differential corpus name # has the shape, and the 'abdul Smith V Ph. D.' spelling stays out of @@ -477,15 +482,18 @@ fields = ["given", "middle", "suffix"] [[change]] issue = "fix(#425/#436/#437) the bound-given reserve runs assign's peel over the joined view" -# 'abdul Smith Jr Ma', 'abdul Smith Ma': rules.md#P5 -- "the join is -# tried on the pieces as it would leave them, assign's trailing peel -# (S2) is read over that, and the name words it leaves are the words -# to spare" and "a word the peel reads as a suffix unjoined must read -# so joined, or the join declines". given 'abdul Smith', family '' -> -# given 'abdul', family 'Smith' (the acronym and the suffix peel as -# they do for 'John Smith Jr Ma'); given 'abdul Smith', family 'Ma' -# -> given 'abdul', family 'Smith', suffix 'Ma' (the join would have -# turned a credential into the family). Both 1.4.0 parity, so the +# 'abdul Smith Jr Ma', 'abdul Smith Ma': P5 as it landed tried the join +# on the pieces as it would leave them and read the trailing peel over +# that, a word the peel read as a suffix unjoined having to read so +# joined. Since #614 rules.md#P5 reads the run first -- "the trailing +# suffix run (S2) and the trailing title run (H5), each read over what +# the other leaves until neither takes anything more, are read once, +# before the join is tried" -- and the #289 note below records how +# 'abdul Smith Ma' has read since 2026-09-18. given 'abdul Smith', +# family '' -> given 'abdul', family 'Smith' (the acronym and the suffix +# peel as they do for 'John Smith Jr Ma'); given 'abdul Smith', family +# 'Ma' -> given 'abdul', family 'Smith', suffix 'Ma' (the join would +# have turned a credential into the family). Both 1.4.0 parity, so the # 1.4.0 ledger has no twin. Rules.md examples; no differential corpus # name has either shape. # @@ -793,14 +801,14 @@ fields = ["family", "suffix", "_ambiguities"] [[change]] issue = "fix(#424) the particle chain stops before the trailing numeral" -# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- -# read as assign will read it (S2): a trailing roman numeral, or a -# bare acronym with words to spare, ends the chain as a suffix word -# does". last 'van der Berg V' -> family 'van der Berg', suffix 'V', -# as 'John Smith V' reads. Shipped since 1.x: the chain's stop asked -# with the suffix-piece test, whose initial veto does not see a bare -# V. A rules.md example; no differential corpus name has the shape. -# The fork's report arrives with the reading. +# 'John van der Berg V': rules.md#P2 -- "a trailing suffix begins -- the +# trailing run S2 reads, once, before the chain is made [...] a trailing +# roman numeral, or a bare acronym with words to spare, ends the chain +# as a suffix word does". last 'van der Berg V' -> family 'van der +# Berg', suffix 'V', as 'John Smith V' reads. Shipped since 1.x: the +# chain's stop asked with the suffix-piece test, whose initial veto does +# not see a bare V. A rules.md example; no differential corpus name has +# the shape. The fork's report arrives with the reading. name_regex = "(?i)^john\\s+van\\s+der\\s+berg\\s+v$" fields = ["family", "suffix", "_ambiguities"] @@ -829,13 +837,14 @@ fields = ["family", "suffix", "_ambiguities"] [[change]] issue = "fix(#424/#445) the maiden walk stops before the trailing numeral, and the lone name word is the family" -# 'John née Jones Smith V': rules.md#M2 -- "takes the words after it -# -- up to any suffix word, or the trailing roman numeral assign -# reads as the suffix (S2) -- as the maiden name". maiden 'Jones -# Smith V' -> maiden 'Jones Smith', suffix 'V'. The walk's stop asked -# with the suffix-piece test too. No 1.4.0 twin: v1 had no maiden -# support, and the fix(#274) rule carries the name there. A rules.md -# example; no differential corpus name has the shape. +# 'John née Jones Smith V': rules.md#M2 -- "It takes them up to the +# trailing run of post-nominals and titles that the end of the name +# reads as if the clause were not written", and 'John V' reads the V as +# the suffix. maiden 'Jones Smith V' -> maiden 'Jones Smith', suffix +# 'V'. The walk's stop asked with the suffix-piece test too. No 1.4.0 +# twin: v1 had no maiden support, and the fix(#274) rule carries the +# name there. A rules.md example; no differential corpus name has the +# shape. # # Renamed for #445 (2026-08-27) rather than widened in place, because # the diff has two causes now and one rule has to explain the whole @@ -1564,10 +1573,10 @@ fields = ["nickname", "maiden"] [[change]] issue = "fix(#335) a marker-led clause leaves the one name word its bare reading" # 'Smith (née Jones)', which rules.md#N3 carries as an Accepted line. -# N3 reads a name that is only a nickname plus one name word as "that -# word is the family name", and a marker-led clause is not a nickname -# clause, so N3 no longer reaches this shape: the word keeps the -# reading the bare "Smith née Jones" gives it -- `given` when this +# rules.md#N3: "A name that is only a nickname and one name word reads +# that word as the family name", and a marker-led clause is not a +# nickname clause, so N3 no longer reaches this shape: the word keeps +# the reading the bare "Smith née Jones" gives it -- `given` when this # rule was written, and `family` since #445 gave the bare spelling # the same reading from the other side. # @@ -1663,7 +1672,7 @@ fields = ["given", "middle", "family", "maiden"] [[change]] issue = "fix(#371) a suffix never begins a name: the Ph./D. merge declines at the head" # 'Ph. D. Van Johnson', 'Ph. D. John Smith', 'Ph. D.', 'Ph. D., Jr.': -# rules.md#S2 -- "A suffix never BEGINS a name: position outranks the +# rules.md#S2 -- "A suffix never OPENS THE STRING: position outranks the # vocabulary match". The v1 fix_phd merge joined the pair wherever it # stood, which is what made a leading credential possible at all, and # it emptied the family doing it: 'Ph. D. Van Johnson' read given @@ -2486,8 +2495,10 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#562) a comma credential run holding a particle chain reads by the name-word count" -# rules.md#C1: "The part is read as its words stand, before any join" -# (#613, 2026-10-06): no particle chain reaches it, so each member's +# rules.md#C1: "A part read as postnominal [...] is read as its words +# stand, before any join, and that reading is final: [...] no particle +# run, connective join or bound given name reaches into it" (#613, +# 2026-10-06): no particle chain reaches it, so each member's # capitals settle it and the run reads whole as the credential run -- # silently since #613, where #562 (2026-10-01) had read the pair by the # count and reported the flip. The family-comma path had read the part as name text: @@ -2510,25 +2521,25 @@ issue = "fix(#563) paired initials after a comma read as the given name unless a # two words before the comma may be one surname ('García Márquez, # G.J.'), so two dotted single letters alone after it stay the given # name and report the fork, where #516 had read them as a credential. -# "Only a credential in front of them, or another word the class -# admits by its dotted shape standing in the same part, makes them the -# credential run": 'John Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' -# move to the run, while 'De La Cruz, M.J. PhD' keeps its initials, -# the degree standing behind them, and so does 'García Márquez, Ms -# G.J.', a title in front ("a suffix word in front of them that is -# not also title vocabulary"), which moves nothing at any baseline. -# The second period is optional ('García Márquez, G.J'). Two pairs -# speaking only for each other make the run and report it behind two -# name words ('John Smith, X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, -# M.J. K.L.' is no longer that case -- 'De La Cruz' is one name word, -# so the count no longer reaches it and it reads as 'Cruz, M.J. K.L.' -# does, given 'M.J.', suffix 'K.L.', reporting both; a listed member -# in front speaks for nothing -# ('García Márquez, Ed G.J.' keeps the family comma, and the fix(#531) -# rule claims what its given part's trailing slot then reads). Against a 2.x baseline the names -# whose roles match the baseline's move only the report, which a -# 1.4.0 comparison cannot see; that ledger lists the three role movers -# alone. +# "Only an unambiguous suffix word in front of them that is not also +# title vocabulary, or another word the class admits by its dotted shape +# standing in the same part, makes them the credential run": 'John +# Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' move to the run, while +# 'De La Cruz, M.J. PhD' keeps its initials, the degree standing behind +# them, and so does 'García Márquez, Ms G.J.', a title in front ("an +# unambiguous suffix word in front of them that is not also title +# vocabulary"), which moves nothing at any baseline. The second period +# is optional ('García Márquez, G.J'). Two pairs speaking only for each +# other make the run and report it behind two name words ('John Smith, +# X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, M.J. K.L.' is no longer +# that case -- 'De La Cruz' is one name word, so the count no longer +# reaches it and it reads as 'Cruz, M.J. K.L.' does, given 'M.J.', +# suffix 'K.L.', reporting both; a listed member in front speaks for +# nothing ('García Márquez, Ed G.J.' keeps the family comma, and the +# fix(#531) rule claims what its given part's trailing slot then reads). +# Against a 2.x baseline the names whose roles match the baseline's move +# only the report, which a 1.4.0 comparison cannot see; that ledger +# lists the three role movers alone. name_regex = "^(?:De La Cruz, M\\.J\\. K\\.L\\.|De La Cruz, M\\.J\\. PhD|García Márquez, G\\.J|García Márquez, G\\.J\\.|García Márquez, PhD G\\.J\\.|John Smith, A\\.B\\.|John Smith, A\\.B\\. Ph\\.D\\.|John Smith, PhD X\\.Y\\.|John Smith, X\\.Y\\. P\\.Q\\.)$" fields = ["family", "given", "middle", "suffix", "title", "_ambiguities"] orders = ["DEFAULT"] @@ -2682,8 +2693,8 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # text whole. # # Reporting a DECLINED fork is #530's stated rule -- the report tracks -# the fork consulted, not the lean -- and rules.md#A1's "a kind is -# worth adding only if a reader would hesitate too" is the standing +# the fork consulted, not the lean -- and AGENTS.md's "A kind is worth +# adding only if a reader would hesitate too" is the standing # objection it answers. It is also the only part of #531 that adds a # report without moving a field, which is why it is a rule of its own # rather than a widening of the one above: `fields` is @@ -2706,12 +2717,13 @@ issue = "fix(#531) capitals take the do collision from the family-comma particle # 'Doe, John DO', alone, because the word is alone in the class: `do` # is the one ambiguous credential that is also particle vocabulary, so # this slot and P6's attachment (rules.md#P6) want the same word. -# Derek's decision, recorded at decisions.md#S2: CAPITALS DECIDE, AND -# THE PARTICLE RULE KEEPS EVERY OTHER SPELLING. An all-caps member in -# a name written in more than one case carries a positive credential -# lean, so the credential reading wins and P6 stands down; every other -# spelling attaches exactly as it did, with P6's own -# `particle-or-given` and no second report. +# Derek's decision, recorded at decisions.md#S2: "CAPITALS DECIDE FOR +# `do`, AND THE PARTICLE RULE KEEPS EVERY OTHER SPELLING". An all-caps +# member in a name written in more than one case carries a positive +# credential lean, so the credential reading wins and P6 stands down +# (and since #544 so does an unambiguous credential in front of it, +# 'doe, jane v phd do'); every other spelling attaches exactly as it +# did, with P6's own `particle-or-given` and no second report. # # THE PAIRING IS THE ARGUMENT and the accepted cost is half of it. In # ONE CASE the rule cannot tell 'NASCIMENTO, EDSON ARANTES DO' from @@ -3964,10 +3976,11 @@ fields = ["_ambiguities", "suffix", "title"] [[change]] # rules.md#S2: "The run at a name's trailing slot is read ONCE, once # the connectives have joined (P3) and before the particle chain (P2) -# and the bound given-name join (P5) are made" (#614, 2026-10-07), and -# rules.md#H5: "a particle chain (P2) stops in front of a trailing -# title rather than taking it into the family name". This baseline -# made the chain first and the title was a word of the family name. +# and the bound given-name join (P5) are made" and "neither join +# reaches a word the run took" (#614, 2026-10-07), and rules.md#H5: "a +# particle chain (P2) stops in front of a trailing title rather than +# taking it into the family name". This baseline made the chain first +# and the title was a word of the family name. issue = "fix(#614) a trailing title is read before the particle chain" name_regex = "^John van der Berg Prof\\.$" fields = ["family", "title"] diff --git a/tools/differential/expected_since_2.2.0.toml b/tools/differential/expected_since_2.2.0.toml index f85fabe0..5d8a0461 100644 --- a/tools/differential/expected_since_2.2.0.toml +++ b/tools/differential/expected_since_2.2.0.toml @@ -155,10 +155,15 @@ fields = ["_ambiguities"] [[change]] issue = "feat(#449) an interpunct transcription declines the script order, so the convention decides it" -# '王·Smith'. rules.md#T3: the 間隔号 marks a transcription and -# suppresses the script_orders lookup whole, so name_order governs and -# the one remaining name word is O5's. Literal-anchored, one corpus -# name. +# '王·Smith'. rules.md#T3: "The interpunct divides a name only between +# two characters of a classified East Asian script; anywhere else it is +# part of the word" -- 'Smith' is Latin, so the name is one word, and +# '王Smith' reads the same. rules.md#W4: "A name written wholly in one +# East Asian script, or in the kana-licensed Japanese repertoire, reads +# family-first whatever order the caller declared" -- a word mixing Han +# and Latin is neither, so no script order applies, name_order governs +# and the one name word is O5's. The issue names the interpunct, which +# is incidental to this name. Literal-anchored, one corpus name. name_regex = "^王·Smith$" fields = ["_ambiguities"] @@ -1081,8 +1086,10 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#562) a comma credential run holding a particle chain reads by the name-word count" -# rules.md#C1: "The part is read as its words stand, before any join" -# (#613, 2026-10-06): no particle chain reaches it, so each member's +# rules.md#C1: "A part read as postnominal [...] is read as its words +# stand, before any join, and that reading is final: [...] no particle +# run, connective join or bound given name reaches into it" (#613, +# 2026-10-06): no particle chain reaches it, so each member's # capitals settle it and the run reads whole as the credential run -- # silently since #613, where #562 (2026-10-01) had read the pair by the # count and reported the flip. The family-comma path had read the part as name text: @@ -1105,25 +1112,25 @@ issue = "fix(#563) paired initials after a comma read as the given name unless a # two words before the comma may be one surname ('García Márquez, # G.J.'), so two dotted single letters alone after it stay the given # name and report the fork, where #516 had read them as a credential. -# "Only a credential in front of them, or another word the class -# admits by its dotted shape standing in the same part, makes them the -# credential run": 'John Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' -# move to the run, while 'De La Cruz, M.J. PhD' keeps its initials, -# the degree standing behind them, and so does 'García Márquez, Ms -# G.J.', a title in front ("a suffix word in front of them that is -# not also title vocabulary"), which moves nothing at any baseline. -# The second period is optional ('García Márquez, G.J'). Two pairs -# speaking only for each other make the run and report it behind two -# name words ('John Smith, X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, -# M.J. K.L.' is no longer that case -- 'De La Cruz' is one name word, -# so the count no longer reaches it and it reads as 'Cruz, M.J. K.L.' -# does, given 'M.J.', suffix 'K.L.', reporting both; a listed member -# in front speaks for nothing -# ('García Márquez, Ed G.J.' keeps the family comma, and the fix(#531) -# rule claims what its given part's trailing slot then reads). Against a 2.x baseline the names -# whose roles match the baseline's move only the report, which a -# 1.4.0 comparison cannot see; that ledger lists the three role movers -# alone. +# "Only an unambiguous suffix word in front of them that is not also +# title vocabulary, or another word the class admits by its dotted shape +# standing in the same part, makes them the credential run": 'John +# Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' move to the run, while +# 'De La Cruz, M.J. PhD' keeps its initials, the degree standing behind +# them, and so does 'García Márquez, Ms G.J.', a title in front ("an +# unambiguous suffix word in front of them that is not also title +# vocabulary"), which moves nothing at any baseline. The second period +# is optional ('García Márquez, G.J'). Two pairs speaking only for each +# other make the run and report it behind two name words ('John Smith, +# X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, M.J. K.L.' is no longer +# that case -- 'De La Cruz' is one name word, so the count no longer +# reaches it and it reads as 'Cruz, M.J. K.L.' does, given 'M.J.', +# suffix 'K.L.', reporting both; a listed member in front speaks for +# nothing ('García Márquez, Ed G.J.' keeps the family comma, and the +# fix(#531) rule claims what its given part's trailing slot then reads). +# Against a 2.x baseline the names whose roles match the baseline's move +# only the report, which a 1.4.0 comparison cannot see; that ledger +# lists the three role movers alone. name_regex = "^(?:De La Cruz, M\\.J\\. K\\.L\\.|De La Cruz, M\\.J\\. PhD|García Márquez, G\\.J|García Márquez, G\\.J\\.|García Márquez, PhD G\\.J\\.|John Smith, A\\.B\\.|John Smith, A\\.B\\. Ph\\.D\\.|John Smith, PhD X\\.Y\\.|John Smith, X\\.Y\\. P\\.Q\\.)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -1269,8 +1276,8 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # text whole. # # Reporting a DECLINED fork is #530's stated rule -- the report tracks -# the fork consulted, not the lean -- and rules.md#A1's "a kind is -# worth adding only if a reader would hesitate too" is the standing +# the fork consulted, not the lean -- and AGENTS.md's "A kind is worth +# adding only if a reader would hesitate too" is the standing # objection it answers. It is also the only part of #531 that adds a # report without moving a field, which is why it is a rule of its own # rather than a widening of the one above: `fields` is @@ -1286,8 +1293,10 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # given part is a title there, 'PhD' is the given name, and the # Title-case 'Ma' ending the part is this slot's declining half, # reported. No role moves from this baseline; #544 keeps it so -# (rules.md#S2: nothing in a part whose leading title run holds a dual -# speaks for the member). +# (rules.md#S2: "A word of both the title and the suffix vocabulary +# standing in the given part's leading title run is a title there and +# speaks for nothing, and no credential behind it speaks for a word of +# this class in that part"). # # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the thirteen movers above. @@ -1300,12 +1309,13 @@ issue = "fix(#531) capitals take the do collision from the family-comma particle # 'Doe, John DO', alone, because the word is alone in the class: `do` # is the one ambiguous credential that is also particle vocabulary, so # this slot and P6's attachment (rules.md#P6) want the same word. -# Derek's decision, recorded at decisions.md#S2: CAPITALS DECIDE, AND -# THE PARTICLE RULE KEEPS EVERY OTHER SPELLING. An all-caps member in -# a name written in more than one case carries a positive credential -# lean, so the credential reading wins and P6 stands down; every other -# spelling attaches exactly as it did, with P6's own -# `particle-or-given` and no second report. +# Derek's decision, recorded at decisions.md#S2: "CAPITALS DECIDE FOR +# `do`, AND THE PARTICLE RULE KEEPS EVERY OTHER SPELLING". An all-caps +# member in a name written in more than one case carries a positive +# credential lean, so the credential reading wins and P6 stands down +# (and since #544 so does an unambiguous credential in front of it, +# 'doe, jane v phd do'); every other spelling attaches exactly as it +# did, with P6's own `particle-or-given` and no second report. # # THE PAIRING IS THE ARGUMENT and the accepted cost is half of it. In # ONE CASE the rule cannot tell 'NASCIMENTO, EDSON ARANTES DO' from @@ -2453,10 +2463,11 @@ fields = ["family", "given", "middle"] [[change]] # rules.md#S2: "The run at a name's trailing slot is read ONCE, once # the connectives have joined (P3) and before the particle chain (P2) -# and the bound given-name join (P5) are made" (#614, 2026-10-07), and -# rules.md#H5: "a particle chain (P2) stops in front of a trailing -# title rather than taking it into the family name". This baseline -# made the chain first and the title was a word of the family name. +# and the bound given-name join (P5) are made" and "neither join +# reaches a word the run took" (#614, 2026-10-07), and rules.md#H5: "a +# particle chain (P2) stops in front of a trailing title rather than +# taking it into the family name". This baseline made the chain first +# and the title was a word of the family name. issue = "fix(#614) a trailing title is read before the particle chain" name_regex = "^John van der Berg Prof\\.$" fields = ["family", "title"] diff --git a/tools/differential/expected_since_2.3.0.toml b/tools/differential/expected_since_2.3.0.toml index ad18a89d..66bfb34f 100644 --- a/tools/differential/expected_since_2.3.0.toml +++ b/tools/differential/expected_since_2.3.0.toml @@ -351,8 +351,10 @@ orders = ["DEFAULT"] [[change]] issue = "fix(#562) a comma credential run holding a particle chain reads by the name-word count" -# rules.md#C1: "The part is read as its words stand, before any join" -# (#613, 2026-10-06): no particle chain reaches it, so each member's +# rules.md#C1: "A part read as postnominal [...] is read as its words +# stand, before any join, and that reading is final: [...] no particle +# run, connective join or bound given name reaches into it" (#613, +# 2026-10-06): no particle chain reaches it, so each member's # capitals settle it and the run reads whole as the credential run -- # silently since #613, where #562 (2026-10-01) had read the pair by the # count and reported the flip. The family-comma path had read the part as name text: @@ -375,25 +377,25 @@ issue = "fix(#563) paired initials after a comma read as the given name unless a # two words before the comma may be one surname ('García Márquez, # G.J.'), so two dotted single letters alone after it stay the given # name and report the fork, where #516 had read them as a credential. -# "Only a credential in front of them, or another word the class -# admits by its dotted shape standing in the same part, makes them the -# credential run": 'John Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' -# move to the run, while 'De La Cruz, M.J. PhD' keeps its initials, -# the degree standing behind them, and so does 'García Márquez, Ms -# G.J.', a title in front ("a suffix word in front of them that is -# not also title vocabulary"), which moves nothing at any baseline. -# The second period is optional ('García Márquez, G.J'). Two pairs -# speaking only for each other make the run and report it behind two -# name words ('John Smith, X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, -# M.J. K.L.' is no longer that case -- 'De La Cruz' is one name word, -# so the count no longer reaches it and it reads as 'Cruz, M.J. K.L.' -# does, given 'M.J.', suffix 'K.L.', reporting both; a listed member -# in front speaks for nothing -# ('García Márquez, Ed G.J.' keeps the family comma, and the fix(#531) -# rule claims what its given part's trailing slot then reads). Against a 2.x baseline the names -# whose roles match the baseline's move only the report, which a -# 1.4.0 comparison cannot see; that ledger lists the three role movers -# alone. +# "Only an unambiguous suffix word in front of them that is not also +# title vocabulary, or another word the class admits by its dotted shape +# standing in the same part, makes them the credential run": 'John +# Smith, PhD X.Y.' and 'John Smith, X.Y. P.Q.' move to the run, while +# 'De La Cruz, M.J. PhD' keeps its initials, the degree standing behind +# them, and so does 'García Márquez, Ms G.J.', a title in front ("an +# unambiguous suffix word in front of them that is not also title +# vocabulary"), which moves nothing at any baseline. The second period +# is optional ('García Márquez, G.J'). Two pairs speaking only for each +# other make the run and report it behind two name words ('John Smith, +# X.Y. P.Q.'). 2026-10-01, #575: 'De La Cruz, M.J. K.L.' is no longer +# that case -- 'De La Cruz' is one name word, so the count no longer +# reaches it and it reads as 'Cruz, M.J. K.L.' does, given 'M.J.', +# suffix 'K.L.', reporting both; a listed member in front speaks for +# nothing ('García Márquez, Ed G.J.' keeps the family comma, and the +# fix(#531) rule claims what its given part's trailing slot then reads). +# Against a 2.x baseline the names whose roles match the baseline's move +# only the report, which a 1.4.0 comparison cannot see; that ledger +# lists the three role movers alone. name_regex = "^(?:De La Cruz, M\\.J\\. K\\.L\\.|De La Cruz, M\\.J\\. PhD|García Márquez, G\\.J|García Márquez, G\\.J\\.|García Márquez, PhD G\\.J\\.|John Smith, A\\.B\\.|John Smith, A\\.B\\. Ph\\.D\\.|John Smith, PhD X\\.Y\\.|John Smith, X\\.Y\\. P\\.Q\\.)$" fields = ["family", "given", "middle", "suffix", "_ambiguities"] orders = ["DEFAULT"] @@ -539,8 +541,8 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # text whole. # # Reporting a DECLINED fork is #530's stated rule -- the report tracks -# the fork consulted, not the lean -- and rules.md#A1's "a kind is -# worth adding only if a reader would hesitate too" is the standing +# the fork consulted, not the lean -- and AGENTS.md's "A kind is worth +# adding only if a reader would hesitate too" is the standing # objection it answers. It is also the only part of #531 that adds a # report without moving a field, which is why it is a rule of its own # rather than a widening of the one above: `fields` is @@ -556,8 +558,10 @@ issue = "fix(#531) a member the writing declines keeps its name reading and repo # given part is a title there, 'PhD' is the given name, and the # Title-case 'Ma' ending the part is this slot's declining half, # reported. No role moves from this baseline; #544 keeps it so -# (rules.md#S2: nothing in a part whose leading title run holds a dual -# speaks for the member). +# (rules.md#S2: "A word of both the title and the suffix vocabulary +# standing in the given part's leading title run is a title there and +# speaks for nothing, and no credential behind it speaks for a word of +# this class in that part"). # # Literal-anchored: the class is the slot's declining half, and a # regex for it would claim the thirteen movers above. @@ -570,12 +574,13 @@ issue = "fix(#531) capitals take the do collision from the family-comma particle # 'Doe, John DO', alone, because the word is alone in the class: `do` # is the one ambiguous credential that is also particle vocabulary, so # this slot and P6's attachment (rules.md#P6) want the same word. -# Derek's decision, recorded at decisions.md#S2: CAPITALS DECIDE, AND -# THE PARTICLE RULE KEEPS EVERY OTHER SPELLING. An all-caps member in -# a name written in more than one case carries a positive credential -# lean, so the credential reading wins and P6 stands down; every other -# spelling attaches exactly as it did, with P6's own -# `particle-or-given` and no second report. +# Derek's decision, recorded at decisions.md#S2: "CAPITALS DECIDE FOR +# `do`, AND THE PARTICLE RULE KEEPS EVERY OTHER SPELLING". An all-caps +# member in a name written in more than one case carries a positive +# credential lean, so the credential reading wins and P6 stands down +# (and since #544 so does an unambiguous credential in front of it, +# 'doe, jane v phd do'); every other spelling attaches exactly as it +# did, with P6's own `particle-or-given` and no second report. # # THE PAIRING IS THE ARGUMENT and the accepted cost is half of it. In # ONE CASE the rule cannot tell 'NASCIMENTO, EDSON ARANTES DO' from @@ -1739,10 +1744,11 @@ fields = ["family", "given", "middle"] [[change]] # rules.md#S2: "The run at a name's trailing slot is read ONCE, once # the connectives have joined (P3) and before the particle chain (P2) -# and the bound given-name join (P5) are made" (#614, 2026-10-07), and -# rules.md#H5: "a particle chain (P2) stops in front of a trailing -# title rather than taking it into the family name". This baseline -# made the chain first and the title was a word of the family name. +# and the bound given-name join (P5) are made" and "neither join +# reaches a word the run took" (#614, 2026-10-07), and rules.md#H5: "a +# particle chain (P2) stops in front of a trailing title rather than +# taking it into the family name". This baseline made the chain first +# and the title was a word of the family name. issue = "fix(#614) a trailing title is read before the particle chain" name_regex = "^John van der Berg Prof\\.$" fields = ["family", "title"]