Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
Show all changes
36 commits
Select commit Hold shift + click to select a range
4f5b956
Add the glued-honorific tail vocabulary (#308)
derek73 Jul 31, 2026
2b0f007
Vet the glued tail set against real surname data (#308)
derek73 Jul 31, 2026
b9c58a2
Add Lexicon.honorific_tails, the glued-peel vocabulary (#308)
derek73 Jul 31, 2026
b72150a
Enforce the honorific-tail subset invariant (#308)
derek73 Jul 31, 2026
9075fde
Peel a glued CJK honorific off the last name token (#308)
derek73 Jul 31, 2026
17f9c39
Never treat a post-nominal as a name site (#308)
derek73 Jul 31, 2026
be97aec
Decline rather than scan past a post-nominal surname site (#308)
derek73 Jul 31, 2026
90be6ba
Pin the suffix-comma peel and name the maiden reach (#308)
derek73 Jul 31, 2026
59aac1f
Let a recognized honorific reach the segmenter, not block it (#308)
derek73 Jul 31, 2026
f71a2e4
Pin that an honorific does not excuse a real boundary (#308)
derek73 Jul 31, 2026
d56fff2
Classify the glued-honorific diffs in the differential (#308)
derek73 Jul 31, 2026
aba6c03
Document the glued honorific peel (#308)
derek73 Jul 31, 2026
1d46fc6
Say what the 君 row actually pins (#308)
derek73 Jul 31, 2026
73fb92f
Finish the #308 staleness sweep (#308)
derek73 Jul 31, 2026
fdd3527
Scope the doctrine paragraph's count to what it counts (#308)
derek73 Jul 31, 2026
227426d
Note the lone-honorific fix and the peel's off-switch (#308)
derek73 Jul 31, 2026
063fd1a
Exempt only the tail the peel manufactured (#308)
derek73 Jul 31, 2026
931ce0e
Ship 박사님, the missing third -님 honorific (#308)
derek73 Jul 31, 2026
cd626cb
Correct six claims the code makes about itself (#308)
derek73 Jul 31, 2026
a527bd6
Pin the v1 shim's honorific_tails translation (#308)
derek73 Jul 31, 2026
288fef2
Say what the peel actually does, in prose (#308)
derek73 Jul 31, 2026
acbfefc
Pin the peel's segment choice and its two boundaries (#308)
derek73 Jul 31, 2026
3e5dad6
Reach the peel from a drawn lexicon (#308)
derek73 Jul 31, 2026
b31f759
Replace the peel docstring's self-refuting no-peel-twice example (#308)
derek73 Jul 31, 2026
dde0c60
Reclassify the FAMILY-comma honorific row to comma-family (#308)
derek73 Jul 31, 2026
d6cd806
Justify the segmenter's spaced-honorific rule by its cost (#308)
derek73 Jul 31, 2026
7819a3d
Scope the spaced-honorific advice to the segmenter path (#308)
derek73 Jul 31, 2026
2189401
Pin _is_post_nominal's strict suffix test with the input it names (#308)
derek73 Jul 31, 2026
b8a9252
Give the spaced-honorific test its missing control (#308)
derek73 Jul 31, 2026
e6a6b69
Correct five accuracy nits around the peel (#308)
derek73 Jul 31, 2026
41588b4
Let _split tag the tail it just built (#308)
derek73 Jul 31, 2026
2c5edcc
Derive the peel test lexicon from the stage's own (#308)
derek73 Jul 31, 2026
6ae3351
Drop a test whose assertion cannot fail (#308)
derek73 Jul 31, 2026
3171267
Pin the 殿 exclusion, the one nothing held (#308)
derek73 Jul 31, 2026
46fdc3e
Put a non-ASCII shape in the scaling benchmark (#308)
derek73 Jul 31, 2026
53611b1
Name the escape hatch that actually works (#308)
derek73 Jul 31, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
Next Next commit
Add Lexicon.honorific_tails, the glued-peel vocabulary (#308)
  • Loading branch information
derek73 committed Jul 31, 2026
commit b9c58a29a1da890ba6fc4a92464165b348130ce9
4 changes: 4 additions & 0 deletions nameparser/_config_shim.py
Original file line number Diff line number Diff line change
Expand Up @@ -986,6 +986,7 @@ def _build_snapshot(self) -> tuple[Lexicon, Policy, _RenderDefaults]:
removal path.
"""
from nameparser.config.maiden_markers import MAIDEN_MARKERS
from nameparser.config.suffixes import GLUED_HONORIFICS
from nameparser.config.surnames import KOREAN_SURNAMES
acronyms = frozenset(self.suffix_acronyms)
particles = frozenset(self.prefixes)
Expand Down Expand Up @@ -1068,6 +1069,9 @@ def _build_snapshot(self) -> tuple[Lexicon, Policy, _RenderDefaults]:
# Unwrapped where maiden_markers above is wrapped: this
# module is born frozen (#293), so no wrap
surnames=KOREAN_SURNAMES,
# likewise no v1 manager: the glued-honorific tail set is
# 2.1 behavior (#308), so it rides in the snapshot only
honorific_tails=frozenset(GLUED_HONORIFICS),
# TupleManager is dict[str, object] (v1 parity: values were
# never statically str-typed); every real entry is a str,
# same assumption _DelimiterManager's sentinel lookup makes
Expand Down
14 changes: 13 additions & 1 deletion nameparser/_lexicon.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
"titles", "given_name_titles", "suffix_acronyms", "suffix_words",
"suffix_acronyms_ambiguous", "particles", "particles_ambiguous",
"conjunctions", "bound_given_names", "maiden_markers", "surnames",
"honorific_tails",
)

#: (marker, base, why) triples. Each marker narrows how entries of its
Expand Down Expand Up @@ -335,6 +336,15 @@ class Lexicon:
#: (:data:`~nameparser.config.surnames.KOREAN_SURNAMES`); Chinese
#: surnames ship in locales.ZH because Han segmentation is opt-in.
surnames: frozenset[str] = frozenset()
#: Honorifics that may be peeled off the END of a name token
#: (#308), matched longest-first: 田中さん splits into 田中 and さん
#: before anything else reads the name. Every entry must also be a
#: :attr:`suffix_words` entry -- the peeled tail is claimed by
#: suffix classification like any other post-nominal. Strictly
#: narrower than the spaced honorific vocabulary, since a glued
#: tail has no token boundary to lean on. Full default list:
#: :data:`~nameparser.config.suffixes.GLUED_HONORIFICS`.
honorific_tails: frozenset[str] = frozenset()
#: Lowercase word -> exact-cased replacement used by capitalized()
#: ("phd" -> "Ph.D."). Pair-valued: change it with
#: dataclasses.replace(), not add()/remove(); read it as a mapping
Expand Down Expand Up @@ -577,7 +587,8 @@ def _default_lexicon() -> Lexicon:
from nameparser.config.maiden_markers import MAIDEN_MARKERS
from nameparser.config.prefixes import NON_FIRST_NAME_PREFIXES, PREFIXES
from nameparser.config.suffixes import (
SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS, SUFFIX_NOT_ACRONYMS,
GLUED_HONORIFICS, SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS,
SUFFIX_NOT_ACRONYMS,
)
from nameparser.config.surnames import KOREAN_SURNAMES
from nameparser.config.titles import FIRST_NAME_TITLES, TITLES
Expand All @@ -602,6 +613,7 @@ def _default_lexicon() -> Lexicon:
# surnames.py is born frozen (#293) -- no call-site wrap needed,
# unlike the v1 modules above (their wraps drop when #293 lands)
surnames=KOREAN_SURNAMES,
honorific_tails=frozenset(GLUED_HONORIFICS),
# pass canonical pair-tuples so this strictly-typed call site never
# feeds a Mapping to the tuple-annotated field; __post_init__
# still tolerates a Mapping at runtime for interactive use
Expand Down
10 changes: 10 additions & 0 deletions tests/v2/test_lexicon.py
Original file line number Diff line number Diff line change
Expand Up @@ -577,3 +577,13 @@ def test_default_lexicon_ships_korean_surnames_only() -> None:
hangul = _SCRIPT_RANGES[Script.HANGUL]
assert all(all(any(lo <= ord(c) <= hi for lo, hi in hangul) for c in s)
and 1 <= len(s) <= 2 for s in lex.surnames)


def test_honorific_tails_is_a_vocab_field() -> None:
lex = Lexicon.empty().add(honorific_tails={"씨", "さん"})
assert lex.honorific_tails == frozenset({"씨", "さん"})
assert lex.remove(honorific_tails={"씨"}).honorific_tails == \
frozenset({"さん"})
merged = (Lexicon(honorific_tails=frozenset({"씨"}))
| Lexicon(honorific_tails=frozenset({"様"})))
assert merged.honorific_tails == frozenset({"씨", "様"})