grc_dep_web_sm / package_functions.py
diyclassics's picture
3.8.3: fix elided lemma lookup (lookup-table-only patch, weights unchanged)
224903a verified
Raw
History Blame Contribute Delete
9.07 kB
# imports
from __future__ import annotations
import string
from typing import Callable
import spacy
from spacy.language import Language
from spacy.tokens import Doc, Token
from spacy.tokenizer import Tokenizer
from spacy.util import (
compile_infix_regex,
compile_prefix_regex,
compile_suffix_regex,
)
# ----- canonical text normalization: surface (token.text) + NORM (token.norm_) ----- #
#
# The grc treebanks encode the elision apostrophe four different ways — Perseus
# U+0313 (a combining-diacritic Betacode artifact), PROIEL/GLAUx U+2019, PTNK
# U+1FBF — which makes tokenization AND evals inconsistent. We canonicalize to
# U+2019 (Unicode Standard §6.2; Nick Nicholas; James Tauber's greek-normalisation)
# and adopt Tauber's `greek-normalisation` for the heavy lifting. This runs
# IDENTICALLY in treebank conversion (gold, all splits) and in the tokenizer below
# (inference) so gold and runtime can never drift.
#
# normalize_surface/normalize_norm/normalize_lookup_key live in latincy-preprocess
# (latincy_preprocess.grc) — the single shared implementation every grc repo now
# imports, rather than each repo defining (or partially reimplementing) its own
# copy. See latincy-preprocess's tests/test_grc_rules.py for the normalization
# contract this depends on.
from latincy_preprocess.grc import normalize_lookup_key, normalize_norm, normalize_surface
# ----- custom tokenizer: canonical normalization + attached elision ----- #
#
# Runtime mirror of the conversion-time normalization: normalise input text to the
# canonical surface form BEFORE tokenizing (so runtime tokenization matches the
# normalized gold corpus and evals align), keep the elision apostrophe attached
# (drop the apostrophe suffix rules so ἔνθ' / δ' stay one token like the gold), and
# set token.norm_ to the isolation form (ἀλλ'→ἀλλά). Crasis handling composes via the
# separate grc_keep_crasis_whole after_creation callback.
# apostrophe suffix patterns in the grc default ruleset (found by inspection):
# r"\'" (ascii) and "’" (U+2019). Removing them stops a trailing elision mark from
# splitting off as its own token.
_APOS_SUFFIX_PATTERNS = {r"\'", "’", "᾿", "ʼ", "'"}
def _grc_defaults():
from spacy.lang.grc import AncientGreekDefaults
return AncientGreekDefaults
def _grc_suffix_search():
"""grc default suffixes minus the apostrophe patterns."""
d = _grc_defaults()
suffixes = [s for s in d.suffixes
if s not in _APOS_SUFFIX_PATTERNS and "’" not in s]
return compile_suffix_regex(suffixes).search
class GreekTokenizer(Tokenizer):
"""grc tokenizer with LatinCy canonical normalization baked in.
Inherits the full spaCy ``lang/grc`` ruleset but (a) normalises input text via
``normalize_surface`` before tokenizing and (b) drops the apostrophe suffix rule so
the (now canonical U+2019) elision mark stays attached. ``token.norm_`` is set to
the ``normalize_norm`` isolation form. Serialises like the base Tokenizer; the
registered factory below reconstructs it on load.
"""
def __init__(self, vocab):
d = _grc_defaults()
super().__init__(
vocab,
rules=d.tokenizer_exceptions,
prefix_search=compile_prefix_regex(d.prefixes).search,
suffix_search=_grc_suffix_search(),
infix_finditer=compile_infix_regex(d.infixes).finditer,
token_match=d.token_match,
url_match=d.url_match,
)
def __call__(self, text: str) -> Doc:
doc = super().__call__(normalize_surface(text))
for tok in doc:
tok.norm_ = normalize_norm(tok.text)
return doc
@spacy.registry.tokenizers("grc_normalizing_tokenizer.v1")
def create_grc_tokenizer() -> Callable[[Language], Tokenizer]:
"""Factory: `[nlp.tokenizer] @tokenizers = "grc_normalizing_tokenizer.v1"`."""
def create_tokenizer(nlp: Language) -> Tokenizer:
return GreekTokenizer(nlp.vocab)
return create_tokenizer
# ----- Crasis tokenization (GLAUx standard: keep crasis whole) ----- #
def _is_crasis_exception_key(key: str) -> bool:
"""True if `key` is a single all-Greek-letter word (i.e. a crasis form).
spaCy's built-in ``grc`` tokenizer ships exceptions that split crasis forms
into their underlying words (κἀγώ → κἀ + γώ, κᾆτα → κ + ᾆτα, τοὔνομα → τοὔ +
νομα, …). GLAUx treats crasis as a single token, so our training corpus keeps
κἀγὼ whole (1075× as one PRON). Dropping these exceptions aligns runtime
tokenization with the gold standard.
Elision exceptions (δ', ἀλλ', παρ') are single-token and their apostrophe
makes ``str.isalpha()`` False, so they are preserved.
"""
return bool(key) and key.isalpha() and all(
"Ͱ" <= c <= "Ͽ" or "ἀ" <= c <= "῿" for c in key
)
@spacy.registry.callbacks("grc_keep_crasis_whole.v1")
def make_keep_crasis_whole() -> Callable[[Language], Language]:
"""`[nlp] after_creation` callback: keep crasis whole (GLAUx standard).
Removes the crasis-splitting exceptions from the tokenizer so forms like
κἀγὼ remain a single token, matching the GLAUx training corpus. This is a
pure tokenizer change baked into the serialized model — no retraining needed.
"""
def keep_crasis_whole(nlp: Language) -> Language:
rules = dict(nlp.tokenizer.rules)
for key, value in list(rules.items()):
if len(value) > 1 and _is_crasis_exception_key(key):
del rules[key]
nlp.tokenizer.rules = rules
return nlp
return keep_crasis_whole
# ----- lookup_lemmatizer ----- #
_LOOKUPS = None
def _get_lookups():
"""Load Greek lemma lookup table from the installed grc-latincy-lookups
package, via spaCy's lookup entry-point system.
Dev/training-time fallback only — see lookup_lemmatizer, which prefers the
table embedded in the model's own vocab.lookups when present. Returns a
spaCy Table object (dict-like, supports .get()).
"""
global _LOOKUPS
if _LOOKUPS is None:
from spacy.lookups import load_lookups
lookups_data = load_lookups(lang="grc", tables=["lemma_lookup"])
_LOOKUPS = lookups_data.get_table("lemma_lookup")
return _LOOKUPS
Token.set_extension("predicted_lemma", default=None, force=True)
@Language.component(name="lookup_lemmatizer")
def lookup_lemmatizer(doc: Doc) -> Doc:
"""Lookup-based lemmatizer for Ancient Greek.
Assigns lemmas using a 1.2M-entry dictionary built from CLTK Morpheus,
UD treebanks, and Wiktionary. Normalizes grave→acute accents at query
time so running-text forms (φονὸς) match citation entries (φονός).
Runs after trainable_lemmatizer: overrides only when a lookup match
exists, preserving the trainable model's output for unseen forms.
Prefers the lemma table baked into this model's own vocab.lookups (done at
packaging time by prepare_package.py / repackage_patch.sh) — published
wheels are self-contained and need no extra pip package at inference time.
Falls back to the pip-installed grc-latincy-lookups package for local
dev/training, before the table has been injected into vocab. Single
function for both cases, rather than the two independently-maintained
lookup_lemmatizer copies (dev vs. packaging-ready) this repo used to carry.
"""
if doc.vocab.lookups.has_table("lemma_lookup"):
lookups = doc.vocab.lookups.get_table("lemma_lookup")
else:
lookups = _get_lookups()
for token in doc:
# Store trainable lemmatizer's prediction
token._.predicted_lemma = token.lemma_
# Skip punctuation
if token.pos_ == "PUNCT" or token.text in string.punctuation:
continue
# Normalize for lookup (grave→acute, strip macron/breve, elision->U+2019)
normalized = normalize_lookup_key(token.text)
# Direct match
if normalized in lookups:
token.lemma_ = lookups[normalized]
continue
# Case-insensitive fallback for capitalized words
if normalized and normalized[0].isupper():
lower = normalized.lower()
if lower in lookups:
token.lemma_ = lookups[lower]
continue
# Elision fallback: for an elided form the (harmonized U+2019) table still
# misses, resolve via the restored isolation form (normalize_norm: δʼ→δέ,
# ἀλλʼ→ἀλλά) and, failing that, use the restored surface — but NEVER leave the
# trainable lemmatizer's apostrophe-token hallucination (lg: ὧδʼ→ὧδʼῖς, ἀλλʼ→̔ἀλλʼ).
# Ambiguous elisions (μʼ = με/μοι) fall back to their canonical surface, not garbage.
if "’" in normalized:
restored = normalize_norm(token.text)
token.lemma_ = lookups[restored] if restored in lookups else restored
return doc