a816b119e7
Backfill ELP vocabulary from FULL lexicons (UniMorph + kaikki.org Wiktionary, real gender/inflections) for 8 languages, 812,894 entries total, in the proven seed-fn format matching the 18 ancient vocabularies: es 72,032 | fr 130,517 | de 144,692 | la 22,590 | it 193,675 | pt 115,772 | ro 86,504 | ca 47,112 4 of these (es fr de la) backfill ELP languages that had morphology but no vocabulary; it/pt/ro/ca are new Romance (need morphology-*.el ports next). Adds lang_profile_* for all 8 + reproducible generators under tests/lang-gen. Vocab is runtime seed data (not in build manifest, like the 18 ancients); seed-fn format validated to compile to C via elc.
563 lines
23 KiB
Python
563 lines
23 KiB
Python
"""morphology_es_full.py — production-grade Spanish morphological generator.
|
|
|
|
NOT a toy. Backed by a real, broad, licensed lexicon:
|
|
|
|
UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0, Wiktionary-derived)
|
|
1,196,245 inflected forms:
|
|
6,695 verb lemmas — full paradigms: indicative (present/preterite/
|
|
imperfect/future), conditional, present & imperfect
|
|
subjunctive, affirmative imperative, formal/informal
|
|
48,353 noun lemmas — WITH inherent gender (N;FEM/MASC;SG/PL)
|
|
16,984 adj lemmas — gender + number paradigms
|
|
|
|
Fallbacks (so we degrade, never crash, on out-of-vocabulary input):
|
|
- verbs : mlconjug3 (ML paradigm model, conjugates ANY Spanish verb) then a
|
|
hand-rolled regular-ending generator
|
|
- nouns : gender heuristic (endings) + regular pluralization
|
|
- adjs : -o/-a gender rule + regular pluralization
|
|
|
|
Every generated form carries a CONFIDENCE flag:
|
|
"lexicon" form came straight from UniMorph (trust: high)
|
|
"model" form came from mlconjug3 (trust: high)
|
|
"rule" form came from a deterministic rule (trust: medium)
|
|
"fallback" we could not inflect; returned lemma as-is (trust: low → FLAG)
|
|
|
|
Public API (used by realizer_es.py):
|
|
conjugate(lemma, mood, tense, person, number, formality="informal") -> (form, conf)
|
|
participle(lemma) -> (form, conf) # past participle (compound tenses)
|
|
gerund(lemma) -> (form, conf)
|
|
noun_gender(lemma) -> "m"|"f"
|
|
inflect_noun(lemma, number) -> (form, conf)
|
|
inflect_adj(lemma, gender, number) -> (form, conf)
|
|
attach_enclitics(verb_form, clitics) -> str # accent-correct enclisis
|
|
lexicon_stats() -> dict
|
|
"""
|
|
import os
|
|
import pickle
|
|
import unicodedata
|
|
|
|
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
_UNIMORPH = os.path.join(_HERE, "data", "spa.unimorph")
|
|
_CACHE = os.path.join(_HERE, "data", "es_morph_cache.pkl")
|
|
|
|
# ── canonical feature keys the realizer speaks, mapped to UniMorph tags ─────────
|
|
# mood/tense pair -> the UniMorph feature substring that identifies it
|
|
_VERB_KEYMAP = {
|
|
("ind", "present"): ("IND", "PRS", None),
|
|
("ind", "preterite"): ("IND", "PST", "PFV"),
|
|
("ind", "imperfect"): ("IND", "PST", "IPFV"),
|
|
("ind", "future"): ("IND", "FUT", None),
|
|
("ind", "conditional"):("COND", None, None),
|
|
("sbjv", "present"): ("SBJV", "PRS", None),
|
|
("sbjv", "imperfect"): ("SBJV", "PST", "LGSPEC1"), # -ra form
|
|
("imp", "present"): ("POS", "IMP", None),
|
|
}
|
|
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
|
_NUMBER = {"singular": "SG", "plural": "PL"}
|
|
|
|
|
|
# ── build / load the compact lexicon ───────────────────────────────────────────
|
|
def _feat_set(tag):
|
|
return set(tag.split(";"))
|
|
|
|
|
|
def _build_cache():
|
|
verbs = {} # (lemma, canonkey) -> form canonkey e.g. "ind|present|1|SG|infm"
|
|
nouns = {} # lemma -> {"g": "m"/"f", "SG": form, "PL": form}
|
|
adjs = {} # lemma -> {("m","SG"): form, ...}
|
|
part = {} # lemma -> masc-sg participle
|
|
ger = {} # lemma -> gerund
|
|
|
|
with open(_UNIMORPH, encoding="utf-8") as fh:
|
|
for line in fh:
|
|
line = line.rstrip("\n")
|
|
if not line or "\t" not in line:
|
|
continue
|
|
parts = line.split("\t")
|
|
if len(parts) != 3:
|
|
continue
|
|
lemma, form, tag = parts
|
|
f = _feat_set(tag)
|
|
head = tag.split(";")[0]
|
|
|
|
if head == "V":
|
|
# skip clitic-bearing rows (we generate clitics ourselves)
|
|
if "PRO" in f:
|
|
continue
|
|
if "V.PTCP" in f and "PST" in f and "MASC" in f and "SG" in f:
|
|
part.setdefault(lemma, form)
|
|
continue
|
|
if "V.CVB" in f or "NFIN" in f or "V.PTCP" in f:
|
|
if "V.CVB" in f:
|
|
ger.setdefault(lemma, form)
|
|
continue
|
|
# identify mood/tense
|
|
mt = None
|
|
for (mood, tense), (a, b, c) in _VERB_KEYMAP.items():
|
|
if a not in f:
|
|
continue
|
|
if b is not None and b not in f:
|
|
continue
|
|
if c is not None and c not in f:
|
|
continue
|
|
# disambiguate IND;PST needing PFV vs IPFV
|
|
if a == "IND" and b == "PST" and c not in f:
|
|
continue
|
|
mt = (mood, tense)
|
|
break
|
|
if mt is None:
|
|
continue
|
|
person = next((p for p in ("1", "2", "3") if p in f), None)
|
|
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
|
if person is None or number is None:
|
|
continue
|
|
formal = "form" if "FORM" in f else ("infm" if "INFM" in f else "any")
|
|
key = f"{mt[0]}|{mt[1]}|{person}|{number}|{formal}"
|
|
verbs.setdefault((lemma, key), form)
|
|
|
|
elif head == "N":
|
|
# substring test handles epicene "MASC+FEM" (-> masc citation)
|
|
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else None)
|
|
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
|
if num is None:
|
|
continue
|
|
# store forms keyed by (gender,number); animate nouns list BOTH
|
|
# genders under one lemma (niño -> niño/niña). Resolve citation
|
|
# gender in a post-pass (gender of the row whose form == lemma).
|
|
d = nouns.setdefault(lemma, {})
|
|
d.setdefault("_rows", []).append((g, num, form))
|
|
|
|
elif head == "ADJ":
|
|
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else "m")
|
|
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
|
if num is None:
|
|
continue
|
|
adjs.setdefault(lemma, {})[(g, num)] = form
|
|
|
|
# post-pass: resolve noun citation gender + default SG/PL forms
|
|
for lemma, d in nouns.items():
|
|
rows = d.pop("_rows", [])
|
|
# citation gender = gender of the row whose form == lemma; else first MASC;
|
|
# else first seen gender.
|
|
cite_g = None
|
|
for g, num, form in rows:
|
|
if form == lemma and g:
|
|
cite_g = g
|
|
break
|
|
if cite_g is None:
|
|
for g, num, form in rows:
|
|
if g == "m":
|
|
cite_g = "m"
|
|
break
|
|
if cite_g is None:
|
|
cite_g = next((g for g, _, _ in rows if g), "m")
|
|
d["g"] = cite_g
|
|
for g, num, form in rows:
|
|
d[(g, num)] = form
|
|
d["SG"] = d.get((cite_g, "SG")) or next((f for g, n, f in rows if n == "SG"), lemma)
|
|
d["PL"] = d.get((cite_g, "PL")) or next((f for g, n, f in rows if n == "PL"), None)
|
|
|
|
# post-pass: UniMorph omits the identity inflection (masc-sg == lemma) for
|
|
# adjectives, so fill it in; without this a fem-sg row wrongly satisfies a
|
|
# masc-sg request (alto -> alta bug).
|
|
for lemma, d in adjs.items():
|
|
d.setdefault(("m", "SG"), lemma)
|
|
|
|
data = {"verbs": verbs, "nouns": nouns, "adjs": adjs, "part": part, "ger": ger}
|
|
try:
|
|
with open(_CACHE, "wb") as fh:
|
|
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
|
except OSError:
|
|
pass
|
|
return data
|
|
|
|
|
|
def _load():
|
|
if os.path.exists(_CACHE) and os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH):
|
|
try:
|
|
with open(_CACHE, "rb") as fh:
|
|
return pickle.load(fh)
|
|
except Exception:
|
|
pass
|
|
return _build_cache()
|
|
|
|
|
|
_LEX = _load()
|
|
_VERBS, _NOUNS, _ADJS, _PART, _GER = (
|
|
_LEX["verbs"], _LEX["nouns"], _LEX["adjs"], _LEX["part"], _LEX["ger"])
|
|
|
|
# ── mlconjug3 fallback (lazy) ───────────────────────────────────────────────────
|
|
_MLC = None
|
|
_MLC_TENSE = { # (mood,tense) -> (mlconjug mood label, tense label)
|
|
("ind", "present"): ("Indicativo", "Indicativo presente"),
|
|
("ind", "preterite"): ("Indicativo", "Indicativo pretérito perfecto simple"),
|
|
("ind", "imperfect"): ("Indicativo", "Indicativo pretérito imperfecto"),
|
|
("ind", "future"): ("Indicativo", "Indicativo futuro"),
|
|
("ind", "conditional"): ("Condicional", "Condicional Condicional"),
|
|
("sbjv", "present"): ("Subjuntivo", "Subjuntivo presente"),
|
|
("sbjv", "imperfect"): ("Subjuntivo", "Subjuntivo pretérito imperfecto 1"),
|
|
("imp", "present"): ("Imperativo", "Imperativo Afirmativo"),
|
|
}
|
|
_MLC_SLOT = { # (person,number) -> mlconjug slot key
|
|
("first", "singular"): "1s", ("second", "singular"): "2s",
|
|
("third", "singular"): "3s", ("first", "plural"): "1p",
|
|
("second", "plural"): "2p", ("third", "plural"): "3p",
|
|
}
|
|
|
|
|
|
def _mlc_conjugate(lemma, mood, tense, person, number):
|
|
global _MLC
|
|
try:
|
|
if _MLC is None:
|
|
from mlconjug3 import Conjugator
|
|
_MLC = Conjugator(language="es")
|
|
v = _MLC.conjugate(lemma)
|
|
if v is None:
|
|
return None
|
|
info = v.conjug_info
|
|
m, t = _MLC_TENSE.get((mood, tense), (None, None))
|
|
if m is None or m not in info or t not in info[m]:
|
|
return None
|
|
block = info[m][t]
|
|
slot = _MLC_SLOT.get((person, number))
|
|
if isinstance(block, dict) and slot in block and block[slot]:
|
|
return block[slot]
|
|
return None
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
# ── regular-ending rule fallback (last resort, deterministic) ───────────────────
|
|
def _vclass(lemma):
|
|
return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else "ar"
|
|
|
|
|
|
def _stem(lemma):
|
|
return lemma[:-2]
|
|
|
|
|
|
_REG = {
|
|
("ind", "present", "ar"): ["o", "as", "a", "amos", "áis", "an"],
|
|
("ind", "present", "er"): ["o", "es", "e", "emos", "éis", "en"],
|
|
("ind", "present", "ir"): ["o", "es", "e", "imos", "ís", "en"],
|
|
("ind", "preterite", "ar"): ["é", "aste", "ó", "amos", "asteis", "aron"],
|
|
("ind", "preterite", "er"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
|
("ind", "preterite", "ir"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
|
("ind", "imperfect", "ar"): ["aba", "abas", "aba", "ábamos", "abais", "aban"],
|
|
("ind", "imperfect", "er"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
|
("ind", "imperfect", "ir"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
|
("sbjv", "present", "ar"): ["e", "es", "e", "emos", "éis", "en"],
|
|
("sbjv", "present", "er"): ["a", "as", "a", "amos", "áis", "an"],
|
|
("sbjv", "present", "ir"): ["a", "as", "a", "amos", "áis", "an"],
|
|
("sbjv", "imperfect", "ar"): ["ara", "aras", "ara", "áramos", "arais", "aran"],
|
|
("sbjv", "imperfect", "er"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
|
("sbjv", "imperfect", "ir"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
|
}
|
|
_FUT = ["é", "ás", "á", "emos", "éis", "án"]
|
|
_COND = ["ía", "ías", "ía", "íamos", "íais", "ían"]
|
|
|
|
|
|
def _slot_idx(person, number):
|
|
base = {"first": 0, "second": 1, "third": 2}[person]
|
|
return base + (0 if number == "singular" else 3)
|
|
|
|
|
|
def _rule_conjugate(lemma, mood, tense, person, number):
|
|
if len(lemma) < 3 or lemma[-2:] not in ("ar", "er", "ir"):
|
|
return None
|
|
vc, st, i = _vclass(lemma), _stem(lemma), _slot_idx(person, number)
|
|
if tense == "future":
|
|
return lemma + _FUT[i]
|
|
if tense == "conditional":
|
|
return lemma + _COND[i]
|
|
table = _REG.get((mood, tense, vc))
|
|
if table:
|
|
return st + table[i]
|
|
if mood == "imp" and tense == "present":
|
|
# affirmative tú imperative = 3sg present indicative
|
|
pres = _REG.get(("ind", "present", vc))
|
|
return st + pres[2] if number == "singular" else st + pres[5]
|
|
return None
|
|
|
|
|
|
# ── PUBLIC: verb conjugation ────────────────────────────────────────────────────
|
|
def conjugate(lemma, mood, tense, person, number, formality="informal"):
|
|
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
|
lemma = lemma.strip().lower()
|
|
p, n = _PERSON.get(person), _NUMBER.get(number)
|
|
formal = "form" if formality == "formal" else "infm"
|
|
if p and n:
|
|
for fkey in (formal, "any", "infm" if formal == "form" else "form"):
|
|
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}|{fkey}"))
|
|
if form:
|
|
return form, "lexicon"
|
|
m = _mlc_conjugate(lemma, mood, tense, person, number)
|
|
if m:
|
|
return m, "model"
|
|
r = _rule_conjugate(lemma, mood, tense, person, number)
|
|
if r:
|
|
return r, "rule"
|
|
return lemma, "fallback"
|
|
|
|
|
|
_IRREG_PART = { # guarantee the common irregular participles
|
|
"escribir": "escrito", "describir": "descrito", "abrir": "abierto",
|
|
"cubrir": "cubierto", "descubrir": "descubierto", "morir": "muerto",
|
|
"poner": "puesto", "ver": "visto", "volver": "vuelto", "devolver": "devuelto",
|
|
"hacer": "hecho", "deshacer": "deshecho", "decir": "dicho", "romper": "roto",
|
|
"resolver": "resuelto", "freír": "frito", "imprimir": "impreso",
|
|
"satisfacer": "satisfecho", "prever": "previsto", "revolver": "revuelto",
|
|
}
|
|
|
|
|
|
def participle(lemma):
|
|
lemma = lemma.strip().lower()
|
|
if lemma in _IRREG_PART:
|
|
return _IRREG_PART[lemma], "lexicon"
|
|
if lemma in _PART:
|
|
return _PART[lemma], "lexicon"
|
|
if lemma.endswith("ar"):
|
|
return lemma[:-2] + "ado", "rule"
|
|
if lemma[-2:] in ("er", "ir"):
|
|
return lemma[:-2] + "ido", "rule"
|
|
return lemma, "fallback"
|
|
|
|
|
|
_IRREG_GER = {"dormir": "durmiendo", "morir": "muriendo", "pedir": "pidiendo",
|
|
"sentir": "sintiendo", "mentir": "mintiendo", "servir": "sirviendo",
|
|
"venir": "viniendo", "decir": "diciendo", "poder": "pudiendo",
|
|
"ir": "yendo", "leer": "leyendo", "creer": "creyendo",
|
|
"oír": "oyendo", "traer": "trayendo", "caer": "cayendo",
|
|
"construir": "construyendo", "huir": "huyendo", "reír": "riendo"}
|
|
|
|
|
|
def gerund(lemma):
|
|
lemma = lemma.strip().lower()
|
|
if lemma in _IRREG_GER:
|
|
return _IRREG_GER[lemma], "lexicon"
|
|
if lemma in _GER:
|
|
return _GER[lemma], "lexicon"
|
|
if lemma.endswith("ar"):
|
|
return lemma[:-2] + "ando", "rule"
|
|
if lemma[-2:] in ("er", "ir"):
|
|
return lemma[:-2] + "iendo", "rule"
|
|
return lemma, "fallback"
|
|
|
|
|
|
# ── PUBLIC: noun gender + number ────────────────────────────────────────────────
|
|
_INVARIANT_PL = {"lunes", "martes", "miércoles", "jueves", "viernes",
|
|
"crisis", "tesis", "análisis", "dosis", "virus", "paraguas"}
|
|
|
|
|
|
def _gender_heuristic(noun):
|
|
for suf, g in (("ión", "f"), ("dad", "f"), ("tad", "f"), ("umbre", "f"),
|
|
("sis", "f"), ("ez", "f"), ("triz", "f"),
|
|
("ema", "m"), ("ama", "m"), ("oma", "m"), ("aje", "m"),
|
|
("or", "m"), ("án", "m"), ("ín", "m")):
|
|
if noun.endswith(suf):
|
|
return g
|
|
if noun.endswith("o"):
|
|
return "m"
|
|
if noun.endswith("a"):
|
|
return "f"
|
|
return "m"
|
|
|
|
|
|
def noun_gender(lemma):
|
|
lemma = lemma.strip().lower()
|
|
d = _NOUNS.get(lemma)
|
|
if d and d.get("g"):
|
|
return d["g"]
|
|
return _gender_heuristic(lemma)
|
|
|
|
|
|
def _regular_plural(noun):
|
|
if noun in _INVARIANT_PL:
|
|
return noun
|
|
if not noun:
|
|
return noun
|
|
last = noun[-1]
|
|
if last == "z":
|
|
return noun[:-1] + "ces"
|
|
if last in "aeiouáéíóú":
|
|
# stressed final vowel í/ú -> +es (rubí->rubíes), else +s
|
|
if last in "íú":
|
|
return noun + "es"
|
|
return noun + "s"
|
|
if last == "s":
|
|
# esdrújula / stress-final handled crudely; most polysyllables invariant
|
|
return noun
|
|
return noun + "es"
|
|
|
|
|
|
def inflect_noun(lemma, number, gender=None):
|
|
lemma = lemma.strip().lower()
|
|
d = _NOUNS.get(lemma)
|
|
num = "SG" if number == "singular" else "PL"
|
|
if d:
|
|
# honor a requested gender for animate nouns (gato -> gata)
|
|
if gender and (gender, num) in d:
|
|
return d[(gender, num)], "lexicon"
|
|
if d.get(num):
|
|
return d[num], "lexicon"
|
|
if number == "singular":
|
|
return lemma, "rule" if not d else "lexicon"
|
|
return _regular_plural(lemma), "rule"
|
|
|
|
|
|
# ── PUBLIC: adjective agreement ─────────────────────────────────────────────────
|
|
_INV_GENDER_ADJ = {"español": "española", "trabajador": "trabajadora",
|
|
"hablador": "habladora", "encantador": "encantadora",
|
|
"alemán": "alemana", "francés": "francesa", "inglés": "inglesa"}
|
|
|
|
|
|
def inflect_adj(lemma, gender, number):
|
|
lemma = lemma.strip().lower()
|
|
d = _ADJS.get(lemma)
|
|
num = "SG" if number == "singular" else "PL"
|
|
if d:
|
|
form = d.get((gender, num))
|
|
if form:
|
|
return form, "lexicon"
|
|
# gender-invariant adjective (grande, feliz, azul): fem == masc.
|
|
# For a missing plural, pluralize this gender's singular form.
|
|
sg = d.get((gender, "SG")) or d.get(("m", "SG")) or lemma
|
|
if number == "plural":
|
|
return _regular_plural(sg), "rule"
|
|
return sg, "lexicon"
|
|
# rule fallback
|
|
a = lemma
|
|
if gender == "f":
|
|
if a in _INV_GENDER_ADJ:
|
|
a = _INV_GENDER_ADJ[a]
|
|
elif a.endswith("o"):
|
|
a = a[:-1] + "a"
|
|
if number == "plural":
|
|
a = _regular_plural(a)
|
|
return a, ("rule" if (a != lemma or gender == "m") else "rule")
|
|
|
|
|
|
# ── PUBLIC: clitic enclisis (dá + me + lo -> dámelo) ────────────────────────────
|
|
def _strip_accents(s):
|
|
return "".join(c for c in unicodedata.normalize("NFD", s)
|
|
if unicodedata.category(c) != "Mn")
|
|
|
|
|
|
def _count_syllables_vowelgroups(word):
|
|
# crude: count vowel groups
|
|
w = _strip_accents(word).lower()
|
|
groups, prev = 0, False
|
|
for ch in w:
|
|
isv = ch in "aeiou"
|
|
if isv and not prev:
|
|
groups += 1
|
|
prev = isv
|
|
return groups
|
|
|
|
|
|
def _host_stress_from_end(word):
|
|
"""Stressed-syllable index counted from the end (1=last) of a verb host."""
|
|
syls = _count_syllables_vowelgroups(word)
|
|
if any(c in "áéíóú" for c in word):
|
|
return None # already carries its own accent
|
|
if word[-2:] in ("ar", "er", "ir"): # infinitive: oxytone
|
|
return 1
|
|
if word.endswith("ndo"): # gerund: paroxytone
|
|
return 2
|
|
if word[-1:] in "aeiouns" and syls >= 2: # default paroxytone
|
|
return 2
|
|
return 1 # monosyllable / consonant-final oxytone
|
|
|
|
|
|
def attach_enclitics(verb_form, clitics):
|
|
"""Append clitic pronouns to a verb (imperative/infinitive/gerund enclisis)
|
|
and add a written accent when the resulting word becomes esdrújula/
|
|
sobreesdrújula (stress >= 3 syllables from the end): dá+me+lo -> dámelo,
|
|
lleva+me -> llévame, but dar+te -> darte and da+me -> dame (no accent)."""
|
|
if not clitics:
|
|
return verb_form
|
|
tail = "".join(clitics)
|
|
if any(c in "áéíóú" for c in verb_form): # host already accented
|
|
return verb_form + tail
|
|
sfe = _host_stress_from_end(verb_form)
|
|
total_sfe = sfe + len(clitics) # each clitic = 1 syllable
|
|
if total_sfe >= 3:
|
|
return _accentuate_nucleus(verb_form, sfe) + tail
|
|
return verb_form + tail
|
|
|
|
|
|
def _accentuate_nucleus(word, sfe):
|
|
"""Put a written accent on the syllable `sfe` positions from the word's end."""
|
|
vowels = "aeiou"
|
|
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
|
if not nuclei or sfe > len(nuclei):
|
|
return word
|
|
i = nuclei[-sfe]
|
|
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
|
return word[:i] + acc[word[i]] + word[i + 1:]
|
|
|
|
|
|
def _accentuate_last_stressed(word):
|
|
# Restore the host's ORIGINAL lexical stress with a written accent.
|
|
# Default Spanish stress: word ending in vowel/n/s -> penultimate syllable;
|
|
# otherwise (e.g. infinitives in -r) -> last syllable.
|
|
vowels = "aeiou"
|
|
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
|
if not nuclei:
|
|
return word
|
|
if word[-1] in "aeiouns" and len(nuclei) >= 2:
|
|
i = nuclei[-2] # paroxytone: penult nucleus
|
|
else:
|
|
i = nuclei[-1] # oxytone / monosyllable: last nucleus
|
|
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
|
return word[:i] + acc[word[i]] + word[i + 1:]
|
|
|
|
|
|
def lexicon_stats():
|
|
return {
|
|
"source": "UniMorph Spanish (github.com/unimorph/spa)",
|
|
"license": "CC-BY-SA 3.0 (Wiktionary-derived)",
|
|
"total_forms": sum(len(v) for v in (_VERBS, _NOUNS, _ADJS)) if False else None,
|
|
"verb_forms": len(_VERBS),
|
|
"verb_lemmas": len({k[0] for k in _VERBS}),
|
|
"noun_lemmas": len(_NOUNS),
|
|
"adj_lemmas": len(_ADJS),
|
|
"participles": len(_PART),
|
|
"gerunds": len(_GER),
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import json
|
|
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
|
tests = [
|
|
("hablar", "ind", "present", "first", "singular", "hablo"),
|
|
("comer", "ind", "present", "third", "plural", "comen"),
|
|
("vivir", "ind", "present", "first", "plural", "vivimos"),
|
|
("ser", "ind", "present", "third", "singular", "es"),
|
|
("ir", "ind", "preterite", "first", "singular", "fui"),
|
|
("tener", "ind", "future", "first", "singular", "tendré"),
|
|
("hacer", "sbjv", "present", "first", "singular", "haga"),
|
|
("dormir", "ind", "present", "first", "singular", "duermo"),
|
|
("pensar", "sbjv", "present", "third", "singular", "piense"),
|
|
("dar", "ind", "preterite", "third", "singular", "dio"),
|
|
("poner", "ind", "conditional", "first", "singular", "pondría"),
|
|
]
|
|
ok = 0
|
|
for lemma, mood, tense, per, num, exp in tests:
|
|
got, conf = conjugate(lemma, mood, tense, per, num)
|
|
flag = "OK " if got == exp else "XX "
|
|
if got == exp:
|
|
ok += 1
|
|
print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]:3} -> {got:14} ({conf}) exp={exp}")
|
|
print(f"verb tests {ok}/{len(tests)}")
|
|
print(" gender casa:", noun_gender("casa"), "| problema:", noun_gender("problema"),
|
|
"| agua:", noun_gender("agua"), "| mano:", noun_gender("mano"))
|
|
print(" plural: luz->", inflect_noun("luz", "plural"), "| rey->", inflect_noun("rey", "plural"))
|
|
print(" adj: rojo/f/pl->", inflect_adj("rojo", "f", "plural"),
|
|
"| feliz/m/pl->", inflect_adj("feliz", "m", "plural"),
|
|
"| grande/f/pl->", inflect_adj("grande", "f", "plural"))
|
|
print(" enclisis: da+[me,lo]->", attach_enclitics("da", ["me", "lo"]),
|
|
"| di+[me]->", attach_enclitics("di", ["me"]),
|
|
"| dar+[se,lo]->", attach_enclitics("dar", ["se", "lo"]))
|