"""morphology_es_full.py — production-grade Spanish morphological generator. NOT a toy. Backed by a real, broad, licensed lexicon: UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0, Wiktionary-derived) 1,196,245 inflected forms: 6,695 verb lemmas — full paradigms: indicative (present/preterite/ imperfect/future), conditional, present & imperfect subjunctive, affirmative imperative, formal/informal 48,353 noun lemmas — WITH inherent gender (N;FEM/MASC;SG/PL) 16,984 adj lemmas — gender + number paradigms Fallbacks (so we degrade, never crash, on out-of-vocabulary input): - verbs : mlconjug3 (ML paradigm model, conjugates ANY Spanish verb) then a hand-rolled regular-ending generator - nouns : gender heuristic (endings) + regular pluralization - adjs : -o/-a gender rule + regular pluralization Every generated form carries a CONFIDENCE flag: "lexicon" form came straight from UniMorph (trust: high) "model" form came from mlconjug3 (trust: high) "rule" form came from a deterministic rule (trust: medium) "fallback" we could not inflect; returned lemma as-is (trust: low → FLAG) Public API (used by realizer_es.py): conjugate(lemma, mood, tense, person, number, formality="informal") -> (form, conf) participle(lemma) -> (form, conf) # past participle (compound tenses) gerund(lemma) -> (form, conf) noun_gender(lemma) -> "m"|"f" inflect_noun(lemma, number) -> (form, conf) inflect_adj(lemma, gender, number) -> (form, conf) attach_enclitics(verb_form, clitics) -> str # accent-correct enclisis lexicon_stats() -> dict """ import os import pickle import unicodedata _HERE = os.path.dirname(os.path.abspath(__file__)) _UNIMORPH = os.path.join(_HERE, "data", "spa.unimorph") _CACHE = os.path.join(_HERE, "data", "es_morph_cache.pkl") # ── canonical feature keys the realizer speaks, mapped to UniMorph tags ───────── # mood/tense pair -> the UniMorph feature substring that identifies it _VERB_KEYMAP = { ("ind", "present"): ("IND", "PRS", None), ("ind", "preterite"): ("IND", "PST", "PFV"), ("ind", "imperfect"): ("IND", "PST", "IPFV"), ("ind", "future"): ("IND", "FUT", None), ("ind", "conditional"):("COND", None, None), ("sbjv", "present"): ("SBJV", "PRS", None), ("sbjv", "imperfect"): ("SBJV", "PST", "LGSPEC1"), # -ra form ("imp", "present"): ("POS", "IMP", None), } _PERSON = {"first": "1", "second": "2", "third": "3"} _NUMBER = {"singular": "SG", "plural": "PL"} # ── build / load the compact lexicon ─────────────────────────────────────────── def _feat_set(tag): return set(tag.split(";")) def _build_cache(): verbs = {} # (lemma, canonkey) -> form canonkey e.g. "ind|present|1|SG|infm" nouns = {} # lemma -> {"g": "m"/"f", "SG": form, "PL": form} adjs = {} # lemma -> {("m","SG"): form, ...} part = {} # lemma -> masc-sg participle ger = {} # lemma -> gerund with open(_UNIMORPH, encoding="utf-8") as fh: for line in fh: line = line.rstrip("\n") if not line or "\t" not in line: continue parts = line.split("\t") if len(parts) != 3: continue lemma, form, tag = parts f = _feat_set(tag) head = tag.split(";")[0] if head == "V": # skip clitic-bearing rows (we generate clitics ourselves) if "PRO" in f: continue if "V.PTCP" in f and "PST" in f and "MASC" in f and "SG" in f: part.setdefault(lemma, form) continue if "V.CVB" in f or "NFIN" in f or "V.PTCP" in f: if "V.CVB" in f: ger.setdefault(lemma, form) continue # identify mood/tense mt = None for (mood, tense), (a, b, c) in _VERB_KEYMAP.items(): if a not in f: continue if b is not None and b not in f: continue if c is not None and c not in f: continue # disambiguate IND;PST needing PFV vs IPFV if a == "IND" and b == "PST" and c not in f: continue mt = (mood, tense) break if mt is None: continue person = next((p for p in ("1", "2", "3") if p in f), None) number = "SG" if "SG" in f else ("PL" if "PL" in f else None) if person is None or number is None: continue formal = "form" if "FORM" in f else ("infm" if "INFM" in f else "any") key = f"{mt[0]}|{mt[1]}|{person}|{number}|{formal}" verbs.setdefault((lemma, key), form) elif head == "N": # substring test handles epicene "MASC+FEM" (-> masc citation) g = "m" if "MASC" in tag else ("f" if "FEM" in tag else None) num = "SG" if "SG" in f else ("PL" if "PL" in f else None) if num is None: continue # store forms keyed by (gender,number); animate nouns list BOTH # genders under one lemma (niño -> niño/niña). Resolve citation # gender in a post-pass (gender of the row whose form == lemma). d = nouns.setdefault(lemma, {}) d.setdefault("_rows", []).append((g, num, form)) elif head == "ADJ": g = "m" if "MASC" in tag else ("f" if "FEM" in tag else "m") num = "SG" if "SG" in f else ("PL" if "PL" in f else None) if num is None: continue adjs.setdefault(lemma, {})[(g, num)] = form # post-pass: resolve noun citation gender + default SG/PL forms for lemma, d in nouns.items(): rows = d.pop("_rows", []) # citation gender = gender of the row whose form == lemma; else first MASC; # else first seen gender. cite_g = None for g, num, form in rows: if form == lemma and g: cite_g = g break if cite_g is None: for g, num, form in rows: if g == "m": cite_g = "m" break if cite_g is None: cite_g = next((g for g, _, _ in rows if g), "m") d["g"] = cite_g for g, num, form in rows: d[(g, num)] = form d["SG"] = d.get((cite_g, "SG")) or next((f for g, n, f in rows if n == "SG"), lemma) d["PL"] = d.get((cite_g, "PL")) or next((f for g, n, f in rows if n == "PL"), None) # post-pass: UniMorph omits the identity inflection (masc-sg == lemma) for # adjectives, so fill it in; without this a fem-sg row wrongly satisfies a # masc-sg request (alto -> alta bug). for lemma, d in adjs.items(): d.setdefault(("m", "SG"), lemma) data = {"verbs": verbs, "nouns": nouns, "adjs": adjs, "part": part, "ger": ger} try: with open(_CACHE, "wb") as fh: pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL) except OSError: pass return data def _load(): if os.path.exists(_CACHE) and os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH): try: with open(_CACHE, "rb") as fh: return pickle.load(fh) except Exception: pass return _build_cache() _LEX = _load() _VERBS, _NOUNS, _ADJS, _PART, _GER = ( _LEX["verbs"], _LEX["nouns"], _LEX["adjs"], _LEX["part"], _LEX["ger"]) # ── mlconjug3 fallback (lazy) ─────────────────────────────────────────────────── _MLC = None _MLC_TENSE = { # (mood,tense) -> (mlconjug mood label, tense label) ("ind", "present"): ("Indicativo", "Indicativo presente"), ("ind", "preterite"): ("Indicativo", "Indicativo pretérito perfecto simple"), ("ind", "imperfect"): ("Indicativo", "Indicativo pretérito imperfecto"), ("ind", "future"): ("Indicativo", "Indicativo futuro"), ("ind", "conditional"): ("Condicional", "Condicional Condicional"), ("sbjv", "present"): ("Subjuntivo", "Subjuntivo presente"), ("sbjv", "imperfect"): ("Subjuntivo", "Subjuntivo pretérito imperfecto 1"), ("imp", "present"): ("Imperativo", "Imperativo Afirmativo"), } _MLC_SLOT = { # (person,number) -> mlconjug slot key ("first", "singular"): "1s", ("second", "singular"): "2s", ("third", "singular"): "3s", ("first", "plural"): "1p", ("second", "plural"): "2p", ("third", "plural"): "3p", } def _mlc_conjugate(lemma, mood, tense, person, number): global _MLC try: if _MLC is None: from mlconjug3 import Conjugator _MLC = Conjugator(language="es") v = _MLC.conjugate(lemma) if v is None: return None info = v.conjug_info m, t = _MLC_TENSE.get((mood, tense), (None, None)) if m is None or m not in info or t not in info[m]: return None block = info[m][t] slot = _MLC_SLOT.get((person, number)) if isinstance(block, dict) and slot in block and block[slot]: return block[slot] return None except Exception: return None # ── regular-ending rule fallback (last resort, deterministic) ─────────────────── def _vclass(lemma): return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else "ar" def _stem(lemma): return lemma[:-2] _REG = { ("ind", "present", "ar"): ["o", "as", "a", "amos", "áis", "an"], ("ind", "present", "er"): ["o", "es", "e", "emos", "éis", "en"], ("ind", "present", "ir"): ["o", "es", "e", "imos", "ís", "en"], ("ind", "preterite", "ar"): ["é", "aste", "ó", "amos", "asteis", "aron"], ("ind", "preterite", "er"): ["í", "iste", "ió", "imos", "isteis", "ieron"], ("ind", "preterite", "ir"): ["í", "iste", "ió", "imos", "isteis", "ieron"], ("ind", "imperfect", "ar"): ["aba", "abas", "aba", "ábamos", "abais", "aban"], ("ind", "imperfect", "er"): ["ía", "ías", "ía", "íamos", "íais", "ían"], ("ind", "imperfect", "ir"): ["ía", "ías", "ía", "íamos", "íais", "ían"], ("sbjv", "present", "ar"): ["e", "es", "e", "emos", "éis", "en"], ("sbjv", "present", "er"): ["a", "as", "a", "amos", "áis", "an"], ("sbjv", "present", "ir"): ["a", "as", "a", "amos", "áis", "an"], ("sbjv", "imperfect", "ar"): ["ara", "aras", "ara", "áramos", "arais", "aran"], ("sbjv", "imperfect", "er"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"], ("sbjv", "imperfect", "ir"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"], } _FUT = ["é", "ás", "á", "emos", "éis", "án"] _COND = ["ía", "ías", "ía", "íamos", "íais", "ían"] def _slot_idx(person, number): base = {"first": 0, "second": 1, "third": 2}[person] return base + (0 if number == "singular" else 3) def _rule_conjugate(lemma, mood, tense, person, number): if len(lemma) < 3 or lemma[-2:] not in ("ar", "er", "ir"): return None vc, st, i = _vclass(lemma), _stem(lemma), _slot_idx(person, number) if tense == "future": return lemma + _FUT[i] if tense == "conditional": return lemma + _COND[i] table = _REG.get((mood, tense, vc)) if table: return st + table[i] if mood == "imp" and tense == "present": # affirmative tú imperative = 3sg present indicative pres = _REG.get(("ind", "present", vc)) return st + pres[2] if number == "singular" else st + pres[5] return None # ── PUBLIC: verb conjugation ──────────────────────────────────────────────────── def conjugate(lemma, mood, tense, person, number, formality="informal"): """Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP.""" lemma = lemma.strip().lower() p, n = _PERSON.get(person), _NUMBER.get(number) formal = "form" if formality == "formal" else "infm" if p and n: for fkey in (formal, "any", "infm" if formal == "form" else "form"): form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}|{fkey}")) if form: return form, "lexicon" m = _mlc_conjugate(lemma, mood, tense, person, number) if m: return m, "model" r = _rule_conjugate(lemma, mood, tense, person, number) if r: return r, "rule" return lemma, "fallback" _IRREG_PART = { # guarantee the common irregular participles "escribir": "escrito", "describir": "descrito", "abrir": "abierto", "cubrir": "cubierto", "descubrir": "descubierto", "morir": "muerto", "poner": "puesto", "ver": "visto", "volver": "vuelto", "devolver": "devuelto", "hacer": "hecho", "deshacer": "deshecho", "decir": "dicho", "romper": "roto", "resolver": "resuelto", "freír": "frito", "imprimir": "impreso", "satisfacer": "satisfecho", "prever": "previsto", "revolver": "revuelto", } def participle(lemma): lemma = lemma.strip().lower() if lemma in _IRREG_PART: return _IRREG_PART[lemma], "lexicon" if lemma in _PART: return _PART[lemma], "lexicon" if lemma.endswith("ar"): return lemma[:-2] + "ado", "rule" if lemma[-2:] in ("er", "ir"): return lemma[:-2] + "ido", "rule" return lemma, "fallback" _IRREG_GER = {"dormir": "durmiendo", "morir": "muriendo", "pedir": "pidiendo", "sentir": "sintiendo", "mentir": "mintiendo", "servir": "sirviendo", "venir": "viniendo", "decir": "diciendo", "poder": "pudiendo", "ir": "yendo", "leer": "leyendo", "creer": "creyendo", "oír": "oyendo", "traer": "trayendo", "caer": "cayendo", "construir": "construyendo", "huir": "huyendo", "reír": "riendo"} def gerund(lemma): lemma = lemma.strip().lower() if lemma in _IRREG_GER: return _IRREG_GER[lemma], "lexicon" if lemma in _GER: return _GER[lemma], "lexicon" if lemma.endswith("ar"): return lemma[:-2] + "ando", "rule" if lemma[-2:] in ("er", "ir"): return lemma[:-2] + "iendo", "rule" return lemma, "fallback" # ── PUBLIC: noun gender + number ──────────────────────────────────────────────── _INVARIANT_PL = {"lunes", "martes", "miércoles", "jueves", "viernes", "crisis", "tesis", "análisis", "dosis", "virus", "paraguas"} def _gender_heuristic(noun): for suf, g in (("ión", "f"), ("dad", "f"), ("tad", "f"), ("umbre", "f"), ("sis", "f"), ("ez", "f"), ("triz", "f"), ("ema", "m"), ("ama", "m"), ("oma", "m"), ("aje", "m"), ("or", "m"), ("án", "m"), ("ín", "m")): if noun.endswith(suf): return g if noun.endswith("o"): return "m" if noun.endswith("a"): return "f" return "m" def noun_gender(lemma): lemma = lemma.strip().lower() d = _NOUNS.get(lemma) if d and d.get("g"): return d["g"] return _gender_heuristic(lemma) def _regular_plural(noun): if noun in _INVARIANT_PL: return noun if not noun: return noun last = noun[-1] if last == "z": return noun[:-1] + "ces" if last in "aeiouáéíóú": # stressed final vowel í/ú -> +es (rubí->rubíes), else +s if last in "íú": return noun + "es" return noun + "s" if last == "s": # esdrújula / stress-final handled crudely; most polysyllables invariant return noun return noun + "es" def inflect_noun(lemma, number, gender=None): lemma = lemma.strip().lower() d = _NOUNS.get(lemma) num = "SG" if number == "singular" else "PL" if d: # honor a requested gender for animate nouns (gato -> gata) if gender and (gender, num) in d: return d[(gender, num)], "lexicon" if d.get(num): return d[num], "lexicon" if number == "singular": return lemma, "rule" if not d else "lexicon" return _regular_plural(lemma), "rule" # ── PUBLIC: adjective agreement ───────────────────────────────────────────────── _INV_GENDER_ADJ = {"español": "española", "trabajador": "trabajadora", "hablador": "habladora", "encantador": "encantadora", "alemán": "alemana", "francés": "francesa", "inglés": "inglesa"} def inflect_adj(lemma, gender, number): lemma = lemma.strip().lower() d = _ADJS.get(lemma) num = "SG" if number == "singular" else "PL" if d: form = d.get((gender, num)) if form: return form, "lexicon" # gender-invariant adjective (grande, feliz, azul): fem == masc. # For a missing plural, pluralize this gender's singular form. sg = d.get((gender, "SG")) or d.get(("m", "SG")) or lemma if number == "plural": return _regular_plural(sg), "rule" return sg, "lexicon" # rule fallback a = lemma if gender == "f": if a in _INV_GENDER_ADJ: a = _INV_GENDER_ADJ[a] elif a.endswith("o"): a = a[:-1] + "a" if number == "plural": a = _regular_plural(a) return a, ("rule" if (a != lemma or gender == "m") else "rule") # ── PUBLIC: clitic enclisis (dá + me + lo -> dámelo) ──────────────────────────── def _strip_accents(s): return "".join(c for c in unicodedata.normalize("NFD", s) if unicodedata.category(c) != "Mn") def _count_syllables_vowelgroups(word): # crude: count vowel groups w = _strip_accents(word).lower() groups, prev = 0, False for ch in w: isv = ch in "aeiou" if isv and not prev: groups += 1 prev = isv return groups def _host_stress_from_end(word): """Stressed-syllable index counted from the end (1=last) of a verb host.""" syls = _count_syllables_vowelgroups(word) if any(c in "áéíóú" for c in word): return None # already carries its own accent if word[-2:] in ("ar", "er", "ir"): # infinitive: oxytone return 1 if word.endswith("ndo"): # gerund: paroxytone return 2 if word[-1:] in "aeiouns" and syls >= 2: # default paroxytone return 2 return 1 # monosyllable / consonant-final oxytone def attach_enclitics(verb_form, clitics): """Append clitic pronouns to a verb (imperative/infinitive/gerund enclisis) and add a written accent when the resulting word becomes esdrújula/ sobreesdrújula (stress >= 3 syllables from the end): dá+me+lo -> dámelo, lleva+me -> llévame, but dar+te -> darte and da+me -> dame (no accent).""" if not clitics: return verb_form tail = "".join(clitics) if any(c in "áéíóú" for c in verb_form): # host already accented return verb_form + tail sfe = _host_stress_from_end(verb_form) total_sfe = sfe + len(clitics) # each clitic = 1 syllable if total_sfe >= 3: return _accentuate_nucleus(verb_form, sfe) + tail return verb_form + tail def _accentuate_nucleus(word, sfe): """Put a written accent on the syllable `sfe` positions from the word's end.""" vowels = "aeiou" nuclei = [i for i, ch in enumerate(word) if ch in vowels] if not nuclei or sfe > len(nuclei): return word i = nuclei[-sfe] acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"} return word[:i] + acc[word[i]] + word[i + 1:] def _accentuate_last_stressed(word): # Restore the host's ORIGINAL lexical stress with a written accent. # Default Spanish stress: word ending in vowel/n/s -> penultimate syllable; # otherwise (e.g. infinitives in -r) -> last syllable. vowels = "aeiou" nuclei = [i for i, ch in enumerate(word) if ch in vowels] if not nuclei: return word if word[-1] in "aeiouns" and len(nuclei) >= 2: i = nuclei[-2] # paroxytone: penult nucleus else: i = nuclei[-1] # oxytone / monosyllable: last nucleus acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"} return word[:i] + acc[word[i]] + word[i + 1:] def lexicon_stats(): return { "source": "UniMorph Spanish (github.com/unimorph/spa)", "license": "CC-BY-SA 3.0 (Wiktionary-derived)", "total_forms": sum(len(v) for v in (_VERBS, _NOUNS, _ADJS)) if False else None, "verb_forms": len(_VERBS), "verb_lemmas": len({k[0] for k in _VERBS}), "noun_lemmas": len(_NOUNS), "adj_lemmas": len(_ADJS), "participles": len(_PART), "gerunds": len(_GER), } if __name__ == "__main__": import json print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False)) tests = [ ("hablar", "ind", "present", "first", "singular", "hablo"), ("comer", "ind", "present", "third", "plural", "comen"), ("vivir", "ind", "present", "first", "plural", "vivimos"), ("ser", "ind", "present", "third", "singular", "es"), ("ir", "ind", "preterite", "first", "singular", "fui"), ("tener", "ind", "future", "first", "singular", "tendré"), ("hacer", "sbjv", "present", "first", "singular", "haga"), ("dormir", "ind", "present", "first", "singular", "duermo"), ("pensar", "sbjv", "present", "third", "singular", "piense"), ("dar", "ind", "preterite", "third", "singular", "dio"), ("poner", "ind", "conditional", "first", "singular", "pondría"), ] ok = 0 for lemma, mood, tense, per, num, exp in tests: got, conf = conjugate(lemma, mood, tense, per, num) flag = "OK " if got == exp else "XX " if got == exp: ok += 1 print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]:3} -> {got:14} ({conf}) exp={exp}") print(f"verb tests {ok}/{len(tests)}") print(" gender casa:", noun_gender("casa"), "| problema:", noun_gender("problema"), "| agua:", noun_gender("agua"), "| mano:", noun_gender("mano")) print(" plural: luz->", inflect_noun("luz", "plural"), "| rey->", inflect_noun("rey", "plural")) print(" adj: rojo/f/pl->", inflect_adj("rojo", "f", "plural"), "| feliz/m/pl->", inflect_adj("feliz", "m", "plural"), "| grande/f/pl->", inflect_adj("grande", "f", "plural")) print(" enclisis: da+[me,lo]->", attach_enclitics("da", ["me", "lo"]), "| di+[me]->", attach_enclitics("di", ["me"]), "| dar+[se,lo]->", attach_enclitics("dar", ["se", "lo"]))