Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c62ea3383c | |||
| ba6e36c3f7 |
@@ -0,0 +1,45 @@
|
||||
;;; lang_profile_es.el — Spanish language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_en / _pt.
|
||||
|
||||
(lang_profile_es
|
||||
(language "Spanish")
|
||||
(iso639 "es")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon); REAL per-noun gender from UniMorph — NOT a heuristic
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; questions by intonation/punctuation, not inversion
|
||||
(question-strategy intonation)
|
||||
(article-selection "el/la/los/las un/una/unos/unas")
|
||||
(stressed-a-rule yes) ; fem sg noun in stressed a-/ha- takes el/un (el agua)
|
||||
(adjective-position postnominal) ; default post; a few prenominal + apocope
|
||||
(adjective-agreement "gender+number")
|
||||
(question-punct inverted) ; opening ¿ ¡ required
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (coordinator quality bar) --------------------
|
||||
(contractions ((de el "del") (a el "al")))
|
||||
(contraction-mandatory yes) ; 'de el'/'a el' MUST surface as del/al
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "haber") ; haber + past participle (invariant -o)
|
||||
(progressive-aux "estar") ; estar + gerund
|
||||
(passive-aux "ser") ; ser + participle (agrees) + por-agent
|
||||
(copula-split "ser/estar") ; permanent vs stage-level
|
||||
(future "infinitive + é/ás/á/emos/éis/án")
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; me te lo la le nos os los las; proclisis/enclisis
|
||||
(clitic-order "se II I III (le+lo -> se lo)")
|
||||
(enclisis "imperative/infinitive/gerund + accent repair (dá+me+lo->dámelo)")
|
||||
(verb-prep-government yes) ; verbs select prep (protestar+contra, escapar+de)
|
||||
|
||||
;; -- SACRED safety bar (shared with en/pt) -------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
+171
-10
@@ -54,6 +54,16 @@ fn es_str_last3(s: String) -> String {
|
||||
// Spanish verbs fall into three conjugation classes defined by the infinitive
|
||||
// ending: -ar, -er, -ir. The stem is the infinitive minus those two characters.
|
||||
|
||||
// Strong-vowel-final test (a/e/o) — used for orthographic y-insertion and the
|
||||
// accented -ído participle (caer->caído/cayó; but ui/iu diphthongs stay plain).
|
||||
fn es_strong_vowel_final(s: String) -> Bool {
|
||||
let c: String = es_str_last_char(s)
|
||||
if str_eq(c, "a") { return true }
|
||||
if str_eq(c, "e") { return true }
|
||||
if str_eq(c, "o") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn es_verb_class(base: String) -> String {
|
||||
if es_str_ends(base, "ar") { return "ar" }
|
||||
if es_str_ends(base, "er") { return "er" }
|
||||
@@ -453,12 +463,18 @@ fn es_regular_preterite(stem: String, vclass: String, slot: Int) -> String {
|
||||
if slot == 4 { return stem + "asteis" }
|
||||
return stem + "aron"
|
||||
}
|
||||
// -er and -ir share the same preterite endings
|
||||
// -er and -ir share the same preterite endings.
|
||||
// Orthographic rule: a vowel-final stem takes -yó/-yeron (caer->cayó,
|
||||
// leer->leyó, creer->creyó) since -ió after a vowel becomes -yó.
|
||||
if slot == 0 { return stem + "í" }
|
||||
if slot == 1 { return stem + "iste" }
|
||||
if slot == 2 { return stem + "ió" }
|
||||
if slot == 2 {
|
||||
if es_strong_vowel_final(stem) { return stem + "yó" }
|
||||
return stem + "ió"
|
||||
}
|
||||
if slot == 3 { return stem + "imos" }
|
||||
if slot == 4 { return stem + "isteis" }
|
||||
if es_strong_vowel_final(stem) { return stem + "yeron" }
|
||||
return stem + "ieron"
|
||||
}
|
||||
|
||||
@@ -640,17 +656,35 @@ fn es_pluralize(noun: String) -> String {
|
||||
if !str_eq(inv, "") {
|
||||
return inv
|
||||
}
|
||||
let last: String = es_str_last_char(noun)
|
||||
// Oxytone nouns ending accented-vowel + n LOSE the written accent in the
|
||||
// plural (canción->canciones, razón->razones, jardín->jardines).
|
||||
// NOTE El strings are byte-indexed; "ón"/"án"/... are 3 bytes (accent=2 +n).
|
||||
if es_str_ends(noun, "ón") { return es_str_drop_last(noun, 3) + "ones" }
|
||||
if es_str_ends(noun, "án") { return es_str_drop_last(noun, 3) + "anes" }
|
||||
if es_str_ends(noun, "én") { return es_str_drop_last(noun, 3) + "enes" }
|
||||
if es_str_ends(noun, "ín") { return es_str_drop_last(noun, 3) + "ines" }
|
||||
if es_str_ends(noun, "ún") { return es_str_drop_last(noun, 3) + "unes" }
|
||||
// Oxytone accented-vowel + s also loses the accent (francés->franceses,
|
||||
// inglés->ingleses). í/ú stay (país->países via the consonant rule).
|
||||
if es_str_ends(noun, "és") { return es_str_drop_last(noun, 3) + "eses" }
|
||||
if es_str_ends(noun, "ás") { return es_str_drop_last(noun, 3) + "ases" }
|
||||
if es_str_ends(noun, "ós") { return es_str_drop_last(noun, 3) + "oses" }
|
||||
// Ends in -z: replace with -ces
|
||||
if str_eq(last, "z") {
|
||||
if es_str_ends(noun, "z") {
|
||||
return es_str_drop_last(noun, 1) + "ces"
|
||||
}
|
||||
// Ends in a vowel: add -s
|
||||
if str_eq(last, "a") { return noun + "s" }
|
||||
if str_eq(last, "e") { return noun + "s" }
|
||||
if str_eq(last, "i") { return noun + "s" }
|
||||
if str_eq(last, "o") { return noun + "s" }
|
||||
if str_eq(last, "u") { return noun + "s" }
|
||||
// Stressed final vowel: á/é/ó -> +s (café->cafés); í/ú -> +es (rubí->rubíes)
|
||||
if es_str_ends(noun, "á") { return noun + "s" }
|
||||
if es_str_ends(noun, "é") { return noun + "s" }
|
||||
if es_str_ends(noun, "ó") { return noun + "s" }
|
||||
if es_str_ends(noun, "í") { return noun + "es" }
|
||||
if es_str_ends(noun, "ú") { return noun + "es" }
|
||||
// Plain final vowel: add -s
|
||||
if es_str_ends(noun, "a") { return noun + "s" }
|
||||
if es_str_ends(noun, "e") { return noun + "s" }
|
||||
if es_str_ends(noun, "i") { return noun + "s" }
|
||||
if es_str_ends(noun, "o") { return noun + "s" }
|
||||
if es_str_ends(noun, "u") { return noun + "s" }
|
||||
// Ends in consonant (including -s for stressed words like autobús): add -es
|
||||
return noun + "es"
|
||||
}
|
||||
@@ -687,6 +721,37 @@ fn es_starts_with_stressed_a(noun: String) -> Bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// es_article_for_gender: the article logic, given gender EXPLICITLY.
|
||||
// The realizer should call this with the REAL per-noun gender from
|
||||
// vocabulary-es.el (form2), NOT the es_gender heuristic — that is what
|
||||
// eliminates the 'el mano' / 'la día' masculine-default error class.
|
||||
fn es_article_for_gender(gender: String, noun: String, definite: String, number: String) -> String {
|
||||
let is_plural: Bool = str_eq(number, "plural")
|
||||
let is_def: Bool = str_eq(definite, "true")
|
||||
|
||||
if is_def {
|
||||
if is_plural {
|
||||
if str_eq(gender, "f") { return "las" }
|
||||
return "los"
|
||||
}
|
||||
if str_eq(gender, "f") {
|
||||
if es_starts_with_stressed_a(noun) { return "el" }
|
||||
return "la"
|
||||
}
|
||||
return "el"
|
||||
}
|
||||
|
||||
if is_plural {
|
||||
if str_eq(gender, "f") { return "unas" }
|
||||
return "unos"
|
||||
}
|
||||
if str_eq(gender, "f") {
|
||||
if es_starts_with_stressed_a(noun) { return "un" }
|
||||
return "una"
|
||||
}
|
||||
return "un"
|
||||
}
|
||||
|
||||
fn es_agree_article(noun: String, definite: String, number: String) -> String {
|
||||
let gender: String = es_gender(noun)
|
||||
let is_plural: Bool = str_eq(number, "plural")
|
||||
@@ -714,3 +779,99 @@ fn es_agree_article(noun: String, definite: String, number: String) -> String {
|
||||
if str_eq(gender, "f") { return "una" }
|
||||
return "un"
|
||||
}
|
||||
|
||||
// ── Past participle (compound tenses: haber + participle) ─────────────────────
|
||||
//
|
||||
// Irregular participle table (transcribed from morphology_es_full._IRREG_PART),
|
||||
// then the regular rule: -ar -> -ado, -er/-ir -> -ido.
|
||||
|
||||
fn es_participle(verb: String) -> String {
|
||||
if str_eq(verb, "escribir") { return "escrito" }
|
||||
if str_eq(verb, "describir") { return "descrito" }
|
||||
if str_eq(verb, "abrir") { return "abierto" }
|
||||
if str_eq(verb, "cubrir") { return "cubierto" }
|
||||
if str_eq(verb, "descubrir") { return "descubierto" }
|
||||
if str_eq(verb, "morir") { return "muerto" }
|
||||
if str_eq(verb, "poner") { return "puesto" }
|
||||
if str_eq(verb, "ver") { return "visto" }
|
||||
if str_eq(verb, "volver") { return "vuelto" }
|
||||
if str_eq(verb, "devolver") { return "devuelto" }
|
||||
if str_eq(verb, "hacer") { return "hecho" }
|
||||
if str_eq(verb, "deshacer") { return "deshecho" }
|
||||
if str_eq(verb, "decir") { return "dicho" }
|
||||
if str_eq(verb, "romper") { return "roto" }
|
||||
if str_eq(verb, "resolver") { return "resuelto" }
|
||||
if str_eq(verb, "freír") { return "frito" }
|
||||
if str_eq(verb, "imprimir") { return "impreso" }
|
||||
if str_eq(verb, "satisfacer") { return "satisfecho" }
|
||||
if str_eq(verb, "prever") { return "previsto" }
|
||||
if str_eq(verb, "revolver") { return "revuelto" }
|
||||
// Accented -ír infinitives (oír->oído, sonreír->sonreído): "ír" is 3 bytes.
|
||||
if es_str_ends(verb, "ír") { return es_str_drop_last(verb, 3) + "ído" }
|
||||
if es_str_ends(verb, "ar") { return es_str_drop_last(verb, 2) + "ado" }
|
||||
// -er/-ir: a strong-vowel stem takes the accented -ído (caer->caído,
|
||||
// leer->leído, poseer->poseído); consonant stems stay -ido; ui/iu
|
||||
// diphthongs (construir->construido) stay plain.
|
||||
if es_str_ends(verb, "er") {
|
||||
let st: String = es_str_drop_last(verb, 2)
|
||||
if es_strong_vowel_final(st) { return st + "ído" }
|
||||
return st + "ido"
|
||||
}
|
||||
if es_str_ends(verb, "ir") {
|
||||
let st: String = es_str_drop_last(verb, 2)
|
||||
if es_strong_vowel_final(st) { return st + "ído" }
|
||||
return st + "ido"
|
||||
}
|
||||
return verb
|
||||
}
|
||||
|
||||
// ── Adjective agreement (gender + number) ─────────────────────────────────────
|
||||
//
|
||||
// Rule-based agreement mirroring morphology_es_full.inflect_adj (rule path):
|
||||
// feminine: -o -> -a; nationality/-dor exceptions add -a; else invariant
|
||||
// plural: reuse es_pluralize on the agreed singular
|
||||
// (Lexicon-backed adjectives in Python may differ; those divergences are what
|
||||
// the parity check surfaces.)
|
||||
|
||||
fn es_adj_feminine(lemma: String) -> String {
|
||||
if str_eq(lemma, "español") { return "española" }
|
||||
if str_eq(lemma, "francés") { return "francesa" }
|
||||
if str_eq(lemma, "inglés") { return "inglesa" }
|
||||
if str_eq(lemma, "alemán") { return "alemana" }
|
||||
if str_eq(lemma, "trabajador") { return "trabajadora" }
|
||||
if str_eq(lemma, "hablador") { return "habladora" }
|
||||
if str_eq(lemma, "encantador") { return "encantadora" }
|
||||
if es_str_ends(lemma, "o") { return es_str_drop_last(lemma, 1) + "a" }
|
||||
return lemma
|
||||
}
|
||||
|
||||
fn es_inflect_adj(lemma: String, gender: String, number: String) -> String {
|
||||
if str_eq(gender, "f") {
|
||||
let fem: String = es_adj_feminine(lemma)
|
||||
if str_eq(number, "plural") { return es_pluralize(fem) }
|
||||
return fem
|
||||
}
|
||||
// masculine (citation)
|
||||
if str_eq(number, "plural") { return es_pluralize(lemma) }
|
||||
return lemma
|
||||
}
|
||||
|
||||
// ── Mandatory preposition + article contraction (de+el->del, a+el->al) ────────
|
||||
//
|
||||
// Transcribed from realizer_es._contract. np_text is the already-realized NP
|
||||
// beginning with "el " when the contraction fires.
|
||||
|
||||
fn es_starts_el(np_text: String) -> Bool {
|
||||
let n: Int = str_len(np_text)
|
||||
if n < 3 { return false }
|
||||
return str_eq(str_slice(np_text, 0, 3), "el ")
|
||||
}
|
||||
|
||||
fn es_contract(prep: String, np_text: String) -> String {
|
||||
if es_starts_el(np_text) {
|
||||
let rest: String = str_slice(np_text, 3, str_len(np_text))
|
||||
if str_eq(prep, "de") { return "del " + rest }
|
||||
if str_eq(prep, "a") { return "al " + rest }
|
||||
}
|
||||
return prep + " " + np_text
|
||||
}
|
||||
|
||||
@@ -0,0 +1,362 @@
|
||||
;;; vocabulary-es.el — Spanish vocabulary for ELP surface realization.
|
||||
;;; Schema: (lemma pos form0 form1 form2 en_translation semantic_hint)
|
||||
;;; Source: UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0),
|
||||
;;; generated by gen_elp_es.py via morphology_es_full (real forms).
|
||||
;;; Verbs: form0=present-ind-3sg form1=preterite-3sg form2=past-participle
|
||||
;;; Nouns: form0=singular form1=plural form2=REAL gender (m/f, from lexicon —
|
||||
;;; NOT an ending heuristic; this is what kills 'el mano'/'la día' errors)
|
||||
;;; Adjs : form0=masc-sg form1=fem-sg form2=masc-pl
|
||||
|
||||
(vocabulary-es
|
||||
|
||||
;; -- function / closed class (incl. mandatory contractions del/al) --------
|
||||
("el" "det" "el" "los" "m" "the" "definite article m.sg")
|
||||
("la" "det" "la" "las" "f" "the" "definite article f.sg")
|
||||
("un" "det" "un" "unos" "m" "a" "indefinite article m.sg")
|
||||
("una" "det" "una" "unas" "f" "a" "indefinite article f.sg")
|
||||
("del" "contraction" "del" "" "" "of the" "de + el (mandatory contraction)")
|
||||
("al" "contraction" "al" "" "" "to the" "a + el (mandatory contraction)")
|
||||
("este" "dem" "este" "estos" "m" "this" "proximal dem m")
|
||||
("esta" "dem" "esta" "estas" "f" "this" "proximal dem f")
|
||||
("no" "neg" "no" "" "" "not/no" "sentential negator (preverbal)")
|
||||
("ninguno" "det" "ningún" "ninguna" "" "none" "negative determiner (apocope ningún m.sg)")
|
||||
("y" "conj" "y" "e" "" "and" "coordinator (e before i-/hi-)")
|
||||
("o" "conj" "o" "u" "" "or" "coordinator (u before o-/ho-)")
|
||||
("pero" "conj" "pero" "" "" "but" "adversative coordinator")
|
||||
("que" "conj" "que" "" "" "that" "complementizer / relative")
|
||||
("si" "conj" "si" "" "" "if" "conditional subordinator")
|
||||
("porque" "conj" "porque" "" "" "because" "causal subordinator")
|
||||
("cuando" "conj" "cuando" "" "" "when" "temporal subordinator")
|
||||
("a" "prep" "a" "" "" "to" "dir-obj (personal a) / dative / allative; a+el=al")
|
||||
("de" "prep" "de" "" "" "of/from" "genitive/ablative government; de+el=del")
|
||||
("en" "prep" "en" "" "" "in/on" "locative")
|
||||
("con" "prep" "con" "" "" "with" "comitative")
|
||||
("por" "prep" "por" "" "" "by/for" "passive agent / cause")
|
||||
("para" "prep" "para" "" "" "for" "purpose/benefactive")
|
||||
("contra" "prep" "contra" "" "" "against" "adversative government (protestar contra)")
|
||||
("sin" "prep" "sin" "" "" "without" "privative")
|
||||
("yo" "pron" "yo" "me" "mi" "I" "1sg subj/obj/poss")
|
||||
("tú" "pron" "tú" "te" "tu" "you" "2sg informal")
|
||||
("usted" "pron" "usted" "lo" "su" "you" "2sg formal (3sg agreement)")
|
||||
("él" "pron" "él" "lo" "su" "he" "3sg m subj/DO-clitic/poss")
|
||||
("ella" "pron" "ella" "la" "su" "she" "3sg f subj/DO-clitic/poss")
|
||||
("nosotros" "pron" "nosotros" "nos" "nuestro" "we" "1pl")
|
||||
("vosotros" "pron" "vosotros" "os" "vuestro" "you" "2pl informal")
|
||||
("ellos" "pron" "ellos" "los" "su" "they" "3pl m")
|
||||
("ellas" "pron" "ellas" "las" "su" "they" "3pl f")
|
||||
("le" "clitic" "le" "les" "" "to-him/her" "dative clitic 3sg/3pl (->se before lo/la)")
|
||||
("se" "clitic" "se" "se" "" "himself/-self" "reflexive / spurious-se (le+lo->se lo)")
|
||||
|
||||
;; -- verbs (form0=pres-3sg form1=pret-3sg form2=past-participle) -----------
|
||||
("abrazar" "verb" "abraza" "abrazó" "abrazado" "abrazar" "ar/regular")
|
||||
("abrir" "verb" "abre" "abrió" "abierto" "abrir" "ir/irregular")
|
||||
("aceptar" "verb" "acepta" "aceptó" "aceptado" "aceptar" "ar/regular")
|
||||
("acordar" "verb" "acuerda" "acordó" "acordado" "acordar" "ar/regular")
|
||||
("amar" "verb" "ama" "amó" "amado" "amar" "ar/regular")
|
||||
("anunciar" "verb" "anuncia" "anunció" "anunciado" "anunciar" "ar/regular")
|
||||
("aprobar" "verb" "aprueba" "aprobó" "aprobado" "aprobar" "ar/regular")
|
||||
("aumentar" "verb" "aumenta" "aumentó" "aumentado" "aumentar" "ar/regular")
|
||||
("ayudar" "verb" "ayuda" "ayudó" "ayudado" "ayudar" "ar/regular")
|
||||
("bajar" "verb" "baja" "bajó" "bajado" "bajar" "ar/regular")
|
||||
("besar" "verb" "besa" "besó" "besado" "besar" "ar/regular")
|
||||
("buscar" "verb" "busca" "buscó" "buscado" "buscar" "ar/regular")
|
||||
("caer" "verb" "cae" "cayó" "caído" "caer" "er/regular")
|
||||
("cambiar" "verb" "cambia" "cambió" "cambiado" "cambiar" "ar/regular")
|
||||
("caminar" "verb" "camina" "caminó" "caminado" "caminar" "ar/regular")
|
||||
("cantar" "verb" "canta" "cantó" "cantado" "cantar" "ar/regular")
|
||||
("celebrar" "verb" "celebra" "celebró" "celebrado" "celebrar" "ar/regular")
|
||||
("cerrar" "verb" "cierra" "cerró" "cerrado" "cerrar" "ar/regular")
|
||||
("comenzar" "verb" "comienza" "comenzó" "comenzado" "comenzar" "ar/regular")
|
||||
("comer" "verb" "come" "comió" "comido" "comer" "er/regular")
|
||||
("condenar" "verb" "condena" "condenó" "condenado" "condenar" "ar/regular")
|
||||
("confirmar" "verb" "confirma" "confirmó" "confirmado" "confirmar" "ar/regular")
|
||||
("conocer" "verb" "conoce" "conoció" "conocido" "conocer" "er/regular")
|
||||
("contar" "verb" "cuenta" "contó" "contado" "contar" "ar/regular")
|
||||
("contratar" "verb" "contrata" "contrató" "contratado" "contratar" "ar/regular")
|
||||
("costar" "verb" "cuesta" "costó" "costado" "costar" "ar/regular")
|
||||
("crear" "verb" "crea" "creó" "creado" "crear" "ar/regular")
|
||||
("crecer" "verb" "crece" "creció" "crecido" "crecer" "er/regular")
|
||||
("creer" "verb" "cree" "creyó" "creído" "creer" "er/regular")
|
||||
("dar" "verb" "da" "dio" "dado" "dar" "ar/irregular")
|
||||
("deber" "verb" "debe" "debió" "debido" "deber" "er/regular")
|
||||
("decir" "verb" "dice" "dijo" "dicho" "decir" "ir/irregular")
|
||||
("dejar" "verb" "deja" "dejó" "dejado" "dejar" "ar/regular")
|
||||
("desaparecer" "verb" "desaparece" "desapareció" "desaparecido" "desaparecer" "er/regular")
|
||||
("descubrir" "verb" "descubre" "descubrió" "descubierto" "descubrir" "ir/irregular")
|
||||
("despedir" "verb" "despide" "despidió" "despedido" "despedir" "ir/regular")
|
||||
("detener" "verb" "detiene" "detuvo" "detenido" "detener" "er/regular")
|
||||
("dimitir" "verb" "dimite" "dimitió" "dimitido" "dimitir" "ir/regular")
|
||||
("doler" "verb" "duele" "dolió" "dolido" "doler" "er/regular")
|
||||
("dormir" "verb" "duerme" "durmió" "dormido" "dormir" "ir/regular")
|
||||
("durar" "verb" "dura" "duró" "durado" "durar" "ar/regular")
|
||||
("empezar" "verb" "empieza" "empezó" "empezado" "empezar" "ar/regular")
|
||||
("encontrar" "verb" "encuentra" "encontró" "encontrado" "encontrar" "ar/regular")
|
||||
("entender" "verb" "entiende" "entendió" "entendido" "entender" "er/regular")
|
||||
("entrar" "verb" "entra" "entró" "entrado" "entrar" "ar/regular")
|
||||
("entregar" "verb" "entrega" "entregó" "entregado" "entregar" "ar/regular")
|
||||
("escapar" "verb" "escapa" "escapó" "escapado" "escapar" "ar/regular")
|
||||
("escribir" "verb" "escribe" "escribió" "escrito" "escribir" "ir/regular")
|
||||
("escuchar" "verb" "escucha" "escuchó" "escuchado" "escuchar" "ar/regular")
|
||||
("esperar" "verb" "espera" "esperó" "esperado" "esperar" "ar/regular")
|
||||
("estar" "verb" "está" "estuvo" "estado" "estar" "ar/irregular")
|
||||
("estudiar" "verb" "estudia" "estudió" "estudiado" "estudiar" "ar/regular")
|
||||
("existir" "verb" "existe" "existió" "existido" "existir" "ir/regular")
|
||||
("explicar" "verb" "explica" "explicó" "explicado" "explicar" "ar/regular")
|
||||
("firmar" "verb" "firma" "firmó" "firmado" "firmar" "ar/regular")
|
||||
("ganar" "verb" "gana" "ganó" "ganado" "ganar" "ar/regular")
|
||||
("gritar" "verb" "grita" "gritó" "gritado" "gritar" "ar/regular")
|
||||
("guardar" "verb" "guarda" "guardó" "guardado" "guardar" "ar/regular")
|
||||
("gustar" "verb" "gusta" "gustó" "gustado" "gustar" "ar/regular")
|
||||
("haber" "verb" "ha" "hubo" "habido" "haber" "er/irregular")
|
||||
("hablar" "verb" "habla" "habló" "hablado" "hablar" "ar/regular")
|
||||
("hacer" "verb" "hace" "hizo" "hecho" "hacer" "er/irregular")
|
||||
("ir" "verb" "va" "fue" "ido" "ir" "ir/irregular")
|
||||
("jugar" "verb" "juega" "jugó" "jugado" "jugar" "ar/regular")
|
||||
("lavar" "verb" "lava" "lavó" "lavado" "lavar" "ar/regular")
|
||||
("leer" "verb" "lee" "leyó" "leído" "leer" "er/regular")
|
||||
("llamar" "verb" "llama" "llamó" "llamado" "llamar" "ar/regular")
|
||||
("llegar" "verb" "llega" "llegó" "llegado" "llegar" "ar/regular")
|
||||
("llevar" "verb" "lleva" "llevó" "llevado" "llevar" "ar/regular")
|
||||
("llover" "verb" "llueve" "llovió" "llovido" "llover" "er/regular")
|
||||
("mirar" "verb" "mira" "miró" "mirado" "mirar" "ar/regular")
|
||||
("morir" "verb" "muere" "murió" "muerto" "morir" "ir/irregular")
|
||||
("mostrar" "verb" "muestra" "mostró" "mostrado" "mostrar" "ar/regular")
|
||||
("nacer" "verb" "nace" "nació" "nacido" "nacer" "er/regular")
|
||||
("ocultar" "verb" "oculta" "ocultó" "ocultado" "ocultar" "ar/regular")
|
||||
("ofrecer" "verb" "ofrece" "ofreció" "ofrecido" "ofrecer" "er/regular")
|
||||
("olvidar" "verb" "olvida" "olvidó" "olvidado" "olvidar" "ar/regular")
|
||||
("oír" "verb" "oye" "oyó" "oído" "oír" "ar/regular")
|
||||
("parecer" "verb" "parece" "pareció" "parecido" "parecer" "er/regular")
|
||||
("pasar" "verb" "pasa" "pasó" "pasado" "pasar" "ar/regular")
|
||||
("pedir" "verb" "pide" "pidió" "pedido" "pedir" "ir/regular")
|
||||
("pensar" "verb" "piensa" "pensó" "pensado" "pensar" "ar/regular")
|
||||
("perder" "verb" "pierde" "perdió" "perdido" "perder" "er/regular")
|
||||
("perdonar" "verb" "perdona" "perdonó" "perdonado" "perdonar" "ar/regular")
|
||||
("poder" "verb" "puede" "pudo" "podido" "poder" "er/irregular")
|
||||
("poner" "verb" "pone" "puso" "puesto" "poner" "er/irregular")
|
||||
("preocupar" "verb" "preocupa" "preocupó" "preocupado" "preocupar" "ar/regular")
|
||||
("producir" "verb" "produce" "produjo" "producido" "producir" "ir/regular")
|
||||
("prometer" "verb" "promete" "prometió" "prometido" "prometer" "er/regular")
|
||||
("proponer" "verb" "propone" "propuso" "propuesto" "proponer" "er/regular")
|
||||
("prosperar" "verb" "prospera" "prosperó" "prosperado" "prosperar" "ar/regular")
|
||||
("protestar" "verb" "protesta" "protestó" "protestado" "protestar" "ar/regular")
|
||||
("quedar" "verb" "queda" "quedó" "quedado" "quedar" "ar/regular")
|
||||
("querer" "verb" "quiere" "quiso" "querido" "querer" "er/irregular")
|
||||
("recordar" "verb" "recuerda" "recordó" "recordado" "recordar" "ar/regular")
|
||||
("recorrer" "verb" "recorre" "recorrió" "recorrido" "recorrer" "er/regular")
|
||||
("regresar" "verb" "regresa" "regresó" "regresado" "regresar" "ar/regular")
|
||||
("renacer" "verb" "renace" "renació" "renacido" "renacer" "er/regular")
|
||||
("saber" "verb" "sabe" "supo" "sabido" "saber" "er/irregular")
|
||||
("salir" "verb" "sale" "salió" "salido" "salir" "ir/irregular")
|
||||
("sentar" "verb" "sienta" "sentó" "sentado" "sentar" "ar/regular")
|
||||
("sentir" "verb" "siente" "sintió" "sentido" "sentir" "ir/regular")
|
||||
("separar" "verb" "separa" "separó" "separado" "separar" "ar/regular")
|
||||
("ser" "verb" "es" "fue" "sido" "ser" "er/irregular")
|
||||
("sonreír" "verb" "sonríe" "sonrió" "sonreído" "sonreír" "ar/regular")
|
||||
("soplar" "verb" "sopla" "sopló" "soplado" "soplar" "ar/regular")
|
||||
("sorprender" "verb" "sorprende" "sorprendió" "sorprendido" "sorprender" "er/regular")
|
||||
("soñar" "verb" "sueña" "soñó" "soñado" "soñar" "ar/regular")
|
||||
("subir" "verb" "sube" "subió" "subido" "subir" "ir/regular")
|
||||
("tener" "verb" "tiene" "tuvo" "tenido" "tener" "er/irregular")
|
||||
("terminar" "verb" "termina" "terminó" "terminado" "terminar" "ar/regular")
|
||||
("tocar" "verb" "toca" "tocó" "tocado" "tocar" "ar/regular")
|
||||
("tomar" "verb" "toma" "tomó" "tomado" "tomar" "ar/regular")
|
||||
("trabajar" "verb" "trabaja" "trabajó" "trabajado" "trabajar" "ar/regular")
|
||||
("vender" "verb" "vende" "vendió" "vendido" "vender" "er/regular")
|
||||
("venir" "verb" "viene" "vino" "venido" "venir" "ir/irregular")
|
||||
("ver" "verb" "ve" "vio" "visto" "ver" "er/irregular")
|
||||
("viajar" "verb" "viaja" "viajó" "viajado" "viajar" "ar/regular")
|
||||
("vivir" "verb" "vive" "vivió" "vivido" "vivir" "ir/regular")
|
||||
("volver" "verb" "vuelve" "volvió" "vuelto" "volver" "er/irregular")
|
||||
|
||||
;; -- nouns (form0=sg form1=pl form2=REAL gender m/f) ----------------------
|
||||
("Barcelona" "noun" "barcelona" "barcelonas" "f" "Barcelona" "gender:heuristic")
|
||||
("Madrid" "noun" "madrid" "madrides" "m" "Madrid" "gender:heuristic")
|
||||
("María" "noun" "maría" "marías" "f" "María" "gender:heuristic")
|
||||
("acuerdo" "noun" "acuerdo" "acuerdos" "m" "acuerdo" "gender:lexicon")
|
||||
("acusado" "noun" "acusado" "acusados" "m" "acusado" "gender:lexicon")
|
||||
("agua" "noun" "agua" "aguas" "f" "agua" "gender:lexicon")
|
||||
("amigo" "noun" "amigo" "amigos" "m" "amigo" "gender:lexicon")
|
||||
("amor" "noun" "amor" "amores" "m" "amor" "gender:lexicon")
|
||||
("anciano" "noun" "anciano" "ancianos" "m" "anciano" "gender:lexicon")
|
||||
("autor" "noun" "autor" "autores" "m" "autor" "gender:lexicon")
|
||||
("ayuda" "noun" "ayuda" "ayudas" "f" "ayuda" "gender:lexicon")
|
||||
("año" "noun" "año" "años" "m" "año" "gender:lexicon")
|
||||
("banco" "noun" "banco" "bancos" "m" "banco" "gender:lexicon")
|
||||
("barco" "noun" "barco" "barcos" "m" "barco" "gender:lexicon")
|
||||
("baño" "noun" "baño" "baños" "m" "baño" "gender:lexicon")
|
||||
("beneficio" "noun" "beneficio" "beneficios" "m" "beneficio" "gender:lexicon")
|
||||
("billete" "noun" "billete" "billetes" "m" "billete" "gender:lexicon")
|
||||
("café" "noun" "café" "cafés" "m" "café" "gender:lexicon")
|
||||
("calma" "noun" "calma" "calmas" "f" "calma" "gender:lexicon")
|
||||
("camino" "noun" "camino" "caminos" "m" "camino" "gender:lexicon")
|
||||
("canción" "noun" "canción" "canciones" "f" "canción" "gender:lexicon")
|
||||
("candidato" "noun" "candidato" "candidatos" "m" "candidato" "gender:lexicon")
|
||||
("casa" "noun" "casa" "casas" "f" "casa" "gender:lexicon")
|
||||
("cena" "noun" "cena" "cenas" "f" "cena" "gender:lexicon")
|
||||
("ciudad" "noun" "ciudad" "ciudades" "f" "ciudad" "gender:lexicon")
|
||||
("ciudadano" "noun" "ciudadano" "ciudadanos" "m" "ciudadano" "gender:lexicon")
|
||||
("color" "noun" "color" "colores" "m" "color" "gender:lexicon")
|
||||
("condición" "noun" "condición" "condiciones" "f" "condición" "gender:lexicon")
|
||||
("corazón" "noun" "corazón" "corazones" "m" "corazón" "gender:lexicon")
|
||||
("costa" "noun" "costa" "costas" "f" "costa" "gender:lexicon")
|
||||
("crisis" "noun" "crisis" "crisis" "f" "crisis" "gender:lexicon")
|
||||
("crédito" "noun" "crédito" "créditos" "m" "crédito" "gender:lexicon")
|
||||
("culpa" "noun" "culpa" "culpas" "f" "culpa" "gender:lexicon")
|
||||
("damnificado" "noun" "damnificado" "damnificados" "m" "damnificado" "gender:lexicon")
|
||||
("demás" "noun" "demás" "demás" "m" "demás" "gender:heuristic")
|
||||
("dinero" "noun" "dinero" "dineros" "m" "dinero" "gender:lexicon")
|
||||
("día" "noun" "día" "días" "m" "día" "gender:lexicon")
|
||||
("economía" "noun" "economía" "economías" "f" "economía" "gender:lexicon")
|
||||
("elección" "noun" "elección" "elecciones" "f" "elección" "gender:lexicon")
|
||||
("empleo" "noun" "empleo" "empleos" "m" "empleo" "gender:lexicon")
|
||||
("empresa" "noun" "empresa" "empresas" "f" "empresa" "gender:lexicon")
|
||||
("equipo" "noun" "equipo" "equipos" "m" "equipo" "gender:lexicon")
|
||||
("estación" "noun" "estación" "estaciones" "f" "estación" "gender:lexicon")
|
||||
("estudiante" "noun" "estudiante" "estudiantes" "m" "estudiante" "gender:lexicon")
|
||||
("experto" "noun" "experto" "expertos" "m" "experto" "gender:lexicon")
|
||||
("fiesta" "noun" "fiesta" "fiestas" "f" "fiesta" "gender:lexicon")
|
||||
("flor" "noun" "flor" "flores" "f" "flor" "gender:lexicon")
|
||||
("foto" "noun" "foto" "fotos" "f" "foto" "gender:lexicon")
|
||||
("frío" "noun" "frío" "fríos" "m" "frío" "gender:lexicon")
|
||||
("fuego" "noun" "fuego" "fuegos" "m" "fuego" "gender:lexicon")
|
||||
("fuerza" "noun" "fuerza" "fuerzas" "f" "fuerza" "gender:lexicon")
|
||||
("gato" "noun" "gato" "gatos" "m" "gato" "gender:lexicon")
|
||||
("gobierno" "noun" "gobierno" "gobiernos" "m" "gobierno" "gender:lexicon")
|
||||
("gusto" "noun" "gusto" "gustos" "m" "gusto" "gender:lexicon")
|
||||
("hermano" "noun" "hermano" "hermanos" "m" "hermano" "gender:lexicon")
|
||||
("hombre" "noun" "hombre" "hombres" "m" "hombre" "gender:lexicon")
|
||||
("jardín" "noun" "jardín" "jardines" "m" "jardín" "gender:lexicon")
|
||||
("juez" "noun" "juez" "jueces" "m" "juez" "gender:lexicon")
|
||||
("juicio" "noun" "juicio" "juicios" "m" "juicio" "gender:lexicon")
|
||||
("lentitud" "noun" "lentitud" "lentitudes" "f" "lentitud" "gender:lexicon")
|
||||
("ley" "noun" "ley" "leyes" "f" "ley" "gender:lexicon")
|
||||
("libertad" "noun" "libertad" "libertades" "f" "libertad" "gender:lexicon")
|
||||
("libro" "noun" "libro" "libros" "m" "libro" "gender:lexicon")
|
||||
("llave" "noun" "llave" "llaves" "f" "llave" "gender:lexicon")
|
||||
("lluvia" "noun" "lluvia" "lluvias" "f" "lluvia" "gender:lexicon")
|
||||
("luna" "noun" "luna" "lunas" "f" "luna" "gender:lexicon")
|
||||
("luz" "noun" "luz" "luces" "f" "luz" "gender:lexicon")
|
||||
("madre" "noun" "madre" "madres" "f" "madre" "gender:lexicon")
|
||||
("mano" "noun" "mano" "manos" "f" "mano" "gender:lexicon")
|
||||
("mapa" "noun" "mapa" "mapas" "m" "mapa" "gender:lexicon")
|
||||
("mar" "noun" "mar" "mares" "m" "mar" "gender:lexicon")
|
||||
("marea" "noun" "marea" "mareas" "f" "marea" "gender:lexicon")
|
||||
("memoria" "noun" "memoria" "memorias" "f" "memoria" "gender:lexicon")
|
||||
("mesa" "noun" "mesa" "mesas" "f" "mesa" "gender:lexicon")
|
||||
("ministro" "noun" "ministro" "ministros" "m" "ministro" "gender:lexicon")
|
||||
("montaña" "noun" "montaña" "montañas" "f" "montaña" "gender:lexicon")
|
||||
("moto" "noun" "moto" "motos" "f" "moto" "gender:lexicon")
|
||||
("mujer" "noun" "mujer" "mujeres" "f" "mujer" "gender:lexicon")
|
||||
("mundo" "noun" "mundo" "mundos" "m" "mundo" "gender:lexicon")
|
||||
("música" "noun" "música" "músicas" "f" "música" "gender:lexicon")
|
||||
("nación" "noun" "nación" "naciones" "f" "nación" "gender:lexicon")
|
||||
("niebla" "noun" "niebla" "nieblas" "f" "niebla" "gender:lexicon")
|
||||
("nieve" "noun" "nieve" "nieves" "f" "nieve" "gender:lexicon")
|
||||
("niña" "noun" "niña" "niñas" "f" "niña" "gender:lexicon")
|
||||
("niño" "noun" "niño" "niños" "m" "niño" "gender:lexicon")
|
||||
("noche" "noun" "noche" "noches" "f" "noche" "gender:lexicon")
|
||||
("nombre" "noun" "nombre" "nombres" "m" "nombre" "gender:lexicon")
|
||||
("ojo" "noun" "ojo" "ojos" "m" "ojo" "gender:heuristic")
|
||||
("orilla" "noun" "orilla" "orillas" "f" "orilla" "gender:lexicon")
|
||||
("paisaje" "noun" "paisaje" "paisajes" "m" "paisaje" "gender:lexicon")
|
||||
("parlamento" "noun" "parlamento" "parlamentos" "m" "parlamento" "gender:lexicon")
|
||||
("parte" "noun" "parte" "partes" "f" "parte" "gender:lexicon")
|
||||
("pasillo" "noun" "pasillo" "pasillos" "m" "pasillo" "gender:lexicon")
|
||||
("país" "noun" "país" "países" "m" "país" "gender:lexicon")
|
||||
("película" "noun" "película" "películas" "f" "película" "gender:lexicon")
|
||||
("perro" "noun" "perro" "perros" "m" "perro" "gender:lexicon")
|
||||
("persona" "noun" "persona" "personas" "f" "persona" "gender:lexicon")
|
||||
("petróleo" "noun" "petróleo" "petróleos" "m" "petróleo" "gender:lexicon")
|
||||
("pez" "noun" "pez" "peces" "f" "pez" "gender:lexicon")
|
||||
("plan" "noun" "plan" "planes" "m" "plan" "gender:lexicon")
|
||||
("policía" "noun" "policía" "policías" "f" "policía" "gender:lexicon")
|
||||
("portavoz" "noun" "portavoz" "portavoces" "m" "portavoz" "gender:lexicon")
|
||||
("portero" "noun" "portero" "porteros" "m" "portero" "gender:lexicon")
|
||||
("precio" "noun" "precio" "precios" "m" "precio" "gender:lexicon")
|
||||
("presidente" "noun" "presidente" "presidentes" "m" "presidente" "gender:lexicon")
|
||||
("problema" "noun" "problema" "problemas" "m" "problema" "gender:lexicon")
|
||||
("programa" "noun" "programa" "programas" "m" "programa" "gender:lexicon")
|
||||
("proyecto" "noun" "proyecto" "proyectos" "m" "proyecto" "gender:lexicon")
|
||||
("puerta" "noun" "puerta" "puertas" "f" "puerta" "gender:lexicon")
|
||||
("pájaro" "noun" "pájaro" "pájaros" "m" "pájaro" "gender:lexicon")
|
||||
("razón" "noun" "razón" "razones" "f" "razón" "gender:lexicon")
|
||||
("raíz" "noun" "raíz" "raíces" "f" "raíz" "gender:lexicon")
|
||||
("recuerdo" "noun" "recuerdo" "recuerdos" "m" "recuerdo" "gender:lexicon")
|
||||
("reforma" "noun" "reforma" "reformas" "f" "reforma" "gender:lexicon")
|
||||
("región" "noun" "región" "regiones" "f" "región" "gender:lexicon")
|
||||
("río" "noun" "río" "ríos" "m" "río" "gender:lexicon")
|
||||
("semilla" "noun" "semilla" "semillas" "f" "semilla" "gender:lexicon")
|
||||
("sendero" "noun" "sendero" "senderos" "m" "sendero" "gender:lexicon")
|
||||
("señor" "noun" "señor" "señores" "m" "señor" "gender:lexicon")
|
||||
("silencio" "noun" "silencio" "silencios" "m" "silencio" "gender:lexicon")
|
||||
("silla" "noun" "silla" "sillas" "f" "silla" "gender:lexicon")
|
||||
("sol" "noun" "sol" "soles" "m" "sol" "gender:lexicon")
|
||||
("tarea" "noun" "tarea" "tareas" "f" "tarea" "gender:lexicon")
|
||||
("tema" "noun" "tema" "temas" "m" "tema" "gender:lexicon")
|
||||
("ti" "noun" "ti" "tis" "m" "ti" "gender:heuristic")
|
||||
("tiempo" "noun" "tiempo" "tiempos" "m" "tiempo" "gender:lexicon")
|
||||
("tipo" "noun" "tipo" "tipos" "m" "tipo" "gender:lexicon")
|
||||
("todo" "noun" "todo" "todos" "m" "todo" "gender:heuristic")
|
||||
("tormenta" "noun" "tormenta" "tormentas" "f" "tormenta" "gender:lexicon")
|
||||
("trabajador" "noun" "trabajador" "trabajadores" "m" "trabajador" "gender:lexicon")
|
||||
("tristeza" "noun" "tristeza" "tristezas" "f" "tristeza" "gender:lexicon")
|
||||
("ventana" "noun" "ventana" "ventanas" "f" "ventana" "gender:lexicon")
|
||||
("verdad" "noun" "verdad" "verdades" "f" "verdad" "gender:lexicon")
|
||||
("viaje" "noun" "viaje" "viajes" "m" "viaje" "gender:lexicon")
|
||||
("victoria" "noun" "victoria" "victorias" "f" "victoria" "gender:lexicon")
|
||||
("vida" "noun" "vida" "vidas" "f" "vida" "gender:lexicon")
|
||||
("viento" "noun" "viento" "vientos" "m" "viento" "gender:lexicon")
|
||||
("voz" "noun" "voz" "voces" "f" "voz" "gender:lexicon")
|
||||
("vuelta" "noun" "vuelta" "vueltas" "f" "vuelta" "gender:lexicon")
|
||||
("árbol" "noun" "árbol" "árboles" "m" "árbol" "gender:lexicon")
|
||||
|
||||
;; -- adjectives (form0=masc-sg form1=fem-sg form2=masc-pl) ----------------
|
||||
("alto" "adj" "alto" "alta" "altos" "alto" "lexicon")
|
||||
("ambos" "adj" "ambos" "ambos" "ambos" "ambos" "rule")
|
||||
("antiguo" "adj" "antiguo" "antigua" "antiguos" "antiguo" "lexicon")
|
||||
("azul" "adj" "azul" "azul" "azules" "azul" "lexicon")
|
||||
("bajo" "adj" "bajo" "baja" "bajos" "bajo" "lexicon")
|
||||
("blanco" "adj" "blanco" "blanca" "blancos" "blanco" "lexicon")
|
||||
("bondadoso" "adj" "bondadoso" "bondadosa" "bondadosos" "bondadoso" "lexicon")
|
||||
("bonito" "adj" "bonito" "bonita" "bonitos" "bonito" "lexicon")
|
||||
("bueno" "adj" "bueno" "buena" "buenos" "bueno" "lexicon")
|
||||
("cansado" "adj" "cansado" "cansada" "cansados" "cansado" "lexicon")
|
||||
("corto" "adj" "corto" "corta" "cortos" "corto" "lexicon")
|
||||
("difícil" "adj" "difícil" "difícil" "difíciles" "difícil" "lexicon")
|
||||
("dos" "adj" "dos" "dos" "dos" "dos" "rule")
|
||||
("económico" "adj" "económico" "económica" "económicos" "económico" "lexicon")
|
||||
("español" "adj" "español" "española" "españoles" "español" "lexicon")
|
||||
("estrecho" "adj" "estrecho" "estrecha" "estrechos" "estrecho" "lexicon")
|
||||
("fascinante" "adj" "fascinante" "fascinante" "fascinantes" "fascinante" "rule")
|
||||
("feliz" "adj" "feliz" "feliz" "felices" "feliz" "lexicon")
|
||||
("francés" "adj" "francés" "francesa" "franceses" "francés" "lexicon")
|
||||
("frío" "adj" "frío" "fría" "fríos" "frío" "lexicon")
|
||||
("fácil" "adj" "fácil" "fácil" "fáciles" "fácil" "lexicon")
|
||||
("grande" "adj" "grande" "grande" "grandes" "grande" "lexicon")
|
||||
("hermoso" "adj" "hermoso" "hermosa" "hermosos" "hermoso" "lexicon")
|
||||
("importante" "adj" "importante" "importante" "importantes" "importante" "lexicon")
|
||||
("inglés" "adj" "inglés" "inglesa" "ingleses" "inglés" "lexicon")
|
||||
("largo" "adj" "largo" "larga" "largos" "largo" "lexicon")
|
||||
("lento" "adj" "lento" "lenta" "lentos" "lento" "lexicon")
|
||||
("malo" "adj" "malo" "mala" "malos" "malo" "lexicon")
|
||||
("mucho" "adj" "mucho" "mucha" "muchos" "mucho" "lexicon")
|
||||
("necesario" "adj" "necesario" "necesaria" "necesarios" "necesario" "lexicon")
|
||||
("negro" "adj" "negro" "negra" "negros" "negro" "lexicon")
|
||||
("nuevo" "adj" "nuevo" "nueva" "nuevos" "nuevo" "lexicon")
|
||||
("olvidado" "adj" "olvidado" "olvidada" "olvidados" "olvidado" "lexicon")
|
||||
("oscuro" "adj" "oscuro" "oscura" "oscuros" "oscuro" "lexicon")
|
||||
("pequeño" "adj" "pequeño" "pequeña" "pequeños" "pequeño" "lexicon")
|
||||
("político" "adj" "político" "política" "políticos" "político" "lexicon")
|
||||
("posible" "adj" "posible" "posible" "posibles" "posible" "lexicon")
|
||||
("rojo" "adj" "rojo" "roja" "rojos" "rojo" "lexicon")
|
||||
("rápido" "adj" "rápido" "rápida" "rápidos" "rápido" "lexicon")
|
||||
("sabio" "adj" "sabio" "sabia" "sabios" "sabio" "lexicon")
|
||||
("silencioso" "adj" "silencioso" "silenciosa" "silenciosos" "silencioso" "lexicon")
|
||||
("social" "adj" "social" "social" "sociales" "social" "lexicon")
|
||||
("trabajador" "adj" "trabajador" "trabajadora" "trabajadores" "trabajador" "lexicon")
|
||||
("triste" "adj" "triste" "triste" "tristes" "triste" "rule")
|
||||
("verde" "adj" "verde" "verde" "verdes" "verde" "lexicon")
|
||||
("viejo" "adj" "viejo" "vieja" "viejos" "viejo" "lexicon")
|
||||
|
||||
)
|
||||
@@ -0,0 +1,262 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""gen_elp_es.py — emit the ELP (.el) port artifacts for Spanish.
|
||||
|
||||
Mirrors gen_elp_en.py. Produces:
|
||||
vocabulary-es.el real generated vocabulary in the established schema
|
||||
[lemma, pos, form0, form1, form2, en_translation, semantic_hint]
|
||||
(same schema as vocabulary-got.el / vocabulary-en.el;
|
||||
UniMorph spa lineage).
|
||||
Verbs : form0=present-ind-3sg form1=preterite-3sg form2=past-participle
|
||||
Nouns : form0=singular form1=plural form2=REAL gender (m/f, lexicon)
|
||||
Adjs : form0=masc-sg form1=fem-sg form2=masc-pl
|
||||
lang_profile_es.el the Spanish profile with the flags the realizer keys on.
|
||||
|
||||
Every form is generated by morphology_es_full (real UniMorph lexicon, not
|
||||
hand-typed), so the .el vocabulary is honest and reproduces the forms the
|
||||
realizer used. CRITICAL (coordinator quality bar): noun gender in form2 is the
|
||||
REAL per-lemma lexicon gender (N;FEM/MASC), NOT an ending heuristic — this is
|
||||
what kills the 'el mano / la día' masculine-default error class.
|
||||
"""
|
||||
import morphology_es_full as M
|
||||
from test_set_es import TESTS
|
||||
from held_out_es import HELD
|
||||
|
||||
# ── core closed class + common content lemmas so the vocab is usable beyond the
|
||||
# validated sentences ──────────────────────────────────────────────────────
|
||||
_CORE_VERBS = ["ser", "estar", "haber", "tener", "hacer", "ir", "ver", "dar",
|
||||
"saber", "poder", "querer", "venir", "decir", "poner", "salir",
|
||||
"hablar", "comer", "vivir", "trabajar", "estudiar", "llegar",
|
||||
"pasar", "deber", "parecer", "quedar", "creer", "dejar", "llevar",
|
||||
"encontrar", "llamar", "pensar", "volver", "conocer", "sentir",
|
||||
"contar", "empezar", "buscar", "esperar", "existir", "entrar",
|
||||
"escribir", "perder", "producir", "recordar", "morir", "nacer",
|
||||
"abrir", "escapar", "soñar", "amar", "caer", "leer", "oír"]
|
||||
_CORE_NOUNS = ["tiempo", "persona", "año", "día", "mano", "mundo", "vida",
|
||||
"hombre", "mujer", "parte", "casa", "país", "problema", "programa",
|
||||
"tema", "mapa", "agua", "foto", "moto", "ciudad", "libertad",
|
||||
"canción", "nación", "flor", "color", "amor", "señor", "viaje",
|
||||
"paisaje", "gato", "perro", "libro", "mesa", "silla", "noche",
|
||||
"luz", "voz", "pez", "raíz", "crisis", "sol", "luna", "mar",
|
||||
"corazón", "flor", "árbol", "camino", "puerta", "ventana"]
|
||||
_CORE_ADJS = ["bueno", "malo", "nuevo", "viejo", "grande", "pequeño", "alto",
|
||||
"bajo", "largo", "corto", "feliz", "triste", "fácil", "difícil",
|
||||
"rápido", "lento", "hermoso", "económico", "político", "social",
|
||||
"azul", "rojo", "verde", "blanco", "negro", "español", "francés",
|
||||
"inglés", "importante", "posible", "necesario", "trabajador"]
|
||||
|
||||
# closed-class function words. Contractions (del/al) and the government notes
|
||||
# are the coordinator's quality bar (mandatory contraction; verb-prep govt).
|
||||
_FUNCTION = [
|
||||
# articles (gender/number agreement is in morphology; these are citation)
|
||||
("el", "det", "el", "los", "m", "the", "definite article m.sg"),
|
||||
("la", "det", "la", "las", "f", "the", "definite article f.sg"),
|
||||
("un", "det", "un", "unos","m", "a", "indefinite article m.sg"),
|
||||
("una", "det", "una", "unas","f", "a", "indefinite article f.sg"),
|
||||
# MANDATORY CONTRACTIONS (prep + el) — del / al
|
||||
("del", "contraction", "del", "", "", "of the", "de + el (mandatory contraction)"),
|
||||
("al", "contraction", "al", "", "", "to the", "a + el (mandatory contraction)"),
|
||||
# demonstratives
|
||||
("este", "dem", "este", "estos", "m", "this", "proximal dem m"),
|
||||
("esta", "dem", "esta", "estas", "f", "this", "proximal dem f"),
|
||||
# negation (SACRED — polarity never dropped)
|
||||
("no", "neg", "no", "", "", "not/no", "sentential negator (preverbal)"),
|
||||
("ninguno", "det", "ningún", "ninguna", "", "none", "negative determiner (apocope ningún m.sg)"),
|
||||
# conjunctions
|
||||
("y", "conj", "y", "e", "", "and", "coordinator (e before i-/hi-)"),
|
||||
("o", "conj", "o", "u", "", "or", "coordinator (u before o-/ho-)"),
|
||||
("pero", "conj", "pero", "", "", "but", "adversative coordinator"),
|
||||
("que", "conj", "que", "", "", "that", "complementizer / relative"),
|
||||
("si", "conj", "si", "", "", "if", "conditional subordinator"),
|
||||
("porque", "conj", "porque", "", "", "because", "causal subordinator"),
|
||||
("cuando", "conj", "cuando", "", "", "when", "temporal subordinator"),
|
||||
# prepositions (government: verbs select these; contraction with el applies to a/de)
|
||||
("a", "prep", "a", "", "", "to", "dir-obj (personal a) / dative / allative; a+el=al"),
|
||||
("de", "prep", "de", "", "", "of/from", "genitive/ablative government; de+el=del"),
|
||||
("en", "prep", "en", "", "", "in/on", "locative"),
|
||||
("con", "prep", "con", "", "", "with", "comitative"),
|
||||
("por", "prep", "por", "", "", "by/for", "passive agent / cause"),
|
||||
("para", "prep", "para", "", "", "for", "purpose/benefactive"),
|
||||
("contra", "prep", "contra", "", "", "against", "adversative government (protestar contra)"),
|
||||
("sin", "prep", "sin", "", "", "without", "privative"),
|
||||
# subject pronouns
|
||||
("yo", "pron", "yo", "me", "mi", "I", "1sg subj/obj/poss"),
|
||||
("tú", "pron", "tú", "te", "tu", "you", "2sg informal"),
|
||||
("usted", "pron", "usted", "lo", "su", "you", "2sg formal (3sg agreement)"),
|
||||
("él", "pron", "él", "lo", "su", "he", "3sg m subj/DO-clitic/poss"),
|
||||
("ella", "pron", "ella", "la", "su", "she", "3sg f subj/DO-clitic/poss"),
|
||||
("nosotros", "pron", "nosotros", "nos", "nuestro", "we", "1pl"),
|
||||
("vosotros", "pron", "vosotros", "os", "vuestro", "you", "2pl informal"),
|
||||
("ellos", "pron", "ellos", "los", "su", "they", "3pl m"),
|
||||
("ellas", "pron", "ellas", "las", "su", "they", "3pl f"),
|
||||
# indirect-object clitics
|
||||
("le", "clitic", "le", "les", "", "to-him/her", "dative clitic 3sg/3pl (->se before lo/la)"),
|
||||
("se", "clitic", "se", "se", "", "himself/-self", "reflexive / spurious-se (le+lo->se lo)"),
|
||||
]
|
||||
|
||||
|
||||
def _walk_collect(spec, verbs, nouns, adjs):
|
||||
"""Recursively collect verb/noun/adj lemmas from a semantic spec."""
|
||||
if isinstance(spec, dict):
|
||||
if spec.get("pred"):
|
||||
verbs.add(spec["pred"])
|
||||
if spec.get("noun"):
|
||||
nouns.add(spec["noun"])
|
||||
if spec.get("adj"):
|
||||
adjs.add(spec["adj"])
|
||||
if spec.get("superlative"):
|
||||
adjs.add(spec["superlative"])
|
||||
if spec.get("from_adj"):
|
||||
adjs.add(spec["from_adj"])
|
||||
# adjs: [{"lemma":..,"pos":..}] | ["lemma", ..]
|
||||
for a in spec.get("adjs", []) or []:
|
||||
adjs.add(a["lemma"] if isinstance(a, dict) else a)
|
||||
for a in spec.get("adj_coord", []) or []:
|
||||
adjs.add(a["lemma"] if isinstance(a, dict) else a)
|
||||
if isinstance(spec.get("pcomp"), dict):
|
||||
pc = spec["pcomp"]
|
||||
if pc.get("adj"):
|
||||
adjs.add(pc["adj"])
|
||||
for a in pc.get("adj_coord", []) or []:
|
||||
adjs.add(a["lemma"] if isinstance(a, dict) else a)
|
||||
for v in spec.values():
|
||||
_walk_collect(v, verbs, nouns, adjs)
|
||||
elif isinstance(spec, list):
|
||||
for it in spec:
|
||||
_walk_collect(it, verbs, nouns, adjs)
|
||||
|
||||
|
||||
def _collect_from_specs():
|
||||
verbs, nouns, adjs = set(), set(), set()
|
||||
for t in TESTS + HELD:
|
||||
_walk_collect(t["spec"], verbs, nouns, adjs)
|
||||
return verbs, nouns, adjs
|
||||
|
||||
|
||||
def _esc(s):
|
||||
return str(s).replace('"', '\\"')
|
||||
|
||||
|
||||
def _row(fields):
|
||||
return " (" + " ".join(f'"{_esc(f)}"' for f in fields) + ")"
|
||||
|
||||
|
||||
def emit_vocabulary(path):
|
||||
v_specs, n_specs, a_specs = _collect_from_specs()
|
||||
verbs = sorted(set(_CORE_VERBS) | v_specs)
|
||||
nouns = sorted(set(_CORE_NOUNS) | n_specs)
|
||||
adjs = sorted(set(_CORE_ADJS) | a_specs)
|
||||
|
||||
lines = [
|
||||
";;; vocabulary-es.el — Spanish vocabulary for ELP surface realization.",
|
||||
";;; Schema: (lemma pos form0 form1 form2 en_translation semantic_hint)",
|
||||
";;; Source: UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0),",
|
||||
";;; generated by gen_elp_es.py via morphology_es_full (real forms).",
|
||||
";;; Verbs: form0=present-ind-3sg form1=preterite-3sg form2=past-participle",
|
||||
";;; Nouns: form0=singular form1=plural form2=REAL gender (m/f, from lexicon —",
|
||||
";;; NOT an ending heuristic; this is what kills 'el mano'/'la día' errors)",
|
||||
";;; Adjs : form0=masc-sg form1=fem-sg form2=masc-pl",
|
||||
"",
|
||||
"(vocabulary-es",
|
||||
"",
|
||||
" ;; -- function / closed class (incl. mandatory contractions del/al) --------",
|
||||
]
|
||||
for f in _FUNCTION:
|
||||
lines.append(_row(f))
|
||||
|
||||
lines.append("")
|
||||
lines.append(" ;; -- verbs (form0=pres-3sg form1=pret-3sg form2=past-participle) -----------")
|
||||
for lem in verbs:
|
||||
f0, c0 = M.conjugate(lem, "ind", "present", "third", "singular")
|
||||
f1, c1 = M.conjugate(lem, "ind", "preterite", "third", "singular")
|
||||
pp, cp = M.participle(lem)
|
||||
vclass = lem[-2:] if lem[-2:] in ("ar", "er", "ir") else "ar"
|
||||
irr = "irregular" if (c0 == "lexicon" and pp in M._IRREG_PART.values()) or \
|
||||
lem in ("ser", "estar", "ir", "haber", "tener", "hacer", "ver",
|
||||
"dar", "saber", "poder", "querer", "venir", "decir",
|
||||
"poner", "salir") else "regular"
|
||||
lines.append(_row([lem, "verb", f0, f1, pp, lem, vclass + "/" + irr]))
|
||||
|
||||
lines.append("")
|
||||
lines.append(" ;; -- nouns (form0=sg form1=pl form2=REAL gender m/f) ----------------------")
|
||||
for lem in nouns:
|
||||
sg, _ = M.inflect_noun(lem, "singular")
|
||||
pl, _ = M.inflect_noun(lem, "plural")
|
||||
g = M.noun_gender(lem)
|
||||
# honesty flag: did gender come from the lexicon, or a heuristic fallback?
|
||||
src = "lexicon" if (lem in M._NOUNS and M._NOUNS[lem].get("g")) else "heuristic"
|
||||
lines.append(_row([lem, "noun", sg, pl, g, lem, "gender:" + src]))
|
||||
|
||||
lines.append("")
|
||||
lines.append(" ;; -- adjectives (form0=masc-sg form1=fem-sg form2=masc-pl) ----------------")
|
||||
for lem in adjs:
|
||||
m_sg, _ = M.inflect_adj(lem, "m", "singular")
|
||||
f_sg, _ = M.inflect_adj(lem, "f", "singular")
|
||||
m_pl, _ = M.inflect_adj(lem, "m", "plural")
|
||||
src = "lexicon" if lem in M._ADJS else "rule"
|
||||
lines.append(_row([lem, "adj", m_sg, f_sg, m_pl, lem, src]))
|
||||
|
||||
lines.append("")
|
||||
lines.append(")")
|
||||
with open(path, "w", encoding="utf-8") as fh:
|
||||
fh.write("\n".join(lines) + "\n")
|
||||
return len(_FUNCTION) + len(verbs) + len(nouns) + len(adjs), len(verbs), len(nouns), len(adjs)
|
||||
|
||||
|
||||
LANG_PROFILE = ''';;; lang_profile_es.el — Spanish language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_en / _pt.
|
||||
|
||||
(lang_profile_es
|
||||
(language "Spanish")
|
||||
(iso639 "es")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon); REAL per-noun gender from UniMorph — NOT a heuristic
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; questions by intonation/punctuation, not inversion
|
||||
(question-strategy intonation)
|
||||
(article-selection "el/la/los/las un/una/unos/unas")
|
||||
(stressed-a-rule yes) ; fem sg noun in stressed a-/ha- takes el/un (el agua)
|
||||
(adjective-position postnominal) ; default post; a few prenominal + apocope
|
||||
(adjective-agreement "gender+number")
|
||||
(question-punct inverted) ; opening ¿ ¡ required
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (coordinator quality bar) --------------------
|
||||
(contractions ((de el "del") (a el "al")))
|
||||
(contraction-mandatory yes) ; 'de el'/'a el' MUST surface as del/al
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "haber") ; haber + past participle (invariant -o)
|
||||
(progressive-aux "estar") ; estar + gerund
|
||||
(passive-aux "ser") ; ser + participle (agrees) + por-agent
|
||||
(copula-split "ser/estar") ; permanent vs stage-level
|
||||
(future "infinitive + é/ás/á/emos/éis/án")
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; me te lo la le nos os los las; proclisis/enclisis
|
||||
(clitic-order "se II I III (le+lo -> se lo)")
|
||||
(enclisis "imperative/infinitive/gerund + accent repair (dá+me+lo->dámelo)")
|
||||
(verb-prep-government yes) ; verbs select prep (protestar+contra, escapar+de)
|
||||
|
||||
;; -- SACRED safety bar (shared with en/pt) -------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
'''
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
voc_path = sys.argv[1] if len(sys.argv) > 1 else "vocabulary-es.el"
|
||||
lp_path = sys.argv[2] if len(sys.argv) > 2 else "lang_profile_es.el"
|
||||
total, nv, nn, na = emit_vocabulary(voc_path)
|
||||
with open(lp_path, "w", encoding="utf-8") as fh:
|
||||
fh.write(LANG_PROFILE)
|
||||
print(f"wrote {voc_path} ({total} entries: {len(_FUNCTION)} fn, {nv} verbs, {nn} nouns, {na} adjs)")
|
||||
print(f"wrote {lp_path}")
|
||||
print("lexicon:", M.lexicon_stats())
|
||||
@@ -0,0 +1,100 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""gen_parity_es.py — emit an El parity program that checks the .el Spanish
|
||||
morphology against the validated Python (UniMorph-backed) realizer.
|
||||
|
||||
Python is the ORACLE. For every lemma in the held-out inventory we embed the
|
||||
Python-produced form, call the corresponding .el function, and the El program
|
||||
prints PASS/FAIL per category. Aggregation is done in bash (grep -c), so no
|
||||
El-side mutable counters are needed.
|
||||
|
||||
Categories:
|
||||
Vpres verb present-ind-3sg es_conjugate(v,present,third,singular)
|
||||
Vpret verb preterite-3sg es_conjugate(v,past,third,singular)
|
||||
Vpart past participle es_participle(v)
|
||||
Npl noun plural es_pluralize(n)
|
||||
Gheur noun gender HEURISTIC es_gender(n) [exposes the bug]
|
||||
Aheur def article via heuristic es_agree_article(n,true,sg) [inherits the bug]
|
||||
Avocab def article via REAL g es_article_for_gender(realg,...) [the fix]
|
||||
Ampl adj masc-plural es_inflect_adj(a,m,plural)
|
||||
Afsg adj fem-singular es_inflect_adj(a,f,singular)
|
||||
Ctr contraction del/al es_contract(prep, np)
|
||||
"""
|
||||
import morphology_es_full as M
|
||||
import realizer_es as R
|
||||
from gen_elp_es import _collect_from_specs, _CORE_VERBS, _CORE_NOUNS, _CORE_ADJS
|
||||
|
||||
|
||||
def _esc(s):
|
||||
return str(s).replace('\\', '\\\\').replace('"', '\\"')
|
||||
|
||||
|
||||
def _check(cat, lemma, el_call, expected):
|
||||
return (f' es_check("{cat}", "{_esc(lemma)}", {el_call}, "{_esc(expected)}")')
|
||||
|
||||
|
||||
def main(out_path):
|
||||
v_specs, n_specs, a_specs = _collect_from_specs()
|
||||
verbs = sorted(set(_CORE_VERBS) | v_specs)
|
||||
nouns = sorted(set(_CORE_NOUNS) | n_specs)
|
||||
adjs = sorted(set(_CORE_ADJS) | a_specs)
|
||||
|
||||
lines = []
|
||||
lines.append("// gen'd parity checks — Python oracle embedded, El functions called.")
|
||||
lines.append("fn es_check(cat: String, lemma: String, got: String, exp: String) {")
|
||||
lines.append(' if str_eq(got, exp) {')
|
||||
lines.append(' println("PASS " + cat)')
|
||||
lines.append(' } else {')
|
||||
lines.append(' println("FAIL " + cat + " " + lemma + " got=" + got + " exp=" + exp)')
|
||||
lines.append(' }')
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
lines.append("fn es_parity() {")
|
||||
|
||||
# verbs
|
||||
for v in verbs:
|
||||
p0 = M.conjugate(v, "ind", "present", "third", "singular")[0]
|
||||
p1 = M.conjugate(v, "ind", "preterite", "third", "singular")[0]
|
||||
pp = M.participle(v)[0]
|
||||
lines.append(_check("Vpres", v, f'es_conjugate("{_esc(v)}", "present", "third", "singular")', p0))
|
||||
lines.append(_check("Vpret", v, f'es_conjugate("{_esc(v)}", "past", "third", "singular")', p1))
|
||||
lines.append(_check("Vpart", v, f'es_participle("{_esc(v)}")', pp))
|
||||
|
||||
# nouns
|
||||
for n in nouns:
|
||||
pl = M.inflect_noun(n, "plural")[0]
|
||||
g = M.noun_gender(n) # REAL lexicon gender
|
||||
art = R._article(g, "singular", "def", n) # oracle article from real gender
|
||||
lines.append(_check("Npl", n, f'es_pluralize("{_esc(n)}")', pl))
|
||||
lines.append(_check("Gheur", n, f'es_gender("{_esc(n)}")', g))
|
||||
lines.append(_check("Aheur", n, f'es_agree_article("{_esc(n)}", "true", "singular")', art))
|
||||
lines.append(_check("Avocab", n, f'es_article_for_gender("{_esc(g)}", "{_esc(n)}", "true", "singular")', art))
|
||||
|
||||
# adjectives
|
||||
for a in adjs:
|
||||
mpl = M.inflect_adj(a, "m", "plural")[0]
|
||||
fsg = M.inflect_adj(a, "f", "singular")[0]
|
||||
lines.append(_check("Ampl", a, f'es_inflect_adj("{_esc(a)}", "m", "plural")', mpl))
|
||||
lines.append(_check("Afsg", a, f'es_inflect_adj("{_esc(a)}", "f", "singular")', fsg))
|
||||
|
||||
# contractions (mandatory)
|
||||
ctr_cases = [("de", "el día"), ("a", "el hombre"), ("de", "el mundo"),
|
||||
("a", "el país"), ("en", "el mar"), ("de", "el año"),
|
||||
("a", "la casa"), ("de", "la ciudad")]
|
||||
for prep, np in ctr_cases:
|
||||
exp = R._contract(prep, np)
|
||||
lines.append(_check("Ctr", prep + "+" + np, f'es_contract("{_esc(prep)}", "{_esc(np)}")', exp))
|
||||
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
lines.append("fn main() {")
|
||||
lines.append(" es_parity()")
|
||||
lines.append("}")
|
||||
|
||||
with open(out_path, "w", encoding="utf-8") as fh:
|
||||
fh.write("\n".join(lines) + "\n")
|
||||
print(f"wrote {out_path} ({len(verbs)} verbs, {len(nouns)} nouns, {len(adjs)} adjs)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
main(sys.argv[1] if len(sys.argv) > 1 else "parity_es_checks.el")
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -3155,6 +3155,37 @@ static void jb_init(JsonBuf* b) {
|
||||
b->buf[0] = '\0';
|
||||
}
|
||||
|
||||
/* jb_init_cap — jb_init with a caller-supplied starting capacity.
|
||||
*
|
||||
* WHY THIS EXISTS (2026-08-11 self-review). jb_init starts at 64 BYTES and
|
||||
* jb_reserve grows by doubling. That is right for the hundreds of small JSON
|
||||
* responses this runtime builds per minute and catastrophic for the one that
|
||||
* is 64 MEGABYTES: serializing the canonical snapshot walked the buffer
|
||||
* 64B → 128B → ... → 128MB, about twenty reallocs, each copying everything
|
||||
* written so far. Roughly 128MB of memcpy per save, and — the part that
|
||||
* actually hurt — a fresh large span from the allocator every time.
|
||||
*
|
||||
* MEASURED (13,129 nodes / 43,400 edges, macOS arm64): RSS climbed +63MB per
|
||||
* snapshot write, linearly, 14 for 14 writes, no plateau — 204MB to 1,028MB.
|
||||
* `leaks` reported only 15KB genuinely unreachable, which is what makes this
|
||||
* subtle: nothing is leaked in the reachable/unreachable sense. engram_save
|
||||
* frees b.buf correctly on every path. The growth is the allocator declining
|
||||
* to return large freed spans to the OS, and the doubling walk guaranteeing
|
||||
* that each save asks for a differently-sized region than the last free made
|
||||
* available. Every durable write path calls this — node create, edge create,
|
||||
* the Hebbian batch write-back — so on the live daemon it grows without bound
|
||||
* until the process dies.
|
||||
*
|
||||
* The fix is to ask for the right size once. With a stable capacity the
|
||||
* allocator hands back the same span on every save and RSS flattens. */
|
||||
static void jb_init_cap(JsonBuf* b, size_t cap) {
|
||||
if (cap < 64) cap = 64;
|
||||
b->cap = cap; b->len = 0;
|
||||
b->buf = malloc(b->cap);
|
||||
if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); }
|
||||
b->buf[0] = '\0';
|
||||
}
|
||||
|
||||
static void jb_reserve(JsonBuf* b, size_t add) {
|
||||
if (b->len + add + 1 > b->cap) {
|
||||
while (b->len + add + 1 > b->cap) b->cap *= 2;
|
||||
@@ -6424,6 +6455,26 @@ static float* _eg_ctx_c = NULL;
|
||||
static int32_t _eg_ctx_dim = 0;
|
||||
static double _eg_act_ctx_cos = -2.0;
|
||||
|
||||
/* Fan-effect gauges (2026-08-11 self-review). Per-call, like ctx_cos: they
|
||||
* describe THIS activation, not process history. Without these the degree
|
||||
* normalization is an unobservable change to the most important scoring path
|
||||
* in the runtime, and "did it do anything" would be unanswerable — which is
|
||||
* exactly the failure the Hebbian learning rate had before it was measured.
|
||||
* fan_mean — mean applied factor over every propagation step. 1.0 means the
|
||||
* correction never bound (graph is flat, or d_ref is above every
|
||||
* pair's geometric mean degree). Falling toward FAN_MIN means
|
||||
* traversal is running through hubs.
|
||||
* fan_min_seen / fan_hits — the worst single penalty and how many steps were
|
||||
* penalized at all, so a low mean caused by one pathological hub
|
||||
* is distinguishable from broad hub saturation.
|
||||
* fan_dref — the live mean degree the correction is calibrated against;
|
||||
* publishing it makes densification visible over time. */
|
||||
static double _eg_act_fan_sum = 0.0;
|
||||
static double _eg_act_fan_min = 1.0;
|
||||
static int64_t _eg_act_fan_n = 0;
|
||||
static int64_t _eg_act_fan_hits = 0;
|
||||
static double _eg_act_fan_dref = 0.0;
|
||||
|
||||
static int _eg_embed_consec_fail = 0;
|
||||
static int64_t _eg_embed_breaker_until = 0;
|
||||
|
||||
@@ -6782,6 +6833,10 @@ typedef struct EngramStore {
|
||||
int* adj_to_len;
|
||||
int adj_dirty; /* 1 = rebuild needed before next BFS */
|
||||
int64_t adj_node_count; /* node_count at time of last adj_rebuild */
|
||||
/* Nodes with degree >= 1 at last adj_rebuild. The denominator for the
|
||||
* fan-effect reference degree — see eg_fan_factor for why isolated nodes
|
||||
* must not be counted. (2026-08-11 self-review) */
|
||||
int64_t adj_connected;
|
||||
} EngramStore;
|
||||
|
||||
static EngramStore* engram_global = NULL;
|
||||
@@ -7145,11 +7200,16 @@ static void engram_adj_rebuild(EngramStore* g) {
|
||||
if (ti >= 0 && g->adj_to[ti])
|
||||
g->adj_to[ti][to_pos[ti]++] = (int)ei;
|
||||
}
|
||||
/* Copy counts */
|
||||
/* Copy counts. Also tally how many nodes have any edge at all — the
|
||||
* fan-effect denominator. Free here, in the O(V) pass that already exists,
|
||||
* rather than as a separate scan. (2026-08-11 self-review) */
|
||||
int64_t connected = 0;
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
g->adj_from_len[i] = from_cnt[i];
|
||||
g->adj_to_len[i] = to_cnt[i];
|
||||
if (from_cnt[i] + to_cnt[i] > 0) connected++;
|
||||
}
|
||||
g->adj_connected = connected;
|
||||
free(from_cnt); free(to_cnt); free(from_pos); free(to_pos);
|
||||
g->adj_node_count = g->node_count;
|
||||
g->adj_dirty = 0;
|
||||
@@ -8297,6 +8357,108 @@ static double engram_activation_dampen(const EngramNode* n) {
|
||||
return 1.0 / (1.0 + log(1.0 + (double)n->activation_count));
|
||||
}
|
||||
|
||||
/* ── ACT-R fan effect: degree normalization for spreading activation ─────────
|
||||
* (2026-08-11 self-review. Closes the other half of a mechanism that has been
|
||||
* half-implemented since the BLL work.)
|
||||
*
|
||||
* THE GAP. This runtime implements ACT-R's base-level learning term
|
||||
* B_i = ln(Σ t_k^-d) (engram_bll_base_level) but never implemented the
|
||||
* ASSOCIATIVE term that goes with it:
|
||||
*
|
||||
* A_i = B_i + Σ_j W_j · S_ji where S_ji = S − ln(fan_j)
|
||||
*
|
||||
* fan_j is the number of things j is associated with. The whole point of the
|
||||
* fan effect (Anderson 1974; Anderson & Reder 1999) is that a source spreads a
|
||||
* FIXED budget of activation across its associations — so being connected to
|
||||
* many things makes each individual connection weaker. Without it, degree is
|
||||
* pure advantage: a node wins retrieval by being popular rather than by being
|
||||
* relevant. That is backwards, and it is what this graph has been doing.
|
||||
*
|
||||
* MEASURED ON THE LIVE STORE (13,129 nodes / 43,400 edges, 2026-08-11):
|
||||
* degree p50=14 p90=34 p95=82 p99=275 max=357 mean=23.3
|
||||
* the top 1% of nodes by degree touch 21.2% of all edges
|
||||
* So the most-connected node had a 25x propagation advantage over the median
|
||||
* node for no reason other than accumulated connections. The top hubs are not
|
||||
* even semantically central — several are duplicate pairs of the same document
|
||||
* left over from the redundancy census of the 2026-08-05 review.
|
||||
*
|
||||
* The hub problem was already recognized twice and patched narrowly both
|
||||
* times: InternalStateEvent nodes were cut out of propagation entirely (see
|
||||
* the frontier loop) and eg_hebb_node_budget caps per-node Hebbian mass. Both
|
||||
* are special cases of this general law. This is the general fix.
|
||||
*
|
||||
* FORM. Symmetric normalization, w / (deg(u)^β · deg(v)^β) with β = 0.5 — the
|
||||
* normalized-Laplacian / GCN form, which penalizes a hub both for sending and
|
||||
* for receiving. Both failure modes are live here: a hub source floods its
|
||||
* neighborhood, and a hub target gets reached by everything regardless of
|
||||
* relevance. Written relative to the graph's own mean degree:
|
||||
*
|
||||
* fan(u,v) = clamp( d_ref / sqrt(deg(u) · deg(v)), FAN_MIN, 1.0 )
|
||||
* d_ref = 2·|E| / |V| (mean degree, O(1), live)
|
||||
*
|
||||
* WHY IT IS CLAMPED AT 1.0 ON TOP — this is the load-bearing safety property,
|
||||
* not a detail. The factor can only ever REDUCE propagation, never amplify it.
|
||||
* Every constant downstream of this multiply is calibrated against today's
|
||||
* activation magnitudes: the 0.02 firing threshold, SPREAD_DECAY = 0.7, the
|
||||
* 0.15 WM promotion threshold, the 24-slot WM cap. A normalization that
|
||||
* boosted low-degree nodes would inflate the frontier, change how many nodes
|
||||
* clear 0.02, and silently recalibrate working memory as a side effect of a
|
||||
* change that was supposed to be about hubs. Capping at 1.0 means every pair
|
||||
* at or below mean degree — the common case — propagates EXACTLY as it does
|
||||
* today, and the only behavior that changes is that above-mean hubs stop
|
||||
* winning on degree alone. Strictly monotone, strictly conservative, and the
|
||||
* blast radius is confined to the nodes the change is aimed at.
|
||||
*
|
||||
* Self-calibrating: d_ref is recomputed from the live graph, so the correction
|
||||
* tracks densification instead of drifting against a constant that was right
|
||||
* in August 2026 and wrong a year later. Change is the signal.
|
||||
*
|
||||
* FAN_MIN = 0.30 bottoms the penalty at ~3.3x rather than the ~15x that raw
|
||||
* 1/deg would give at max degree. Same reasoning as ENGRAM_QGATE_FLOOR: damp
|
||||
* the uninformative path, never sever it. A hub is usually a hub for a reason;
|
||||
* it just should not also get a free win.
|
||||
*
|
||||
* Sources: Anderson & Reder 1999 (fan effect, S=1.6-2.0, d=0.5) ·
|
||||
* arXiv:2405.14831 HippoRAG (node specificity) · Systems 9(2):22
|
||||
* (normalized-Laplacian spreading activation) · arXiv:2606.30133 (β is a
|
||||
* low-sensitivity knob; gating and fan normalization carry the effect). */
|
||||
/* FAN_MIN 0.50, not the 0.30 this shipped as on the first build. Measured on
|
||||
* the live graph, β=0.5 with a 0.30 floor damped 96% of propagation steps to a
|
||||
* mean factor of 0.34 — and that number is not a bug in the correction, it is
|
||||
* an honest measurement of how hub-dominated traversal here actually is. But a
|
||||
* ~3x near-uniform damp is a bigger global change than one A/B run justifies,
|
||||
* and it cost a working-memory promotion (5 → 4) on the one query measured
|
||||
* cleanly. A 0.50 floor keeps the full mechanism and the whole [0.5, 1.0]
|
||||
* dynamic range for separating hubs from non-hubs, at half the blast radius.
|
||||
* The fan_mean / fan_hits gauges make the next review's tuning evidence-based
|
||||
* rather than another guess: loosen it when the data says WM can afford it. */
|
||||
#define ENGRAM_FAN_MIN 0.50
|
||||
|
||||
/* eg_node_degree — total (in + out) degree from the adjacency index. The index
|
||||
* is rebuilt at the top of engram_activate whenever topology changed, so this
|
||||
* is current. adj_node_count is the count at BUILD time and can lag
|
||||
* node_count; out-of-range indices report 0 and are treated as unpenalized. */
|
||||
static int eg_node_degree(const EngramStore* g, int64_t idx) {
|
||||
if (idx < 0 || idx >= g->adj_node_count) return 0;
|
||||
if (!g->adj_from_len || !g->adj_to_len) return 0;
|
||||
return g->adj_from_len[idx] + g->adj_to_len[idx];
|
||||
}
|
||||
|
||||
static double eg_fan_factor(const EngramStore* g, double d_ref,
|
||||
int64_t u_idx, int64_t v_idx) {
|
||||
if (d_ref <= 0.0) return 1.0;
|
||||
int du = eg_node_degree(g, u_idx);
|
||||
int dv = eg_node_degree(g, v_idx);
|
||||
/* Degree 0 is only reachable when the adjacency index is stale or absent;
|
||||
* an actually-isolated node is never on the frontier. Do not penalize what
|
||||
* we cannot measure. */
|
||||
if (du <= 0 || dv <= 0) return 1.0;
|
||||
double f = d_ref / sqrt((double)du * (double)dv);
|
||||
if (f > 1.0) return 1.0; /* never amplify — see above */
|
||||
if (f < ENGRAM_FAN_MIN) return ENGRAM_FAN_MIN;
|
||||
return f;
|
||||
}
|
||||
|
||||
/* Temporal proximity bonus: boost propagation along edges connecting
|
||||
* co-temporal nodes. Returns a multiplier bonus in [0, 0.2]. */
|
||||
static double engram_temporal_proximity_bonus(int64_t node_created,
|
||||
@@ -8464,6 +8626,8 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
||||
* miss nearly all events between beats; see the definition site).
|
||||
* ctx_cos stays per-call: it is a gauge of THIS query vs the centroid. */
|
||||
_eg_act_ctx_cos = -2.0;
|
||||
_eg_act_fan_sum = 0.0; _eg_act_fan_min = 1.0;
|
||||
_eg_act_fan_n = 0; _eg_act_fan_hits = 0;
|
||||
|
||||
/* ── Embedding backfill + query embedding (2026-07-24, bl-b2d1c944) ──
|
||||
* Backfill: embed up to N un-embedded eligible nodes per call, newest
|
||||
@@ -8696,6 +8860,29 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
||||
ftail++;
|
||||
}
|
||||
const double SPREAD_DECAY = 0.7;
|
||||
/* Reference degree for the fan-effect correction: mean degree over
|
||||
* CONNECTED nodes, 2|E| / |{v : deg(v) > 0}|. O(1) — adj_connected is
|
||||
* tallied during adjacency rebuild.
|
||||
*
|
||||
* NOT 2|E|/|V|. That was the first cut and instrumentation caught it
|
||||
* immediately: on the live graph it gives d_ref = 6.61, while the median
|
||||
* degree of a node that actually has edges is 14. Isolated nodes cannot
|
||||
* be on the frontier — spreading activation only ever traverses connected
|
||||
* ones — so including them in the denominator deflates the reference below
|
||||
* anything traversal will ever see, and the correction pins to
|
||||
* ENGRAM_FAN_MIN on every step. Measured on the first build:
|
||||
* fan_mean 0.3026 with fan_hits 579/579 — a uniform 0.30 multiplier, which
|
||||
* is not a fan effect at all. It is just a weaker SPREAD_DECAY, and it
|
||||
* would have quietly recalibrated the 0.02 firing threshold and WM
|
||||
* competition while appearing to be a targeted change.
|
||||
*
|
||||
* Over connected nodes the reference is ~23, above the median, so typical
|
||||
* traversal rides the 1.0 cap unchanged and only genuine hubs are damped
|
||||
* — which is the whole intent. The gauge that caught this is the reason it
|
||||
* was worth adding the gauge. */
|
||||
const double FAN_DREF = (g->adj_connected > 0)
|
||||
? (2.0 * (double)g->edge_count / (double)g->adj_connected) : 0.0;
|
||||
_eg_act_fan_dref = FAN_DREF;
|
||||
while (fhead < ftail) {
|
||||
Frontier f = fr[fhead++];
|
||||
if (f.hops >= max_depth) continue;
|
||||
@@ -8767,11 +8954,24 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
||||
double c = cosq[oi] > 0.0 ? cosq[oi] : 0.0;
|
||||
qgate = ENGRAM_QGATE_FLOOR + (1.0 - ENGRAM_QGATE_FLOOR) * c;
|
||||
}
|
||||
/* ── ACT-R fan effect (2026-08-11 self-review) ──
|
||||
* Symmetric degree normalization over the (source, target) pair.
|
||||
* The query gate above prunes branches that are semantically
|
||||
* irrelevant; this prunes branches that are merely POPULAR. They
|
||||
* are different failure modes — a duplicate document with 357
|
||||
* edges can be highly cosine-similar to the query and still be
|
||||
* the wrong thing to spread through. Only ever <= 1.0, so it
|
||||
* cannot inflate the frontier. See eg_fan_factor. */
|
||||
double fan = eg_fan_factor(g, FAN_DREF, cur, oi);
|
||||
_eg_act_fan_sum += fan;
|
||||
_eg_act_fan_n++;
|
||||
if (fan < 1.0) _eg_act_fan_hits++;
|
||||
if (fan < _eg_act_fan_min) _eg_act_fan_min = fan;
|
||||
/* eg_edge_eff_weight, not e->weight: edges that have repeatedly
|
||||
* carried co-activated pairs propagate more strongly. Identity on
|
||||
* an unlearned edge. (2026-08-04 self-review.) */
|
||||
double new_act = f.act * eg_edge_eff_weight(e) * SPREAD_DECAY
|
||||
* (1.0 + tbonus) * tdecay * dampen * qgate;
|
||||
* (1.0 + tbonus) * tdecay * dampen * qgate * fan;
|
||||
/* Firing threshold per classic spreading-activation: sub-threshold
|
||||
* activation neither updates the target nor enqueues it, so weak
|
||||
* signals die out instead of flooding the whole graph with tiny
|
||||
@@ -9744,11 +9944,24 @@ static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e) {
|
||||
jb_putc(b, '}');
|
||||
}
|
||||
|
||||
/* Size of the last snapshot this process serialized. Seeds the next save's
|
||||
* buffer so the doubling walk never runs on the big document. See jb_init_cap
|
||||
* for the measurement that motivated it. (2026-08-11 self-review) */
|
||||
static size_t _eg_save_cap_hint = 0;
|
||||
|
||||
el_val_t engram_save(el_val_t path) {
|
||||
const char* p = EL_CSTR(path);
|
||||
if (!p || !*p) return 0;
|
||||
EngramStore* g = engram_get();
|
||||
JsonBuf b; jb_init(&b);
|
||||
/* Pre-size from the previous save plus 12.5% headroom, so ordinary growth
|
||||
* between snapshots does not trigger a realloc and the request size stays
|
||||
* stable enough for the allocator to reuse the same span. First save of
|
||||
* the process has no hint and starts at 1MB — still 14 doublings better
|
||||
* than 64 bytes. */
|
||||
JsonBuf b;
|
||||
jb_init_cap(&b, _eg_save_cap_hint
|
||||
? _eg_save_cap_hint + (_eg_save_cap_hint >> 3) + 1024
|
||||
: (size_t)1 << 20);
|
||||
jb_puts(&b, "{\"nodes\":[");
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
if (i > 0) jb_putc(&b, ',');
|
||||
@@ -9788,6 +10001,10 @@ el_val_t engram_save(el_val_t path) {
|
||||
jb_putc(&b, '}');
|
||||
}
|
||||
jb_puts(&b, "]}");
|
||||
/* Remember the size BEFORE the write: the hint is about how much buffer
|
||||
* the next serialization needs, which is a property of the graph, not of
|
||||
* whether this particular fopen succeeded. */
|
||||
_eg_save_cap_hint = b.len;
|
||||
FILE* f = fopen(p, "wb");
|
||||
if (!f) { free(b.buf); return 0; }
|
||||
size_t w = fwrite(b.buf, 1, b.len, f);
|
||||
@@ -10714,8 +10931,11 @@ el_val_t engram_act_stats_json(void) {
|
||||
}
|
||||
/* 768, not 512: the write-back gauges added 2026-08-07 push the worst-case
|
||||
* rendering past the old bound, and snprintf would truncate the JSON into
|
||||
* an unparseable tail rather than fail loudly. */
|
||||
char buf[896];
|
||||
* an unparseable tail rather than fail loudly.
|
||||
* 1152, not 896: the five fan-effect gauges added 2026-08-11 add ~90 bytes
|
||||
* worst-case. Same reasoning — headroom is cheaper than a truncated tail
|
||||
* that every downstream JSON parser rejects as a whole. */
|
||||
char buf[1152];
|
||||
/* ctx_cos (2026-07-29): cos(query, context centroid) at the LAST
|
||||
* activate call, measured before the query was folded in. ~1.0 =
|
||||
* context aligned with current query; low = divergence (expected at
|
||||
@@ -10736,7 +10956,14 @@ el_val_t engram_act_stats_json(void) {
|
||||
* any climb means a write path is mangling text again. Cheap
|
||||
* (counted at creation) — the full census lives in
|
||||
* engram_text_health_json. (2026-08-08 self-review) */
|
||||
"\"txt_damaged\":%lld}",
|
||||
"\"txt_damaged\":%lld,"
|
||||
/* Fan-effect gauges (2026-08-11 self-review) — see the
|
||||
* _eg_act_fan_* definitions. fan_mean == 1.0 with fan_hits == 0
|
||||
* means the degree correction never bound on the last activation;
|
||||
* a mean drifting toward ENGRAM_FAN_MIN means traversal is
|
||||
* running through hubs and the correction is doing work. */
|
||||
"\"fan_mean\":%.4f,\"fan_min\":%.4f,\"fan_hits\":%lld,"
|
||||
"\"fan_steps\":%lld,\"fan_dref\":%.2f}",
|
||||
(long long)_eg_act_wm_evicted,
|
||||
(long long)_eg_act_breakthroughs,
|
||||
breaker_open, _eg_embed_consec_fail,
|
||||
@@ -10748,7 +10975,10 @@ el_val_t engram_act_stats_json(void) {
|
||||
(long long)_eg_hebb_wb_dropped,
|
||||
(long long)_eg_act_dup_seeds, (long long)_eg_act_dup_wm,
|
||||
(long long)_eg_act_dup_wm_global,
|
||||
(long long)_eg_txt_write_damaged);
|
||||
(long long)_eg_txt_write_damaged,
|
||||
(_eg_act_fan_n > 0 ? _eg_act_fan_sum / (double)_eg_act_fan_n : 1.0),
|
||||
_eg_act_fan_min, (long long)_eg_act_fan_hits,
|
||||
(long long)_eg_act_fan_n, _eg_act_fan_dref);
|
||||
return el_wrap_str(el_strdup(buf));
|
||||
}
|
||||
|
||||
@@ -10881,6 +11111,285 @@ el_val_t engram_label_df(el_val_t term) {
|
||||
return (el_val_t)df;
|
||||
}
|
||||
|
||||
/* ── Salient-term extraction (2026-08-13 self-review) ────────────────────────
|
||||
* THE MEASUREMENT. auto_term_empty_streak, the counter added by the 2026-08-06
|
||||
* review precisely to catch this class of silent death, read 50 and climbing.
|
||||
* Fifty consecutive curiosity scans in which the soul's dynamic seeding path
|
||||
* produced NOTHING and the loop fell back to its four hardcoded rotating
|
||||
* phrases. Dumping the live WM top says why in one look:
|
||||
*
|
||||
* Memory 0.390 memory:remembered
|
||||
* Memory 0.378 memory:remembered
|
||||
* Memory 0.377 memory:remembered
|
||||
* Memory 0.373 memory:remembered
|
||||
* Memory 0.370 memory:remembered
|
||||
*
|
||||
* Every slot at the top of working memory is a Memory node, and every Memory
|
||||
* node written by remember() carries the sentinel label "memory:remembered".
|
||||
* auto_term_try_slot reads the LABEL and only the label; the colon-no-space
|
||||
* guard (correctly) rejects sentinels as carrying no seed signal; so the
|
||||
* extractor had nothing to work with and returned empty, forever.
|
||||
*
|
||||
* THE ACTUAL DEFECT is not the sentinel guard — that guard is right. It is
|
||||
* that the extractor was built against Knowledge nodes, which have real
|
||||
* titles, and is structurally blind to the node type that in fact dominates
|
||||
* working memory. The label is not the content. A Memory node's topic is in
|
||||
* its text; the runtime just never looked there.
|
||||
*
|
||||
* WHY NOT ANOTHER GUARD. The extractor's whole history is guards: genre words
|
||||
* (07-23), quoted titles (07-25), English stopwords (07-30), label-df
|
||||
* (08-03). Four reviews, four blocklists, each written after watching a flood
|
||||
* happen. That is a losing shape, and 08-03 said so explicitly before adding
|
||||
* the fifth. The reason it keeps recurring is the algorithm underneath:
|
||||
* TAKE THE FIRST WORD, THEN CHECK WHETHER IT IS ACCEPTABLE. A first-word
|
||||
* extractor has no notion of term quality, so quality has to be bolted on as
|
||||
* rejection, and rejection can only encode the past.
|
||||
*
|
||||
* THE FIX is to invert it: score EVERY candidate token in the text and take
|
||||
* the argmax. Then term quality is the selection criterion rather than a
|
||||
* veto, and a bad token does not need to be on a list to lose — it only needs
|
||||
* a better token in the same text, which is the common case.
|
||||
*
|
||||
* SCORING (YAKE, Campos et al., Information Sciences 509:257-289, 2020 —
|
||||
* lightweight unsupervised single-document keyword extraction). YAKE scores
|
||||
* candidates on casing, position, frequency, context relatedness and sentence
|
||||
* dispersion, and beats RAKE/TextRank/SingleRank across twenty datasets. Two
|
||||
* of its five features port directly and cheaply; the other three are
|
||||
* within-document proxies for a corpus YAKE deliberately does not have. This
|
||||
* system DOES have the corpus — 12.7k labelled nodes — so real IDF is
|
||||
* substituted where YAKE has to approximate:
|
||||
*
|
||||
* score(t) = idf(t) · position(t) · casing(t)
|
||||
*
|
||||
* idf = ln((N+1)/(df+1)) real corpus specificity (Spärck
|
||||
* Jones 1972), strictly better than
|
||||
* YAKE's TF-based stand-in
|
||||
* position = 1/ln(e + i) YAKE T_Position: earlier tokens are
|
||||
* more topical. Keeps the old
|
||||
* first-word bias as a SOFT preference
|
||||
* instead of an absolute rule
|
||||
* casing = 1.30 acronym / 1.15 capitalised / 1.00 otherwise
|
||||
* YAKE T_Case
|
||||
*
|
||||
* THE min_df GATE. The df ceiling (08-03) rejects corpus-frequent markup and
|
||||
* sentinels. A floor was added alongside it for an independent reason: a term
|
||||
* appearing in ZERO labels cannot lexically reach anything, so it is a bad
|
||||
* seed however specific it looks.
|
||||
*
|
||||
* An earlier draft of this comment claimed the floor also subsumes the 73
|
||||
* hand-listed stopwords that 08-03 measured label-df as missing (Whose:0,
|
||||
* Would:0, Could:0). MEASURED, AND THAT CLAIM IS FALSE. Under word-boundary
|
||||
* df on the live store, function words are rare in labels but not absent:
|
||||
* about:2, whole:1, them:2, head:2. They clear a floor of 1. What actually
|
||||
* keeps them from winning is the argmax itself — they carry no position
|
||||
* advantage and lose to a topical term in the same text on every node
|
||||
* measured. The stopword list therefore STAYS as a real defense for the
|
||||
* Title-case cases, not as vestigial belt-and-braces. Recording the
|
||||
* correction rather than the tidier story: the floor buys lexical
|
||||
* reachability, the argmax buys quality, and the list still earns its keep.
|
||||
*
|
||||
* TABU IS APPLIED DURING THE ARGMAX, not after it. The old code picked a term
|
||||
* and then discarded it if it was tabu, which turned inhibition-of-return
|
||||
* into another source of empty scans. Excluding tabu terms from the candidate
|
||||
* set instead yields the best NON-TABU term, so rotation costs quality rather
|
||||
* than costing the whole scan.
|
||||
*
|
||||
* COST. One pass over g->nodes scoring all candidates at once (12.7k labels ×
|
||||
* <=32 candidates, short strings, good locality), twice per 30 s scan.
|
||||
*
|
||||
* POLICY LIVES IN THE SOUL. Thresholds arrive as arguments; the runtime
|
||||
* measures and ranks, awareness.el decides. Same split as engram_label_df.
|
||||
*
|
||||
* Returns the winning token, or "" when the node is missing, has no usable
|
||||
* text, or every candidate is gated out — "" remains the honest signal that
|
||||
* this slot yielded no seed, and auto_term_empty_streak still counts it. */
|
||||
#define ENGRAM_ST_MAXCAND 32
|
||||
#define ENGRAM_ST_TOKLEN 64
|
||||
#define ENGRAM_ST_SCANCHARS 400
|
||||
|
||||
/* Trim leading/trailing non-alphanumerics, then accept only tokens whose core
|
||||
* is alphanumeric plus '-' and '_' with at least 3 letters. This subsumes the
|
||||
* quoted-title guard (2026-07-25) and the "<!--" flood (2026-08-03)
|
||||
* structurally: markup and punctuation-bearing tokens never become
|
||||
* candidates, rather than being blocklisted after the fact. */
|
||||
static int eg_st_clean_token(const char* raw, size_t rawlen,
|
||||
char* out, size_t outcap) {
|
||||
size_t s = 0, e = rawlen;
|
||||
while (s < e && !isalnum((unsigned char)raw[s])) s++;
|
||||
while (e > s && !isalnum((unsigned char)raw[e - 1])) e--;
|
||||
size_t len = e - s;
|
||||
if (len < 4 || len >= outcap) return 0;
|
||||
int alpha = 0;
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
unsigned char c = (unsigned char)raw[s + i];
|
||||
if (isalpha(c)) alpha++;
|
||||
else if (!isdigit(c) && c != '-' && c != '_') return 0;
|
||||
}
|
||||
if (alpha < 3) return 0;
|
||||
memcpy(out, raw + s, len);
|
||||
out[len] = '\0';
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ENGRAM_ST_DEBUG=1 dumps the full scored candidate set to stderr. One
|
||||
* cached branch in production. This exists because the first live run of this
|
||||
* function returned five ALL-CAPS terms in a row and there was no way to see
|
||||
* whether that was the corpus or the casing weight without guessing — the
|
||||
* lesson this system keeps relearning. */
|
||||
static int _eg_st_debug(void) {
|
||||
static int v = -1;
|
||||
if (v < 0) { const char* e = getenv("ENGRAM_ST_DEBUG"); v = (e && *e == '1'); }
|
||||
return v;
|
||||
}
|
||||
|
||||
/* Word-boundary document frequency. engram_label_df uses istr_contains, i.e.
|
||||
* SUBSTRING matching, and that is the wrong estimator for term specificity on
|
||||
* short tokens: "them" hits inside "theme" and "anthem", "about" and "whole"
|
||||
* come back with df 2 and 1 rather than 0. That matters here specifically
|
||||
* because the min_df floor is what rejects English function words, and it can
|
||||
* only do that job if their df is honestly zero. Substring df quietly handed
|
||||
* them a survival ticket. Measured on the live store before this fix, "whole"
|
||||
* (df=1, idf=8.76) and "about" (df=2, idf=8.36) were outscoring real topical
|
||||
* terms and losing only on position — one node whose text happened to open
|
||||
* with a function word would have seeded on it.
|
||||
*
|
||||
* engram_label_df keeps substring semantics: it is a separate published
|
||||
* measure with existing callers, and changing it underneath them is not this
|
||||
* change's business. */
|
||||
static int eg_st_label_has_word(const char* hay, const char* word) {
|
||||
size_t wl = strlen(word);
|
||||
for (const char* p = hay; *p; p++) {
|
||||
if (strncasecmp(p, word, wl) != 0) continue;
|
||||
char before = (p == hay) ? '\0' : p[-1];
|
||||
char after = p[wl];
|
||||
if (before && (isalnum((unsigned char)before) || before == '_')) continue;
|
||||
if (after && (isalnum((unsigned char)after) || after == '_')) continue;
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* YAKE T_Case, adapted to this corpus. YAKE up-weights all-caps tokens
|
||||
* because in ordinary prose an acronym is rare and carries topic. That
|
||||
* assumption does not hold here: memory content written by remember()
|
||||
* conventionally OPENS WITH AN ALL-CAPS HEADER ("FRAME-ROUTER UPGRADE —
|
||||
* RESULTS", "THE GAP", "CENSUS"), so a flat acronym bonus systematically
|
||||
* hands the seed to whatever word the header happens to start with and lets
|
||||
* casing override the specificity signal it is supposed to only nudge.
|
||||
* Measured on the live store: the first five WM nodes returned PRIMING,
|
||||
* CONVERSATION, OCCUPATION, RELATIONAL, SELF-OCCUPATION — every one an
|
||||
* all-caps header word, none chosen on its merits.
|
||||
*
|
||||
* Genuine acronyms are SHORT (VBD, CCR, MCP, HTTP); shouty headers are long
|
||||
* words that happen to be capitalised. So the acronym bonus is restricted to
|
||||
* tokens of <= 5 characters, where all-caps is actually evidence of an
|
||||
* acronym rather than evidence of a heading. Longer all-caps tokens fall
|
||||
* through to the ordinary Title-case nudge — they still compete, they just
|
||||
* compete on specificity instead of on volume. */
|
||||
static double eg_st_casing(const char* t) {
|
||||
int upper = 0, lower = 0;
|
||||
size_t len = 0;
|
||||
for (const char* q = t; *q; q++, len++) {
|
||||
if (isupper((unsigned char)*q)) upper++;
|
||||
else if (islower((unsigned char)*q)) lower++;
|
||||
}
|
||||
if (lower == 0 && upper >= 2 && len <= 5) return 1.30; /* acronym */
|
||||
if (isupper((unsigned char)t[0])) return 1.15; /* Title/hdr */
|
||||
return 1.0;
|
||||
}
|
||||
|
||||
el_val_t engram_salient_term(el_val_t node_id, el_val_t max_df_v,
|
||||
el_val_t min_df_v, el_val_t tabu_v) {
|
||||
EngramStore* g = engram_get();
|
||||
int64_t ix = engram_find_node_index(EL_CSTR(node_id));
|
||||
if (ix < 0) return el_wrap_str(el_strdup(""));
|
||||
EngramNode* n = &g->nodes[ix];
|
||||
|
||||
int64_t max_df = (int64_t)max_df_v;
|
||||
int64_t min_df = (int64_t)min_df_v;
|
||||
if (max_df <= 0) max_df = g->node_count;
|
||||
if (min_df < 0) min_df = 0;
|
||||
const char* tabu = EL_CSTR(tabu_v);
|
||||
|
||||
/* Source selection. Prefer the label — it is a curated title when it is
|
||||
* one. Fall back to content when the label is absent or a sentinel
|
||||
* ("memory:remembered": a colon and no space). This single line is what
|
||||
* makes Memory nodes visible to the extractor at all. */
|
||||
const char* src = n->label;
|
||||
if (!src || !*src) {
|
||||
src = n->content;
|
||||
} else if (strchr(src, ':') != NULL && strchr(src, ' ') == NULL) {
|
||||
src = n->content;
|
||||
}
|
||||
if (!src || !*src) return el_wrap_str(el_strdup(""));
|
||||
|
||||
/* Collect distinct candidates from the head of the text. */
|
||||
char cand[ENGRAM_ST_MAXCAND][ENGRAM_ST_TOKLEN];
|
||||
int pos[ENGRAM_ST_MAXCAND];
|
||||
int64_t df[ENGRAM_ST_MAXCAND];
|
||||
int ncand = 0, tokidx = 0;
|
||||
|
||||
const char* p = src;
|
||||
const char* lim = src + strnlen(src, ENGRAM_ST_SCANCHARS);
|
||||
while (p < lim && ncand < ENGRAM_ST_MAXCAND) {
|
||||
while (p < lim && isspace((unsigned char)*p)) p++;
|
||||
if (p >= lim) break;
|
||||
const char* tk = p;
|
||||
while (p < lim && !isspace((unsigned char)*p)) p++;
|
||||
char buf[ENGRAM_ST_TOKLEN];
|
||||
int slot = tokidx++;
|
||||
if (!eg_st_clean_token(tk, (size_t)(p - tk), buf, sizeof(buf))) continue;
|
||||
|
||||
/* Tabu exclusion, applied here so the argmax runs over eligible
|
||||
* terms only. tabu arrives pipe-delimited: "|t0|t1|t2|t3|". */
|
||||
if (tabu && *tabu) {
|
||||
char pat[ENGRAM_ST_TOKLEN + 2];
|
||||
snprintf(pat, sizeof(pat), "|%s|", buf);
|
||||
if (istr_contains(tabu, pat)) continue;
|
||||
}
|
||||
int dup = 0;
|
||||
for (int i = 0; i < ncand; i++)
|
||||
if (strcasecmp(cand[i], buf) == 0) { dup = 1; break; }
|
||||
if (dup) continue;
|
||||
|
||||
memcpy(cand[ncand], buf, strlen(buf) + 1);
|
||||
pos[ncand] = slot;
|
||||
df[ncand] = 0;
|
||||
ncand++;
|
||||
}
|
||||
if (ncand == 0) return el_wrap_str(el_strdup(""));
|
||||
|
||||
/* One pass over the store, all candidates at once. */
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
const char* lbl = g->nodes[i].label;
|
||||
if (!lbl || !*lbl) continue;
|
||||
for (int c = 0; c < ncand; c++)
|
||||
if (eg_st_label_has_word(lbl, cand[c])) df[c]++;
|
||||
}
|
||||
|
||||
/* Argmax over idf · position · casing, subject to the df band. */
|
||||
int best = -1;
|
||||
double best_score = 0.0;
|
||||
for (int c = 0; c < ncand; c++) {
|
||||
if (df[c] > max_df) continue;
|
||||
if (df[c] < min_df) continue;
|
||||
double idf = log(((double)g->node_count + 1.0) / ((double)df[c] + 1.0));
|
||||
if (idf <= 0.0) continue;
|
||||
double position = 1.0 / log(2.718281828459045 + (double)pos[c]);
|
||||
double casing = eg_st_casing(cand[c]);
|
||||
double score = idf * position * casing;
|
||||
if (_eg_st_debug()) {
|
||||
fprintf(stderr, " cand %-24s df=%-5lld idf=%.2f pos=%d p=%.2f "
|
||||
"case=%.2f score=%.3f\n",
|
||||
cand[c], (long long)df[c], idf, pos[c], position,
|
||||
casing, score);
|
||||
}
|
||||
if (score > best_score) { best_score = score; best = c; }
|
||||
}
|
||||
if (best < 0) return el_wrap_str(el_strdup(""));
|
||||
return el_wrap_str(el_strdup(cand[best]));
|
||||
}
|
||||
|
||||
/* engram_embed_backfill — explicitly drive the lazy embedding backfill.
|
||||
* (2026-07-25 self-review.) The per-activate backfill (8 nodes/call) only
|
||||
* runs inside engram_activate, and on the authoritative HTTP store nothing
|
||||
|
||||
@@ -628,6 +628,13 @@ el_val_t engram_hebb_drain_json(el_val_t max);
|
||||
/* Document frequency of a term across node labels — term-specificity signal
|
||||
* for curiosity seed selection. (2026-08-03 self-review.) */
|
||||
el_val_t engram_label_df(el_val_t term);
|
||||
/* Best curiosity seed from one node: argmax over idf·position·casing across
|
||||
* the candidate tokens of its label, falling back to its content when the
|
||||
* label is a sentinel. Excludes pipe-delimited tabu terms during selection
|
||||
* and gates candidates to the df band [min_df, max_df]. Returns "" when
|
||||
* nothing qualifies. (2026-08-13 self-review.) */
|
||||
el_val_t engram_salient_term(el_val_t node_id, el_val_t max_df,
|
||||
el_val_t min_df, el_val_t tabu);
|
||||
el_val_t engram_embed_backfill(el_val_t count);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* Working memory introspection — count, mean weight, and top-N snapshot.
|
||||
|
||||
Reference in New Issue
Block a user