Files
el/elp/src/morphology-txb.el
2026-05-02 22:15:25 -05:00

309 lines
13 KiB
EmacsLisp

// morphology-txb.el - Tocharian B morphology for the NLG engine.
//
// Implements Tocharian B verb conjugation and noun declension for the ca. 500-1000 CE
// period. Designed as a companion to morphology.el and called by the engine when
// the language profile code is "txb".
//
// Language profile: code=txb, name=Tocharian B, morph_type=fusional, word_order=SOV,
// question_strategy=particle, script=latin, family=tocharian.
//
// Tocharian B is an extinct Indo-European language attested in the Tarim Basin
// (modern Xinjiang, China). Most surviving texts are Buddhist in content.
// The transliteration used here follows standard scholarly convention (Sieg, Siegling,
// Winter); subscript dots are omitted in identifiers but retained in string literals.
//
// Verb conjugation covered:
// Tenses: present (class I endings), imperfect not implemented
// Persons: first/second/third x singular/plural (slots 0-5)
// Irregulars: käm- (to come), yä- (to go), wes-/ste (to be),
// lyut- (to see), wak- (to speak)
// Canonical map: "be" -> "ste" (3sg) / "wes" (other)
//
// Noun declension covered:
// Masculine o-stem: nom/acc/gen/dat x sg/pl
// Feminine ā-stem: nom/acc/gen/dat x sg/pl
// (The full 8-case paradigm is simplified to 4 cases for the engine.)
// Number: singular, plural
// Articles: none Tocharian B has no article system
//
// Depends on: morphology.el (str_eq, str_len, str_slice, str_ends_with)
// String helpers
fn txb_drop(s: String, n: Int) -> String {
let len: Int = str_len(s)
if n >= len { return "" }
return str_slice(s, 0, len - n)
}
fn txb_ends(s: String, suf: String) -> Bool {
return str_ends_with(s, suf)
}
// Person/number slot
//
// Maps person x number to a 0-based paradigm slot.
// 0 = 1st singular
// 1 = 2nd singular
// 2 = 3rd singular (most frequently attested)
// 3 = 1st plural
// 4 = 2nd plural
// 5 = 3rd plural
fn txb_slot(person: String, number: String) -> Int {
if str_eq(person, "first") {
if str_eq(number, "singular") { return 0 }
return 3
}
if str_eq(person, "second") {
if str_eq(number, "singular") { return 1 }
return 4
}
// third
if str_eq(number, "singular") { return 2 }
return 5
}
// Present class I endings
//
// Tocharian B present class I (most regular verbs). The full paradigm is
// complex with active/medio-passive distinction; only the active series is
// implemented here.
//
// Active present class I endings applied to stem:
// Sg: 1st -au 2nd -ät 3rd -em
// Pl: 1st -emane 2nd -em 3rd -em
//
// Note: the plural endings show extensive syncretism in the attested corpus.
fn txb_pres1_suffix(slot: Int) -> String {
if slot == 0 { return "au" }
if slot == 1 { return "ät" }
if slot == 2 { return "em" }
if slot == 3 { return "emane" }
if slot == 4 { return "em" }
return "em"
}
// Irregular verb tables
// käm- (to come) suppletive present paradigm
// kam, käm, käm, kamnäṃ, kamnäṃ, kamnäṃ
fn txb_kam_present(slot: Int) -> String {
if slot == 0 { return "kam" }
if slot == 1 { return "käm" }
if slot == 2 { return "käm" }
if slot == 3 { return "kamnäṃ" }
if slot == 4 { return "kamnäṃ" }
return "kamnäṃ"
}
// yä- (to go) class IV verb; short stem
// yau, yät, yäm, ymäṃ, ymäṃ, yänmäṃ
fn txb_ya_present(slot: Int) -> String {
if slot == 0 { return "yau" }
if slot == 1 { return "yät" }
if slot == 2 { return "yäm" }
if slot == 3 { return "ymäṃ" }
if slot == 4 { return "ymäṃ" }
return "yänmäṃ"
}
// wes- / ste (to be) the copula/existential "be" verb
// 3sg form "ste" is by far the most attested; elsewhere "wes" is used.
// Simplified paradigm: "ste" for 3sg, "wes" for all other slots.
fn txb_wes_present(slot: Int) -> String {
if slot == 2 { return "ste" }
return "wes"
}
// lyut- (to see) class I regular stem, fully attested
// lyutau, lyutät, lyutem, lyutemane, lyutem, lyutem
fn txb_lyut_present(slot: Int) -> String {
if slot == 0 { return "lyutau" }
if slot == 1 { return "lyutät" }
if slot == 2 { return "lyutem" }
if slot == 3 { return "lyutemane" }
if slot == 4 { return "lyutem" }
return "lyutem"
}
// wak- (to speak) class I regular stem
// wakau, wakät, wakem, wakemane, wakem, wakem
fn txb_wak_present(slot: Int) -> String {
if slot == 0 { return "wakau" }
if slot == 1 { return "wakät" }
if slot == 2 { return "wakem" }
if slot == 3 { return "wakemane" }
if slot == 4 { return "wakem" }
return "wakem"
}
// Canonical verb mapping
//
// Maps English semantic labels to Tocharian B citation forms (verbal stems).
fn txb_map_canonical(verb: String) -> String {
if str_eq(verb, "be") { return "wes" }
if str_eq(verb, "come") { return "käm" }
if str_eq(verb, "go") { return "" }
if str_eq(verb, "see") { return "lyut" }
if str_eq(verb, "speak") { return "wak" }
if str_eq(verb, "say") { return "wak" }
return verb
}
// txb_conjugate: main conjugation entry point
//
// verb: Tocharian B stem or English canonical label
// tense: "present" (only present is implemented; others return base form)
// person: "first" | "second" | "third"
// number: "singular" | "plural"
//
// Returns the inflected form. For unknown verbs the stem is returned as-is
// the corpus is small enough that most productive verbs are known to the engine.
fn txb_conjugate(verb: String, tense: String, person: String, number: String) -> String {
let v: String = txb_map_canonical(verb)
let slot: Int = txb_slot(person, number)
// wes-/ste (to be)
if str_eq(v, "wes") {
if str_eq(tense, "present") { return txb_wes_present(slot) }
return v
}
// käm- (to come)
if str_eq(v, "käm") {
if str_eq(tense, "present") { return txb_kam_present(slot) }
return v
}
// yä- (to go)
if str_eq(v, "") {
if str_eq(tense, "present") { return txb_ya_present(slot) }
return v
}
// lyut- (to see)
if str_eq(v, "lyut") {
if str_eq(tense, "present") { return txb_lyut_present(slot) }
return v
}
// wak- (to speak)
if str_eq(v, "wak") {
if str_eq(tense, "present") { return txb_wak_present(slot) }
return v
}
// Regular class I verb
//
// Apply present class I endings directly to the stem.
if str_eq(tense, "present") { return v + txb_pres1_suffix(slot) }
// Unknown tense: return base form
return v
}
// Masculine o-stem declension
//
// Tocharian B masculine o-stem endings (simplified 4-case system):
// Singular: nom -e, acc -e, gen -entse, dat -ene
// Plural: nom -i, acc -i, gen -entwetse, dat -ene
//
// The o-stem masculine is the most common nominal class. The nom/acc syncretism
// in the singular is a characteristic Tocharian B feature.
fn txb_decline_masc(noun: String, gram_case: String, number: String) -> String {
if str_eq(number, "singular") {
if str_eq(gram_case, "nominative") { return noun + "e" }
if str_eq(gram_case, "accusative") { return noun + "e" }
if str_eq(gram_case, "genitive") { return noun + "entse" }
if str_eq(gram_case, "dative") { return noun + "ene" }
return noun + "e"
}
// plural
if str_eq(gram_case, "nominative") { return noun + "i" }
if str_eq(gram_case, "accusative") { return noun + "i" }
if str_eq(gram_case, "genitive") { return noun + "entwetse" }
if str_eq(gram_case, "dative") { return noun + "ene" }
return noun + "i"
}
// Feminine ā-stem declension
//
// Tocharian B feminine ā-stem endings (simplified 4-case system):
// Singular: nom -a, acc -a, gen -antse, dat -ane
// Plural: nom -ä, acc -ä, gen -antse, dat -ane
//
// The ā-stem feminine shows the same nom/acc syncretism as the masculine.
fn txb_decline_fem(noun: String, gram_case: String, number: String) -> String {
if str_eq(number, "singular") {
if str_eq(gram_case, "nominative") { return noun + "a" }
if str_eq(gram_case, "accusative") { return noun + "a" }
if str_eq(gram_case, "genitive") { return noun + "antse" }
if str_eq(gram_case, "dative") { return noun + "ane" }
return noun + "a"
}
// plural
if str_eq(gram_case, "nominative") { return noun + "ä" }
if str_eq(gram_case, "accusative") { return noun + "ä" }
if str_eq(gram_case, "genitive") { return noun + "antse" }
if str_eq(gram_case, "dative") { return noun + "ane" }
return noun + "ä"
}
// Gender detection heuristic
//
// In Tocharian B the gender of a noun must ideally be supplied by the lexicon.
// Gender was originally inherited from Proto-Indo-European but many neuters
// merged into masculine. This heuristic defaults to masculine; known feminine
// items should be supplied via the vocabulary layer.
fn txb_detect_gender(noun: String) -> String {
// No reliable phonological gender markers in Tocharian B stems.
// Default: masculine.
return "masculine"
}
// txb_decline: main declension entry point
//
// noun: Tocharian B noun stem (uninflected base form)
// gram_case: "nominative" | "accusative" | "genitive" | "dative"
// number: "singular" | "plural"
//
// Returns the inflected form. Falls back to the bare stem on unknown input.
fn txb_decline(noun: String, gram_case: String, number: String) -> String {
let gender: String = txb_detect_gender(noun)
if str_eq(gender, "feminine") { return txb_decline_fem(noun, gram_case, number) }
// masculine (default)
return txb_decline_masc(noun, gram_case, number)
}
// txb_noun_phrase: noun phrase builder
//
// Tocharian B has no article system. Definiteness was not grammatically marked
// by a separate morpheme context and word order served that function.
// This function therefore ignores the "definite" parameter and returns the
// declined noun form directly.
//
// noun: Tocharian B noun stem
// gram_case: "nominative" | "accusative" | "genitive" | "dative"
// number: "singular" | "plural"
// definite: "true" | "false" (ignored no articles in Tocharian B)
//
// Returns the declined noun form.
fn txb_noun_phrase(noun: String, gram_case: String, number: String, definite: String) -> String {
return txb_decline(noun, gram_case, number)
}