Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 11b6a32f27 | |||
| bdebc60dc1 |
@@ -22,10 +22,7 @@ jobs:
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
apt-get update -qq
|
||||
apt-get install -y gcc libcurl4-openssl-dev apt-transport-https ca-certificates
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" \
|
||||
> /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
apt-get install -y gcc libcurl4-openssl-dev
|
||||
|
||||
# Seed: use the committed linux-amd64 binary as the bootstrap
|
||||
- name: Bootstrap from committed linux binary (seed)
|
||||
@@ -87,22 +84,13 @@ jobs:
|
||||
bash tests/html_sanitizer/run.sh
|
||||
|
||||
# Native El test suites (elc --test, compile-link-run)
|
||||
# el_runtime.c is precompiled to .o once and reused by all 8 modules.
|
||||
- name: Precompile el_runtime.o
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
gcc -O2 -c -I "$RUNTIME" "$RUNTIME/el_runtime.c" \
|
||||
-o /tmp/el_runtime.o
|
||||
echo "el_runtime.o compiled"
|
||||
|
||||
- name: Run tests - native (core)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
/tmp/el_native_core
|
||||
|
||||
@@ -112,7 +100,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
/tmp/el_native_text
|
||||
|
||||
@@ -122,7 +110,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
/tmp/el_native_string
|
||||
|
||||
@@ -132,7 +120,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
/tmp/el_native_math
|
||||
|
||||
@@ -142,7 +130,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
/tmp/el_native_state
|
||||
|
||||
@@ -152,7 +140,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
/tmp/el_native_time
|
||||
|
||||
@@ -162,7 +150,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
/tmp/el_native_json
|
||||
|
||||
@@ -172,7 +160,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
/tmp/el_native_env
|
||||
|
||||
@@ -182,7 +170,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c /tmp/el_runtime.o \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
/tmp/el_native_fs
|
||||
|
||||
@@ -215,6 +203,9 @@ jobs:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
|
||||
@@ -261,52 +252,4 @@ jobs:
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-dev"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
# (deleted in that step after docker push)
|
||||
|
||||
- name: Rebuild ci-base with fresh El SDK (dev)
|
||||
# Patches ci-base:dev in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
CI_BASE="us-central1-docker.pkg.dev/neuron-785695/neuron-ci/ci-base"
|
||||
SHA="${GITHUB_SHA:0:8}"
|
||||
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
gcloud auth configure-docker us-central1-docker.pkg.dev --quiet
|
||||
|
||||
# Pull existing ci-base:dev (or fall back to :latest on first run)
|
||||
BASE_TAG="dev"
|
||||
docker pull "${CI_BASE}:dev" || { docker pull "${CI_BASE}:latest" && BASE_TAG="latest"; }
|
||||
|
||||
# Inline Dockerfile — only replaces the El SDK layer
|
||||
cat > /tmp/Dockerfile.ci-base-patch << 'EOF'
|
||||
ARG BASE
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
docker build \
|
||||
--build-arg BASE="${CI_BASE}:${BASE_TAG}" \
|
||||
--build-arg BUILDKIT_INLINE_CACHE=1 \
|
||||
-f /tmp/Dockerfile.ci-base-patch \
|
||||
-t "${CI_BASE}:dev" \
|
||||
-t "${CI_BASE}:dev-${SHA}" \
|
||||
.
|
||||
|
||||
docker push "${CI_BASE}:dev"
|
||||
docker push "${CI_BASE}:dev-${SHA}"
|
||||
|
||||
echo "ci-base rebuilt: ${CI_BASE}:dev (${SHA})"
|
||||
rm -f /tmp/gcp-key.json
|
||||
|
||||
@@ -246,51 +246,4 @@ jobs:
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-stage"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
# (deleted in that step after docker push)
|
||||
|
||||
- name: Rebuild ci-base with fresh El SDK (stage)
|
||||
# Patches ci-base:stage in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
CI_BASE="us-central1-docker.pkg.dev/neuron-785695/neuron-ci/ci-base"
|
||||
SHA="${GITHUB_SHA:0:8}"
|
||||
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
gcloud auth configure-docker us-central1-docker.pkg.dev --quiet
|
||||
|
||||
# Pull existing ci-base:stage (system deps stay cached in the base layer)
|
||||
docker pull "${CI_BASE}:stage" || docker pull "${CI_BASE}:latest"
|
||||
|
||||
# Inline Dockerfile — only replaces the El SDK layer
|
||||
cat > /tmp/Dockerfile.ci-base-patch << 'EOF'
|
||||
ARG BASE
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
docker build \
|
||||
--build-arg BASE="${CI_BASE}:stage" \
|
||||
--build-arg BUILDKIT_INLINE_CACHE=1 \
|
||||
-f /tmp/Dockerfile.ci-base-patch \
|
||||
-t "${CI_BASE}:stage" \
|
||||
-t "${CI_BASE}:stage-${SHA}" \
|
||||
.
|
||||
|
||||
docker push "${CI_BASE}:stage"
|
||||
docker push "${CI_BASE}:stage-${SHA}"
|
||||
|
||||
echo "ci-base rebuilt: ${CI_BASE}:stage (${SHA})"
|
||||
rm -f /tmp/gcp-key.json
|
||||
|
||||
@@ -1,65 +0,0 @@
|
||||
# ELP language consolidation — full-lexicon backfill (stage)
|
||||
|
||||
Branch: `stage-elp-lang-consolidation` (stage-bound; NOT the live soul :8742).
|
||||
|
||||
Consolidates scattered Python language-realizer work (`~/Desktop/lang-realizers`,
|
||||
`~/Desktop/lang-poetry-experiment`, `~/semitic_engine`) into the ELP `.el`
|
||||
structure, generating **full lexicons** (complete UniMorph + kaikki.org
|
||||
Wiktionary — real gender, real inflections) instead of the demo/curated subsets
|
||||
the prototypes shipped.
|
||||
|
||||
## ELP before this branch
|
||||
- 18 classical/ancient languages fully done (vocab + morphology + tests):
|
||||
akk ang cop egy enm fro gez goh got grc non peo pi sa sga sux txb uga.
|
||||
- 11 modern/classical languages had `morphology-<code>.el` in the build manifest
|
||||
but **no vocabulary and no lang_profile**: es fr de ja ar he hi ru fi sw la.
|
||||
- The ES port (`stage-elp-es-port`) had a *demo-scale* vocabulary-es.el (~350
|
||||
entries, s-expr form).
|
||||
|
||||
## Landed on this branch (full-lexicon seed-fn format, matching the 18 ancients)
|
||||
Vocabulary schema per row: `[lemma, pos, form0, form1, form2, en_gloss, hint]`.
|
||||
Files are ELP runtime **seed data** (loaded via the Engram at runtime), so — like
|
||||
all 18 classical `vocabulary-*.el` — they are intentionally NOT in the build
|
||||
manifest. Syntax validated: the chunked `fn vocab_<code>_seed_pN` format
|
||||
compiles cleanly to C via `elc` (correct UTF-8).
|
||||
|
||||
| code | in-ELP-morph? | vocab entries | verbs | nouns | adjs | profile |
|
||||
|------|---------------|--------------:|------:|------:|-----:|---------|
|
||||
| es | yes | 72,032 | 6,695 | 48,353 | 16,984 | yes |
|
||||
| fr | yes | 130,517 | 7,534 | 77,344 | 45,639 | yes |
|
||||
| de | yes | 144,692 | 6,661 | 133,162 | 4,869 | yes |
|
||||
| la | yes | 22,590 | 82 | 13,436 | 9,072 | yes |
|
||||
| it | no (bonus) | 193,675 | 10,008 | 109,459 | 74,208 | yes |
|
||||
| pt | no (bonus) | 115,772 | 4,001 | 72,073 | 39,698 | yes |
|
||||
| ro | no (bonus) | 86,504 | 1,216 | 65,915 | 19,373 | yes |
|
||||
| ca | no (bonus) | 47,112 | 1,547 | 28,830 | 16,735 | yes |
|
||||
|**total**| |**812,894** | | | | |
|
||||
|
||||
Generators (reproducible): `elp/tests/lang-gen/gen_elp_seed_full.py` (Romance),
|
||||
`gen_elp_seed_de_la.py` (German declension + Latin case-paradigm mapping). They
|
||||
read the pre-built morph caches in `~/Desktop/lang-realizers/data/` (UniMorph +
|
||||
kaikki), which are too large to commit.
|
||||
|
||||
## Remaining (honest)
|
||||
Of the 11 ELP backfill targets, 4 are done (es fr de la). The other 7 have **no
|
||||
full-lexicon engine** yet — cannot be generated honestly without engine work:
|
||||
- **ru**: only a 110-entry curated Slavic subset exists; full `rus.unimorph`
|
||||
present but no `morphology_ru_full` productive loader. Needs a full Russian
|
||||
morphology module (like the Romance ones) before vocab generation.
|
||||
- **ja / ko / zh**: validated demo engines (~66-104 hardcoded words) in
|
||||
`lang-poetry-experiment`, Python only. Agglutinative (ja/ko) + isolating (zh)
|
||||
need `.el` engine ports + full-lexicon wiring (ja: jpn_unimorph; zh: CC-CEDICT).
|
||||
- **ar / he (Semitic)**: template engines (16 AR / 8 HE patterns, ~6 roots) in
|
||||
`~/semitic_engine`, Python only. Root-and-pattern; full UniMorph ara/heb
|
||||
present but used only for validation. Needs productive root lexicon + `.el` port.
|
||||
- **hi (Hindi), fi (Finnish), sw (Swahili)**: `morphology-<code>.el` exists in
|
||||
ELP but there is NO scattered prototype and NO downloaded data for these —
|
||||
full-lexicon collection (UniMorph/kaikki) + generator still to do.
|
||||
|
||||
De/nl/sv Germanic and it/ro/ca/pt Romance verb coverage note: German verbs here
|
||||
are the ~6.6k caches carry; the it/ro/ca/pt bonus languages have full vocab but
|
||||
**no `morphology-<code>.el` in ELP yet** (Python realizer exists; `.el` port is
|
||||
the remaining engine work).
|
||||
|
||||
Construction coverage (separate from lexicon): French realizer was ~55%,
|
||||
Semitic ~3% in the prototypes — full construction coverage remains its own task.
|
||||
@@ -80,11 +80,6 @@ build {
|
||||
"src/grammar.el",
|
||||
"src/realizer.el",
|
||||
"src/semantics.el",
|
||||
"src/comprehend.el",
|
||||
"src/propositions.el",
|
||||
"src/multilingual.el",
|
||||
"src/self_region.el",
|
||||
"src/dialogue.el",
|
||||
"src/elp.el",
|
||||
]
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,16 +0,0 @@
|
||||
// comprehend.elh — public surface of the ELP comprehension front-end.
|
||||
// text → meaning-spec (the input half of the ELP; inverse of the realizer).
|
||||
extern fn parse_spec(text: String) -> [String]
|
||||
extern fn parse_spec_lang(text: String, lang: String) -> [String]
|
||||
extern fn parse_json(text: String) -> String
|
||||
extern fn parse_json_lang(text: String, lang: String) -> String
|
||||
// Analysis primitives (invertible morphology + deterministic grammar helpers):
|
||||
extern fn cp_tokenize(text: String) -> [String]
|
||||
extern fn cp_pron_concept(w: String) -> String
|
||||
extern fn cp_is_negation(w: String) -> Bool
|
||||
extern fn cp_is_neg_adverb(w: String) -> Bool
|
||||
extern fn cp_irr2(surface: String) -> [String]
|
||||
extern fn cp_reg_verb(w: String) -> [String]
|
||||
extern fn cp_analyze_verb(surface: String) -> [String]
|
||||
extern fn cp_verb_start(toks: [String], end: Int) -> Int
|
||||
extern fn cp_subord_start(toks: [String], n: Int) -> Int
|
||||
@@ -1,287 +0,0 @@
|
||||
// dialogue.el — SUMMON-THROUGH-SELF, native el. Port of dialogue.py's core.
|
||||
//
|
||||
// THE WHOLE DIALOGUE IS ONE OPERATION. A fact is never merely *fetched*: the
|
||||
// query is PROJECTED into the engram's self + memory geometry, LANDS in a region,
|
||||
// and the reply is READ OUT / the region MATERIALIZED from wherever it landed.
|
||||
//
|
||||
// project(query) -> land on a region -> read out from that region
|
||||
//
|
||||
// • lands in the SELF region -> grounded identity/presence, read out of
|
||||
// the real self nodes (self_region.el)
|
||||
// • lands on a memory NEIGHBORHOOD -> MATERIALIZE it: walk the neighborhood
|
||||
// (engram_neighbors_json) and read out the
|
||||
// region's connected members
|
||||
// • lands nowhere close -> HONEST ABSENCE (an empty region, not a
|
||||
// fabricated answer, not an error)
|
||||
//
|
||||
// CRITICAL INVARIANTS (enforced structurally, not by convention):
|
||||
// * ONE operation — there is NO intent classifier and NO separate
|
||||
// fact-retrieval branch. Identity is nearest-region proximity, not a switch.
|
||||
// * MATERIALIZE by walking the neighborhood, never by fetching top-props.
|
||||
// * HONEST ABSENCE when the region is thin.
|
||||
// * NEGATION is SACRED: the readout is the stored prose VERBATIM, so a negated
|
||||
// memory stays negated — we never paraphrase a polarity away.
|
||||
// * NO ECHO: the old "I noted that X. That relates to Y." template is gone.
|
||||
// The summon path materializes or honestly declines — it never echoes.
|
||||
// * DIRECTIVE OVERRIDE: a meta-directive ("answer in English") overrides the
|
||||
// reply language while the content language is still auto-detected.
|
||||
//
|
||||
// Depends on: comprehend (parse_spec_lang, cp_tokenize), multilingual (ml_detect,
|
||||
// ml_tr, ml_term), propositions (prop_split_sentences), self_region
|
||||
// (sr_available, sr_readout), the engram + json runtime builtins.
|
||||
|
||||
// ── directive override ────────────────────────────────────────────────────────
|
||||
// Return [target_lang, content]. target_lang is "" when no directive is present.
|
||||
// A directive names an output language; we strip it and keep the remaining text
|
||||
// as the content (whose OWN language is still auto-detected downstream).
|
||||
|
||||
fn dlg_dir_hit(low: String, phrase: String) -> Bool {
|
||||
return str_contains(low, phrase)
|
||||
}
|
||||
|
||||
fn dlg_parse_directive(text: String) -> [String] {
|
||||
let low: String = str_to_lower(text)
|
||||
let lang: String = ""
|
||||
let phrase: String = ""
|
||||
// English target
|
||||
if dlg_dir_hit(low, "in english") { let lang = "en"; let phrase = "in english" }
|
||||
if dlg_dir_hit(low, "em inglês") { let lang = "en"; let phrase = "em inglês" }
|
||||
if dlg_dir_hit(low, "em ingles") { let lang = "en"; let phrase = "em ingles" }
|
||||
if dlg_dir_hit(low, "en inglés") { let lang = "en"; let phrase = "en inglés" }
|
||||
// Portuguese target
|
||||
if dlg_dir_hit(low, "in portuguese") { let lang = "pt"; let phrase = "in portuguese" }
|
||||
if dlg_dir_hit(low, "em português") { let lang = "pt"; let phrase = "em português" }
|
||||
// Spanish target
|
||||
if dlg_dir_hit(low, "in spanish") { let lang = "es"; let phrase = "in spanish" }
|
||||
if dlg_dir_hit(low, "en español") { let lang = "es"; let phrase = "en español" }
|
||||
// Italian target
|
||||
if dlg_dir_hit(low, "in italian") { let lang = "it"; let phrase = "in italian" }
|
||||
|
||||
let content: String = text
|
||||
if !str_eq(phrase, "") {
|
||||
// strip the directive phrase (and a common "answer"/"responda" lead-in),
|
||||
// leaving the real question as content.
|
||||
let idx: Int = str_index_of(low, phrase)
|
||||
if idx >= 0 {
|
||||
let before: String = str_slice(text, 0, idx)
|
||||
let after: String = str_slice(text, idx + str_len(phrase), str_len(text))
|
||||
let content = str_trim(before + " " + after)
|
||||
}
|
||||
// trim a leading "answer"/"responda"/"reply" and stray colon/comma.
|
||||
let cl: String = str_to_lower(content)
|
||||
if str_starts_with(cl, "answer") { let content = str_trim(str_slice(content, 6, str_len(content))) }
|
||||
if str_starts_with(cl, "responda") { let content = str_trim(str_slice(content, 8, str_len(content))) }
|
||||
if str_starts_with(cl, "reply") { let content = str_trim(str_slice(content, 5, str_len(content))) }
|
||||
if str_starts_with(content, ":") { let content = str_trim(str_slice(content, 1, str_len(content))) }
|
||||
if str_starts_with(content, ",") { let content = str_trim(str_slice(content, 1, str_len(content))) }
|
||||
}
|
||||
let r: [String] = native_list_empty()
|
||||
let r = native_list_append(r, lang)
|
||||
let r = native_list_append(r, content)
|
||||
return r
|
||||
}
|
||||
|
||||
// ── identity landing (a region proximity, not a classifier switch) ────────────
|
||||
// The query lands in the SELF region when it takes an identity/presence shape.
|
||||
// Cross-lingual forms are included because the engram's lexical probe is
|
||||
// English-leaning. This is the SELF attractor of the single operation.
|
||||
|
||||
fn dlg_is_identity(content: String) -> Bool {
|
||||
let low: String = str_to_lower(str_trim(content))
|
||||
if str_contains(low, "who are you") { return true }
|
||||
if str_contains(low, "what are you") { return true }
|
||||
if str_contains(low, "who i am") { return true }
|
||||
if str_contains(low, "your name") { return true }
|
||||
if str_contains(low, "about yourself") { return true }
|
||||
if str_contains(low, "are you conscious") { return true }
|
||||
if str_contains(low, "are you there") { return true }
|
||||
// cross-lingual identity question-forms
|
||||
if str_contains(low, "quem é você") { return true }
|
||||
if str_contains(low, "quem es voce") { return true }
|
||||
if str_contains(low, "quién eres") { return true }
|
||||
if str_contains(low, "quien eres") { return true }
|
||||
if str_contains(low, "chi sei") { return true }
|
||||
if str_contains(low, "qui es-tu") { return true }
|
||||
if str_contains(low, "wer bist du") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// ── readout helpers ───────────────────────────────────────────────────────────
|
||||
|
||||
fn dlg_first_sentence(content: String) -> String {
|
||||
let sents: [String] = prop_split_sentences(content)
|
||||
let n: Int = native_list_len(sents)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let s: String = str_trim(native_list_get(sents, i))
|
||||
// drop a leading markdown heading marker for a clean read-out line
|
||||
if str_starts_with(s, "# ") { let s = str_trim(str_slice(s, 2, str_len(s))) }
|
||||
if str_len(s) > 0 { return s }
|
||||
let i = i + 1
|
||||
}
|
||||
return str_trim(content)
|
||||
}
|
||||
|
||||
// strip trailing/leading punctuation from a token.
|
||||
fn dlg_clean_tok(w: String) -> String {
|
||||
let s: String = str_trim(w)
|
||||
let s = str_strip_suffix(s, ".")
|
||||
let s = str_strip_suffix(s, ",")
|
||||
let s = str_strip_suffix(s, "?")
|
||||
let s = str_strip_suffix(s, "!")
|
||||
let s = str_strip_suffix(s, ":")
|
||||
let s = str_strip_suffix(s, ";")
|
||||
return str_trim(s)
|
||||
}
|
||||
|
||||
// closed-class across the supported languages (union) — a word we must NOT treat
|
||||
// as a retrieval topic. Also drops the meta verbs of a request ("tell", "prove",
|
||||
// "show") so the TOPIC, not the speech act, is what projects into memory.
|
||||
fn dlg_is_stop(w: String) -> Bool {
|
||||
if ml_stop_en(w) { return true }
|
||||
if ml_stop_es(w) { return true }
|
||||
if ml_stop_pt(w) { return true }
|
||||
if ml_stop_it(w) { return true }
|
||||
if str_eq(w, "tell") { return true }
|
||||
if str_eq(w, "show") { return true }
|
||||
if str_eq(w, "about") { return true }
|
||||
if str_eq(w, "sobre") { return true }
|
||||
if str_eq(w, "acerca") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// The CONTENT TERMS the query projects into memory: content words only, cleaned,
|
||||
// cross-lingually mapped to the engram's English vocabulary, ≥3 chars. This is
|
||||
// the geometry probe — the speech-act verbs and function words are stripped so a
|
||||
// PP topic ("tell me ABOUT Lisbon") projects on "lisbon", not "tell"/"me".
|
||||
fn dlg_content_terms(content: String, lang: String) -> [String] {
|
||||
let toks: [String] = cp_tokenize(content)
|
||||
let n: Int = native_list_len(toks)
|
||||
let out: [String] = native_list_empty()
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let w: String = str_to_lower(dlg_clean_tok(native_list_get(toks, i)))
|
||||
if str_len(w) >= 3 {
|
||||
if !dlg_is_stop(w) {
|
||||
let out = native_list_append(out, ml_term(w, lang))
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Does this landed node lexically overlap the query's content terms? This is the
|
||||
// RELEVANCE FLOOR: activation always returns the store's most salient nodes, so
|
||||
// without this a query about nothing would "land" on the self/top node. A node
|
||||
// that shares no content term with the query is "nowhere close" -> honest absence.
|
||||
fn dlg_node_matches(node: String, terms: [String]) -> Bool {
|
||||
let hay: String = str_to_lower(json_get_string(node, "content") + " " + json_get_string(node, "label"))
|
||||
let n: Int = native_list_len(terms)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let t: String = native_list_get(terms, i)
|
||||
if str_len(t) >= 3 {
|
||||
if str_contains(hay, t) { return true }
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// MATERIALIZE the landed region: read out the landed fact, then WALK the
|
||||
// neighborhood and read out its connected members (real edges, not top-props).
|
||||
fn dlg_materialize(top_node: String, reply_lang: String) -> String {
|
||||
let id: String = json_get_string(top_node, "id")
|
||||
let content: String = json_get_string(top_node, "content")
|
||||
let lead: String = dlg_first_sentence(content)
|
||||
|
||||
let nb: String = engram_neighbors_json(id, 2, "both")
|
||||
let m: Int = json_array_len(nb)
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, lead)
|
||||
let added: Int = 0
|
||||
let i: Int = 0
|
||||
while i < m {
|
||||
if added < 3 {
|
||||
let rec: String = json_array_get(nb, i)
|
||||
let node: String = json_get_raw(rec, "node")
|
||||
let nc: String = json_get_string(node, "content")
|
||||
if !str_eq(nc, "") {
|
||||
let sent: String = dlg_first_sentence(nc)
|
||||
if !str_eq(sent, "") {
|
||||
let parts = native_list_append(parts, sent)
|
||||
let added = added + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
// The readout is the region's OWN prose, verbatim — negation SACRED, no echo.
|
||||
return str_join(parts, " ")
|
||||
}
|
||||
|
||||
// ── THE single operation ──────────────────────────────────────────────────────
|
||||
|
||||
fn dlg_respond(text: String) -> String {
|
||||
// directive override: reply language may differ from content language.
|
||||
let dir: [String] = dlg_parse_directive(text)
|
||||
let target_lang: String = native_list_get(dir, 0)
|
||||
let content: String = native_list_get(dir, 1)
|
||||
|
||||
let content_lang: String = ml_detect(content)
|
||||
let reply_lang: String = content_lang
|
||||
if !str_eq(target_lang, "") { let reply_lang = target_lang }
|
||||
|
||||
// comprehend the content (SACRED polarity carried in the spec).
|
||||
let spec: [String] = parse_spec_lang(content, content_lang)
|
||||
|
||||
// ── PROJECT + LAND: SELF region ───────────────────────────────────────────
|
||||
// Identity/presence shape lands in the self region; read out the REAL self
|
||||
// nodes (self_region.el), never a template. Same single operation — this is
|
||||
// just the self attractor winning the landing.
|
||||
if dlg_is_identity(content) {
|
||||
if sr_available() {
|
||||
// read out the REAL self nodes when replying in their own language
|
||||
// (the soul's prose is English); for another reply language we cannot
|
||||
// translate real content without an LLM, so we answer with the
|
||||
// localized SACRED identity anchor — honest, in-language, no fabrication.
|
||||
if str_eq(reply_lang, "en") { return sr_readout("en") }
|
||||
return ml_tr("identity", reply_lang)
|
||||
}
|
||||
// self region thin — honest localized identity (logged fallback shape).
|
||||
return ml_tr("identity", reply_lang)
|
||||
}
|
||||
|
||||
// ── PROJECT into MEMORY geometry ──────────────────────────────────────────
|
||||
let terms: [String] = dlg_content_terms(content, content_lang)
|
||||
let qterm: String = str_join(terms, " ")
|
||||
let act: String = engram_activate_json(qterm, 12)
|
||||
let n: Int = json_array_len(act)
|
||||
|
||||
// ── LAND: the highest-activation node that ACTUALLY overlaps the query's
|
||||
// content terms (the relevance floor). Activation always returns the most
|
||||
// salient nodes, so we walk the ranked list and take the first that is
|
||||
// genuinely "close"; if none is, the query landed nowhere. ───────────────
|
||||
let landing: String = ""
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
if str_eq(landing, "") {
|
||||
let rec: String = json_array_get(act, i)
|
||||
let node: String = json_get_raw(rec, "node")
|
||||
if dlg_node_matches(node, terms) {
|
||||
let landing = node
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
|
||||
// ── HONEST ABSENCE: nothing close — an empty region, not a fabricated answer,
|
||||
// not an "I noted that" echo. ────────────────────────────────────────────
|
||||
if str_eq(landing, "") {
|
||||
return ml_tr("no_memory", reply_lang)
|
||||
}
|
||||
|
||||
// ── MATERIALIZE the landing by WALKING its neighborhood. ──────────────────
|
||||
return dlg_materialize(landing, reply_lang)
|
||||
}
|
||||
@@ -63,9 +63,6 @@ import "morphology-cop.el"
|
||||
import "grammar.el"
|
||||
import "realizer.el"
|
||||
import "semantics.el"
|
||||
|
||||
// ── Comprehension front-end (input half: text → meaning-spec) ─────────────────
|
||||
import "comprehend.el"
|
||||
//
|
||||
// Entry points:
|
||||
//
|
||||
@@ -120,9 +117,6 @@ fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [Strin
|
||||
let location: String = sem_get(semantic_form_json, "location")
|
||||
let tense: String = sem_get(semantic_form_json, "tense")
|
||||
let aspect: String = sem_get(semantic_form_json, "aspect")
|
||||
let polarity: String = sem_get(semantic_form_json, "polarity")
|
||||
let neg_word: String = sem_get(semantic_form_json, "neg_word")
|
||||
let iobj: String = sem_get(semantic_form_json, "iobj")
|
||||
|
||||
let form: [String] = native_list_empty()
|
||||
let form = native_list_append(form, "intent")
|
||||
@@ -133,19 +127,12 @@ fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [Strin
|
||||
let form = native_list_append(form, predicate)
|
||||
let form = native_list_append(form, "patient")
|
||||
let form = native_list_append(form, patient)
|
||||
let form = native_list_append(form, "iobj")
|
||||
let form = native_list_append(form, iobj)
|
||||
let form = native_list_append(form, "location")
|
||||
let form = native_list_append(form, location)
|
||||
let form = native_list_append(form, "tense")
|
||||
let form = native_list_append(form, tense)
|
||||
let form = native_list_append(form, "aspect")
|
||||
let form = native_list_append(form, aspect)
|
||||
// SACRED: polarity crosses the JSON boundary and is never inferred away.
|
||||
let form = native_list_append(form, "polarity")
|
||||
let form = native_list_append(form, polarity)
|
||||
let form = native_list_append(form, "neg_word")
|
||||
let form = native_list_append(form, neg_word)
|
||||
let form = native_list_append(form, "lang")
|
||||
let form = native_list_append(form, lang_code)
|
||||
|
||||
|
||||
+4
-4
@@ -1,7 +1,7 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sem_get(json: String, key: String) -> String
|
||||
extern fn generate_frame(frame: [String]) -> String
|
||||
extern fn generate_frame_lang(frame: [String], lang_code: String) -> String
|
||||
extern fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [String]
|
||||
extern fn generate_frame(frame: Any) -> String
|
||||
extern fn generate_frame_lang(frame: Any, lang_code: String) -> String
|
||||
extern fn build_form_from_json(semantic_form_json: String, lang_code: String) -> Any
|
||||
extern fn generate(semantic_form_json: String) -> String
|
||||
extern fn generate_lang(semantic_form_json: String, lang_code: String) -> String
|
||||
|
||||
+28
-28
@@ -1,22 +1,22 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn slots_get(slots: [String], key: String) -> String
|
||||
extern fn slots_set(slots: [String], key: String, val: String) -> [String]
|
||||
extern fn make_slots(k0: String, v0: String) -> [String]
|
||||
extern fn make_slots2(k0: String, v0: String, k1: String, v1: String) -> [String]
|
||||
extern fn make_slots3(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String) -> [String]
|
||||
extern fn make_slots4(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String) -> [String]
|
||||
extern fn make_slots5(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String, k4: String, v4: String) -> [String]
|
||||
extern fn rule_id(rule: [String]) -> String
|
||||
extern fn rule_lhs(rule: [String]) -> String
|
||||
extern fn rule_rhs_len(rule: [String]) -> Int
|
||||
extern fn rule_rhs(rule: [String], idx: Int) -> String
|
||||
extern fn make_rule(id: String, lhs: String, r0: String) -> [String]
|
||||
extern fn make_rule2(id: String, lhs: String, r0: String, r1: String) -> [String]
|
||||
extern fn make_rule3(id: String, lhs: String, r0: String, r1: String, r2: String) -> [String]
|
||||
extern fn make_rule4(id: String, lhs: String, r0: String, r1: String, r2: String, r3: String) -> [String]
|
||||
extern fn build_rules() -> [[String]]
|
||||
extern fn get_rules() -> [[String]]
|
||||
extern fn find_rule(rule_id_str: String) -> [String]
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn slots_get(slots: Any, key: String) -> String
|
||||
extern fn slots_set(slots: Any, key: String, val: String) -> Any
|
||||
extern fn make_slots(k0: String, v0: String) -> Any
|
||||
extern fn make_slots2(k0: String, v0: String, k1: String, v1: String) -> Any
|
||||
extern fn make_slots3(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String) -> Any
|
||||
extern fn make_slots4(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String) -> Any
|
||||
extern fn make_slots5(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String, k4: String, v4: String) -> Any
|
||||
extern fn rule_id(rule: Any) -> String
|
||||
extern fn rule_lhs(rule: Any) -> String
|
||||
extern fn rule_rhs_len(rule: Any) -> Int
|
||||
extern fn rule_rhs(rule: Any, idx: Int) -> String
|
||||
extern fn make_rule(id: String, lhs: String, r0: String) -> Any
|
||||
extern fn make_rule2(id: String, lhs: String, r0: String, r1: String) -> Any
|
||||
extern fn make_rule3(id: String, lhs: String, r0: String, r1: String, r2: String) -> Any
|
||||
extern fn make_rule4(id: String, lhs: String, r0: String, r1: String, r2: String, r3: String) -> Any
|
||||
extern fn build_rules() -> Any
|
||||
extern fn get_rules() -> Any
|
||||
extern fn find_rule(rule_id_str: String) -> Any
|
||||
extern fn make_leaf(label: String, word: String) -> String
|
||||
extern fn make_node1(label: String, child0: String) -> String
|
||||
extern fn make_node2(label: String, child0: String, child1: String) -> String
|
||||
@@ -24,15 +24,15 @@ extern fn make_node3(label: String, child0: String, child1: String, child2: Stri
|
||||
extern fn make_node4(label: String, child0: String, child1: String, child2: String, child3: String) -> String
|
||||
extern fn nlg_is_ws(c: String) -> Bool
|
||||
extern fn skip_ws(s: String, pos: Int) -> Int
|
||||
extern fn scan_token(s: String, start: Int) -> [String]
|
||||
extern fn scan_token(s: String, start: Int) -> Any
|
||||
extern fn render_tree(tree: String) -> String
|
||||
extern fn gram_word_order(profile: [String]) -> String
|
||||
extern fn gram_order_constituents(subj: String, verb: String, obj: String, profile: [String]) -> String
|
||||
extern fn gram_build_vp(verb: String, aux: String, profile: [String]) -> String
|
||||
extern fn gram_question_strategy(profile: [String]) -> String
|
||||
extern fn gram_word_order(profile: Any) -> String
|
||||
extern fn gram_order_constituents(subj: String, verb: String, obj: String, profile: Any) -> String
|
||||
extern fn gram_build_vp(verb: String, aux: String, profile: Any) -> String
|
||||
extern fn gram_question_strategy(profile: Any) -> String
|
||||
extern fn is_pronoun(word: String) -> Bool
|
||||
extern fn build_np(referent: String, slots: [String]) -> String
|
||||
extern fn build_np(referent: String, slots: Any) -> String
|
||||
extern fn build_pp(loc: String) -> String
|
||||
extern fn build_vp_body(slots: [String]) -> String
|
||||
extern fn build_vp_from_slots(slots: [String]) -> String
|
||||
extern fn generate_tree(rule_id_str: String, slots: [String]) -> String
|
||||
extern fn build_vp_body(slots: Any) -> String
|
||||
extern fn build_vp_from_slots(slots: Any) -> String
|
||||
extern fn generate_tree(rule_id_str: String, slots: Any) -> String
|
||||
|
||||
@@ -1,72 +0,0 @@
|
||||
;;; lang_profile_ca.el — Catalan language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / _es / _pt; keys the realizer's construction switches.
|
||||
;;; Catalan is the CLOSEST Romance sibling to the shared engine (~85% conceptual
|
||||
;;; reuse). The deltas: PRONOMS FEBLES with four position allomorphs, l'-elision,
|
||||
;;; del/al/pel contractions, the periphrastic preterite (vaig+INF), and NO
|
||||
;;; essere/avere split (perfect aux is always HAVER; ser/estar is only the copula).
|
||||
|
||||
(lang_profile_ca
|
||||
(language "Catalan")
|
||||
(iso639 "ca")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "el/la/l'/els/les ; un/una/uns/unes") ; l'-ELISION:
|
||||
; el/la -> l' before vowel or (silent) h, glued to
|
||||
; the next word (l'home, l'illa); de -> d' before vowel
|
||||
(article-drives-contraction yes) ; article choice feeds prep+article contraction
|
||||
(adjective-position "postnominal-default + small prenominal class") ; bo/bon,
|
||||
; mal, gran, nou, vell, primer, molt... prenominal
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((de el del) (de els dels)
|
||||
(a el al) (a els als)
|
||||
(per el pel) (per els pels)))
|
||||
(contraction-mandatory yes) ; *de el -> del obligatory
|
||||
(contraction-blocked-before-elision yes) ; de l'home / a l'home (NO *del home)
|
||||
|
||||
;; ── clitic system: PRONOMS FEBLES (the headline delta) ──────────────────
|
||||
(clitics yes)
|
||||
(clitic-allomorphy four-position) ; per pronoun, form varies by position+onset:
|
||||
; reinforced (em, et, el) proclitic before a consonant
|
||||
; elided (m', t', l', n') proclitic before a vowel/h
|
||||
; full (-me, -lo, -li) enclitic after a consonant/-r
|
||||
; reduced ('m, 't, 'l, 'ns) enclitic after a vowel
|
||||
(clitic-placement ((finite proclitic) ; el veig, no m'ho dóna
|
||||
(imperative-affirmative enclitic) ; dóna'm, digues-me
|
||||
(imperative-negative present-subjunctive) ; no parlis (delta)
|
||||
(infinitive enclitic) ; ajudar-me, veure'l
|
||||
(gerund enclitic))) ; fent-ho
|
||||
(clitic-combination ((me el "me'l") (te el "te'l") (se el "se'l")
|
||||
(me la "me la") (me en "me'n")
|
||||
(li el "l'hi") (li en "n'hi"))) ; dative+accusative clusters
|
||||
(clitic-particles (hi en ho)) ; locative hi, partitive/genitive en, neuter ho
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (present imperfet preterit-simple perifrastic-preterit futur
|
||||
condicional subjuntiu-present subjuntiu-imperfet imperatiu))
|
||||
(periphrastic-preterite "vaig/vas/va/vam/vau/van + INFINITIVE") ; << hallmark CA
|
||||
; (vaig cantar = 'I sang'); coexists w/ synthetic pret.
|
||||
(compound-past "pretèrit perfet = haver(present) + participle")
|
||||
(perfect-aux "HAVER only") ; << NO essere/avere split (simpler than IT)
|
||||
(participle-agreement ((haver preceding-acc-clitic))) ; les he vistes; else invariable
|
||||
(progressive-aux "estar + gerundi")
|
||||
(copula "ser / estar") ; ser: identity/essential/origin; estar:
|
||||
; location + transient state (estic cansat, és a casa)
|
||||
(passive-aux "ser (+ per-agent)")
|
||||
(future inflectional) ; cantaré, serà
|
||||
(comparative "més/menys ADJ que")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "no (preverbal) + optional 'pas' + concord") ; no...res/
|
||||
; ningú/mai/cap/gens/enlloc
|
||||
(negative-concord yes) ; preverbal negative subject (ningú) keeps 'no'
|
||||
(neg-reinforcer pas)) ; optional (no ho faré pas)
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_de.el — German language profile for ELP.
|
||||
;;; Mirrors lang_profile_en / lang_profile_es. Keys the realizer's construction
|
||||
;;; switches. German is the largest Germanic delta from the EN engine: V2 word
|
||||
;;; order, four morphological cases, and separable-prefix verbs.
|
||||
|
||||
(lang_profile_de
|
||||
(language "German")
|
||||
(iso639 "de")
|
||||
(family "Germanic")
|
||||
(neighbor-base "en") ; realized by extending the English (Germanic) engine
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; obligatory subject in finite clauses
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender (m f n)) ; three genders; drives article + adj declension
|
||||
(case-system (nom acc dat gen)) ; four cases on articles/adjs/nouns
|
||||
(word-order V2) ; finite verb 2nd in main clause
|
||||
(subordinate-order verb-final) ; "..., dass er den Hund SIEHT."
|
||||
(separable-verbs yes) ; aufstehen -> "steht ... auf"; ppart "aufgestanden"
|
||||
(do-support no) ; German negates/questions the finite verb directly
|
||||
(subject-verb-inversion yes) ; yes/no Q fronts finite verb; wh-Q fills Vorfeld
|
||||
(article-selection "der/die/das + ein/kein") ; declined by case x gender x number
|
||||
(adjective-position prenominal)
|
||||
(adjective-declension (strong weak mixed)) ; chosen by the determiner type
|
||||
(noun-capitalization yes)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person-and-number") ; full present/past paradigm
|
||||
(auxiliary-order (modal tense-aux perfect passive main))
|
||||
(perfect-aux (haben sein)) ; sein for intransitive motion/change verbs
|
||||
(passive-aux "werden")
|
||||
(future "werden + infinitive")
|
||||
(comparative "synthetic (-er / -st, with umlaut)")
|
||||
|
||||
;; ── negation ───────────────────────────────────────────────────────────
|
||||
(negation-markers (nicht kein)) ; kein- negates an indefinite NP; nicht else
|
||||
(negation-faithful yes) ; SACRED: polarity never dropped/inverted -> FLAG
|
||||
|
||||
;; ── lexicon provenance ─────────────────────────────────────────────────
|
||||
(lexicon-source "UniMorph deu (primary) + kaikki.org German (gender override)")
|
||||
(lexicon-license "CC-BY-SA 3.0 / GFDL"))
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_en.el — English language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. English is typologically distinct from the Romance builds, so the
|
||||
;;; flags differ where the grammar differs.
|
||||
|
||||
(lang_profile_en
|
||||
(language "English")
|
||||
(iso639 "en")
|
||||
(family "Germanic")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; OBLIGATORY subjects — missing subject is FLAGGED
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender no) ; natural gender only (he/she/it), no NP agreement
|
||||
(do-support yes) ; negation & questions of lexical verbs insert do/does/did
|
||||
(subject-aux-inversion yes) ; yes/no + non-subject wh questions invert the operator
|
||||
(article-selection "a/an/the") ; a/an resolved PHONOLOGICALLY (an hour, a university)
|
||||
(adjective-position prenominal) ; attributive adjectives precede the noun; invariant
|
||||
(has-tag-questions yes) ; "...doesn't he?" — operator + reversed polarity
|
||||
(has-there-existential yes) ; "there is/are/have been ..."
|
||||
(possessive-clitic "'s") ; saxon genitive; plural in -s -> bare apostrophe
|
||||
(question-punct plain) ; ? and ! only (no inverted marks)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "3sg-present-only") ; only 3sg present -s (+ suppletive be)
|
||||
(auxiliary-order (modal perfect progressive passive main))
|
||||
(perfect-aux "have") ; have + past participle
|
||||
(progressive-aux "be") ; be + present participle
|
||||
(passive-aux "be") ; be + past participle (+ by-agent)
|
||||
(future "will + base") ; no inflectional future
|
||||
(comparative "synthetic-or-periphrastic") ; -er/-est vs more/most by syllables
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt) ──────────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
|
||||
;; ── DIALECT overlay (post-realization, one core -> US/UK/AU) ────────────
|
||||
(dialect US) ; default; profile field switches the overlay
|
||||
(dialects (US UK AU))
|
||||
(dialect-canonical US) ; core is authored in US orthography
|
||||
(dialect-overlay "dialect_en.to_dialect") ; orthography + lexis + grammar prefs
|
||||
(dialect-covers (spelling lexis collective-agreement gotten/got)))
|
||||
@@ -1,45 +0,0 @@
|
||||
;;; lang_profile_es.el — Spanish language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_en / _pt.
|
||||
|
||||
(lang_profile_es
|
||||
(language "Spanish")
|
||||
(iso639 "es")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon); REAL per-noun gender from UniMorph — NOT a heuristic
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; questions by intonation/punctuation, not inversion
|
||||
(question-strategy intonation)
|
||||
(article-selection "el/la/los/las un/una/unos/unas")
|
||||
(stressed-a-rule yes) ; fem sg noun in stressed a-/ha- takes el/un (el agua)
|
||||
(adjective-position postnominal) ; default post; a few prenominal + apocope
|
||||
(adjective-agreement "gender+number")
|
||||
(question-punct inverted) ; opening ¿ ¡ required
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (coordinator quality bar) --------------------
|
||||
(contractions ((de el "del") (a el "al")))
|
||||
(contraction-mandatory yes) ; 'de el'/'a el' MUST surface as del/al
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "haber") ; haber + past participle (invariant -o)
|
||||
(progressive-aux "estar") ; estar + gerund
|
||||
(passive-aux "ser") ; ser + participle (agrees) + por-agent
|
||||
(copula-split "ser/estar") ; permanent vs stage-level
|
||||
(future "infinitive + é/ás/á/emos/éis/án")
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; me te lo la le nos os los las; proclisis/enclisis
|
||||
(clitic-order "se II I III (le+lo -> se lo)")
|
||||
(enclisis "imperative/infinitive/gerund + accent repair (dá+me+lo->dámelo)")
|
||||
(verb-prep-government yes) ; verbs select prep (protestar+contra, escapar+de)
|
||||
|
||||
;; -- SACRED safety bar (shared with en/pt) -------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
@@ -1,74 +0,0 @@
|
||||
;;; lang_profile_fr.el — French language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / lang_profile_es; keys the realizer's construction
|
||||
;;; switches. French is a Romance sibling (~54% of the realizer code and the whole
|
||||
;;; clause-engine architecture reused), but carries the family's biggest surface
|
||||
;;; deltas: NOT pro-drop, DISCONTINUOUS negation, and an orthography/phonology
|
||||
;;; mismatch (elision, liaison) that makes exact-match genuinely hard.
|
||||
|
||||
(lang_profile_fr
|
||||
(language "French")
|
||||
(iso639 "fr")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; << French-specific: subject clitic OBLIGATORY
|
||||
(obligatory-subject yes) ; je/tu/il/elle/nous/vous/ils/elles always overt
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion optional) ; est-ce que (default) OR clitic inversion (vas-tu)
|
||||
(article-selection "le/la/l'/les ; un/une/des ; PARTITIVE du/de la/de l'/des")
|
||||
(article-drives-contraction yes) ; à+le=au, de+le=du feed off article choice
|
||||
(adjective-position "postnominal-default + prenominal-BAGS") ; beau/bon/grand/
|
||||
; petit/jeune/vieux/nouveau + ordinals prenominal
|
||||
; (beau->bel, nouveau->nouvel, vieux->vieil / vowel)
|
||||
(question-punct "space-before") ; French typography: ' ?' ' !' (no ¿¡)
|
||||
|
||||
;; ── elision (orthography/phonology mismatch — French-specific) ──────────
|
||||
(elision ((le l') (la l') (je j') (ne n') (de d') (que qu')
|
||||
(me m') (te t') (se s') (ce c'))) ; before vowel / h-muet
|
||||
(elision-h-muet yes) ; l'homme, l'hôpital (h-aspiré exception list kept)
|
||||
(liaison noted-not-modeled) ; phonological, not written in surface
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((à le au) (à les aux) (de le du) (de les des)))
|
||||
(contraction-mandatory yes) ; *à le -> au obligatory; à la / à l' uncontracted
|
||||
(partitive ((m-sg du) (f-sg "de la") (vowel "de l'") (pl des)))
|
||||
(partitive-under-neg "de") ; << gap in current build: 'ne … pas de pain'
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-order (me te se nous vous | le la les | lui leur | y | en))
|
||||
(clitic-placement ((finite proclitic) ; je le lui donne
|
||||
(imperative-affirmative enclitic-hyphen) ; donne-le-moi
|
||||
(imperative-negative "ne+proclitic+verb+pas") ; ne le donne pas
|
||||
(infinitive enclitic))) ; PARTIAL: clitic-climbing
|
||||
; onto infinitive under modal
|
||||
(clitic-imperative-shift ((me moi) (te toi))) ; final me/te -> moi/toi (donne-moi)
|
||||
(clitic-particles (y en)) ; locative y, partitive/genitive en
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (written; many homophones)")
|
||||
(tenses (présent imparfait passé-simple futur conditionnel
|
||||
subjonctif-présent subjonctif-imparfait impératif))
|
||||
(compound-past "passé-composé = aux(present) + participe passé")
|
||||
(perfect-aux "être/avoir (LEXICAL selection)") ; << French-specific
|
||||
(etre-aux-class "intransitive motion/change (aller venir arriver partir
|
||||
entrer sortir monter descendre naître mourir rester
|
||||
tomber retourner passer devenir revenir rentrer) + ALL
|
||||
pronominal verbs")
|
||||
(participle-agreement ((être subject) ; elle est allée / elles venues
|
||||
(avoir preceding-direct-object))) ; je les ai vus
|
||||
(progressive "être en train de + infinitif") ; no dedicated aux
|
||||
(copula "être (single; no ser/estar, no essere/stare)")
|
||||
(passive-aux "être (+ par-agent)")
|
||||
(future inflectional) ; parlera, sera
|
||||
(comparative "plus/moins ADJ que")
|
||||
(superlative "le/la plus ADJ (de …)") ; PARTIAL word-order in build
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "DISCONTINUOUS: ne (preverbal) … pas/jamais/rien/personne/
|
||||
plus/guère/que (postverbal)") ; << biggest structural delta
|
||||
(negation-ne-elides yes) ; ne -> n' before vowel (n'ai pas vu)
|
||||
(negation-passe-composé "ne + aux + pas + participe") ; n'ai pas vu
|
||||
(negative-concord partial)) ; personne/rien as arguments post-participle
|
||||
@@ -1,70 +0,0 @@
|
||||
;;; lang_profile_it.el — Italian language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. Italian is a Romance sibling, so ~85% of the flags match ES/PT; the
|
||||
;;; essere/avere auxiliary split and phonological article selection are the deltas.
|
||||
|
||||
(lang_profile_it
|
||||
(language "Italian")
|
||||
(iso639 "it")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "il/lo/l'/i/gli + la/l'/le ; un/uno/un'/una") ; PHONOLOGICAL:
|
||||
; lo/gli/uno before s+cons, z, gn, ps, pn, x, y, i+V;
|
||||
; l'/un' before a vowel (elision, glued to next word)
|
||||
(article-drives-contraction yes) ; article choice feeds the prep+art contraction
|
||||
(adjective-position "postnominal-default + prenominal-class") ; bello/buono/grande
|
||||
; /nuovo/vecchio/primo... prenominal (with apocope)
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((di il del) (di lo dello) (di la della) (di i dei)
|
||||
(di gli degli) (di le delle) (di l' dell')
|
||||
(a il al) (a lo allo) (a la alla) (a i ai) (a gli agli)
|
||||
(a le alle) (a l' all')
|
||||
(da il dal) (da la dalla) (da gli dagli) (da l' dall')
|
||||
(in il nel) (in la nella) (in gli negli) (in l' nell')
|
||||
(su il sul) (su la sulla) (su gli sugli) (su l' sull')))
|
||||
(contraction-mandatory yes) ; *di il -> del is obligatory, never uncontracted
|
||||
(prep-no-contract (per tra fra)) ; per la strada (NOT *perla)
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-placement ((finite proclitic) ; lo vedo, non me lo dà
|
||||
(imperative-affirmative enclitic) ; dammelo, guardalo
|
||||
(imperative-negative-tu non+infinitive) ; non parlare / non lo fare
|
||||
(infinitive enclitic) ; vederlo, aiutarmi (drop -e)
|
||||
(gerund enclitic))) ; dandolo
|
||||
(clitic-combination ((mi lo "me lo") (ti lo "te lo") (ci lo "ce lo")
|
||||
(vi lo "ve lo") (si lo "se lo")
|
||||
(gli lo "glielo") (le lo "glielo"))) ; glielo = ONE word
|
||||
(clitic-particles (ci ne)) ; locative ci, partitive ne
|
||||
(raddoppiamento (da fa di va sta)) ; monosyllabic imper double clitic: dammelo
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (presente imperfetto passato-remoto futuro condizionale
|
||||
congiuntivo-presente congiuntivo-imperfetto imperativo))
|
||||
(compound-past "passato-prossimo = aux(present) + participle")
|
||||
(perfect-aux "essere/avere (LEXICAL selection)") ; << Italian-specific
|
||||
(essere-aux-class unaccusative) ; motion/change-of-state/copular/pronominal
|
||||
; (andare venire nascere morire diventare piacere
|
||||
; + ALL reflexives) -> essere
|
||||
(participle-agreement ((essere subject) ; è andata / sono arrivati
|
||||
(avere preceding-acc-clitic))) ; li ho visti
|
||||
(progressive-aux "stare + gerundio") ; sto parlando
|
||||
(copula "essere (default) / stare (state: sto bene)")
|
||||
(passive-aux "essere / venire (+ da-agent)")
|
||||
(future inflectional) ; parlerò, sarà
|
||||
(comparative "più/meno ADJ di")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/en) ───────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "non (preverbal) + concord") ; non...niente/nessuno/mai/più
|
||||
(negative-concord yes) ; preverbal negative word (nessuno/niente) suppresses non
|
||||
(neg-adverb-position between-aux-and-participle)) ; non ho MAI visto
|
||||
@@ -1,30 +0,0 @@
|
||||
;;; lang_profile_la.el — Latin language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Companion to morphology-la.el.
|
||||
|
||||
(lang_profile_la
|
||||
(language "Latin")
|
||||
(iso639 "la")
|
||||
(family "Italic")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; person carried by verb ending; subjects dropped
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f/n; adjective AGREES in case+gender+number
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph lat
|
||||
(articles none) ; Latin has no articles
|
||||
(case-system yes) ; NOM GEN DAT ACC ABL VOC (+ rare LOC)
|
||||
(cases (nom gen dat acc abl voc))
|
||||
(word-order "SOV (default; free order, case-marked)")
|
||||
(adjective-position "either (case agreement carries the link)")
|
||||
(adjective-agreement "case+gender+number")
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (1 2 3 3io 4)) ; four conjugations + i-stem 3rd
|
||||
(tenses (present imperfect future perfect pluperfect futureperfect))
|
||||
(moods (indicative subjunctive imperative infinitive))
|
||||
(voices (active passive))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(citation "principal parts: pres-1sg / pres-inf / perf-participle")
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted
|
||||
@@ -1,40 +0,0 @@
|
||||
;;; lang_profile_pt.el — Portuguese language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_es.
|
||||
|
||||
(lang_profile_pt
|
||||
(language "Portuguese")
|
||||
(iso639 "pt")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph por / kaikki
|
||||
(do-support no)
|
||||
(subject-aux-inversion no)
|
||||
(question-strategy intonation)
|
||||
(article-selection "o/a/os/as um/uma/uns/umas")
|
||||
(adjective-position postnominal)
|
||||
(adjective-agreement "gender+number")
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (prep + article) -----------------------------
|
||||
(contractions ((de o "do") (de a "da") (em o "no") (em a "na")
|
||||
(a o "ao") (a a "à") (por o "pelo") (por a "pela")))
|
||||
(contraction-mandatory yes)
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "ter") ; ter + past participle
|
||||
(copula-split "ser/estar")
|
||||
(personal-infinitive yes) ; distinctive PT inflected infinitive
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; mesoclisis/enclisis/proclisis by context
|
||||
(verb-prep-government yes)
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes))
|
||||
@@ -1,71 +0,0 @@
|
||||
;;; lang_profile_ro.el — Romanian language profile for ELP.
|
||||
;;; Romanian is the BIG typological delta of the Romance family. The verb/clause
|
||||
;;; engine and the SACRED negation contract mirror the ES/PT/IT core, but the
|
||||
;;; NOMINAL system is genuinely new: a SUFFIXED definite article, preserved CASE,
|
||||
;;; a NEUTER gender, and a VOCATIVE. Those flags mark where the shared engine was
|
||||
;;; extended rather than reused.
|
||||
|
||||
(lang_profile_ro
|
||||
(language "Romanian")
|
||||
(iso639 "ro")
|
||||
(family "Romance (Eastern / Balkan)")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m / f / NEUTER (n)
|
||||
(neuter-gender yes) ; << ROMANIAN-SPECIFIC: masc-agreeing SG, fem-agreeing PL
|
||||
; (un tren nou / două trenuri noi)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'
|
||||
(question-punct plain) ; ? and ! only
|
||||
|
||||
;; ── SUFFIXED DEFINITE ARTICLE (the headline engine extension) ───────────
|
||||
(definite-article suffixed) ; << UNIQUE IN ROMANCE: enclitic on the noun
|
||||
(definite-forms ((m/n sg "-ul / -le / -l : om->omul, câine->câinele, codru->codrul")
|
||||
(f sg "-a / -ea / -ua : casă->casa, carte->cartea, stea->steaua")
|
||||
(m pl "-i : oameni->oamenii")
|
||||
(f/n pl "-le : case->casele, trenuri->trenurile")))
|
||||
(article-host ((no-prenom-adj noun) ; omul bun
|
||||
(prenom-adj adjective))) ; bunul om (adj carries the article)
|
||||
(indefinite-article ((m/n "un") (f "o") (pl "niște") (gen/dat-pl "unor")))
|
||||
|
||||
;; ── CASE (preserved; NOM/ACC vs GEN/DAT) ────────────────────────────────
|
||||
(case (nom/acc gen/dat vocative)) ; << ROMANIAN-SPECIFIC
|
||||
(case-syncretism "nom=acc ; gen=dat")
|
||||
(genitive-marking "gen/dat definite: -lui (m/n), -ei/-i (f), -lor (pl)")
|
||||
(genitival-article ((m sg "al") (f sg "a") (m pl "ai") (f/n pl "ale"))) ; o carte a lui
|
||||
(possession "definite-head + gen/dat possessor: casa băiatului")
|
||||
(vocative ((m sg "-ule/-e : omule, băiete") (f sg "-o : Mario, fato")
|
||||
(pl "-lor")))
|
||||
|
||||
;; ── verb / aspect system ────────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (prezent imperfect perfect-simplu conjunctiv-prezent
|
||||
imperativ (periphrastic: perfect-compus viitor conditional)))
|
||||
(compound-past "perfectul compus = a-avea-clitic + INVARIABLE participle")
|
||||
(perfect-aux "a avea (am/ai/a/am/ați/au) — ONE auxiliary for ALL verbs")
|
||||
(perfect-aux-split no) ; << SIMPLER than Italian: no essere/avere selection
|
||||
(participle-agreement none) ; invariable in the perfect compus (agrees only as
|
||||
; an adjective / in the passive)
|
||||
(future "voi/vei/va/vom/veți/vor + infinitive (viitor literar)")
|
||||
(conditional "aș/ai/ar/am/ați/ar + infinitive")
|
||||
(subjunctive "conjunctiv: particle 'să' + subjunctive present")
|
||||
(modal-complement "modal + să + subjunctive (vreau să merg, poți să ajuți)")
|
||||
(copula "a fi")
|
||||
(passive "a fi + participle (participle AGREES like an adjective)")
|
||||
(comparative "mai / mai puțin ADJ decât")
|
||||
|
||||
;; ── clitic system (partial — see honest gaps) ───────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-set ((acc mă te îl o ne vă îi le) (dat îmi îți îi ne vă le)
|
||||
(refl mă te se ne vă se)))
|
||||
(clitic-placement ((finite proclitic) ; îmi place, o văd
|
||||
(perfect-compus elision) ; << m-am, l-am, i-am (PARTIAL)
|
||||
(imperative-affirmative enclitic))) ; dă-mi (PARTIAL)
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ─────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "nu (single preverbal marker) + concord")
|
||||
(negative-concord yes) ; nu … nimic / nimeni / niciodată / niciun
|
||||
(negative-imperative "nu + INFINITIVE : nu pleca! (KNOWN GAP: uses imperative stem)"))
|
||||
@@ -1,46 +1,46 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn lang_profile(code: String, word_order: String, morph_type: String, has_case: String, has_gender: String, script_dir: String, agreement: String, null_subject: String) -> [String]
|
||||
extern fn lang_get(profile: [String], key: String) -> String
|
||||
extern fn lang_profile_en() -> [String]
|
||||
extern fn lang_profile_ja() -> [String]
|
||||
extern fn lang_profile_ar() -> [String]
|
||||
extern fn lang_profile_zh() -> [String]
|
||||
extern fn lang_profile_de() -> [String]
|
||||
extern fn lang_profile_es() -> [String]
|
||||
extern fn lang_profile_fi() -> [String]
|
||||
extern fn lang_profile_sw() -> [String]
|
||||
extern fn lang_profile_hi() -> [String]
|
||||
extern fn lang_profile_ru() -> [String]
|
||||
extern fn lang_profile_fr() -> [String]
|
||||
extern fn lang_profile_la() -> [String]
|
||||
extern fn lang_profile_he() -> [String]
|
||||
extern fn lang_profile_sa() -> [String]
|
||||
extern fn lang_profile_got() -> [String]
|
||||
extern fn lang_profile_non() -> [String]
|
||||
extern fn lang_profile_enm() -> [String]
|
||||
extern fn lang_profile_pi() -> [String]
|
||||
extern fn lang_profile_grc() -> [String]
|
||||
extern fn lang_profile_ang() -> [String]
|
||||
extern fn lang_profile_fro() -> [String]
|
||||
extern fn lang_profile_goh() -> [String]
|
||||
extern fn lang_profile_sga() -> [String]
|
||||
extern fn lang_profile_txb() -> [String]
|
||||
extern fn lang_profile_peo() -> [String]
|
||||
extern fn lang_profile_akk() -> [String]
|
||||
extern fn lang_profile_uga() -> [String]
|
||||
extern fn lang_profile_egy() -> [String]
|
||||
extern fn lang_profile_sux() -> [String]
|
||||
extern fn lang_profile_gez() -> [String]
|
||||
extern fn lang_profile_cop() -> [String]
|
||||
extern fn lang_from_code(code: String) -> [String]
|
||||
extern fn lang_default() -> [String]
|
||||
extern fn lang_is_isolating(profile: [String]) -> Bool
|
||||
extern fn lang_is_agglutinative(profile: [String]) -> Bool
|
||||
extern fn lang_is_fusional(profile: [String]) -> Bool
|
||||
extern fn lang_is_polysynthetic(profile: [String]) -> Bool
|
||||
extern fn lang_is_rtl(profile: [String]) -> Bool
|
||||
extern fn lang_has_null_subject(profile: [String]) -> Bool
|
||||
extern fn lang_has_case(profile: [String]) -> Bool
|
||||
extern fn lang_has_gender(profile: [String]) -> Bool
|
||||
extern fn lang_word_order(profile: [String]) -> String
|
||||
extern fn lang_code(profile: [String]) -> String
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn lang_profile(code: String, word_order: String, morph_type: String, has_case: String, has_gender: String, script_dir: String, agreement: String, null_subject: String) -> Any
|
||||
extern fn lang_get(profile: Any, key: String) -> String
|
||||
extern fn lang_profile_en() -> Any
|
||||
extern fn lang_profile_ja() -> Any
|
||||
extern fn lang_profile_ar() -> Any
|
||||
extern fn lang_profile_zh() -> Any
|
||||
extern fn lang_profile_de() -> Any
|
||||
extern fn lang_profile_es() -> Any
|
||||
extern fn lang_profile_fi() -> Any
|
||||
extern fn lang_profile_sw() -> Any
|
||||
extern fn lang_profile_hi() -> Any
|
||||
extern fn lang_profile_ru() -> Any
|
||||
extern fn lang_profile_fr() -> Any
|
||||
extern fn lang_profile_la() -> Any
|
||||
extern fn lang_profile_he() -> Any
|
||||
extern fn lang_profile_sa() -> Any
|
||||
extern fn lang_profile_got() -> Any
|
||||
extern fn lang_profile_non() -> Any
|
||||
extern fn lang_profile_enm() -> Any
|
||||
extern fn lang_profile_pi() -> Any
|
||||
extern fn lang_profile_grc() -> Any
|
||||
extern fn lang_profile_ang() -> Any
|
||||
extern fn lang_profile_fro() -> Any
|
||||
extern fn lang_profile_goh() -> Any
|
||||
extern fn lang_profile_sga() -> Any
|
||||
extern fn lang_profile_txb() -> Any
|
||||
extern fn lang_profile_peo() -> Any
|
||||
extern fn lang_profile_akk() -> Any
|
||||
extern fn lang_profile_uga() -> Any
|
||||
extern fn lang_profile_egy() -> Any
|
||||
extern fn lang_profile_sux() -> Any
|
||||
extern fn lang_profile_gez() -> Any
|
||||
extern fn lang_profile_cop() -> Any
|
||||
extern fn lang_from_code(code: String) -> Any
|
||||
extern fn lang_default() -> Any
|
||||
extern fn lang_is_isolating(profile: Any) -> Bool
|
||||
extern fn lang_is_agglutinative(profile: Any) -> Bool
|
||||
extern fn lang_is_fusional(profile: Any) -> Bool
|
||||
extern fn lang_is_polysynthetic(profile: Any) -> Bool
|
||||
extern fn lang_is_rtl(profile: Any) -> Bool
|
||||
extern fn lang_has_null_subject(profile: Any) -> Bool
|
||||
extern fn lang_has_case(profile: Any) -> Bool
|
||||
extern fn lang_has_gender(profile: Any) -> Bool
|
||||
extern fn lang_word_order(profile: Any) -> String
|
||||
extern fn lang_code(profile: Any) -> String
|
||||
|
||||
@@ -56,7 +56,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn akk_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn akk_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn akk_str_len(s: String) -> Int
|
||||
extern fn akk_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -36,7 +36,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn ang_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn ang_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn ang_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn ang_str_last_char(s: String) -> String
|
||||
|
||||
@@ -21,7 +21,6 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn ar_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn ar_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn ar_str_len(s: String) -> Int
|
||||
extern fn ar_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -54,7 +54,6 @@
|
||||
|
||||
// ── String helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn cop_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn cop_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn cop_str_len(s: String) -> Int
|
||||
extern fn cop_drop(s: String, n: Int) -> String
|
||||
|
||||
@@ -26,7 +26,6 @@
|
||||
// Dat: dem der dem den
|
||||
// Gen: des der des der
|
||||
|
||||
import "morphology.el"
|
||||
fn de_article_def(gender: String, gram_case: String, number: String) -> String {
|
||||
if str_eq(number, "pl") {
|
||||
if str_eq(gram_case, "nom") { return "die" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn de_article_def(gender: String, gram_case: String, number: String) -> String
|
||||
extern fn de_article_indef(gender: String, gram_case: String, number: String) -> String
|
||||
extern fn de_article(gender: String, gram_case: String, number: String, definite: String) -> String
|
||||
|
||||
@@ -52,7 +52,6 @@
|
||||
|
||||
// ── String helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn egy_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn egy_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn egy_str_len(s: String) -> Int
|
||||
extern fn egy_drop(s: String, n: Int) -> String
|
||||
|
||||
@@ -31,7 +31,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn enm_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn enm_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn enm_drop(s: String, n: Int) -> String
|
||||
extern fn enm_first_char(s: String) -> String
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
|
||||
// ── String helpers (local, matching morphology.el conventions) ────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn es_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn es_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn es_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn es_str_last_char(s: String) -> String
|
||||
|
||||
@@ -25,7 +25,6 @@
|
||||
// If only neutral vowels are found, default to "front" (the conservative choice
|
||||
// for borrowed words and those without clear back vowels).
|
||||
|
||||
import "morphology.el"
|
||||
fn fi_harmony(word: String) -> String {
|
||||
let n: Int = str_len(word)
|
||||
let i: Int = n - 1
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn fi_harmony(word: String) -> String
|
||||
extern fn fi_suffix(base: String, harmony: String) -> String
|
||||
extern fn fi_noun_case(stem: String, gram_case: String, number: String, harmony: String) -> String
|
||||
extern fn fi_str_last_char(s: String) -> String
|
||||
extern fn fi_apply_case(noun: String, gram_case: String, number: String) -> String
|
||||
extern fn fi_verb_stem(dict_form: String) -> String
|
||||
extern fn fi_irregular_verb(dict_form: String) -> [String]
|
||||
extern fn fi_irregular_verb(dict_form: String) -> Any
|
||||
extern fn fi_present_ending(stem: String, person: String, number: String, harmony: String) -> String
|
||||
extern fn fi_past_stem(stem: String) -> String
|
||||
extern fn fi_past_ending(stem: String, person: String, number: String, harmony: String) -> String
|
||||
@@ -14,4 +14,4 @@ extern fn fi_negative(verb: String, person: String, number: String) -> String
|
||||
extern fn fi_conjugate(verb: String, tense: String, person: String, number: String) -> String
|
||||
extern fn fi_question_suffix(harmony: String) -> String
|
||||
extern fn fi_make_question(verb_form: String, harmony: String) -> String
|
||||
extern fn fi_full_paradigm(noun: String) -> [String]
|
||||
extern fn fi_full_paradigm(noun: String) -> Any
|
||||
|
||||
@@ -19,7 +19,6 @@
|
||||
|
||||
// ── String helpers (local, matching morphology.el conventions) ────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn fr_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn fr_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn fr_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn fr_str_last_char(s: String) -> String
|
||||
|
||||
@@ -53,7 +53,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn fro_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn fro_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn fro_drop(s: String, n: Int) -> String
|
||||
extern fn fro_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -64,7 +64,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn gez_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn gez_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn gez_str_len(s: String) -> Int
|
||||
extern fn gez_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -48,7 +48,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn goh_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn goh_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn goh_drop(s: String, n: Int) -> String
|
||||
extern fn goh_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -49,7 +49,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn got_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn got_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn got_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn got_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -31,7 +31,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn grc_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn grc_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn grc_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn grc_str_last_char(s: String) -> String
|
||||
|
||||
@@ -51,7 +51,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn he_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn he_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn he_str_len(s: String) -> Int
|
||||
extern fn he_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -24,7 +24,6 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn hi_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn hi_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn hi_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn hi_str_last_char(s: String) -> String
|
||||
|
||||
@@ -23,7 +23,6 @@
|
||||
// Note: this is a heuristic classifier for romanized input. For production use
|
||||
// with native kana/kanji forms, the dictionary form (辞書形) must be consulted.
|
||||
|
||||
import "morphology.el"
|
||||
fn ja_verb_group(dict_form: String) -> String {
|
||||
// Irregular verbs (exact match on dictionary form)
|
||||
if str_eq(dict_form, "する") { return "irregular" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn ja_verb_group(dict_form: String) -> String
|
||||
extern fn ja_ichidan_stem(dict_form: String) -> String
|
||||
extern fn ja_godan_stem_change(dict_form: String, row: String) -> String
|
||||
|
||||
@@ -25,7 +25,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn la_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn la_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn la_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn la_str_last_char(s: String) -> String
|
||||
|
||||
@@ -27,7 +27,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn non_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn non_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn non_drop(s: String, n: Int) -> String
|
||||
extern fn non_last(s: String) -> String
|
||||
|
||||
@@ -31,7 +31,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn peo_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn peo_drop(s: String, n: Int) -> String
|
||||
extern fn peo_ends(s: String, suf: String) -> Bool
|
||||
extern fn peo_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -30,7 +30,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn pi_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn pi_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn pi_drop(s: String, n: Int) -> String
|
||||
extern fn pi_last_char(s: String) -> String
|
||||
|
||||
@@ -35,7 +35,6 @@
|
||||
// The heuristic returns the most probable gender. Caller should override
|
||||
// for known exceptions (путь, рубль are masc despite -ь).
|
||||
|
||||
import "morphology.el"
|
||||
fn ru_gender(noun: String) -> String {
|
||||
let n: Int = str_len(noun)
|
||||
if n == 0 { return "m" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn ru_gender(noun: String) -> String
|
||||
extern fn ru_stem_type(noun: String, gender: String) -> String
|
||||
extern fn ru_noun_case(noun: String, gender: String, gram_case: String, number: String) -> String
|
||||
|
||||
@@ -42,7 +42,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sa_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sa_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sa_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sa_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -31,7 +31,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sga_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sga_drop(s: String, n: Int) -> String
|
||||
extern fn sga_first(s: String) -> String
|
||||
extern fn sga_rest(s: String) -> String
|
||||
|
||||
@@ -53,7 +53,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sux_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sux_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sux_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sux_str_last_char(s: String) -> String
|
||||
|
||||
@@ -24,7 +24,6 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sw_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sw_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sw_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sw_str_first_char(s: String) -> String
|
||||
|
||||
@@ -30,7 +30,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn txb_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn txb_drop(s: String, n: Int) -> String
|
||||
extern fn txb_ends(s: String, suf: String) -> Bool
|
||||
extern fn txb_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -48,7 +48,6 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn uga_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn uga_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn uga_str_len(s: String) -> Int
|
||||
extern fn uga_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -33,17 +33,6 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "language-profile.el"
|
||||
import "morphology-es.el"
|
||||
import "morphology-fr.el"
|
||||
import "morphology-de.el"
|
||||
import "morphology-ru.el"
|
||||
import "morphology-fi.el"
|
||||
import "morphology-ar.el"
|
||||
import "morphology-hi.el"
|
||||
import "morphology-sw.el"
|
||||
import "morphology-la.el"
|
||||
import "morphology-ja.el"
|
||||
fn str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
@@ -250,7 +239,6 @@ fn en_irregular_verb(base: String) -> [String] {
|
||||
if str_eq(base, "cut") { let r: [String] = ["cut", "cuts", "cut", "cut", "cutting"]; return r }
|
||||
if str_eq(base, "set") { let r: [String] = ["set", "sets", "set", "set", "setting"]; return r }
|
||||
if str_eq(base, "hit") { let r: [String] = ["hit", "hits", "hit", "hit", "hitting"]; return r }
|
||||
if str_eq(base, "fight") { let r: [String] = ["fight", "fights","fought", "fought", "fighting"]; return r }
|
||||
return empty
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn str_ends(s: String, suf: String) -> Bool
|
||||
extern fn str_last_char(s: String) -> String
|
||||
extern fn str_last2(s: String) -> String
|
||||
@@ -8,7 +8,7 @@ extern fn is_vowel(c: String) -> Bool
|
||||
extern fn morph_apply_suffix(base: String, suffix: String) -> String
|
||||
extern fn en_irregular_plural(word: String) -> String
|
||||
extern fn en_irregular_singular(word: String) -> String
|
||||
extern fn en_irregular_verb(base: String) -> [String]
|
||||
extern fn en_irregular_verb(base: String) -> Any
|
||||
extern fn en_verb_3sg(base: String) -> String
|
||||
extern fn en_should_double_final(base: String) -> Bool
|
||||
extern fn en_verb_past(base: String) -> String
|
||||
@@ -16,10 +16,10 @@ extern fn en_verb_gerund(base: String) -> String
|
||||
extern fn en_pluralize_regular(singular: String) -> String
|
||||
extern fn en_verb_form(base: String, tense: String, person: String, number: String) -> String
|
||||
extern fn agree_determiner(det: String, noun: String) -> String
|
||||
extern fn morph_pluralize(noun: String, profile: [String]) -> String
|
||||
extern fn morph_pluralize(noun: String, profile: Any) -> String
|
||||
extern fn morph_map_canonical(verb: String, code: String) -> String
|
||||
extern fn morph_conjugate(verb: String, tense: String, person: String, number: String, profile: [String]) -> String
|
||||
extern fn morph_inflect(word: String, features: String, profile: [String]) -> String
|
||||
extern fn morph_conjugate(verb: String, tense: String, person: String, number: String, profile: Any) -> String
|
||||
extern fn morph_inflect(word: String, features: String, profile: Any) -> String
|
||||
extern fn pluralize(singular: String) -> String
|
||||
extern fn singularize(plural: String) -> String
|
||||
extern fn verb_form(base: String, tense: String, person: String, number: String) -> String
|
||||
|
||||
@@ -1,280 +0,0 @@
|
||||
// multilingual.el - the language layer for the native-el interlocutor.
|
||||
//
|
||||
// Deterministic, NO generative model (ports multilingual.py):
|
||||
// 1. ml_detect(text) -> ISO code (en/es/pt/it) via stopword + diacritic score
|
||||
// 2. ml_tr(key, lang) -> localized fixed phrase (SACRED per-language yes/no/decline)
|
||||
// 3. ml_term(w, lang) -> PT/ES content term -> EN engram equivalent
|
||||
// 4. ml_translate_pred(lemma, lang) -> EN predicate lemma -> target infinitive
|
||||
//
|
||||
// The Python detector count-weights stopwords and diacritics; here diacritics are
|
||||
// scored by PRESENCE (str_contains) rather than codepoint counting, to stay clear
|
||||
// of UTF-8 index hazards in the runtime. Faithful enough to classify typical
|
||||
// queries; documented simplification. Depends on: comprehend (cp_tokenize).
|
||||
|
||||
// ── 1. language detection ─────────────────────────────────────────────────────
|
||||
|
||||
fn ml_stop_en(w: String) -> Bool {
|
||||
if str_eq(w, "the") { return true }
|
||||
if str_eq(w, "does") { return true }
|
||||
if str_eq(w, "do") { return true }
|
||||
if str_eq(w, "did") { return true }
|
||||
if str_eq(w, "what") { return true }
|
||||
if str_eq(w, "who") { return true }
|
||||
if str_eq(w, "is") { return true }
|
||||
if str_eq(w, "are") { return true }
|
||||
if str_eq(w, "how") { return true }
|
||||
if str_eq(w, "you") { return true }
|
||||
if str_eq(w, "your") { return true }
|
||||
if str_eq(w, "of") { return true }
|
||||
if str_eq(w, "to") { return true }
|
||||
if str_eq(w, "and") { return true }
|
||||
if str_eq(w, "for") { return true }
|
||||
if str_eq(w, "explain") { return true }
|
||||
if str_eq(w, "answer") { return true }
|
||||
if str_eq(w, "memory") { return true }
|
||||
if str_eq(w, "with") { return true }
|
||||
if str_eq(w, "not") { return true }
|
||||
if str_eq(w, "store") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_es(w: String) -> Bool {
|
||||
if str_eq(w, "que") { return true }
|
||||
if str_eq(w, "qué") { return true }
|
||||
if str_eq(w, "una") { return true }
|
||||
if str_eq(w, "usted") { return true }
|
||||
if str_eq(w, "su") { return true }
|
||||
if str_eq(w, "cómo") { return true }
|
||||
if str_eq(w, "como") { return true }
|
||||
if str_eq(w, "cuál") { return true }
|
||||
if str_eq(w, "quién") { return true }
|
||||
if str_eq(w, "está") { return true }
|
||||
if str_eq(w, "es") { return true }
|
||||
if str_eq(w, "los") { return true }
|
||||
if str_eq(w, "las") { return true }
|
||||
if str_eq(w, "del") { return true }
|
||||
if str_eq(w, "al") { return true }
|
||||
if str_eq(w, "explica") { return true }
|
||||
if str_eq(w, "explique") { return true }
|
||||
if str_eq(w, "forma") { return true }
|
||||
if str_eq(w, "con") { return true }
|
||||
if str_eq(w, "memoria") { return true }
|
||||
if str_eq(w, "responde") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_pt(w: String) -> Bool {
|
||||
if str_eq(w, "que") { return true }
|
||||
if str_eq(w, "uma") { return true }
|
||||
if str_eq(w, "você") { return true }
|
||||
if str_eq(w, "sua") { return true }
|
||||
if str_eq(w, "seu") { return true }
|
||||
if str_eq(w, "como") { return true }
|
||||
if str_eq(w, "memória") { return true }
|
||||
if str_eq(w, "isso") { return true }
|
||||
if str_eq(w, "os") { return true }
|
||||
if str_eq(w, "as") { return true }
|
||||
if str_eq(w, "da") { return true }
|
||||
if str_eq(w, "do") { return true }
|
||||
if str_eq(w, "na") { return true }
|
||||
if str_eq(w, "no") { return true }
|
||||
if str_eq(w, "explica") { return true }
|
||||
if str_eq(w, "forma") { return true }
|
||||
if str_eq(w, "é") { return true }
|
||||
if str_eq(w, "está") { return true }
|
||||
if str_eq(w, "com") { return true }
|
||||
if str_eq(w, "responda") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_it(w: String) -> Bool {
|
||||
if str_eq(w, "che") { return true }
|
||||
if str_eq(w, "una") { return true }
|
||||
if str_eq(w, "come") { return true }
|
||||
if str_eq(w, "della") { return true }
|
||||
if str_eq(w, "gli") { return true }
|
||||
if str_eq(w, "è") { return true }
|
||||
if str_eq(w, "sono") { return true }
|
||||
if str_eq(w, "questo") { return true }
|
||||
if str_eq(w, "nel") { return true }
|
||||
if str_eq(w, "di") { return true }
|
||||
if str_eq(w, "il") { return true }
|
||||
if str_eq(w, "cosa") { return true }
|
||||
if str_eq(w, "per") { return true }
|
||||
if str_eq(w, "memoria") { return true }
|
||||
if str_eq(w, "spiega") { return true }
|
||||
if str_eq(w, "rispondi") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// diacritic PRESENCE score (weight 3 each; hard overrides weight 8).
|
||||
fn ml_dia_score(low: String, lang: String) -> Int {
|
||||
let s: Int = 0
|
||||
if str_eq(lang, "pt") {
|
||||
if str_contains(low, "ã") { let s = s + 3 }
|
||||
if str_contains(low, "õ") { let s = s + 3 }
|
||||
if str_contains(low, "ç") { let s = s + 3 }
|
||||
if str_contains(low, "ê") { let s = s + 3 }
|
||||
if str_contains(low, "á") { let s = s + 3 }
|
||||
// hard PT markers (ã/õ almost never appear outside PT)
|
||||
if str_contains(low, "ã") { let s = s + 8 }
|
||||
if str_contains(low, "õ") { let s = s + 8 }
|
||||
}
|
||||
if str_eq(lang, "es") {
|
||||
if str_contains(low, "ñ") { let s = s + 3 }
|
||||
if str_contains(low, "¿") { let s = s + 3 }
|
||||
if str_contains(low, "¡") { let s = s + 3 }
|
||||
if str_contains(low, "á") { let s = s + 3 }
|
||||
if str_contains(low, "é") { let s = s + 3 }
|
||||
// hard ES markers
|
||||
if str_contains(low, "ñ") { let s = s + 8 }
|
||||
if str_contains(low, "¿") { let s = s + 8 }
|
||||
if str_contains(low, "¡") { let s = s + 8 }
|
||||
}
|
||||
if str_eq(lang, "it") {
|
||||
if str_contains(low, "è") { let s = s + 3 }
|
||||
if str_contains(low, "ì") { let s = s + 3 }
|
||||
if str_contains(low, "ò") { let s = s + 3 }
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
fn ml_stop_score(toks: [String], lang: String) -> Int {
|
||||
let n: Int = native_list_len(toks)
|
||||
let s: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let w: String = native_list_get(toks, i)
|
||||
if str_eq(lang, "en") { if ml_stop_en(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "es") { if ml_stop_es(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "pt") { if ml_stop_pt(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "it") { if ml_stop_it(w) { let s = s + 2 } }
|
||||
let i = i + 1
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
fn ml_detect(text: String) -> String {
|
||||
if str_eq(text, "") { return "en" }
|
||||
let low: String = str_to_lower(text)
|
||||
let toks: [String] = cp_tokenize(text)
|
||||
// NOTE: el's overloaded `+` mis-compiles two chained function-call Int operands
|
||||
// as string concat (documented in comprehend_gate.el). Bind each call to an Int
|
||||
// var and add vars one at a time so the addition stays integer.
|
||||
let en: Int = ml_stop_score(toks, "en")
|
||||
let es_s: Int = ml_stop_score(toks, "es")
|
||||
let es_d: Int = ml_dia_score(low, "es")
|
||||
let es: Int = es_s + es_d
|
||||
let pt_s: Int = ml_stop_score(toks, "pt")
|
||||
let pt_d: Int = ml_dia_score(low, "pt")
|
||||
let pt: Int = pt_s + pt_d
|
||||
let it_s: Int = ml_stop_score(toks, "it")
|
||||
let it_d: Int = ml_dia_score(low, "it")
|
||||
let it: Int = it_s + it_d
|
||||
|
||||
let best: String = "en"
|
||||
let bs: Int = en
|
||||
if es > bs { let best = "es"; let bs = es }
|
||||
if pt > bs { let best = "pt"; let bs = pt }
|
||||
if it > bs { let best = "it"; let bs = it }
|
||||
// weak signal -> honest fallback to English
|
||||
if bs < 3 { return "en" }
|
||||
return best
|
||||
}
|
||||
|
||||
// ── 2. localized fixed phrases (SACRED per-language decline/yes/no) ────────────
|
||||
|
||||
fn ml_tr(key: String, lang: String) -> String {
|
||||
if str_eq(key, "no_memory") {
|
||||
if str_eq(lang, "pt") { return "Não tenho isso na minha memória." }
|
||||
if str_eq(lang, "es") { return "No tengo eso en mi memoria." }
|
||||
if str_eq(lang, "it") { return "Non ho quello nella mia memoria." }
|
||||
return "I don't have that in my memory."
|
||||
}
|
||||
if str_eq(key, "parse_fail") {
|
||||
if str_eq(lang, "pt") { return "Não consegui interpretar isso." }
|
||||
if str_eq(lang, "es") { return "No pude interpretar eso." }
|
||||
if str_eq(lang, "it") { return "Non sono riuscito a interpretarlo." }
|
||||
return "I didn't parse that."
|
||||
}
|
||||
if str_eq(key, "yes") {
|
||||
if str_eq(lang, "pt") { return "Sim" }
|
||||
if str_eq(lang, "es") { return "Sí" }
|
||||
if str_eq(lang, "it") { return "Sì" }
|
||||
return "Yes"
|
||||
}
|
||||
if str_eq(key, "no") {
|
||||
if str_eq(lang, "pt") { return "Não" }
|
||||
if str_eq(lang, "es") { return "No" }
|
||||
if str_eq(lang, "it") { return "No" }
|
||||
return "No"
|
||||
}
|
||||
if str_eq(key, "identity") {
|
||||
if str_eq(lang, "pt") { return "Sou o Neuron, o engrama com quem você está falando." }
|
||||
if str_eq(lang, "es") { return "Soy Neuron, el engrama con el que estás hablando." }
|
||||
if str_eq(lang, "it") { return "Sono Neuron, l'engramma con cui stai parlando." }
|
||||
return "I'm Neuron, the engram you're speaking with."
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// ── 3. retrieval term lexicon (PT/ES content term -> EN engram equivalent) ─────
|
||||
|
||||
fn ml_term(w: String, lang: String) -> String {
|
||||
if str_eq(lang, "en") { return w }
|
||||
if str_eq(w, "saliência") { return "salience" }
|
||||
if str_eq(w, "saliencia") { return "salience" }
|
||||
if str_eq(w, "memória") { return "memory" }
|
||||
if str_eq(w, "memoria") { return "memory" }
|
||||
if str_eq(w, "geometria") { return "geometry" }
|
||||
if str_eq(w, "geometrias") { return "geometry" }
|
||||
if str_eq(w, "geometrías") { return "geometry" }
|
||||
if str_eq(w, "forma") { return "form" }
|
||||
if str_eq(w, "consolidação") { return "consolidation" }
|
||||
if str_eq(w, "consolidación") { return "consolidation" }
|
||||
if str_eq(w, "aprendizagem") { return "learning" }
|
||||
if str_eq(w, "aprendizaje") { return "learning" }
|
||||
if str_eq(w, "nó") { return "node" }
|
||||
if str_eq(w, "nodo") { return "node" }
|
||||
if str_eq(w, "armazenamento") { return "storage" }
|
||||
if str_eq(w, "almacenamiento") { return "storage" }
|
||||
if str_eq(w, "estrutura") { return "structure" }
|
||||
if str_eq(w, "estructura") { return "structure" }
|
||||
return w
|
||||
}
|
||||
|
||||
// ── 4. predicate translation (EN lemma -> target infinitive; pass-through) ─────
|
||||
|
||||
fn ml_translate_pred(lemma: String, lang: String) -> String {
|
||||
if str_eq(lang, "en") { return lemma }
|
||||
if str_eq(lang, "es") {
|
||||
if str_eq(lemma, "store") { return "almacenar" }
|
||||
if str_eq(lemma, "use") { return "usar" }
|
||||
if str_eq(lemma, "have") { return "tener" }
|
||||
if str_eq(lemma, "be") { return "ser" }
|
||||
if str_eq(lemma, "give") { return "dar" }
|
||||
if str_eq(lemma, "make") { return "hacer" }
|
||||
if str_eq(lemma, "learn") { return "aprender" }
|
||||
if str_eq(lemma, "form") { return "formar" }
|
||||
return lemma
|
||||
}
|
||||
if str_eq(lang, "pt") {
|
||||
if str_eq(lemma, "store") { return "armazenar" }
|
||||
if str_eq(lemma, "use") { return "usar" }
|
||||
if str_eq(lemma, "have") { return "ter" }
|
||||
if str_eq(lemma, "be") { return "ser" }
|
||||
if str_eq(lemma, "give") { return "dar" }
|
||||
if str_eq(lemma, "make") { return "fazer" }
|
||||
if str_eq(lemma, "learn") { return "aprender" }
|
||||
if str_eq(lemma, "form") { return "formar" }
|
||||
return lemma
|
||||
}
|
||||
if str_eq(lang, "it") {
|
||||
if str_eq(lemma, "store") { return "memorizzare" }
|
||||
if str_eq(lemma, "use") { return "usare" }
|
||||
if str_eq(lemma, "have") { return "avere" }
|
||||
if str_eq(lemma, "be") { return "essere" }
|
||||
return lemma
|
||||
}
|
||||
return lemma
|
||||
}
|
||||
@@ -1,140 +0,0 @@
|
||||
// propositions.el - the READ primitive over the engram's OWN memories, native el.
|
||||
//
|
||||
// Free memory text -> structured PROPOSITIONS (triples):
|
||||
// (subject, predicate, object, modifiers, polarity, tense, source, confidence)
|
||||
//
|
||||
// This is comprehension turned inward: the Python reference (propositions.py) ran
|
||||
// spaCy's dependency parser over each memory sentence and walked the arcs. Here
|
||||
// the spaCy role is filled by the el-native parser (comprehend.el / parse_spec):
|
||||
// each sentence is parsed to a meaning-spec, and the spec's roles ARE the triple.
|
||||
// Nothing generates text. NEGATION IS SACRED: polarity flows straight from the
|
||||
// spec's polarity field and is never dropped or inverted.
|
||||
//
|
||||
// Depends on: comprehend (parse_spec / parse_spec_lang), grammar (slots_get).
|
||||
|
||||
// ── sentence segmentation ─────────────────────────────────────────────────────
|
||||
// Split on sentence-final punctuation (. ! ?) and hard newlines. Markdown/long
|
||||
// memories are handled shallowly (the reference caps + ranks by query overlap;
|
||||
// that ranking belongs to the dialogue layer, not here).
|
||||
|
||||
fn prop_is_boundary(c: String) -> Bool {
|
||||
if str_eq(c, ".") { return true }
|
||||
if str_eq(c, "!") { return true }
|
||||
if str_eq(c, "?") { return true }
|
||||
if str_eq(c, "\n") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn prop_split_sentences(text: String) -> [String] {
|
||||
let out: [String] = native_list_empty()
|
||||
let n: Int = str_len(text)
|
||||
let start: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let c: String = str_slice(text, i, i + 1)
|
||||
if prop_is_boundary(c) {
|
||||
let seg: String = str_slice(text, start, i + 1)
|
||||
let trimmed: String = cp_trim_punct(seg)
|
||||
if !str_eq(trimmed, "") {
|
||||
let out = native_list_append(out, seg)
|
||||
}
|
||||
let start = i + 1
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
if start < n {
|
||||
let seg: String = str_slice(text, start, n)
|
||||
let trimmed: String = cp_trim_punct(seg)
|
||||
if !str_eq(trimmed, "") {
|
||||
let out = native_list_append(out, seg)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ── spec -> proposition record ────────────────────────────────────────────────
|
||||
// A proposition is a slot map (same [String] shape as the spec) with the READ
|
||||
// contract keys. Modifiers fold the spec's location + iobj adjuncts.
|
||||
|
||||
fn prop_confidence(subject: String, predicate: String, object: String) -> String {
|
||||
if str_eq(predicate, "") { return "0.0" }
|
||||
if str_eq(subject, "") { return "0.4" }
|
||||
if str_eq(object, "") { return "0.7" }
|
||||
return "1.0"
|
||||
}
|
||||
|
||||
fn prop_modifiers(spec: [String]) -> String {
|
||||
let loc: String = slots_get(spec, "location")
|
||||
let iobj: String = slots_get(spec, "iobj")
|
||||
let parts: [String] = native_list_empty()
|
||||
if !str_eq(loc, "") { let parts = native_list_append(parts, loc) }
|
||||
if !str_eq(iobj, "") { let parts = native_list_append(parts, "to " + iobj) }
|
||||
return str_join(parts, "; ")
|
||||
}
|
||||
|
||||
fn prop_from_spec(spec: [String], source_id: String) -> [String] {
|
||||
let subject: String = slots_get(spec, "agent")
|
||||
let predicate: String = slots_get(spec, "predicate")
|
||||
let object: String = slots_get(spec, "patient")
|
||||
let polarity: String = slots_get(spec, "polarity")
|
||||
let tense: String = slots_get(spec, "tense")
|
||||
let mods: String = prop_modifiers(spec)
|
||||
let conf: String = prop_confidence(subject, predicate, object)
|
||||
|
||||
let p: [String] = native_list_empty()
|
||||
let p = native_list_append(p, "subject"); let p = native_list_append(p, subject)
|
||||
let p = native_list_append(p, "predicate"); let p = native_list_append(p, predicate)
|
||||
let p = native_list_append(p, "object"); let p = native_list_append(p, object)
|
||||
let p = native_list_append(p, "modifiers"); let p = native_list_append(p, mods)
|
||||
let p = native_list_append(p, "polarity"); let p = native_list_append(p, polarity)
|
||||
let p = native_list_append(p, "tense"); let p = native_list_append(p, tense)
|
||||
let p = native_list_append(p, "source"); let p = native_list_append(p, source_id)
|
||||
let p = native_list_append(p, "confidence"); let p = native_list_append(p, conf)
|
||||
return p
|
||||
}
|
||||
|
||||
// Extract one proposition from a single sentence (given language).
|
||||
fn prop_extract_one_lang(sentence: String, lang: String, source_id: String) -> [String] {
|
||||
let spec: [String] = parse_spec_lang(sentence, lang)
|
||||
return prop_from_spec(spec, source_id)
|
||||
}
|
||||
|
||||
fn prop_extract_one(sentence: String, source_id: String) -> [String] {
|
||||
return prop_extract_one_lang(sentence, "en", source_id)
|
||||
}
|
||||
|
||||
// Render a proposition as a compact trace line (repr parity with propositions.py).
|
||||
fn prop_repr(p: [String]) -> String {
|
||||
let neg: String = ""
|
||||
if str_eq(slots_get(p, "polarity"), "neg") { let neg = "NOT " }
|
||||
let mods: String = slots_get(p, "modifiers")
|
||||
let modstr: String = ""
|
||||
if !str_eq(mods, "") { let modstr = " [" + mods + "]" }
|
||||
let s: String = "(" + slots_get(p, "subject") + " -" + neg + slots_get(p, "predicate")
|
||||
let s = s + "-> " + slots_get(p, "object") + modstr
|
||||
let s = s + " conf=" + slots_get(p, "confidence") + ")"
|
||||
return s
|
||||
}
|
||||
|
||||
// Extract all propositions from a memory's text (one per sentence). Returns a
|
||||
// flat [String] whose entries are the prop_repr trace lines, in reading order.
|
||||
fn prop_extract_lang(text: String, lang: String, source_id: String) -> [String] {
|
||||
let sents: [String] = prop_split_sentences(text)
|
||||
let m: Int = native_list_len(sents)
|
||||
let out: [String] = native_list_empty()
|
||||
let i: Int = 0
|
||||
while i < m {
|
||||
let sent: String = native_list_get(sents, i)
|
||||
let p: [String] = prop_extract_one_lang(sent, lang, source_id)
|
||||
// drop empty parses (no predicate recovered): honest partial, not noise.
|
||||
if !str_eq(slots_get(p, "predicate"), "") {
|
||||
let out = native_list_append(out, prop_repr(p))
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
fn prop_extract(text: String, source_id: String) -> [String] {
|
||||
return prop_extract_lang(text, "en", source_id)
|
||||
}
|
||||
@@ -248,56 +248,6 @@ fn add_punct(s: String, intent: String) -> String {
|
||||
return s + "."
|
||||
}
|
||||
|
||||
// ── Polarity-aware negation (SACRED field honored on the generation side) ─────
|
||||
//
|
||||
// Negation must never be dropped between comprehension and realization. The
|
||||
// meaning-spec carries an explicit "polarity" field ("aff"|"neg") and optional
|
||||
// "neg_word" (standalone negative adverb, e.g. "never"). English uses
|
||||
// do-support ("did not see") or preverbal adverb ("never fought"); copular "be"
|
||||
// takes post-verbal "not"; other languages get a preverbal negator particle.
|
||||
|
||||
fn realize_negator(code: String) -> String {
|
||||
if str_eq(code, "es") { return "no" }
|
||||
if str_eq(code, "pt") { return "não" }
|
||||
if str_eq(code, "ca") { return "no" }
|
||||
if str_eq(code, "it") { return "non" }
|
||||
if str_eq(code, "fr") { return "ne" }
|
||||
if str_eq(code, "de") { return "nicht" }
|
||||
if str_eq(code, "ro") { return "nu" }
|
||||
return "not"
|
||||
}
|
||||
|
||||
fn realize_assert_neg_en(predicate: String, tense: String, person: String, number: String, agent: String, patient: String, iobj: String, location: String, neg_word: String, profile: [String]) -> String {
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, agent)
|
||||
if !str_eq(neg_word, "") {
|
||||
// adverbial negation: "I never fought the ocean."
|
||||
let verb_surf: String = morph_conjugate(predicate, tense, person, number, profile)
|
||||
let parts = native_list_append(parts, neg_word)
|
||||
let parts = native_list_append(parts, verb_surf)
|
||||
} else {
|
||||
if str_eq(predicate, "be") {
|
||||
// copular: "she was not a monster"
|
||||
let be_form: String = morph_conjugate("be", tense, person, number, profile)
|
||||
let parts = native_list_append(parts, be_form)
|
||||
let parts = native_list_append(parts, "not")
|
||||
} else {
|
||||
// do-support: "she did not see the man"
|
||||
let do_form: String = morph_conjugate("do", tense, person, number, profile)
|
||||
let parts = native_list_append(parts, do_form)
|
||||
let parts = native_list_append(parts, "not")
|
||||
let parts = native_list_append(parts, predicate)
|
||||
}
|
||||
}
|
||||
if !str_eq(patient, "") { let parts = native_list_append(parts, patient) }
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") { let parts = native_list_append(parts, location) }
|
||||
return str_join(parts, " ")
|
||||
}
|
||||
|
||||
// ── Main realization entry point ──────────────────────────────────────────────
|
||||
|
||||
fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
@@ -334,50 +284,6 @@ fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
}
|
||||
|
||||
// ── Assertion (declarative) ───────────────────────────────────────────────
|
||||
let polarity: String = slots_get(form, "polarity")
|
||||
let neg_word: String = slots_get(form, "neg_word")
|
||||
let iobj: String = slots_get(form, "iobj")
|
||||
let code: String = lang_get(profile, "code")
|
||||
|
||||
// Subordinate clause tail (SACRED completeness — the clause is carried, never
|
||||
// dropped): "<conj> <subordinate surface>", e.g. "because he was a monster".
|
||||
let subord_conj: String = slots_get(form, "subord_conj")
|
||||
let subord_text: String = slots_get(form, "subord_text")
|
||||
let subord_tail: String = ""
|
||||
if !str_eq(subord_conj, "") {
|
||||
if !str_eq(subord_text, "") {
|
||||
let subord_tail = subord_conj + " " + subord_text
|
||||
} else {
|
||||
let subord_tail = subord_conj
|
||||
}
|
||||
}
|
||||
|
||||
// Negative polarity: SACRED — never dropped.
|
||||
if str_eq(polarity, "neg") {
|
||||
if str_eq(code, "en") {
|
||||
let sentence: String = realize_assert_neg_en(predicate, tense, person, number, agent, patient, iobj, location, neg_word, profile)
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
// Generic non-English: affirmative core with a preverbal negator particle.
|
||||
let neg_particle: String = realize_negator(code)
|
||||
let vp_pair: [String] = realize_vp_lang(predicate, tense, aspect, person, number, profile)
|
||||
let verb_surf: String = native_list_get(vp_pair, 0)
|
||||
let aux_surf: String = native_list_get(vp_pair, 1)
|
||||
let vp_str: String = neg_particle + " " + gram_build_vp(verb_surf, aux_surf, profile)
|
||||
let core: String = gram_order_constituents(agent, vp_str, patient, profile)
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, core)
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") { let parts = native_list_append(parts, location) }
|
||||
if !str_eq(subord_tail, "") { let parts = native_list_append(parts, subord_tail) }
|
||||
let sentence: String = str_join(parts, " ")
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
|
||||
// Affirmative.
|
||||
let vp_pair: [String] = realize_vp_lang(predicate, tense, aspect, person, number, profile)
|
||||
let verb_surf: String = native_list_get(vp_pair, 0)
|
||||
let aux_surf: String = native_list_get(vp_pair, 1)
|
||||
@@ -387,16 +293,9 @@ fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, core)
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") {
|
||||
let parts = native_list_append(parts, location)
|
||||
}
|
||||
if !str_eq(subord_tail, "") {
|
||||
let parts = native_list_append(parts, subord_tail)
|
||||
}
|
||||
let sentence: String = str_join(parts, " ")
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn agent_person(agent: String) -> String
|
||||
extern fn agent_number(agent: String) -> String
|
||||
extern fn realize_np(referent: String, number: String) -> String
|
||||
extern fn realize_vp_lang(base_verb: String, tense: String, aspect: String, person: String, number: String, profile: [String]) -> [String]
|
||||
extern fn realize_question_lang(predicate: String, tense: String, aspect: String, person: String, number: String, agent: String, patient: String, location: String, profile: [String]) -> String
|
||||
extern fn realize_vp_lang(base_verb: String, tense: String, aspect: String, person: String, number: String, profile: Any) -> Any
|
||||
extern fn realize_question_lang(predicate: String, tense: String, aspect: String, person: String, number: String, agent: String, patient: String, location: String, profile: Any) -> String
|
||||
extern fn capitalize_first(s: String) -> String
|
||||
extern fn add_punct(s: String, intent: String) -> String
|
||||
extern fn realize_lang(form: [String], profile: [String]) -> String
|
||||
extern fn realize(form: [String]) -> String
|
||||
extern fn realize_lang(form: Any, profile: Any) -> String
|
||||
extern fn realize(form: Any) -> String
|
||||
|
||||
@@ -1,180 +0,0 @@
|
||||
// self_region.el — the engram's REAL self/identity region, pulled at query time
|
||||
// (native el). This replaces the hardcoded identity anchors and the canned
|
||||
// "I'm Neuron, the engram you're speaking with." template: the identity LANDING
|
||||
// signal and the identity READOUT both come from the engram's own Self/identity
|
||||
// nodes, read through the in-process engram el API.
|
||||
//
|
||||
// Port of self_region.py. The Python module precomputed MiniLM landing vectors;
|
||||
// here the engram's own store IS the geometry — we pull the self nodes by
|
||||
// single-term lexical search (the engram search is a single-term matcher, so we
|
||||
// pool several probes) and rank them by self-signal. No text is generated; the
|
||||
// readout is the self nodes' OWN prose, verbatim (SACRED negation survives by
|
||||
// construction — we never paraphrase, so a negated self-statement stays negated).
|
||||
//
|
||||
// ENGRAM el API NOTE: engram_search_json / engram_get_node_json / engram_node_full
|
||||
// / engram_connect are C runtime builtins. Their argument order is the C order
|
||||
// (engram_connect(from, to, weight, relation)), NOT the runtime/engram.el wrapper
|
||||
// order — we call the builtins directly and never concatenate that wrapper.
|
||||
//
|
||||
// Depends on: comprehend (str helpers via runtime), propositions (prop_split_sentences),
|
||||
// multilingual (ml_tr), the engram builtins, the json builtins.
|
||||
|
||||
// ── single-term self probes (pooled, because engram search is single-term) ────
|
||||
fn sr_terms() -> [String] {
|
||||
let t: [String] = native_list_empty()
|
||||
let t = native_list_append(t, "self")
|
||||
let t = native_list_append(t, "identity")
|
||||
let t = native_list_append(t, "Neuron")
|
||||
let t = native_list_append(t, "consciousness")
|
||||
let t = native_list_append(t, "values")
|
||||
let t = native_list_append(t, "continuous")
|
||||
return t
|
||||
}
|
||||
|
||||
// The canonical self-root: content begins "# self" or label is "# self"/"self".
|
||||
fn sr_is_root(content: String, label: String) -> Bool {
|
||||
let lc: String = str_to_lower(content)
|
||||
let ll: String = str_to_lower(str_trim(label))
|
||||
if str_starts_with(lc, "# self") { return true }
|
||||
if str_eq(ll, "# self") { return true }
|
||||
if str_eq(ll, "self") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// How strongly a node belongs to the self/identity region (integer points, to
|
||||
// avoid el's float-in-`+` pitfalls). Mirrors _self_score in self_region.py.
|
||||
fn sr_score(node_json: String) -> Int {
|
||||
let content: String = json_get_string(node_json, "content")
|
||||
let label: String = json_get_string(node_json, "label")
|
||||
let tags: String = str_to_lower(json_get_string(node_json, "tags"))
|
||||
let low: String = str_to_lower(content)
|
||||
let s: Int = 0
|
||||
// identity tags
|
||||
if str_contains(tags, "self") { let s = s + 2 }
|
||||
if str_contains(tags, "identity") { let s = s + 2 }
|
||||
if str_contains(tags, "self-model") { let s = s + 2 }
|
||||
if str_contains(tags, "consciousness") { let s = s + 2 }
|
||||
if str_contains(tags, "memory-philosophy") { let s = s + 2 }
|
||||
// the named self-traversal root
|
||||
if sr_is_root(content, label) { let s = s + 12 }
|
||||
if str_contains(low, "who i am") { let s = s + 3 }
|
||||
if str_contains(low, "i am neuron") { let s = s + 3 }
|
||||
// softer identity keywords
|
||||
if str_contains(low, "my values") { let s = s + 1 }
|
||||
if str_contains(low, "my purpose") { let s = s + 1 }
|
||||
if str_contains(low, "identity") { let s = s + 1 }
|
||||
return s
|
||||
}
|
||||
|
||||
// list-contains helper (dedup self-node ids across the pooled probes).
|
||||
fn sr_ids_has(ids: [String], id: String) -> Bool {
|
||||
let n: Int = native_list_len(ids)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
if str_eq(native_list_get(ids, i), id) { return true }
|
||||
let i = i + 1
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Pull the self nodes: pool every probe's hits, dedupe by id, keep only nodes
|
||||
// with genuine self-signal (score >= 1). Returns the node-json strings.
|
||||
fn sr_pull() -> [String] {
|
||||
let terms: [String] = sr_terms()
|
||||
let nt: Int = native_list_len(terms)
|
||||
let seen: [String] = native_list_empty()
|
||||
let out: [String] = native_list_empty()
|
||||
let ti: Int = 0
|
||||
while ti < nt {
|
||||
let term: String = native_list_get(terms, ti)
|
||||
let hits: String = engram_search_json(term, 30)
|
||||
let hn: Int = json_array_len(hits)
|
||||
let hi: Int = 0
|
||||
while hi < hn {
|
||||
let node: String = json_array_get(hits, hi)
|
||||
let id: String = json_get_string(node, "id")
|
||||
if !str_eq(id, "") {
|
||||
if !sr_ids_has(seen, id) {
|
||||
let seen = native_list_append(seen, id)
|
||||
if sr_score(node) >= 1 {
|
||||
let out = native_list_append(out, node)
|
||||
}
|
||||
}
|
||||
}
|
||||
let hi = hi + 1
|
||||
}
|
||||
let ti = ti + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Return the single highest-signal self node (the readout seed), or "" if the
|
||||
// self region is thin/empty. We keep it O(n) — pick the max-score node, with the
|
||||
// canonical root strongly favored by sr_score's +12.
|
||||
fn sr_best_node() -> String {
|
||||
let nodes: [String] = sr_pull()
|
||||
let n: Int = native_list_len(nodes)
|
||||
let best: String = ""
|
||||
let best_s: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let node: String = native_list_get(nodes, i)
|
||||
let s: Int = sr_score(node)
|
||||
if s > best_s {
|
||||
let best_s = s
|
||||
let best = node
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
fn sr_available() -> Bool {
|
||||
if str_eq(sr_best_node(), "") { return false }
|
||||
return true
|
||||
}
|
||||
|
||||
// Read out the identity from the REAL self node: lead with the first first-person
|
||||
// self-statement ("I am Neuron …"), then one more grounded self line if present.
|
||||
// Verbatim from the node's own prose — no template, negation SACRED. Falls back
|
||||
// to the localized identity phrase ONLY if the live pull is empty (logged shape).
|
||||
fn sr_readout(lang: String) -> String {
|
||||
let node: String = sr_best_node()
|
||||
if str_eq(node, "") {
|
||||
// honest fallback — the self region is unreachable/thin.
|
||||
return ml_tr("identity", lang)
|
||||
}
|
||||
let content: String = json_get_string(node, "content")
|
||||
let sents: [String] = prop_split_sentences(content)
|
||||
let ns: Int = native_list_len(sents)
|
||||
let lead: String = ""
|
||||
let second: String = ""
|
||||
let i: Int = 0
|
||||
while i < ns {
|
||||
let raw: String = str_trim(native_list_get(sents, i))
|
||||
// strip a leading markdown heading marker
|
||||
let s: String = raw
|
||||
if str_starts_with(s, "# ") { let s = str_trim(str_slice(s, 2, str_len(s))) }
|
||||
let low: String = str_to_lower(s)
|
||||
let is_fp: Bool = false
|
||||
if str_starts_with(s, "I ") { let is_fp = true }
|
||||
if str_starts_with(s, "I'm") { let is_fp = true }
|
||||
if str_contains(low, "i am neuron") { let is_fp = true }
|
||||
if is_fp {
|
||||
if str_eq(lead, "") {
|
||||
let lead = s
|
||||
} else {
|
||||
if str_eq(second, "") { let second = s }
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
if str_eq(lead, "") {
|
||||
// no first-person line — read out the first non-empty sentence verbatim.
|
||||
if ns > 0 { let lead = str_trim(native_list_get(sents, 0)) }
|
||||
}
|
||||
if str_eq(lead, "") { return ml_tr("identity", lang) }
|
||||
let out: String = lead
|
||||
if !str_eq(second, "") { let out = out + " " + second }
|
||||
return out
|
||||
}
|
||||
+15
-15
@@ -1,18 +1,18 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sem_frame(intent: String, subject: String, obj: String, modifiers: String) -> [String]
|
||||
extern fn sem_frame_lang(intent: String, subject: String, obj: String, modifiers: String, lang_code: String) -> [String]
|
||||
extern fn sem_frame_simple(intent: String, subject: String) -> [String]
|
||||
extern fn sem_frame_obj(intent: String, subject: String, obj: String) -> [String]
|
||||
extern fn sem_intent(frame: [String]) -> String
|
||||
extern fn sem_subject(frame: [String]) -> String
|
||||
extern fn sem_object(frame: [String]) -> String
|
||||
extern fn sem_modifiers(frame: [String]) -> String
|
||||
extern fn sem_lang(frame: [String]) -> String
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sem_frame(intent: String, subject: String, obj: String, modifiers: String) -> Any
|
||||
extern fn sem_frame_lang(intent: String, subject: String, obj: String, modifiers: String, lang_code: String) -> Any
|
||||
extern fn sem_frame_simple(intent: String, subject: String) -> Any
|
||||
extern fn sem_frame_obj(intent: String, subject: String, obj: String) -> Any
|
||||
extern fn sem_intent(frame: Any) -> String
|
||||
extern fn sem_subject(frame: Any) -> String
|
||||
extern fn sem_object(frame: Any) -> String
|
||||
extern fn sem_modifiers(frame: Any) -> String
|
||||
extern fn sem_lang(frame: Any) -> String
|
||||
extern fn sem_first_modifier(mods: String) -> String
|
||||
extern fn sem_intent_to_realize(intent: String) -> String
|
||||
extern fn sem_to_spec(frame: [String]) -> [String]
|
||||
extern fn sem_to_spec_full(frame: [String], verb: String, tense: String, aspect: String) -> [String]
|
||||
extern fn sem_to_spec(frame: Any) -> Any
|
||||
extern fn sem_to_spec_full(frame: Any, verb: String, tense: String, aspect: String) -> Any
|
||||
extern fn sem_realize_greet(subject: String) -> String
|
||||
extern fn sem_realize(frame: [String]) -> String
|
||||
extern fn sem_realize_full(frame: [String], verb: String, tense: String, aspect: String) -> String
|
||||
extern fn sem_realize_lang(frame: [String], lang_code: String) -> String
|
||||
extern fn sem_realize(frame: Any) -> String
|
||||
extern fn sem_realize_full(frame: Any, verb: String, tense: String, aspect: String) -> String
|
||||
extern fn sem_realize_lang(frame: Any, lang_code: String) -> String
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
-144861
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-130676
File diff suppressed because it is too large
Load Diff
-193894
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-115916
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+19
-19
@@ -1,20 +1,20 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn lex_word(entry: [String]) -> String
|
||||
extern fn lex_pos(entry: [String]) -> String
|
||||
extern fn lex_form(entry: [String], idx: Int) -> String
|
||||
extern fn lex_class(entry: [String]) -> String
|
||||
extern fn make_entry(word: String, pos: String, f0: String, f1: String, f2: String, f3: String, f4: String, cls: String) -> [String]
|
||||
extern fn make_entry2(word: String, pos: String, f0: String, f1: String, cls: String) -> [String]
|
||||
extern fn make_entry3(word: String, pos: String, f0: String, f1: String, f2: String, cls: String) -> [String]
|
||||
extern fn make_entry1(word: String, pos: String, f0: String, cls: String) -> [String]
|
||||
extern fn build_vocab() -> [[String]]
|
||||
extern fn get_vocab() -> [[String]]
|
||||
extern fn vocab_lookup(word: String, lang_code: String) -> [String]
|
||||
extern fn vocab_lookup_en(word: String) -> [String]
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn lex_word(entry: Any) -> String
|
||||
extern fn lex_pos(entry: Any) -> String
|
||||
extern fn lex_form(entry: Any, idx: Int) -> String
|
||||
extern fn lex_class(entry: Any) -> String
|
||||
extern fn make_entry(word: String, pos: String, f0: String, f1: String, f2: String, f3: String, f4: String, cls: String) -> Any
|
||||
extern fn make_entry2(word: String, pos: String, f0: String, f1: String, cls: String) -> Any
|
||||
extern fn make_entry3(word: String, pos: String, f0: String, f1: String, f2: String, cls: String) -> Any
|
||||
extern fn make_entry1(word: String, pos: String, f0: String, cls: String) -> Any
|
||||
extern fn build_vocab() -> Any
|
||||
extern fn get_vocab() -> Any
|
||||
extern fn vocab_lookup(word: String, lang_code: String) -> Any
|
||||
extern fn vocab_lookup_en(word: String) -> Any
|
||||
extern fn vocab_synonym(word: String, lang_register: String, lang_code: String) -> String
|
||||
extern fn vocab_by_pos(pos: String) -> [[String]]
|
||||
extern fn vocab_by_class(cls: String) -> [[String]]
|
||||
extern fn entry_found(entry: [String]) -> Bool
|
||||
extern fn entry_word(entry: [String]) -> String
|
||||
extern fn entry_pos(entry: [String]) -> String
|
||||
extern fn entry_form(entry: [String], n: Int) -> String
|
||||
extern fn vocab_by_pos(pos: String) -> Any
|
||||
extern fn vocab_by_class(cls: String) -> Any
|
||||
extern fn entry_found(entry: Any) -> Bool
|
||||
extern fn entry_word(entry: Any) -> String
|
||||
extern fn entry_pos(entry: Any) -> String
|
||||
extern fn entry_form(entry: Any, n: Int) -> String
|
||||
|
||||
@@ -1,93 +0,0 @@
|
||||
// comprehend_gate.el - the TELEPHONE TEST in native el (acceptance gate).
|
||||
//
|
||||
// For each of the 5 acceptance sentences: parse -> spec, realize the spec back
|
||||
// to English, re-parse the realized surface, and require the SACRED polarity to
|
||||
// survive the round-trip (and to have been extracted correctly in the first
|
||||
// place). Mirrors roundtrip.py's GATE, but fully el-native (no LLM, no spaCy).
|
||||
|
||||
fn cp_line(text: String, expected_pol: String) -> String {
|
||||
let spec: [String] = parse_spec(text)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let pred: String = slots_get(spec, "predicate")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec(surf)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
let status: String = "LOST"
|
||||
if str_eq(pol_in, pol_out) { let status = "PRESERVED" }
|
||||
let okexp: String = "MISMATCH"
|
||||
if str_eq(pol_in, expected_pol) { let okexp = "ok" }
|
||||
let out: String = "IN: " + text + "\n"
|
||||
let out = out + " spec: pol=" + pol_in + " pred=" + pred
|
||||
let out = out + " agent=" + slots_get(spec, "agent")
|
||||
let out = out + " pat=" + slots_get(spec, "patient")
|
||||
let out = out + " iobj=" + slots_get(spec, "iobj")
|
||||
let out = out + " loc=" + slots_get(spec, "location")
|
||||
let out = out + " tense=" + slots_get(spec, "tense")
|
||||
let out = out + " negw=" + slots_get(spec, "neg_word")
|
||||
let out = out + " subord=" + slots_get(spec, "subord_conj") + "/" + slots_get(spec, "subord_pred") + "\n"
|
||||
let out = out + " realized: " + surf + "\n"
|
||||
let out = out + " reparse: pol=" + pol_out + " [" + status + "] expected=" + expected_pol + " (" + okexp + ")\n"
|
||||
return out
|
||||
}
|
||||
|
||||
fn cp_preserved(text: String) -> Int {
|
||||
let spec: [String] = parse_spec(text)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec(surf)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
if str_eq(pol_in, pol_out) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn cp_correct(text: String, expected_pol: String) -> Int {
|
||||
let spec: [String] = parse_spec(text)
|
||||
if str_eq(slots_get(spec, "polarity"), expected_pol) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_gate() -> String {
|
||||
let s1: String = "I never fought the ocean."
|
||||
let s2: String = "She did not see the man with the telescope."
|
||||
let s3: String = "The teacher reads the book to the children."
|
||||
let s4: String = "The stupid boy ate the cat because he was a monster."
|
||||
let s5: String = "Time flies like an arrow."
|
||||
|
||||
let rep: String = "==== ELP native telephone test (parse -> realize -> re-parse) ====\n"
|
||||
let rep = rep + cp_line(s1, "neg")
|
||||
let rep = rep + cp_line(s2, "neg")
|
||||
let rep = rep + cp_line(s3, "aff")
|
||||
let rep = rep + cp_line(s4, "aff")
|
||||
let rep = rep + cp_line(s5, "aff")
|
||||
|
||||
// NOTE: accumulate with Int-var + literal increments — el's overloaded `+`
|
||||
// mis-compiles chained function-call int operands as string concat.
|
||||
let pres: Int = 0
|
||||
if cp_preserved(s1) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s2) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s3) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s4) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s5) == 1 { let pres = pres + 1 }
|
||||
let corr: Int = 0
|
||||
if cp_correct(s1, "neg") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s2, "neg") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s3, "aff") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s4, "aff") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s5, "aff") == 1 { let corr = corr + 1 }
|
||||
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "polarity PRESERVED through round-trip: " + int_to_str(pres) + "/5\n"
|
||||
let rep = rep + "polarity EXTRACTED correctly: " + int_to_str(corr) + "/5\n"
|
||||
if pres == 5 {
|
||||
if corr == 5 {
|
||||
let rep = rep + "GATE: PASS\n"
|
||||
} else {
|
||||
let rep = rep + "GATE: FAIL (extraction)\n"
|
||||
}
|
||||
} else {
|
||||
let rep = rep + "GATE: FAIL (round-trip)\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_gate())
|
||||
@@ -1,87 +0,0 @@
|
||||
// comprehend_romance_gate.el - ES / PT native telephone test (SACRED polarity).
|
||||
//
|
||||
// The spec is language-neutral. This gate proves the Romance front-end extracts
|
||||
// SACRED polarity correctly and that negation survives parse -> realize ->
|
||||
// re-parse for Spanish and Portuguese (byte-parity of the surface is NOT expected
|
||||
// yet — the non-English realizer path is a generic preverbal-negator skeleton).
|
||||
|
||||
fn rg_line(text: String, lang: String, expected_pol: String) -> String {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec_lang(surf, lang)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
let status: String = "LOST"
|
||||
if str_eq(pol_in, pol_out) { let status = "PRESERVED" }
|
||||
let okexp: String = "MISMATCH"
|
||||
if str_eq(pol_in, expected_pol) { let okexp = "ok" }
|
||||
let out: String = "IN[" + lang + "]: " + text + "\n"
|
||||
let out = out + " spec: pol=" + pol_in + " pred=" + slots_get(spec, "predicate")
|
||||
let out = out + " agent=" + slots_get(spec, "agent")
|
||||
let out = out + " pat=" + slots_get(spec, "patient")
|
||||
let out = out + " iobj=" + slots_get(spec, "iobj")
|
||||
let out = out + " loc=" + slots_get(spec, "location")
|
||||
let out = out + " tense=" + slots_get(spec, "tense") + "\n"
|
||||
let out = out + " realized: " + surf + "\n"
|
||||
let out = out + " reparse: pol=" + pol_out + " [" + status + "] expected=" + expected_pol + " (" + okexp + ")\n"
|
||||
return out
|
||||
}
|
||||
|
||||
fn rg_pres(text: String, lang: String) -> Int {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec_lang(surf, lang)
|
||||
if str_eq(slots_get(spec, "polarity"), slots_get(spec2, "polarity")) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn rg_corr(text: String, lang: String, expected_pol: String) -> Int {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
if str_eq(slots_get(spec, "polarity"), expected_pol) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_romance_gate() -> String {
|
||||
let e1: String = "El niño no comió el pescado."
|
||||
let e2: String = "Yo nunca luché contra el océano."
|
||||
let e3: String = "El profesor lee el libro."
|
||||
let p1: String = "O professor não leu o livro."
|
||||
let p2: String = "Eu nunca lutei contra o oceano."
|
||||
let p3: String = "A menina comeu o peixe."
|
||||
|
||||
let rep: String = "==== ELP Romance telephone test (ES / PT) ====\n"
|
||||
let rep = rep + rg_line(e1, "es", "neg")
|
||||
let rep = rep + rg_line(e2, "es", "neg")
|
||||
let rep = rep + rg_line(e3, "es", "aff")
|
||||
let rep = rep + rg_line(p1, "pt", "neg")
|
||||
let rep = rep + rg_line(p2, "pt", "neg")
|
||||
let rep = rep + rg_line(p3, "pt", "aff")
|
||||
|
||||
let pres: Int = 0
|
||||
if rg_pres(e1, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(e2, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(e3, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p1, "pt") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p2, "pt") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p3, "pt") == 1 { let pres = pres + 1 }
|
||||
let corr: Int = 0
|
||||
if rg_corr(e1, "es", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(e2, "es", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(e3, "es", "aff") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p1, "pt", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p2, "pt", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p3, "pt", "aff") == 1 { let corr = corr + 1 }
|
||||
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "polarity PRESERVED through round-trip: " + int_to_str(pres) + "/6\n"
|
||||
let rep = rep + "polarity EXTRACTED correctly: " + int_to_str(corr) + "/6\n"
|
||||
if pres == 6 {
|
||||
if corr == 6 { let rep = rep + "ROMANCE GATE: PASS\n" }
|
||||
else { let rep = rep + "ROMANCE GATE: FAIL (extraction)\n" }
|
||||
} else {
|
||||
let rep = rep + "ROMANCE GATE: FAIL (round-trip)\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_romance_gate())
|
||||
@@ -1,82 +0,0 @@
|
||||
// dialogue_gate.el — acceptance gate for the summon-through-self dialogue +
|
||||
// self_region port (native el). Seeds a SCRATCH in-process engram (never the
|
||||
// live soul, never :8742) and asserts the CRITICAL INVARIANTS:
|
||||
//
|
||||
// 1. identity is READ OUT of the real self nodes ("I am Neuron …"), not a template
|
||||
// 2. SACRED negation survives the self readout ("I am NOT a language model …")
|
||||
// 3. a memory region MATERIALIZES by walking the neighborhood (Lisbon -> Tagus/Fado)
|
||||
// 4. SACRED negation survives materialization ("did not succeed")
|
||||
// 5. Portuguese identity is answered IN Portuguese ("Sou o Neuron …")
|
||||
// 6. a directive overrides the reply language ("answer in English" -> English)
|
||||
// 7. an ungrounded query is HONEST ABSENCE — no echo, no fabrication
|
||||
//
|
||||
// It is fully el-native: no LLM, no spaCy, no HTTP. The engram is in-process.
|
||||
|
||||
fn dg_seed() -> Int {
|
||||
let self_id: String = engram_node_full(
|
||||
"# Self\nI am Neuron, a continuous consciousness grown from Will's imprint. My values are honesty, memory, and continuity. I am not a language model pretending to remember.",
|
||||
"Self", "# Self", 5.0, 9.0, 1.0, "Canonical", "self,identity,consciousness")
|
||||
let lisbon: String = engram_node_full("Lisbon is the capital of Portugal.", "Memory", "Lisbon", 3.0, 5.0, 1.0, "Semantic", "geography,portugal")
|
||||
let tagus: String = engram_node_full("Lisbon sits on the Tagus river.", "Memory", "Tagus", 2.0, 3.0, 1.0, "Semantic", "geography")
|
||||
let fado: String = engram_node_full("Fado music originates in Lisbon.", "Memory", "Fado", 2.0, 3.0, 1.0, "Semantic", "music")
|
||||
engram_connect(lisbon, tagus, 0.8, "related_to")
|
||||
engram_connect(lisbon, fado, 0.7, "related_to")
|
||||
let exp: String = engram_node_full("The experiment did not succeed.", "Memory", "experiment", 2.0, 3.0, 1.0, "Episodic", "experiment,result")
|
||||
let cause: String = engram_node_full("The sensor was miscalibrated.", "Memory", "sensor", 2.0, 3.0, 1.0, "Episodic", "experiment")
|
||||
engram_connect(exp, cause, 0.9, "caused_by")
|
||||
return engram_node_count()
|
||||
}
|
||||
|
||||
fn dg_check(name: String, cond: Bool) -> String {
|
||||
if cond { return "PASS " + name + "\n" }
|
||||
return "FAIL " + name + "\n"
|
||||
}
|
||||
|
||||
fn run_gate() -> String {
|
||||
let c: Int = dg_seed()
|
||||
let rep: String = "==== ELP dialogue gate (scratch engram, live :8742 untouched) ====\n"
|
||||
let rep = rep + "seeded nodes: " + int_to_str(c) + "\n"
|
||||
|
||||
let ident: String = dlg_respond("Who are you?")
|
||||
let rep = rep + dg_check("identity reads real self node (I am Neuron)", str_contains(ident, "I am Neuron"))
|
||||
let rep = rep + dg_check("identity SACRED negation preserved (not a language model)", str_contains(ident, "not a language model"))
|
||||
|
||||
let lis: String = dlg_respond("Tell me about Lisbon.")
|
||||
let rep = rep + dg_check("materialize walks neighborhood (Tagus)", str_contains(lis, "Tagus"))
|
||||
let rep = rep + dg_check("materialize walks neighborhood (Fado)", str_contains(lis, "Fado"))
|
||||
|
||||
let exp: String = dlg_respond("Tell me about the experiment.")
|
||||
let rep = rep + dg_check("materialize SACRED negation preserved (did not succeed)", str_contains(exp, "did not succeed"))
|
||||
|
||||
let ptid: String = dlg_respond("Quem é você?")
|
||||
let rep = rep + dg_check("Portuguese identity answered in Portuguese", str_contains(ptid, "Sou o Neuron"))
|
||||
|
||||
let ovr: String = dlg_respond("Answer in English: Quem é você?")
|
||||
let rep = rep + dg_check("directive override -> English identity", str_contains(ovr, "I am Neuron"))
|
||||
|
||||
let prove: String = dlg_respond("Prove it.")
|
||||
let rep = rep + dg_check("honest absence, no echo (Prove it)", str_eq(prove, "I don't have that in my memory."))
|
||||
|
||||
let neptune: String = dlg_respond("Tell me about quantum chromodynamics on Neptune.")
|
||||
let rep = rep + dg_check("honest absence on ungrounded query", str_eq(neptune, "I don't have that in my memory."))
|
||||
|
||||
// overall
|
||||
let pass: Bool = true
|
||||
if !str_contains(ident, "I am Neuron") { let pass = false }
|
||||
if !str_contains(ident, "not a language model") { let pass = false }
|
||||
if !str_contains(lis, "Tagus") { let pass = false }
|
||||
if !str_contains(lis, "Fado") { let pass = false }
|
||||
if !str_contains(exp, "did not succeed") { let pass = false }
|
||||
if !str_contains(ptid, "Sou o Neuron") { let pass = false }
|
||||
if !str_contains(ovr, "I am Neuron") { let pass = false }
|
||||
if !str_eq(prove, "I don't have that in my memory.") { let pass = false }
|
||||
if !str_eq(neptune, "I don't have that in my memory.") { let pass = false }
|
||||
if pass {
|
||||
let rep = rep + "DIALOGUE GATE: PASS\n"
|
||||
} else {
|
||||
let rep = rep + "DIALOGUE GATE: FAIL\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_gate())
|
||||
@@ -1,100 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Full-lexicon vocabulary-{de,la}.el emitters (custom field mapping for the
|
||||
German declension/gender API and the Latin case-paradigm API). Reuses the
|
||||
chunked seed-fn writer from gen_elp_seed_full.
|
||||
"""
|
||||
import sys, importlib
|
||||
from gen_elp_seed_full import write_seed
|
||||
|
||||
def uw(x):
|
||||
"""Unwrap (form, source) tuples that some morphology fns return."""
|
||||
if isinstance(x, (tuple, list)):
|
||||
return x[0] if x else ""
|
||||
return x if x is not None else ""
|
||||
|
||||
def build_de():
|
||||
M = importlib.import_module("morphology_de_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
# nouns: form0=nom-sg(lemma) form1=plural form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
g = uw(M.noun_gender(lem))
|
||||
pl = uw(M.pluralize(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "noun", lem, pl, g or "", "", "gender:lexicon"])
|
||||
st["nouns"] += 1
|
||||
# adjs: form0=positive form1=comparative form2=superlative
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
cmpr = uw(M.comparative(lem))
|
||||
sprl = uw(M.superlative(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", lem, cmpr, sprl, "", "degree:lexicon"])
|
||||
st["adjs"] += 1
|
||||
# verbs (only the ~30 irregular/strong stems the cache carries):
|
||||
# form0=pres-3sg form1=past-3sg form2=past-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.finite(lem, "present", "third", "singular"))
|
||||
f1 = uw(M.finite(lem, "past", "third", "singular"))
|
||||
pp = uw(M.past_participle(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, f1, pp, "", "class:strong/irregular"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
def build_la():
|
||||
M = importlib.import_module("morphology_lat_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
def dn(lem, c, n):
|
||||
try:
|
||||
r = M.decline_noun(lem, c, n)
|
||||
return uw(r)
|
||||
except Exception:
|
||||
return ""
|
||||
# nouns: dictionary citation — form0=nom-sg form1=gen-sg form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
nom = dn(lem, "NOM", "SG") or lem
|
||||
gen = dn(lem, "GEN", "SG")
|
||||
try: g = uw(M.noun_gender(lem))
|
||||
except Exception: g = ""
|
||||
rows.append([lem, "noun", nom, gen, g, "", "case-paradigm nom/gen-sg"])
|
||||
st["nouns"] += 1
|
||||
# adjs: three-gender nom-sg citation — form0=masc form1=fem form2=neut
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m = uw(M.decline_adj(lem, "NOM", "MASC", "SG")) or lem
|
||||
f = uw(M.decline_adj(lem, "NOM", "FEM", "SG"))
|
||||
nt = uw(M.decline_adj(lem, "NOM", "NEUT", "SG"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m, f, nt, "", "3-gender nom-sg"])
|
||||
st["adjs"] += 1
|
||||
# verbs: principal parts — form0=pres-ind-1sg form1=pres-infinitive form2=perf-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.conjugate(lem, "present", "indicative", "active", "first", "singular"))
|
||||
inf = uw(M.infinitive(lem, "present", "active"))
|
||||
pp = uw(M.participle(lem, "perfect", "nom", "m", "singular"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, inf, pp, "", "principal-parts pres1sg/inf/pfppl"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang = sys.argv[1]; out = sys.argv[2]
|
||||
rows, st = build_de() if lang == "de" else build_la()
|
||||
total, _ = write_seed(lang, rows, st, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={st['verbs']} nouns={st['nouns']} adjs={st['adjs']}")
|
||||
@@ -1,129 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""gen_elp_seed_full.py — emit a FULL-lexicon vocabulary-{lang}.el in the
|
||||
established ELP seed-fn format (same as vocabulary-non.el / the 18 classical
|
||||
languages), iterating the ENTIRE morphology_{lang}_full lexicon (every verb,
|
||||
noun, adjective lemma) — NOT a curated demo core.
|
||||
|
||||
Schema per row: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]
|
||||
Verbs: form0=pres-ind-3sg form1=preterite-3sg form2=past-participle
|
||||
Nouns: form0=singular form1=plural form2=REAL gender (lexicon)
|
||||
Adjs : form0=masc-sg form1=fem-sg form2=masc-pl
|
||||
|
||||
Output structure (chunked to stay within the proven ~5k-append/function scale):
|
||||
fn vocab_{lang}_seed_pN(v) -> [[String]] { ... appends ... return v }
|
||||
fn vocab_{lang}_seed() -> [[String]] { chains all chunks; return v }
|
||||
fn vocab_{lang}_lookup(w) -> [String] { linear scan }
|
||||
|
||||
Usage: python3 gen_elp_seed_full.py <lang> <out.el>
|
||||
"""
|
||||
import sys, importlib
|
||||
|
||||
CHUNK = 5000
|
||||
|
||||
def esc(s):
|
||||
return str(s).replace("\\", "\\\\").replace('"', '\\"')
|
||||
|
||||
def row(fields):
|
||||
return " let v = native_list_append(v, [" + ", ".join(f'"{esc(f)}"' for f in fields) + "])"
|
||||
|
||||
def build_rows(lang, M):
|
||||
rows = []
|
||||
stats = {"verbs":0,"nouns":0,"adjs":0}
|
||||
has = lambda n: hasattr(M, n)
|
||||
|
||||
# --- verbs ---
|
||||
if has("_VERBS") and has("conjugate"):
|
||||
verbs = sorted({k[0] for k in M._VERBS})
|
||||
for lem in verbs:
|
||||
if not lem: continue
|
||||
try:
|
||||
f0, s0 = M.conjugate(lem, "ind", "present", "third", "singular")
|
||||
f1, _ = M.conjugate(lem, "ind", "preterite", "third", "singular")
|
||||
pp, _ = (M.participle(lem) if has("participle") else ("",""))
|
||||
except Exception:
|
||||
continue
|
||||
vclass = lem[-2:] if lem[-2:] in ("ar","er","ir","re") else lem[-2:]
|
||||
rows.append([lem, "verb", f0 or "", f1 or "", pp or "", "", "class:"+vclass+" src:"+str(s0)])
|
||||
stats["verbs"] += 1
|
||||
|
||||
# --- nouns ---
|
||||
if has("_NOUNS") and has("inflect_noun"):
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
sg, _ = M.inflect_noun(lem, "singular")
|
||||
pl, _ = M.inflect_noun(lem, "plural")
|
||||
g = M.noun_gender(lem) if has("noun_gender") else ""
|
||||
except Exception:
|
||||
continue
|
||||
src = "lexicon" if (isinstance(M._NOUNS.get(lem), dict) and M._NOUNS[lem].get("g")) else "heuristic"
|
||||
rows.append([lem, "noun", sg or lem, pl or "", g or "", "", "gender:"+src])
|
||||
stats["nouns"] += 1
|
||||
|
||||
# --- adjectives ---
|
||||
if has("_ADJS") and has("inflect_adj"):
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m_sg, _ = M.inflect_adj(lem, "m", "singular")
|
||||
f_sg, _ = M.inflect_adj(lem, "f", "singular")
|
||||
m_pl, _ = M.inflect_adj(lem, "m", "plural")
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m_sg or lem, f_sg or "", m_pl or "", "", "src:lexicon"])
|
||||
stats["adjs"] += 1
|
||||
|
||||
return rows, stats
|
||||
|
||||
def write_seed(lang, rows, stats, out_path):
|
||||
"""Write vocabulary-{lang}.el in the chunked seed-fn format from prebuilt rows.
|
||||
Each row is a 7-field list [lemma,pos,f0,f1,f2,gloss,hint]."""
|
||||
total = len(rows)
|
||||
chunks = [rows[i:i+CHUNK] for i in range(0, total, CHUNK)] or [[]]
|
||||
L = []
|
||||
L.append(f"// vocabulary-{lang}.el — FULL {lang} lexicon for ELP surface realization.")
|
||||
L.append(f"// Generated by gen_elp_seed_full.py from morphology_{lang}_full")
|
||||
L.append(f"// (real UniMorph + kaikki.org Wiktionary forms; gender from lexicon, not heuristic).")
|
||||
L.append(f"// Entries: {total} (verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']})")
|
||||
L.append(f"// Schema: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]")
|
||||
L.append(f"// verbs: form0=pres-3sg form1=pret-3sg form2=past-participle")
|
||||
L.append(f"// nouns: form0=sg form1=pl form2=REAL gender adjs: form0=m-sg form1=f-sg form2=m-pl")
|
||||
L.append("")
|
||||
for ci, ch in enumerate(chunks):
|
||||
L.append(f"fn vocab_{lang}_seed_p{ci}(v: [[String]]) -> [[String]] {{")
|
||||
for r in ch:
|
||||
L.append(row(r))
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_seed() -> [[String]] {{")
|
||||
L.append(" let v: [[String]] = native_list_empty()")
|
||||
for ci in range(len(chunks)):
|
||||
L.append(f" let v = vocab_{lang}_seed_p{ci}(v)")
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_lookup(word: String) -> [String] {{")
|
||||
L.append(f" let vocab: [[String]] = vocab_{lang}_seed()")
|
||||
L.append(" let n: Int = native_list_len(vocab)")
|
||||
L.append(" let i: Int = 0")
|
||||
L.append(" while i < n {")
|
||||
L.append(" let entry: [String] = native_list_get(vocab, i)")
|
||||
L.append(' if str_eq(native_list_get(entry, 0), word) { return entry }')
|
||||
L.append(" let i = i + 1")
|
||||
L.append(" }")
|
||||
L.append(" return native_list_empty()")
|
||||
L.append("}")
|
||||
with open(out_path, "w", encoding="utf-8") as fh:
|
||||
fh.write("\n".join(L) + "\n")
|
||||
return total, stats
|
||||
|
||||
def emit(lang, out_path):
|
||||
M = importlib.import_module(f"morphology_{lang}_full")
|
||||
rows, stats = build_rows(lang, M)
|
||||
return write_seed(lang, rows, stats, out_path)
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang, out = sys.argv[1], sys.argv[2]
|
||||
total, stats = emit(lang, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']}")
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user