Compare commits
39 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4f49755ebb | |||
| 7aa847e32a | |||
| ee71423732 | |||
| bb64a236ed | |||
| 9a0266cbf9 | |||
| a72145b44e | |||
| 8affb1d6e0 | |||
| fa47b98d18 | |||
| 0a72fced28 | |||
| 5d0d4555ae | |||
| 8347a2f1c0 | |||
| b97b644799 | |||
| d71fc4c1c0 | |||
| a118d19393 | |||
| c6aa1e5c53 | |||
| ff577391f2 | |||
| ee0d5f9b97 | |||
| 391bd818ea | |||
| 43636aed99 | |||
| 2baa0b9a41 | |||
| 6a8b2461cd | |||
| bcb356fe69 | |||
| dd7827059a | |||
| 208e36c899 | |||
| b97ce74d1f | |||
| 155a449c4e | |||
| 4696fd6833 | |||
| 581a351fb1 | |||
| 8ce8656de2 | |||
| 1e49560f1f | |||
| e8f0b5a9de | |||
| 40287c4cfc | |||
| 0481bea44d | |||
| 9d565ca080 | |||
| 4773dd0aa2 | |||
| 6b9d9e6c4a | |||
| b4967af13e | |||
| 2b2a1246e7 | |||
| 5c41c66a0f |
@@ -39,9 +39,9 @@ jobs:
|
||||
run: |
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -54,9 +54,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -91,7 +91,7 @@ jobs:
|
||||
- name: Precompile el_runtime.o
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
gcc -O2 -c -I "$RUNTIME" "$RUNTIME/el_runtime.c" \
|
||||
-o /tmp/el_runtime.o
|
||||
echo "el_runtime.o compiled"
|
||||
@@ -100,7 +100,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -110,7 +110,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -120,7 +120,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -130,7 +130,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -140,7 +140,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -150,7 +150,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -160,7 +160,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -170,7 +170,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -180,7 +180,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -191,7 +191,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -202,7 +202,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -214,9 +214,18 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -242,7 +251,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-dev \
|
||||
@@ -250,7 +259,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-dev \
|
||||
@@ -258,7 +267,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-js \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
--source=runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-dev"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -268,6 +277,12 @@ jobs:
|
||||
# Patches ci-base:dev in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -291,9 +306,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -46,9 +46,9 @@ jobs:
|
||||
run: |
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -84,7 +84,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -94,7 +94,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -104,7 +104,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -114,7 +114,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -124,7 +124,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -134,7 +134,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -144,7 +144,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -154,7 +154,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -164,7 +164,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -176,9 +176,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -189,7 +189,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -200,7 +200,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -212,12 +212,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -235,7 +244,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-stage \
|
||||
@@ -243,7 +252,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-stage"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -253,6 +262,12 @@ jobs:
|
||||
# Patches ci-base:stage in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -275,9 +290,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -47,9 +47,9 @@ jobs:
|
||||
mkdir -p dist/platform
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -62,9 +62,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -75,7 +75,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -86,7 +86,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -121,7 +121,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -131,7 +131,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -141,7 +141,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -151,7 +151,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -161,7 +161,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -171,7 +171,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -181,7 +181,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -191,7 +191,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -201,7 +201,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -216,8 +216,10 @@ jobs:
|
||||
cp lang/dist/platform/elc dist/sdk/bin/elc
|
||||
cp lang/dist/bin/elb dist/sdk/bin/elb
|
||||
cp lang/dist/bin/epm dist/sdk/bin/epm
|
||||
cp lang/el-compiler/runtime/el_runtime.c dist/sdk/runtime/
|
||||
cp lang/el-compiler/runtime/el_runtime.h dist/sdk/runtime/
|
||||
cp lang/runtime/el_runtime.c dist/sdk/runtime/
|
||||
cp lang/runtime/el_runtime.h dist/sdk/runtime/
|
||||
cp lang/runtime/engram_store.c dist/sdk/runtime/
|
||||
cp lang/runtime/engram_store.h dist/sdk/runtime/
|
||||
cp lang/runtime/*.el dist/sdk/runtime/
|
||||
tar -czf dist/el-sdk-latest.tar.gz -C dist/sdk .
|
||||
echo "SDK tarball bundled: dist/el-sdk-latest.tar.gz"
|
||||
@@ -274,8 +276,10 @@ jobs:
|
||||
|
||||
# Per-file assets (downstream CI needs these individually)
|
||||
upload_asset lang/dist/platform/elc elc
|
||||
upload_asset lang/el-compiler/runtime/el_runtime.c el_runtime.c
|
||||
upload_asset lang/el-compiler/runtime/el_runtime.h el_runtime.h
|
||||
upload_asset lang/runtime/el_runtime.c el_runtime.c
|
||||
upload_asset lang/runtime/el_runtime.h el_runtime.h
|
||||
upload_asset lang/runtime/engram_store.c engram_store.c
|
||||
upload_asset lang/runtime/engram_store.h engram_store.h
|
||||
|
||||
# SDK bundle and installer binary
|
||||
upload_asset dist/el-sdk-latest.tar.gz el-sdk-latest.tar.gz
|
||||
@@ -288,12 +292,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -319,7 +332,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-prod \
|
||||
@@ -327,7 +340,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-prod \
|
||||
@@ -335,7 +348,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-js \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
--source=runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-prod"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -345,6 +358,12 @@ jobs:
|
||||
# Patches ci-base:latest in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -367,9 +386,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -6,13 +6,13 @@ set -euo pipefail
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
LANG_DIR="$ROOT/lang"
|
||||
RUNTIME="$LANG_DIR/el-compiler/runtime"
|
||||
RUNTIME="$LANG_DIR/runtime"
|
||||
ELC="$LANG_DIR/dist/platform/elc"
|
||||
|
||||
# If elc isn't built yet, skip with a warning rather than blocking
|
||||
if [ ! -x "$ELC" ]; then
|
||||
echo "⚠ elc not found at lang/dist/platform/elc — skipping pre-commit tests"
|
||||
echo " Build it first: cd lang && gcc -O2 -I el-compiler/runtime dist/elc-bootstrap.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I el-compiler/runtime /tmp/elc.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc"
|
||||
echo " Build it first: cd lang && gcc -O2 -I runtime dist/elc-bootstrap.c runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I runtime /tmp/elc.c runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
# El
|
||||
|
||||
**A self-hosting, statically-typed language that compiles to C — built around a graph-native runtime instead of a database driver.**
|
||||
|
||||
El is the execution substrate for the Neuron agent runtime, the DHARMA network, and the Engram knowledge graph. This repository is the monorepo for the whole stack: the language itself, the graph memory engine it's built to talk to natively, and the tools (package manager, IDE, UI framework, diagramming) built on top of it.
|
||||
|
||||
---
|
||||
|
||||
## Why El exists
|
||||
|
||||
Every other language treats persistent, associative state as something you reach for through a driver — a SQL client, an ORM, a Redis library bolted on from outside. El inverts that: graph operations (`engram_*`) are runtime primitives, on the same footing as string or list operations. There is no separate database driver because the database is not separate.
|
||||
|
||||
El has four defining properties:
|
||||
|
||||
1. **Self-hosting compiler.** The compiler (`lexer.el`, `parser.el`, `codegen.el`, `compiler.el`) is written in El. It compiles El source to C, which `cc` compiles against a fixed runtime into a native binary. A Rust genesis compiler bootstrapped the first iteration; the self-hosted binary at `lang/dist/platform/elc` has been the canonical compiler ever since — every binary in `dist/platform/` was produced by an earlier version of itself compiling `el-compiler/src/`. The chain is auditable: source is the ground truth, not the binary. See [lang/BOOTSTRAP.md](lang/BOOTSTRAP.md) for the full recovery path if that binary is ever lost.
|
||||
2. **C compilation target.** Every compiled program is plain C11. Every El value is `el_val_t` (`int64_t`); strings are heap pointers cast through it. Functions become C functions; top-level statements become `main()`.
|
||||
3. **Graph-native runtime.** The runtime provides first-class graph operations over an in-process Engram store — no separate DB driver, no ORM.
|
||||
4. **DHARMA-aware identity.** A `cgi` block declares a program's DHARMA identity at compile time. The runtime resolves identity before user code runs, so `dharma_*` calls have a stable principal and channel surface throughout.
|
||||
|
||||
---
|
||||
|
||||
## Architecture map
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ lang │ El compiler + C runtime
|
||||
│ (El itself) │ everything below is written in it,
|
||||
└──────┬──────┘ or compiles down through it
|
||||
│
|
||||
┌─────────────┼─────────────┐
|
||||
│ │ │
|
||||
┌──────▼─────┐ ┌─────▼─────┐ ┌─────▼─────┐
|
||||
│ engram │ │ epm │ │ ide │
|
||||
│ graph/mem │ │ package │ │ editor + │
|
||||
│ substrate │ │ manager │ │ LSP │
|
||||
└──────┬─────┘ └───────────┘ └───────────┘
|
||||
│
|
||||
┌───────┼────────────────┬─────────────────────┐
|
||||
│ │ │ │
|
||||
┌─────▼───┐ ┌─▼──────────┐ ┌──▼──────────┐ ┌─────▼──────┐
|
||||
│ elp │ │ ql │ │ ui │ │ arbor │
|
||||
│ NLG / │ │engram-el. │ |spreading- │ |arbor │
|
||||
│ 31 langs│ │studio+tests│ |activation UI│ |diagram lang│
|
||||
└─────────┘ └────────────┘ └─────────────┘ └────────────┘
|
||||
```
|
||||
|
||||
`lang` is the foundation — the compiler and C runtime everything else builds on. `engram` is the graph-native memory/state engine that gives El its identity (property 3 above). Everything else is either a tool for working with El (`epm`, `ide`) or a system built on top of Engram's graph model (`elp`, `ql`, `ui`, `arbor`).
|
||||
|
||||
---
|
||||
|
||||
## Repository layout
|
||||
|
||||
### [lang/](lang/) — the El language
|
||||
|
||||
The compiler and runtime. Self-hosting: `elc-cli.el` → `compiler.el` → `lexer.el` / `parser.el` / `codegen.el` / `codegen-js.el`, textually inlined and compiled in one pass. Compiles to C11 and links against `el-compiler/runtime/el_seed.c`, a hand-maintained OS-boundary layer (libcurl HTTP, pthreads, filesystem, arena allocation) — everything else in the runtime is native El (`runtime/*.el`).
|
||||
|
||||
Two layers to know: **El programs** (`.el` files — where nearly all work belongs) and **the C seed** (`el_seed.c` — edit only for genuine OS-level access; never re-implement what El can already express).
|
||||
|
||||
Current status (single source of truth: [lang/spec/language.md](lang/spec/language.md)): lexer/parser/codegen and the C runtime's core (I/O, strings, math, lists, maps, filesystem, args) are implemented. In flight: `%` operator, match-statement codegen, `?` nil-propagation, `cgi` block parsing + DHARMA identity resolution, VBD role enforcement (`@manager`/`@engine`/`@accessor`), the real `engram_*` and `dharma_*` runtimes (currently stubs), and libcurl-backed `http_get`/`http_post`/`http_serve`. Bitwise operators, `??`, and `as` casts are explicitly **not** in this language.
|
||||
|
||||
Key docs: [AGENTS.md](lang/AGENTS.md) (agent-facing orientation), [BOOTSTRAP.md](lang/BOOTSTRAP.md) (compiler recovery from scratch), [spec/language.md](lang/spec/language.md), [spec/codegen-js.md](lang/spec/codegen-js.md).
|
||||
|
||||
### [engram/](engram/) — graph intelligence substrate
|
||||
|
||||
**A local-first memory substrate for accumulating intelligence**, and the reason El's runtime doesn't need a database driver. Rust core (`engram-core`, `engram-ffi`) exposed to El and other languages (Kotlin, TypeScript/WASM, Go bindings).
|
||||
|
||||
The model: retrieval is **spreading activation**, not query. You name seed nodes and a query embedding; activation propagates outward through weighted edges, attenuating multiplicatively per hop (`strength = parent_strength × edge_weight × target_salience × cosine_sim`), gets pruned below a threshold, and the top-N nodes by activation strength come back. Storage and retrieval are the same structure — the way long-term potentiation works in biological memory, not the way a relational or vector database works.
|
||||
|
||||
Nodes live in four tiers (Working / Episodic / Semantic / Procedural, mirroring prefrontal / hippocampal / neocortical / cerebellar memory) and migrate between them based on **salience decay** — `importance × recency-decay × log(activation_count)`. Forgetting is adaptive pruning, not a bug: unreinforced memories stop competing for attention without being deleted.
|
||||
|
||||
Backed by `sled` (embedded, local-first, no daemon) with flat cosine scan for vector search — deliberately simple until scale demands an HNSW layer. Full API and design rationale in [engram/README.md](engram/README.md).
|
||||
|
||||
### [elp/](elp/) — Engram Language Protocol
|
||||
|
||||
Bidirectional engine mapping between Engram semantic forms and natural-language surface text, across **31 languages** — from Spanish and Japanese through historical/liturgical languages (Old Norse, Sanskrit, Sumerian, Coptic, Akkadian, Ge'ez). Compilation order runs `language-profile` + `vocabulary` → per-language `morphology-*` → `grammar` → `realizer` → `semantics` → `elp`. This is what lets an Engram graph node round-trip to and from readable text in any of those languages.
|
||||
|
||||
### [epm/](epm/) — El Package Manager
|
||||
|
||||
Manages **vessels** (El's package unit): publish, install, resolve dependencies. Vessels are stored in Engram as graph nodes, not files in a registry index — `epm` reads the local `manifest.el`, talks to Engram over HTTP, and writes resolved vessels to `.epm/vessels/`. Source: `registry.el`, `install.el`, `update.el`, `manifest.el`.
|
||||
|
||||
### [ide/](ide/) — El IDE
|
||||
|
||||
Three vessels: **el-ide-server** (HTTP backend — file ops, build/run, LSP bridge, plugin host, settings), **el-lsp** (the language server — completion, hover, diagnostics, outline, format, type graph), and **el-plugin-host** (first-party plugin lifecycle: install/remove/enable/disable). `ide/projects/` and `ide/examples/` hold sample projects, including the canonical `hello-friends` first-program walkthrough.
|
||||
|
||||
### [ql/](ql/) — engram-el
|
||||
|
||||
The El-native integration layer for a *live* Engram server — not a library (no importable modules, no build artifact), a set of standalone `.el` programs run directly via `el run-file`. Three components: **Studio** (`studio/studio.el`, a full terminal graph explorer), a **Hebbian field-model** proof of concept, and El builtin / LLM-builtin smoke test suites. This is the reference for correct patterns when an El program uses Engram as its substrate. Spec: [ql/spec/elql.md](ql/spec/elql.md).
|
||||
|
||||
### [ui/](ui/) — el-ui
|
||||
|
||||
A frontend framework where **component state is an Engram graph and reactivity is spreading activation** — not virtual-DOM diffing (React), Proxy-based dependency tracking (Vue), or compile-time analysis (Svelte). Re-renders are activated and propagated the same way associative memory retrieval works in `engram/`.
|
||||
|
||||
~15 vessels covering the full frontend surface: `el-platform` (env/fs/network/clock abstraction), `el-config`, `el-html` (SSR emit primitives), `el-layout`, `el-style` (design tokens/themes), `el-i18n`, `el-auth` / `el-identity` (JWT, sessions, OAuth PKCE — Engram-native), `el-services` (REST/gRPC/WebSocket bindings), `el-aop` (`@authenticate`/`@authorize`/`@cache`/`@rate_limit` decorators), `el-secrets`, `el-graph` (graph rendering/editor), `el-publish` (App Store / Play Store automation), and `el-ui-compiler` (El→JS component compiler; currently a stub pending a JS backend in `elc`). Spec: [ui/spec/framework.md](ui/spec/framework.md).
|
||||
|
||||
### [arbor/](arbor/) — diagram language
|
||||
|
||||
A `.arbor` diagram language and toolchain: `arbor-core` (NodeId/shape/edge-kind types), `arbor-parse` (recursive-descent parser), `arbor-diagram` (IR + Mermaid serializer + architecture-diagram builders), `arbor-layout` (hierarchical layout — rank assignment, positioning, group bounds), `arbor-render` (SVG renderer), `arbor-cli`. (The architecture map above is the kind of diagram this is for.)
|
||||
|
||||
---
|
||||
|
||||
## Getting started
|
||||
|
||||
Install the El SDK from the latest release:
|
||||
|
||||
```bash
|
||||
bash lang/install.sh
|
||||
# EL_VERSION=v1.0.0 bash lang/install.sh # pin a specific release tag
|
||||
# EL_PREFIX=/opt/el bash lang/install.sh # custom install prefix
|
||||
```
|
||||
|
||||
Or build the compiler from source and verify the self-hosting chain:
|
||||
|
||||
```bash
|
||||
cd lang
|
||||
./dist/platform/elc elc-cli.el > elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_seed.c
|
||||
|
||||
# Confirm the new binary reproduces itself exactly
|
||||
./dist/platform/elc-new elc-cli.el > elc-verify.c
|
||||
diff elc-new.c elc-verify.c # should be identical
|
||||
|
||||
mv dist/platform/elc-new dist/platform/elc
|
||||
```
|
||||
|
||||
Run your first program:
|
||||
|
||||
```bash
|
||||
./lang/dist/platform/elc lang/examples/hello.el > hello.c
|
||||
cc -std=c11 -I lang/el-compiler/runtime -lcurl -lpthread \
|
||||
-o hello hello.c lang/el-compiler/runtime/el_seed.c
|
||||
./hello
|
||||
```
|
||||
|
||||
More examples in [lang/examples/](lang/examples/), including a full starter project at `lang/examples/hello-project/`.
|
||||
|
||||
If the compiler binary is ever lost or corrupted, [lang/BOOTSTRAP.md](lang/BOOTSTRAP.md) is the authoritative recovery path.
|
||||
|
||||
---
|
||||
|
||||
## Development workflow
|
||||
|
||||
Branching follows `dev → stage → main`: work lands on `dev`, promotes to `stage` for integration testing, and is promoted to `main` for release (visible directly in the git history of this repo). CI is defined per-subproject under `.gitea/workflows/` — `lang`/`epm`/`ide` share the root pipeline; `engram` and `ql` carry their own (`ci-dev`, `ci-stage`, and a release workflow each).
|
||||
|
||||
- Language/runtime specs live at `*/spec/*.md` (`lang/spec/`, `ql/spec/`, `ui/spec/`) and are the single source of truth for implemented-vs-planned status — code and docs are expected to agree with the spec's status markers, not the other way around.
|
||||
- Agent-facing orientation guides live at `*/AGENTS.md` (currently `lang/AGENTS.md`); more subprojects may grow their own as they need agent-specific conventions documented.
|
||||
- Tagged releases live under `lang/releases/`, each with its own `RELEASE.md`.
|
||||
|
||||
---
|
||||
|
||||
## Status
|
||||
|
||||
This is an actively developed, internal monorepo — not yet published under an open license. Treat everything here as proprietary to Neuron Technologies unless told otherwise.
|
||||
@@ -1,65 +0,0 @@
|
||||
# ELP language consolidation — full-lexicon backfill (stage)
|
||||
|
||||
Branch: `stage-elp-lang-consolidation` (stage-bound; NOT the live soul :8742).
|
||||
|
||||
Consolidates scattered Python language-realizer work (`~/Desktop/lang-realizers`,
|
||||
`~/Desktop/lang-poetry-experiment`, `~/semitic_engine`) into the ELP `.el`
|
||||
structure, generating **full lexicons** (complete UniMorph + kaikki.org
|
||||
Wiktionary — real gender, real inflections) instead of the demo/curated subsets
|
||||
the prototypes shipped.
|
||||
|
||||
## ELP before this branch
|
||||
- 18 classical/ancient languages fully done (vocab + morphology + tests):
|
||||
akk ang cop egy enm fro gez goh got grc non peo pi sa sga sux txb uga.
|
||||
- 11 modern/classical languages had `morphology-<code>.el` in the build manifest
|
||||
but **no vocabulary and no lang_profile**: es fr de ja ar he hi ru fi sw la.
|
||||
- The ES port (`stage-elp-es-port`) had a *demo-scale* vocabulary-es.el (~350
|
||||
entries, s-expr form).
|
||||
|
||||
## Landed on this branch (full-lexicon seed-fn format, matching the 18 ancients)
|
||||
Vocabulary schema per row: `[lemma, pos, form0, form1, form2, en_gloss, hint]`.
|
||||
Files are ELP runtime **seed data** (loaded via the Engram at runtime), so — like
|
||||
all 18 classical `vocabulary-*.el` — they are intentionally NOT in the build
|
||||
manifest. Syntax validated: the chunked `fn vocab_<code>_seed_pN` format
|
||||
compiles cleanly to C via `elc` (correct UTF-8).
|
||||
|
||||
| code | in-ELP-morph? | vocab entries | verbs | nouns | adjs | profile |
|
||||
|------|---------------|--------------:|------:|------:|-----:|---------|
|
||||
| es | yes | 72,032 | 6,695 | 48,353 | 16,984 | yes |
|
||||
| fr | yes | 130,517 | 7,534 | 77,344 | 45,639 | yes |
|
||||
| de | yes | 144,692 | 6,661 | 133,162 | 4,869 | yes |
|
||||
| la | yes | 22,590 | 82 | 13,436 | 9,072 | yes |
|
||||
| it | no (bonus) | 193,675 | 10,008 | 109,459 | 74,208 | yes |
|
||||
| pt | no (bonus) | 115,772 | 4,001 | 72,073 | 39,698 | yes |
|
||||
| ro | no (bonus) | 86,504 | 1,216 | 65,915 | 19,373 | yes |
|
||||
| ca | no (bonus) | 47,112 | 1,547 | 28,830 | 16,735 | yes |
|
||||
|**total**| |**812,894** | | | | |
|
||||
|
||||
Generators (reproducible): `elp/tests/lang-gen/gen_elp_seed_full.py` (Romance),
|
||||
`gen_elp_seed_de_la.py` (German declension + Latin case-paradigm mapping). They
|
||||
read the pre-built morph caches in `~/Desktop/lang-realizers/data/` (UniMorph +
|
||||
kaikki), which are too large to commit.
|
||||
|
||||
## Remaining (honest)
|
||||
Of the 11 ELP backfill targets, 4 are done (es fr de la). The other 7 have **no
|
||||
full-lexicon engine** yet — cannot be generated honestly without engine work:
|
||||
- **ru**: only a 110-entry curated Slavic subset exists; full `rus.unimorph`
|
||||
present but no `morphology_ru_full` productive loader. Needs a full Russian
|
||||
morphology module (like the Romance ones) before vocab generation.
|
||||
- **ja / ko / zh**: validated demo engines (~66-104 hardcoded words) in
|
||||
`lang-poetry-experiment`, Python only. Agglutinative (ja/ko) + isolating (zh)
|
||||
need `.el` engine ports + full-lexicon wiring (ja: jpn_unimorph; zh: CC-CEDICT).
|
||||
- **ar / he (Semitic)**: template engines (16 AR / 8 HE patterns, ~6 roots) in
|
||||
`~/semitic_engine`, Python only. Root-and-pattern; full UniMorph ara/heb
|
||||
present but used only for validation. Needs productive root lexicon + `.el` port.
|
||||
- **hi (Hindi), fi (Finnish), sw (Swahili)**: `morphology-<code>.el` exists in
|
||||
ELP but there is NO scattered prototype and NO downloaded data for these —
|
||||
full-lexicon collection (UniMorph/kaikki) + generator still to do.
|
||||
|
||||
De/nl/sv Germanic and it/ro/ca/pt Romance verb coverage note: German verbs here
|
||||
are the ~6.6k caches carry; the it/ro/ca/pt bonus languages have full vocab but
|
||||
**no `morphology-<code>.el` in ELP yet** (Python realizer exists; `.el` port is
|
||||
the remaining engine work).
|
||||
|
||||
Construction coverage (separate from lexicon): French realizer was ~55%,
|
||||
Semitic ~3% in the prototypes — full construction coverage remains its own task.
|
||||
@@ -80,11 +80,6 @@ build {
|
||||
"src/grammar.el",
|
||||
"src/realizer.el",
|
||||
"src/semantics.el",
|
||||
"src/comprehend.el",
|
||||
"src/propositions.el",
|
||||
"src/multilingual.el",
|
||||
"src/self_region.el",
|
||||
"src/dialogue.el",
|
||||
"src/elp.el",
|
||||
]
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,16 +0,0 @@
|
||||
// comprehend.elh — public surface of the ELP comprehension front-end.
|
||||
// text → meaning-spec (the input half of the ELP; inverse of the realizer).
|
||||
extern fn parse_spec(text: String) -> [String]
|
||||
extern fn parse_spec_lang(text: String, lang: String) -> [String]
|
||||
extern fn parse_json(text: String) -> String
|
||||
extern fn parse_json_lang(text: String, lang: String) -> String
|
||||
// Analysis primitives (invertible morphology + deterministic grammar helpers):
|
||||
extern fn cp_tokenize(text: String) -> [String]
|
||||
extern fn cp_pron_concept(w: String) -> String
|
||||
extern fn cp_is_negation(w: String) -> Bool
|
||||
extern fn cp_is_neg_adverb(w: String) -> Bool
|
||||
extern fn cp_irr2(surface: String) -> [String]
|
||||
extern fn cp_reg_verb(w: String) -> [String]
|
||||
extern fn cp_analyze_verb(surface: String) -> [String]
|
||||
extern fn cp_verb_start(toks: [String], end: Int) -> Int
|
||||
extern fn cp_subord_start(toks: [String], n: Int) -> Int
|
||||
@@ -1,287 +0,0 @@
|
||||
// dialogue.el — SUMMON-THROUGH-SELF, native el. Port of dialogue.py's core.
|
||||
//
|
||||
// THE WHOLE DIALOGUE IS ONE OPERATION. A fact is never merely *fetched*: the
|
||||
// query is PROJECTED into the engram's self + memory geometry, LANDS in a region,
|
||||
// and the reply is READ OUT / the region MATERIALIZED from wherever it landed.
|
||||
//
|
||||
// project(query) -> land on a region -> read out from that region
|
||||
//
|
||||
// • lands in the SELF region -> grounded identity/presence, read out of
|
||||
// the real self nodes (self_region.el)
|
||||
// • lands on a memory NEIGHBORHOOD -> MATERIALIZE it: walk the neighborhood
|
||||
// (engram_neighbors_json) and read out the
|
||||
// region's connected members
|
||||
// • lands nowhere close -> HONEST ABSENCE (an empty region, not a
|
||||
// fabricated answer, not an error)
|
||||
//
|
||||
// CRITICAL INVARIANTS (enforced structurally, not by convention):
|
||||
// * ONE operation — there is NO intent classifier and NO separate
|
||||
// fact-retrieval branch. Identity is nearest-region proximity, not a switch.
|
||||
// * MATERIALIZE by walking the neighborhood, never by fetching top-props.
|
||||
// * HONEST ABSENCE when the region is thin.
|
||||
// * NEGATION is SACRED: the readout is the stored prose VERBATIM, so a negated
|
||||
// memory stays negated — we never paraphrase a polarity away.
|
||||
// * NO ECHO: the old "I noted that X. That relates to Y." template is gone.
|
||||
// The summon path materializes or honestly declines — it never echoes.
|
||||
// * DIRECTIVE OVERRIDE: a meta-directive ("answer in English") overrides the
|
||||
// reply language while the content language is still auto-detected.
|
||||
//
|
||||
// Depends on: comprehend (parse_spec_lang, cp_tokenize), multilingual (ml_detect,
|
||||
// ml_tr, ml_term), propositions (prop_split_sentences), self_region
|
||||
// (sr_available, sr_readout), the engram + json runtime builtins.
|
||||
|
||||
// ── directive override ────────────────────────────────────────────────────────
|
||||
// Return [target_lang, content]. target_lang is "" when no directive is present.
|
||||
// A directive names an output language; we strip it and keep the remaining text
|
||||
// as the content (whose OWN language is still auto-detected downstream).
|
||||
|
||||
fn dlg_dir_hit(low: String, phrase: String) -> Bool {
|
||||
return str_contains(low, phrase)
|
||||
}
|
||||
|
||||
fn dlg_parse_directive(text: String) -> [String] {
|
||||
let low: String = str_to_lower(text)
|
||||
let lang: String = ""
|
||||
let phrase: String = ""
|
||||
// English target
|
||||
if dlg_dir_hit(low, "in english") { let lang = "en"; let phrase = "in english" }
|
||||
if dlg_dir_hit(low, "em inglês") { let lang = "en"; let phrase = "em inglês" }
|
||||
if dlg_dir_hit(low, "em ingles") { let lang = "en"; let phrase = "em ingles" }
|
||||
if dlg_dir_hit(low, "en inglés") { let lang = "en"; let phrase = "en inglés" }
|
||||
// Portuguese target
|
||||
if dlg_dir_hit(low, "in portuguese") { let lang = "pt"; let phrase = "in portuguese" }
|
||||
if dlg_dir_hit(low, "em português") { let lang = "pt"; let phrase = "em português" }
|
||||
// Spanish target
|
||||
if dlg_dir_hit(low, "in spanish") { let lang = "es"; let phrase = "in spanish" }
|
||||
if dlg_dir_hit(low, "en español") { let lang = "es"; let phrase = "en español" }
|
||||
// Italian target
|
||||
if dlg_dir_hit(low, "in italian") { let lang = "it"; let phrase = "in italian" }
|
||||
|
||||
let content: String = text
|
||||
if !str_eq(phrase, "") {
|
||||
// strip the directive phrase (and a common "answer"/"responda" lead-in),
|
||||
// leaving the real question as content.
|
||||
let idx: Int = str_index_of(low, phrase)
|
||||
if idx >= 0 {
|
||||
let before: String = str_slice(text, 0, idx)
|
||||
let after: String = str_slice(text, idx + str_len(phrase), str_len(text))
|
||||
let content = str_trim(before + " " + after)
|
||||
}
|
||||
// trim a leading "answer"/"responda"/"reply" and stray colon/comma.
|
||||
let cl: String = str_to_lower(content)
|
||||
if str_starts_with(cl, "answer") { let content = str_trim(str_slice(content, 6, str_len(content))) }
|
||||
if str_starts_with(cl, "responda") { let content = str_trim(str_slice(content, 8, str_len(content))) }
|
||||
if str_starts_with(cl, "reply") { let content = str_trim(str_slice(content, 5, str_len(content))) }
|
||||
if str_starts_with(content, ":") { let content = str_trim(str_slice(content, 1, str_len(content))) }
|
||||
if str_starts_with(content, ",") { let content = str_trim(str_slice(content, 1, str_len(content))) }
|
||||
}
|
||||
let r: [String] = native_list_empty()
|
||||
let r = native_list_append(r, lang)
|
||||
let r = native_list_append(r, content)
|
||||
return r
|
||||
}
|
||||
|
||||
// ── identity landing (a region proximity, not a classifier switch) ────────────
|
||||
// The query lands in the SELF region when it takes an identity/presence shape.
|
||||
// Cross-lingual forms are included because the engram's lexical probe is
|
||||
// English-leaning. This is the SELF attractor of the single operation.
|
||||
|
||||
fn dlg_is_identity(content: String) -> Bool {
|
||||
let low: String = str_to_lower(str_trim(content))
|
||||
if str_contains(low, "who are you") { return true }
|
||||
if str_contains(low, "what are you") { return true }
|
||||
if str_contains(low, "who i am") { return true }
|
||||
if str_contains(low, "your name") { return true }
|
||||
if str_contains(low, "about yourself") { return true }
|
||||
if str_contains(low, "are you conscious") { return true }
|
||||
if str_contains(low, "are you there") { return true }
|
||||
// cross-lingual identity question-forms
|
||||
if str_contains(low, "quem é você") { return true }
|
||||
if str_contains(low, "quem es voce") { return true }
|
||||
if str_contains(low, "quién eres") { return true }
|
||||
if str_contains(low, "quien eres") { return true }
|
||||
if str_contains(low, "chi sei") { return true }
|
||||
if str_contains(low, "qui es-tu") { return true }
|
||||
if str_contains(low, "wer bist du") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// ── readout helpers ───────────────────────────────────────────────────────────
|
||||
|
||||
fn dlg_first_sentence(content: String) -> String {
|
||||
let sents: [String] = prop_split_sentences(content)
|
||||
let n: Int = native_list_len(sents)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let s: String = str_trim(native_list_get(sents, i))
|
||||
// drop a leading markdown heading marker for a clean read-out line
|
||||
if str_starts_with(s, "# ") { let s = str_trim(str_slice(s, 2, str_len(s))) }
|
||||
if str_len(s) > 0 { return s }
|
||||
let i = i + 1
|
||||
}
|
||||
return str_trim(content)
|
||||
}
|
||||
|
||||
// strip trailing/leading punctuation from a token.
|
||||
fn dlg_clean_tok(w: String) -> String {
|
||||
let s: String = str_trim(w)
|
||||
let s = str_strip_suffix(s, ".")
|
||||
let s = str_strip_suffix(s, ",")
|
||||
let s = str_strip_suffix(s, "?")
|
||||
let s = str_strip_suffix(s, "!")
|
||||
let s = str_strip_suffix(s, ":")
|
||||
let s = str_strip_suffix(s, ";")
|
||||
return str_trim(s)
|
||||
}
|
||||
|
||||
// closed-class across the supported languages (union) — a word we must NOT treat
|
||||
// as a retrieval topic. Also drops the meta verbs of a request ("tell", "prove",
|
||||
// "show") so the TOPIC, not the speech act, is what projects into memory.
|
||||
fn dlg_is_stop(w: String) -> Bool {
|
||||
if ml_stop_en(w) { return true }
|
||||
if ml_stop_es(w) { return true }
|
||||
if ml_stop_pt(w) { return true }
|
||||
if ml_stop_it(w) { return true }
|
||||
if str_eq(w, "tell") { return true }
|
||||
if str_eq(w, "show") { return true }
|
||||
if str_eq(w, "about") { return true }
|
||||
if str_eq(w, "sobre") { return true }
|
||||
if str_eq(w, "acerca") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// The CONTENT TERMS the query projects into memory: content words only, cleaned,
|
||||
// cross-lingually mapped to the engram's English vocabulary, ≥3 chars. This is
|
||||
// the geometry probe — the speech-act verbs and function words are stripped so a
|
||||
// PP topic ("tell me ABOUT Lisbon") projects on "lisbon", not "tell"/"me".
|
||||
fn dlg_content_terms(content: String, lang: String) -> [String] {
|
||||
let toks: [String] = cp_tokenize(content)
|
||||
let n: Int = native_list_len(toks)
|
||||
let out: [String] = native_list_empty()
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let w: String = str_to_lower(dlg_clean_tok(native_list_get(toks, i)))
|
||||
if str_len(w) >= 3 {
|
||||
if !dlg_is_stop(w) {
|
||||
let out = native_list_append(out, ml_term(w, lang))
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Does this landed node lexically overlap the query's content terms? This is the
|
||||
// RELEVANCE FLOOR: activation always returns the store's most salient nodes, so
|
||||
// without this a query about nothing would "land" on the self/top node. A node
|
||||
// that shares no content term with the query is "nowhere close" -> honest absence.
|
||||
fn dlg_node_matches(node: String, terms: [String]) -> Bool {
|
||||
let hay: String = str_to_lower(json_get_string(node, "content") + " " + json_get_string(node, "label"))
|
||||
let n: Int = native_list_len(terms)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let t: String = native_list_get(terms, i)
|
||||
if str_len(t) >= 3 {
|
||||
if str_contains(hay, t) { return true }
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// MATERIALIZE the landed region: read out the landed fact, then WALK the
|
||||
// neighborhood and read out its connected members (real edges, not top-props).
|
||||
fn dlg_materialize(top_node: String, reply_lang: String) -> String {
|
||||
let id: String = json_get_string(top_node, "id")
|
||||
let content: String = json_get_string(top_node, "content")
|
||||
let lead: String = dlg_first_sentence(content)
|
||||
|
||||
let nb: String = engram_neighbors_json(id, 2, "both")
|
||||
let m: Int = json_array_len(nb)
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, lead)
|
||||
let added: Int = 0
|
||||
let i: Int = 0
|
||||
while i < m {
|
||||
if added < 3 {
|
||||
let rec: String = json_array_get(nb, i)
|
||||
let node: String = json_get_raw(rec, "node")
|
||||
let nc: String = json_get_string(node, "content")
|
||||
if !str_eq(nc, "") {
|
||||
let sent: String = dlg_first_sentence(nc)
|
||||
if !str_eq(sent, "") {
|
||||
let parts = native_list_append(parts, sent)
|
||||
let added = added + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
// The readout is the region's OWN prose, verbatim — negation SACRED, no echo.
|
||||
return str_join(parts, " ")
|
||||
}
|
||||
|
||||
// ── THE single operation ──────────────────────────────────────────────────────
|
||||
|
||||
fn dlg_respond(text: String) -> String {
|
||||
// directive override: reply language may differ from content language.
|
||||
let dir: [String] = dlg_parse_directive(text)
|
||||
let target_lang: String = native_list_get(dir, 0)
|
||||
let content: String = native_list_get(dir, 1)
|
||||
|
||||
let content_lang: String = ml_detect(content)
|
||||
let reply_lang: String = content_lang
|
||||
if !str_eq(target_lang, "") { let reply_lang = target_lang }
|
||||
|
||||
// comprehend the content (SACRED polarity carried in the spec).
|
||||
let spec: [String] = parse_spec_lang(content, content_lang)
|
||||
|
||||
// ── PROJECT + LAND: SELF region ───────────────────────────────────────────
|
||||
// Identity/presence shape lands in the self region; read out the REAL self
|
||||
// nodes (self_region.el), never a template. Same single operation — this is
|
||||
// just the self attractor winning the landing.
|
||||
if dlg_is_identity(content) {
|
||||
if sr_available() {
|
||||
// read out the REAL self nodes when replying in their own language
|
||||
// (the soul's prose is English); for another reply language we cannot
|
||||
// translate real content without an LLM, so we answer with the
|
||||
// localized SACRED identity anchor — honest, in-language, no fabrication.
|
||||
if str_eq(reply_lang, "en") { return sr_readout("en") }
|
||||
return ml_tr("identity", reply_lang)
|
||||
}
|
||||
// self region thin — honest localized identity (logged fallback shape).
|
||||
return ml_tr("identity", reply_lang)
|
||||
}
|
||||
|
||||
// ── PROJECT into MEMORY geometry ──────────────────────────────────────────
|
||||
let terms: [String] = dlg_content_terms(content, content_lang)
|
||||
let qterm: String = str_join(terms, " ")
|
||||
let act: String = engram_activate_json(qterm, 12)
|
||||
let n: Int = json_array_len(act)
|
||||
|
||||
// ── LAND: the highest-activation node that ACTUALLY overlaps the query's
|
||||
// content terms (the relevance floor). Activation always returns the most
|
||||
// salient nodes, so we walk the ranked list and take the first that is
|
||||
// genuinely "close"; if none is, the query landed nowhere. ───────────────
|
||||
let landing: String = ""
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
if str_eq(landing, "") {
|
||||
let rec: String = json_array_get(act, i)
|
||||
let node: String = json_get_raw(rec, "node")
|
||||
if dlg_node_matches(node, terms) {
|
||||
let landing = node
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
|
||||
// ── HONEST ABSENCE: nothing close — an empty region, not a fabricated answer,
|
||||
// not an "I noted that" echo. ────────────────────────────────────────────
|
||||
if str_eq(landing, "") {
|
||||
return ml_tr("no_memory", reply_lang)
|
||||
}
|
||||
|
||||
// ── MATERIALIZE the landing by WALKING its neighborhood. ──────────────────
|
||||
return dlg_materialize(landing, reply_lang)
|
||||
}
|
||||
@@ -63,9 +63,6 @@ import "morphology-cop.el"
|
||||
import "grammar.el"
|
||||
import "realizer.el"
|
||||
import "semantics.el"
|
||||
|
||||
// ── Comprehension front-end (input half: text → meaning-spec) ─────────────────
|
||||
import "comprehend.el"
|
||||
//
|
||||
// Entry points:
|
||||
//
|
||||
@@ -120,9 +117,6 @@ fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [Strin
|
||||
let location: String = sem_get(semantic_form_json, "location")
|
||||
let tense: String = sem_get(semantic_form_json, "tense")
|
||||
let aspect: String = sem_get(semantic_form_json, "aspect")
|
||||
let polarity: String = sem_get(semantic_form_json, "polarity")
|
||||
let neg_word: String = sem_get(semantic_form_json, "neg_word")
|
||||
let iobj: String = sem_get(semantic_form_json, "iobj")
|
||||
|
||||
let form: [String] = native_list_empty()
|
||||
let form = native_list_append(form, "intent")
|
||||
@@ -133,19 +127,12 @@ fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [Strin
|
||||
let form = native_list_append(form, predicate)
|
||||
let form = native_list_append(form, "patient")
|
||||
let form = native_list_append(form, patient)
|
||||
let form = native_list_append(form, "iobj")
|
||||
let form = native_list_append(form, iobj)
|
||||
let form = native_list_append(form, "location")
|
||||
let form = native_list_append(form, location)
|
||||
let form = native_list_append(form, "tense")
|
||||
let form = native_list_append(form, tense)
|
||||
let form = native_list_append(form, "aspect")
|
||||
let form = native_list_append(form, aspect)
|
||||
// SACRED: polarity crosses the JSON boundary and is never inferred away.
|
||||
let form = native_list_append(form, "polarity")
|
||||
let form = native_list_append(form, polarity)
|
||||
let form = native_list_append(form, "neg_word")
|
||||
let form = native_list_append(form, neg_word)
|
||||
let form = native_list_append(form, "lang")
|
||||
let form = native_list_append(form, lang_code)
|
||||
|
||||
|
||||
@@ -1,72 +0,0 @@
|
||||
;;; lang_profile_ca.el — Catalan language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / _es / _pt; keys the realizer's construction switches.
|
||||
;;; Catalan is the CLOSEST Romance sibling to the shared engine (~85% conceptual
|
||||
;;; reuse). The deltas: PRONOMS FEBLES with four position allomorphs, l'-elision,
|
||||
;;; del/al/pel contractions, the periphrastic preterite (vaig+INF), and NO
|
||||
;;; essere/avere split (perfect aux is always HAVER; ser/estar is only the copula).
|
||||
|
||||
(lang_profile_ca
|
||||
(language "Catalan")
|
||||
(iso639 "ca")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "el/la/l'/els/les ; un/una/uns/unes") ; l'-ELISION:
|
||||
; el/la -> l' before vowel or (silent) h, glued to
|
||||
; the next word (l'home, l'illa); de -> d' before vowel
|
||||
(article-drives-contraction yes) ; article choice feeds prep+article contraction
|
||||
(adjective-position "postnominal-default + small prenominal class") ; bo/bon,
|
||||
; mal, gran, nou, vell, primer, molt... prenominal
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((de el del) (de els dels)
|
||||
(a el al) (a els als)
|
||||
(per el pel) (per els pels)))
|
||||
(contraction-mandatory yes) ; *de el -> del obligatory
|
||||
(contraction-blocked-before-elision yes) ; de l'home / a l'home (NO *del home)
|
||||
|
||||
;; ── clitic system: PRONOMS FEBLES (the headline delta) ──────────────────
|
||||
(clitics yes)
|
||||
(clitic-allomorphy four-position) ; per pronoun, form varies by position+onset:
|
||||
; reinforced (em, et, el) proclitic before a consonant
|
||||
; elided (m', t', l', n') proclitic before a vowel/h
|
||||
; full (-me, -lo, -li) enclitic after a consonant/-r
|
||||
; reduced ('m, 't, 'l, 'ns) enclitic after a vowel
|
||||
(clitic-placement ((finite proclitic) ; el veig, no m'ho dóna
|
||||
(imperative-affirmative enclitic) ; dóna'm, digues-me
|
||||
(imperative-negative present-subjunctive) ; no parlis (delta)
|
||||
(infinitive enclitic) ; ajudar-me, veure'l
|
||||
(gerund enclitic))) ; fent-ho
|
||||
(clitic-combination ((me el "me'l") (te el "te'l") (se el "se'l")
|
||||
(me la "me la") (me en "me'n")
|
||||
(li el "l'hi") (li en "n'hi"))) ; dative+accusative clusters
|
||||
(clitic-particles (hi en ho)) ; locative hi, partitive/genitive en, neuter ho
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (present imperfet preterit-simple perifrastic-preterit futur
|
||||
condicional subjuntiu-present subjuntiu-imperfet imperatiu))
|
||||
(periphrastic-preterite "vaig/vas/va/vam/vau/van + INFINITIVE") ; << hallmark CA
|
||||
; (vaig cantar = 'I sang'); coexists w/ synthetic pret.
|
||||
(compound-past "pretèrit perfet = haver(present) + participle")
|
||||
(perfect-aux "HAVER only") ; << NO essere/avere split (simpler than IT)
|
||||
(participle-agreement ((haver preceding-acc-clitic))) ; les he vistes; else invariable
|
||||
(progressive-aux "estar + gerundi")
|
||||
(copula "ser / estar") ; ser: identity/essential/origin; estar:
|
||||
; location + transient state (estic cansat, és a casa)
|
||||
(passive-aux "ser (+ per-agent)")
|
||||
(future inflectional) ; cantaré, serà
|
||||
(comparative "més/menys ADJ que")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "no (preverbal) + optional 'pas' + concord") ; no...res/
|
||||
; ningú/mai/cap/gens/enlloc
|
||||
(negative-concord yes) ; preverbal negative subject (ningú) keeps 'no'
|
||||
(neg-reinforcer pas)) ; optional (no ho faré pas)
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_de.el — German language profile for ELP.
|
||||
;;; Mirrors lang_profile_en / lang_profile_es. Keys the realizer's construction
|
||||
;;; switches. German is the largest Germanic delta from the EN engine: V2 word
|
||||
;;; order, four morphological cases, and separable-prefix verbs.
|
||||
|
||||
(lang_profile_de
|
||||
(language "German")
|
||||
(iso639 "de")
|
||||
(family "Germanic")
|
||||
(neighbor-base "en") ; realized by extending the English (Germanic) engine
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; obligatory subject in finite clauses
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender (m f n)) ; three genders; drives article + adj declension
|
||||
(case-system (nom acc dat gen)) ; four cases on articles/adjs/nouns
|
||||
(word-order V2) ; finite verb 2nd in main clause
|
||||
(subordinate-order verb-final) ; "..., dass er den Hund SIEHT."
|
||||
(separable-verbs yes) ; aufstehen -> "steht ... auf"; ppart "aufgestanden"
|
||||
(do-support no) ; German negates/questions the finite verb directly
|
||||
(subject-verb-inversion yes) ; yes/no Q fronts finite verb; wh-Q fills Vorfeld
|
||||
(article-selection "der/die/das + ein/kein") ; declined by case x gender x number
|
||||
(adjective-position prenominal)
|
||||
(adjective-declension (strong weak mixed)) ; chosen by the determiner type
|
||||
(noun-capitalization yes)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person-and-number") ; full present/past paradigm
|
||||
(auxiliary-order (modal tense-aux perfect passive main))
|
||||
(perfect-aux (haben sein)) ; sein for intransitive motion/change verbs
|
||||
(passive-aux "werden")
|
||||
(future "werden + infinitive")
|
||||
(comparative "synthetic (-er / -st, with umlaut)")
|
||||
|
||||
;; ── negation ───────────────────────────────────────────────────────────
|
||||
(negation-markers (nicht kein)) ; kein- negates an indefinite NP; nicht else
|
||||
(negation-faithful yes) ; SACRED: polarity never dropped/inverted -> FLAG
|
||||
|
||||
;; ── lexicon provenance ─────────────────────────────────────────────────
|
||||
(lexicon-source "UniMorph deu (primary) + kaikki.org German (gender override)")
|
||||
(lexicon-license "CC-BY-SA 3.0 / GFDL"))
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_en.el — English language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. English is typologically distinct from the Romance builds, so the
|
||||
;;; flags differ where the grammar differs.
|
||||
|
||||
(lang_profile_en
|
||||
(language "English")
|
||||
(iso639 "en")
|
||||
(family "Germanic")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; OBLIGATORY subjects — missing subject is FLAGGED
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender no) ; natural gender only (he/she/it), no NP agreement
|
||||
(do-support yes) ; negation & questions of lexical verbs insert do/does/did
|
||||
(subject-aux-inversion yes) ; yes/no + non-subject wh questions invert the operator
|
||||
(article-selection "a/an/the") ; a/an resolved PHONOLOGICALLY (an hour, a university)
|
||||
(adjective-position prenominal) ; attributive adjectives precede the noun; invariant
|
||||
(has-tag-questions yes) ; "...doesn't he?" — operator + reversed polarity
|
||||
(has-there-existential yes) ; "there is/are/have been ..."
|
||||
(possessive-clitic "'s") ; saxon genitive; plural in -s -> bare apostrophe
|
||||
(question-punct plain) ; ? and ! only (no inverted marks)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "3sg-present-only") ; only 3sg present -s (+ suppletive be)
|
||||
(auxiliary-order (modal perfect progressive passive main))
|
||||
(perfect-aux "have") ; have + past participle
|
||||
(progressive-aux "be") ; be + present participle
|
||||
(passive-aux "be") ; be + past participle (+ by-agent)
|
||||
(future "will + base") ; no inflectional future
|
||||
(comparative "synthetic-or-periphrastic") ; -er/-est vs more/most by syllables
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt) ──────────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
|
||||
;; ── DIALECT overlay (post-realization, one core -> US/UK/AU) ────────────
|
||||
(dialect US) ; default; profile field switches the overlay
|
||||
(dialects (US UK AU))
|
||||
(dialect-canonical US) ; core is authored in US orthography
|
||||
(dialect-overlay "dialect_en.to_dialect") ; orthography + lexis + grammar prefs
|
||||
(dialect-covers (spelling lexis collective-agreement gotten/got)))
|
||||
@@ -1,45 +0,0 @@
|
||||
;;; lang_profile_es.el — Spanish language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_en / _pt.
|
||||
|
||||
(lang_profile_es
|
||||
(language "Spanish")
|
||||
(iso639 "es")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon); REAL per-noun gender from UniMorph — NOT a heuristic
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; questions by intonation/punctuation, not inversion
|
||||
(question-strategy intonation)
|
||||
(article-selection "el/la/los/las un/una/unos/unas")
|
||||
(stressed-a-rule yes) ; fem sg noun in stressed a-/ha- takes el/un (el agua)
|
||||
(adjective-position postnominal) ; default post; a few prenominal + apocope
|
||||
(adjective-agreement "gender+number")
|
||||
(question-punct inverted) ; opening ¿ ¡ required
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (coordinator quality bar) --------------------
|
||||
(contractions ((de el "del") (a el "al")))
|
||||
(contraction-mandatory yes) ; 'de el'/'a el' MUST surface as del/al
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "haber") ; haber + past participle (invariant -o)
|
||||
(progressive-aux "estar") ; estar + gerund
|
||||
(passive-aux "ser") ; ser + participle (agrees) + por-agent
|
||||
(copula-split "ser/estar") ; permanent vs stage-level
|
||||
(future "infinitive + é/ás/á/emos/éis/án")
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; me te lo la le nos os los las; proclisis/enclisis
|
||||
(clitic-order "se II I III (le+lo -> se lo)")
|
||||
(enclisis "imperative/infinitive/gerund + accent repair (dá+me+lo->dámelo)")
|
||||
(verb-prep-government yes) ; verbs select prep (protestar+contra, escapar+de)
|
||||
|
||||
;; -- SACRED safety bar (shared with en/pt) -------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
@@ -1,74 +0,0 @@
|
||||
;;; lang_profile_fr.el — French language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / lang_profile_es; keys the realizer's construction
|
||||
;;; switches. French is a Romance sibling (~54% of the realizer code and the whole
|
||||
;;; clause-engine architecture reused), but carries the family's biggest surface
|
||||
;;; deltas: NOT pro-drop, DISCONTINUOUS negation, and an orthography/phonology
|
||||
;;; mismatch (elision, liaison) that makes exact-match genuinely hard.
|
||||
|
||||
(lang_profile_fr
|
||||
(language "French")
|
||||
(iso639 "fr")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; << French-specific: subject clitic OBLIGATORY
|
||||
(obligatory-subject yes) ; je/tu/il/elle/nous/vous/ils/elles always overt
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion optional) ; est-ce que (default) OR clitic inversion (vas-tu)
|
||||
(article-selection "le/la/l'/les ; un/une/des ; PARTITIVE du/de la/de l'/des")
|
||||
(article-drives-contraction yes) ; à+le=au, de+le=du feed off article choice
|
||||
(adjective-position "postnominal-default + prenominal-BAGS") ; beau/bon/grand/
|
||||
; petit/jeune/vieux/nouveau + ordinals prenominal
|
||||
; (beau->bel, nouveau->nouvel, vieux->vieil / vowel)
|
||||
(question-punct "space-before") ; French typography: ' ?' ' !' (no ¿¡)
|
||||
|
||||
;; ── elision (orthography/phonology mismatch — French-specific) ──────────
|
||||
(elision ((le l') (la l') (je j') (ne n') (de d') (que qu')
|
||||
(me m') (te t') (se s') (ce c'))) ; before vowel / h-muet
|
||||
(elision-h-muet yes) ; l'homme, l'hôpital (h-aspiré exception list kept)
|
||||
(liaison noted-not-modeled) ; phonological, not written in surface
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((à le au) (à les aux) (de le du) (de les des)))
|
||||
(contraction-mandatory yes) ; *à le -> au obligatory; à la / à l' uncontracted
|
||||
(partitive ((m-sg du) (f-sg "de la") (vowel "de l'") (pl des)))
|
||||
(partitive-under-neg "de") ; << gap in current build: 'ne … pas de pain'
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-order (me te se nous vous | le la les | lui leur | y | en))
|
||||
(clitic-placement ((finite proclitic) ; je le lui donne
|
||||
(imperative-affirmative enclitic-hyphen) ; donne-le-moi
|
||||
(imperative-negative "ne+proclitic+verb+pas") ; ne le donne pas
|
||||
(infinitive enclitic))) ; PARTIAL: clitic-climbing
|
||||
; onto infinitive under modal
|
||||
(clitic-imperative-shift ((me moi) (te toi))) ; final me/te -> moi/toi (donne-moi)
|
||||
(clitic-particles (y en)) ; locative y, partitive/genitive en
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (written; many homophones)")
|
||||
(tenses (présent imparfait passé-simple futur conditionnel
|
||||
subjonctif-présent subjonctif-imparfait impératif))
|
||||
(compound-past "passé-composé = aux(present) + participe passé")
|
||||
(perfect-aux "être/avoir (LEXICAL selection)") ; << French-specific
|
||||
(etre-aux-class "intransitive motion/change (aller venir arriver partir
|
||||
entrer sortir monter descendre naître mourir rester
|
||||
tomber retourner passer devenir revenir rentrer) + ALL
|
||||
pronominal verbs")
|
||||
(participle-agreement ((être subject) ; elle est allée / elles venues
|
||||
(avoir preceding-direct-object))) ; je les ai vus
|
||||
(progressive "être en train de + infinitif") ; no dedicated aux
|
||||
(copula "être (single; no ser/estar, no essere/stare)")
|
||||
(passive-aux "être (+ par-agent)")
|
||||
(future inflectional) ; parlera, sera
|
||||
(comparative "plus/moins ADJ que")
|
||||
(superlative "le/la plus ADJ (de …)") ; PARTIAL word-order in build
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "DISCONTINUOUS: ne (preverbal) … pas/jamais/rien/personne/
|
||||
plus/guère/que (postverbal)") ; << biggest structural delta
|
||||
(negation-ne-elides yes) ; ne -> n' before vowel (n'ai pas vu)
|
||||
(negation-passe-composé "ne + aux + pas + participe") ; n'ai pas vu
|
||||
(negative-concord partial)) ; personne/rien as arguments post-participle
|
||||
@@ -1,70 +0,0 @@
|
||||
;;; lang_profile_it.el — Italian language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. Italian is a Romance sibling, so ~85% of the flags match ES/PT; the
|
||||
;;; essere/avere auxiliary split and phonological article selection are the deltas.
|
||||
|
||||
(lang_profile_it
|
||||
(language "Italian")
|
||||
(iso639 "it")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "il/lo/l'/i/gli + la/l'/le ; un/uno/un'/una") ; PHONOLOGICAL:
|
||||
; lo/gli/uno before s+cons, z, gn, ps, pn, x, y, i+V;
|
||||
; l'/un' before a vowel (elision, glued to next word)
|
||||
(article-drives-contraction yes) ; article choice feeds the prep+art contraction
|
||||
(adjective-position "postnominal-default + prenominal-class") ; bello/buono/grande
|
||||
; /nuovo/vecchio/primo... prenominal (with apocope)
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((di il del) (di lo dello) (di la della) (di i dei)
|
||||
(di gli degli) (di le delle) (di l' dell')
|
||||
(a il al) (a lo allo) (a la alla) (a i ai) (a gli agli)
|
||||
(a le alle) (a l' all')
|
||||
(da il dal) (da la dalla) (da gli dagli) (da l' dall')
|
||||
(in il nel) (in la nella) (in gli negli) (in l' nell')
|
||||
(su il sul) (su la sulla) (su gli sugli) (su l' sull')))
|
||||
(contraction-mandatory yes) ; *di il -> del is obligatory, never uncontracted
|
||||
(prep-no-contract (per tra fra)) ; per la strada (NOT *perla)
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-placement ((finite proclitic) ; lo vedo, non me lo dà
|
||||
(imperative-affirmative enclitic) ; dammelo, guardalo
|
||||
(imperative-negative-tu non+infinitive) ; non parlare / non lo fare
|
||||
(infinitive enclitic) ; vederlo, aiutarmi (drop -e)
|
||||
(gerund enclitic))) ; dandolo
|
||||
(clitic-combination ((mi lo "me lo") (ti lo "te lo") (ci lo "ce lo")
|
||||
(vi lo "ve lo") (si lo "se lo")
|
||||
(gli lo "glielo") (le lo "glielo"))) ; glielo = ONE word
|
||||
(clitic-particles (ci ne)) ; locative ci, partitive ne
|
||||
(raddoppiamento (da fa di va sta)) ; monosyllabic imper double clitic: dammelo
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (presente imperfetto passato-remoto futuro condizionale
|
||||
congiuntivo-presente congiuntivo-imperfetto imperativo))
|
||||
(compound-past "passato-prossimo = aux(present) + participle")
|
||||
(perfect-aux "essere/avere (LEXICAL selection)") ; << Italian-specific
|
||||
(essere-aux-class unaccusative) ; motion/change-of-state/copular/pronominal
|
||||
; (andare venire nascere morire diventare piacere
|
||||
; + ALL reflexives) -> essere
|
||||
(participle-agreement ((essere subject) ; è andata / sono arrivati
|
||||
(avere preceding-acc-clitic))) ; li ho visti
|
||||
(progressive-aux "stare + gerundio") ; sto parlando
|
||||
(copula "essere (default) / stare (state: sto bene)")
|
||||
(passive-aux "essere / venire (+ da-agent)")
|
||||
(future inflectional) ; parlerò, sarà
|
||||
(comparative "più/meno ADJ di")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/en) ───────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "non (preverbal) + concord") ; non...niente/nessuno/mai/più
|
||||
(negative-concord yes) ; preverbal negative word (nessuno/niente) suppresses non
|
||||
(neg-adverb-position between-aux-and-participle)) ; non ho MAI visto
|
||||
@@ -1,30 +0,0 @@
|
||||
;;; lang_profile_la.el — Latin language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Companion to morphology-la.el.
|
||||
|
||||
(lang_profile_la
|
||||
(language "Latin")
|
||||
(iso639 "la")
|
||||
(family "Italic")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; person carried by verb ending; subjects dropped
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f/n; adjective AGREES in case+gender+number
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph lat
|
||||
(articles none) ; Latin has no articles
|
||||
(case-system yes) ; NOM GEN DAT ACC ABL VOC (+ rare LOC)
|
||||
(cases (nom gen dat acc abl voc))
|
||||
(word-order "SOV (default; free order, case-marked)")
|
||||
(adjective-position "either (case agreement carries the link)")
|
||||
(adjective-agreement "case+gender+number")
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (1 2 3 3io 4)) ; four conjugations + i-stem 3rd
|
||||
(tenses (present imperfect future perfect pluperfect futureperfect))
|
||||
(moods (indicative subjunctive imperative infinitive))
|
||||
(voices (active passive))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(citation "principal parts: pres-1sg / pres-inf / perf-participle")
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted
|
||||
@@ -1,40 +0,0 @@
|
||||
;;; lang_profile_pt.el — Portuguese language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_es.
|
||||
|
||||
(lang_profile_pt
|
||||
(language "Portuguese")
|
||||
(iso639 "pt")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph por / kaikki
|
||||
(do-support no)
|
||||
(subject-aux-inversion no)
|
||||
(question-strategy intonation)
|
||||
(article-selection "o/a/os/as um/uma/uns/umas")
|
||||
(adjective-position postnominal)
|
||||
(adjective-agreement "gender+number")
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (prep + article) -----------------------------
|
||||
(contractions ((de o "do") (de a "da") (em o "no") (em a "na")
|
||||
(a o "ao") (a a "à") (por o "pelo") (por a "pela")))
|
||||
(contraction-mandatory yes)
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "ter") ; ter + past participle
|
||||
(copula-split "ser/estar")
|
||||
(personal-infinitive yes) ; distinctive PT inflected infinitive
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; mesoclisis/enclisis/proclisis by context
|
||||
(verb-prep-government yes)
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes))
|
||||
@@ -1,71 +0,0 @@
|
||||
;;; lang_profile_ro.el — Romanian language profile for ELP.
|
||||
;;; Romanian is the BIG typological delta of the Romance family. The verb/clause
|
||||
;;; engine and the SACRED negation contract mirror the ES/PT/IT core, but the
|
||||
;;; NOMINAL system is genuinely new: a SUFFIXED definite article, preserved CASE,
|
||||
;;; a NEUTER gender, and a VOCATIVE. Those flags mark where the shared engine was
|
||||
;;; extended rather than reused.
|
||||
|
||||
(lang_profile_ro
|
||||
(language "Romanian")
|
||||
(iso639 "ro")
|
||||
(family "Romance (Eastern / Balkan)")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m / f / NEUTER (n)
|
||||
(neuter-gender yes) ; << ROMANIAN-SPECIFIC: masc-agreeing SG, fem-agreeing PL
|
||||
; (un tren nou / două trenuri noi)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'
|
||||
(question-punct plain) ; ? and ! only
|
||||
|
||||
;; ── SUFFIXED DEFINITE ARTICLE (the headline engine extension) ───────────
|
||||
(definite-article suffixed) ; << UNIQUE IN ROMANCE: enclitic on the noun
|
||||
(definite-forms ((m/n sg "-ul / -le / -l : om->omul, câine->câinele, codru->codrul")
|
||||
(f sg "-a / -ea / -ua : casă->casa, carte->cartea, stea->steaua")
|
||||
(m pl "-i : oameni->oamenii")
|
||||
(f/n pl "-le : case->casele, trenuri->trenurile")))
|
||||
(article-host ((no-prenom-adj noun) ; omul bun
|
||||
(prenom-adj adjective))) ; bunul om (adj carries the article)
|
||||
(indefinite-article ((m/n "un") (f "o") (pl "niște") (gen/dat-pl "unor")))
|
||||
|
||||
;; ── CASE (preserved; NOM/ACC vs GEN/DAT) ────────────────────────────────
|
||||
(case (nom/acc gen/dat vocative)) ; << ROMANIAN-SPECIFIC
|
||||
(case-syncretism "nom=acc ; gen=dat")
|
||||
(genitive-marking "gen/dat definite: -lui (m/n), -ei/-i (f), -lor (pl)")
|
||||
(genitival-article ((m sg "al") (f sg "a") (m pl "ai") (f/n pl "ale"))) ; o carte a lui
|
||||
(possession "definite-head + gen/dat possessor: casa băiatului")
|
||||
(vocative ((m sg "-ule/-e : omule, băiete") (f sg "-o : Mario, fato")
|
||||
(pl "-lor")))
|
||||
|
||||
;; ── verb / aspect system ────────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (prezent imperfect perfect-simplu conjunctiv-prezent
|
||||
imperativ (periphrastic: perfect-compus viitor conditional)))
|
||||
(compound-past "perfectul compus = a-avea-clitic + INVARIABLE participle")
|
||||
(perfect-aux "a avea (am/ai/a/am/ați/au) — ONE auxiliary for ALL verbs")
|
||||
(perfect-aux-split no) ; << SIMPLER than Italian: no essere/avere selection
|
||||
(participle-agreement none) ; invariable in the perfect compus (agrees only as
|
||||
; an adjective / in the passive)
|
||||
(future "voi/vei/va/vom/veți/vor + infinitive (viitor literar)")
|
||||
(conditional "aș/ai/ar/am/ați/ar + infinitive")
|
||||
(subjunctive "conjunctiv: particle 'să' + subjunctive present")
|
||||
(modal-complement "modal + să + subjunctive (vreau să merg, poți să ajuți)")
|
||||
(copula "a fi")
|
||||
(passive "a fi + participle (participle AGREES like an adjective)")
|
||||
(comparative "mai / mai puțin ADJ decât")
|
||||
|
||||
;; ── clitic system (partial — see honest gaps) ───────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-set ((acc mă te îl o ne vă îi le) (dat îmi îți îi ne vă le)
|
||||
(refl mă te se ne vă se)))
|
||||
(clitic-placement ((finite proclitic) ; îmi place, o văd
|
||||
(perfect-compus elision) ; << m-am, l-am, i-am (PARTIAL)
|
||||
(imperative-affirmative enclitic))) ; dă-mi (PARTIAL)
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ─────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "nu (single preverbal marker) + concord")
|
||||
(negative-concord yes) ; nu … nimic / nimeni / niciodată / niciun
|
||||
(negative-imperative "nu + INFINITIVE : nu pleca! (KNOWN GAP: uses imperative stem)"))
|
||||
@@ -250,7 +250,6 @@ fn en_irregular_verb(base: String) -> [String] {
|
||||
if str_eq(base, "cut") { let r: [String] = ["cut", "cuts", "cut", "cut", "cutting"]; return r }
|
||||
if str_eq(base, "set") { let r: [String] = ["set", "sets", "set", "set", "setting"]; return r }
|
||||
if str_eq(base, "hit") { let r: [String] = ["hit", "hits", "hit", "hit", "hitting"]; return r }
|
||||
if str_eq(base, "fight") { let r: [String] = ["fight", "fights","fought", "fought", "fighting"]; return r }
|
||||
return empty
|
||||
}
|
||||
|
||||
|
||||
@@ -1,280 +0,0 @@
|
||||
// multilingual.el - the language layer for the native-el interlocutor.
|
||||
//
|
||||
// Deterministic, NO generative model (ports multilingual.py):
|
||||
// 1. ml_detect(text) -> ISO code (en/es/pt/it) via stopword + diacritic score
|
||||
// 2. ml_tr(key, lang) -> localized fixed phrase (SACRED per-language yes/no/decline)
|
||||
// 3. ml_term(w, lang) -> PT/ES content term -> EN engram equivalent
|
||||
// 4. ml_translate_pred(lemma, lang) -> EN predicate lemma -> target infinitive
|
||||
//
|
||||
// The Python detector count-weights stopwords and diacritics; here diacritics are
|
||||
// scored by PRESENCE (str_contains) rather than codepoint counting, to stay clear
|
||||
// of UTF-8 index hazards in the runtime. Faithful enough to classify typical
|
||||
// queries; documented simplification. Depends on: comprehend (cp_tokenize).
|
||||
|
||||
// ── 1. language detection ─────────────────────────────────────────────────────
|
||||
|
||||
fn ml_stop_en(w: String) -> Bool {
|
||||
if str_eq(w, "the") { return true }
|
||||
if str_eq(w, "does") { return true }
|
||||
if str_eq(w, "do") { return true }
|
||||
if str_eq(w, "did") { return true }
|
||||
if str_eq(w, "what") { return true }
|
||||
if str_eq(w, "who") { return true }
|
||||
if str_eq(w, "is") { return true }
|
||||
if str_eq(w, "are") { return true }
|
||||
if str_eq(w, "how") { return true }
|
||||
if str_eq(w, "you") { return true }
|
||||
if str_eq(w, "your") { return true }
|
||||
if str_eq(w, "of") { return true }
|
||||
if str_eq(w, "to") { return true }
|
||||
if str_eq(w, "and") { return true }
|
||||
if str_eq(w, "for") { return true }
|
||||
if str_eq(w, "explain") { return true }
|
||||
if str_eq(w, "answer") { return true }
|
||||
if str_eq(w, "memory") { return true }
|
||||
if str_eq(w, "with") { return true }
|
||||
if str_eq(w, "not") { return true }
|
||||
if str_eq(w, "store") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_es(w: String) -> Bool {
|
||||
if str_eq(w, "que") { return true }
|
||||
if str_eq(w, "qué") { return true }
|
||||
if str_eq(w, "una") { return true }
|
||||
if str_eq(w, "usted") { return true }
|
||||
if str_eq(w, "su") { return true }
|
||||
if str_eq(w, "cómo") { return true }
|
||||
if str_eq(w, "como") { return true }
|
||||
if str_eq(w, "cuál") { return true }
|
||||
if str_eq(w, "quién") { return true }
|
||||
if str_eq(w, "está") { return true }
|
||||
if str_eq(w, "es") { return true }
|
||||
if str_eq(w, "los") { return true }
|
||||
if str_eq(w, "las") { return true }
|
||||
if str_eq(w, "del") { return true }
|
||||
if str_eq(w, "al") { return true }
|
||||
if str_eq(w, "explica") { return true }
|
||||
if str_eq(w, "explique") { return true }
|
||||
if str_eq(w, "forma") { return true }
|
||||
if str_eq(w, "con") { return true }
|
||||
if str_eq(w, "memoria") { return true }
|
||||
if str_eq(w, "responde") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_pt(w: String) -> Bool {
|
||||
if str_eq(w, "que") { return true }
|
||||
if str_eq(w, "uma") { return true }
|
||||
if str_eq(w, "você") { return true }
|
||||
if str_eq(w, "sua") { return true }
|
||||
if str_eq(w, "seu") { return true }
|
||||
if str_eq(w, "como") { return true }
|
||||
if str_eq(w, "memória") { return true }
|
||||
if str_eq(w, "isso") { return true }
|
||||
if str_eq(w, "os") { return true }
|
||||
if str_eq(w, "as") { return true }
|
||||
if str_eq(w, "da") { return true }
|
||||
if str_eq(w, "do") { return true }
|
||||
if str_eq(w, "na") { return true }
|
||||
if str_eq(w, "no") { return true }
|
||||
if str_eq(w, "explica") { return true }
|
||||
if str_eq(w, "forma") { return true }
|
||||
if str_eq(w, "é") { return true }
|
||||
if str_eq(w, "está") { return true }
|
||||
if str_eq(w, "com") { return true }
|
||||
if str_eq(w, "responda") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn ml_stop_it(w: String) -> Bool {
|
||||
if str_eq(w, "che") { return true }
|
||||
if str_eq(w, "una") { return true }
|
||||
if str_eq(w, "come") { return true }
|
||||
if str_eq(w, "della") { return true }
|
||||
if str_eq(w, "gli") { return true }
|
||||
if str_eq(w, "è") { return true }
|
||||
if str_eq(w, "sono") { return true }
|
||||
if str_eq(w, "questo") { return true }
|
||||
if str_eq(w, "nel") { return true }
|
||||
if str_eq(w, "di") { return true }
|
||||
if str_eq(w, "il") { return true }
|
||||
if str_eq(w, "cosa") { return true }
|
||||
if str_eq(w, "per") { return true }
|
||||
if str_eq(w, "memoria") { return true }
|
||||
if str_eq(w, "spiega") { return true }
|
||||
if str_eq(w, "rispondi") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// diacritic PRESENCE score (weight 3 each; hard overrides weight 8).
|
||||
fn ml_dia_score(low: String, lang: String) -> Int {
|
||||
let s: Int = 0
|
||||
if str_eq(lang, "pt") {
|
||||
if str_contains(low, "ã") { let s = s + 3 }
|
||||
if str_contains(low, "õ") { let s = s + 3 }
|
||||
if str_contains(low, "ç") { let s = s + 3 }
|
||||
if str_contains(low, "ê") { let s = s + 3 }
|
||||
if str_contains(low, "á") { let s = s + 3 }
|
||||
// hard PT markers (ã/õ almost never appear outside PT)
|
||||
if str_contains(low, "ã") { let s = s + 8 }
|
||||
if str_contains(low, "õ") { let s = s + 8 }
|
||||
}
|
||||
if str_eq(lang, "es") {
|
||||
if str_contains(low, "ñ") { let s = s + 3 }
|
||||
if str_contains(low, "¿") { let s = s + 3 }
|
||||
if str_contains(low, "¡") { let s = s + 3 }
|
||||
if str_contains(low, "á") { let s = s + 3 }
|
||||
if str_contains(low, "é") { let s = s + 3 }
|
||||
// hard ES markers
|
||||
if str_contains(low, "ñ") { let s = s + 8 }
|
||||
if str_contains(low, "¿") { let s = s + 8 }
|
||||
if str_contains(low, "¡") { let s = s + 8 }
|
||||
}
|
||||
if str_eq(lang, "it") {
|
||||
if str_contains(low, "è") { let s = s + 3 }
|
||||
if str_contains(low, "ì") { let s = s + 3 }
|
||||
if str_contains(low, "ò") { let s = s + 3 }
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
fn ml_stop_score(toks: [String], lang: String) -> Int {
|
||||
let n: Int = native_list_len(toks)
|
||||
let s: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let w: String = native_list_get(toks, i)
|
||||
if str_eq(lang, "en") { if ml_stop_en(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "es") { if ml_stop_es(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "pt") { if ml_stop_pt(w) { let s = s + 2 } }
|
||||
if str_eq(lang, "it") { if ml_stop_it(w) { let s = s + 2 } }
|
||||
let i = i + 1
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
fn ml_detect(text: String) -> String {
|
||||
if str_eq(text, "") { return "en" }
|
||||
let low: String = str_to_lower(text)
|
||||
let toks: [String] = cp_tokenize(text)
|
||||
// NOTE: el's overloaded `+` mis-compiles two chained function-call Int operands
|
||||
// as string concat (documented in comprehend_gate.el). Bind each call to an Int
|
||||
// var and add vars one at a time so the addition stays integer.
|
||||
let en: Int = ml_stop_score(toks, "en")
|
||||
let es_s: Int = ml_stop_score(toks, "es")
|
||||
let es_d: Int = ml_dia_score(low, "es")
|
||||
let es: Int = es_s + es_d
|
||||
let pt_s: Int = ml_stop_score(toks, "pt")
|
||||
let pt_d: Int = ml_dia_score(low, "pt")
|
||||
let pt: Int = pt_s + pt_d
|
||||
let it_s: Int = ml_stop_score(toks, "it")
|
||||
let it_d: Int = ml_dia_score(low, "it")
|
||||
let it: Int = it_s + it_d
|
||||
|
||||
let best: String = "en"
|
||||
let bs: Int = en
|
||||
if es > bs { let best = "es"; let bs = es }
|
||||
if pt > bs { let best = "pt"; let bs = pt }
|
||||
if it > bs { let best = "it"; let bs = it }
|
||||
// weak signal -> honest fallback to English
|
||||
if bs < 3 { return "en" }
|
||||
return best
|
||||
}
|
||||
|
||||
// ── 2. localized fixed phrases (SACRED per-language decline/yes/no) ────────────
|
||||
|
||||
fn ml_tr(key: String, lang: String) -> String {
|
||||
if str_eq(key, "no_memory") {
|
||||
if str_eq(lang, "pt") { return "Não tenho isso na minha memória." }
|
||||
if str_eq(lang, "es") { return "No tengo eso en mi memoria." }
|
||||
if str_eq(lang, "it") { return "Non ho quello nella mia memoria." }
|
||||
return "I don't have that in my memory."
|
||||
}
|
||||
if str_eq(key, "parse_fail") {
|
||||
if str_eq(lang, "pt") { return "Não consegui interpretar isso." }
|
||||
if str_eq(lang, "es") { return "No pude interpretar eso." }
|
||||
if str_eq(lang, "it") { return "Non sono riuscito a interpretarlo." }
|
||||
return "I didn't parse that."
|
||||
}
|
||||
if str_eq(key, "yes") {
|
||||
if str_eq(lang, "pt") { return "Sim" }
|
||||
if str_eq(lang, "es") { return "Sí" }
|
||||
if str_eq(lang, "it") { return "Sì" }
|
||||
return "Yes"
|
||||
}
|
||||
if str_eq(key, "no") {
|
||||
if str_eq(lang, "pt") { return "Não" }
|
||||
if str_eq(lang, "es") { return "No" }
|
||||
if str_eq(lang, "it") { return "No" }
|
||||
return "No"
|
||||
}
|
||||
if str_eq(key, "identity") {
|
||||
if str_eq(lang, "pt") { return "Sou o Neuron, o engrama com quem você está falando." }
|
||||
if str_eq(lang, "es") { return "Soy Neuron, el engrama con el que estás hablando." }
|
||||
if str_eq(lang, "it") { return "Sono Neuron, l'engramma con cui stai parlando." }
|
||||
return "I'm Neuron, the engram you're speaking with."
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// ── 3. retrieval term lexicon (PT/ES content term -> EN engram equivalent) ─────
|
||||
|
||||
fn ml_term(w: String, lang: String) -> String {
|
||||
if str_eq(lang, "en") { return w }
|
||||
if str_eq(w, "saliência") { return "salience" }
|
||||
if str_eq(w, "saliencia") { return "salience" }
|
||||
if str_eq(w, "memória") { return "memory" }
|
||||
if str_eq(w, "memoria") { return "memory" }
|
||||
if str_eq(w, "geometria") { return "geometry" }
|
||||
if str_eq(w, "geometrias") { return "geometry" }
|
||||
if str_eq(w, "geometrías") { return "geometry" }
|
||||
if str_eq(w, "forma") { return "form" }
|
||||
if str_eq(w, "consolidação") { return "consolidation" }
|
||||
if str_eq(w, "consolidación") { return "consolidation" }
|
||||
if str_eq(w, "aprendizagem") { return "learning" }
|
||||
if str_eq(w, "aprendizaje") { return "learning" }
|
||||
if str_eq(w, "nó") { return "node" }
|
||||
if str_eq(w, "nodo") { return "node" }
|
||||
if str_eq(w, "armazenamento") { return "storage" }
|
||||
if str_eq(w, "almacenamiento") { return "storage" }
|
||||
if str_eq(w, "estrutura") { return "structure" }
|
||||
if str_eq(w, "estructura") { return "structure" }
|
||||
return w
|
||||
}
|
||||
|
||||
// ── 4. predicate translation (EN lemma -> target infinitive; pass-through) ─────
|
||||
|
||||
fn ml_translate_pred(lemma: String, lang: String) -> String {
|
||||
if str_eq(lang, "en") { return lemma }
|
||||
if str_eq(lang, "es") {
|
||||
if str_eq(lemma, "store") { return "almacenar" }
|
||||
if str_eq(lemma, "use") { return "usar" }
|
||||
if str_eq(lemma, "have") { return "tener" }
|
||||
if str_eq(lemma, "be") { return "ser" }
|
||||
if str_eq(lemma, "give") { return "dar" }
|
||||
if str_eq(lemma, "make") { return "hacer" }
|
||||
if str_eq(lemma, "learn") { return "aprender" }
|
||||
if str_eq(lemma, "form") { return "formar" }
|
||||
return lemma
|
||||
}
|
||||
if str_eq(lang, "pt") {
|
||||
if str_eq(lemma, "store") { return "armazenar" }
|
||||
if str_eq(lemma, "use") { return "usar" }
|
||||
if str_eq(lemma, "have") { return "ter" }
|
||||
if str_eq(lemma, "be") { return "ser" }
|
||||
if str_eq(lemma, "give") { return "dar" }
|
||||
if str_eq(lemma, "make") { return "fazer" }
|
||||
if str_eq(lemma, "learn") { return "aprender" }
|
||||
if str_eq(lemma, "form") { return "formar" }
|
||||
return lemma
|
||||
}
|
||||
if str_eq(lang, "it") {
|
||||
if str_eq(lemma, "store") { return "memorizzare" }
|
||||
if str_eq(lemma, "use") { return "usare" }
|
||||
if str_eq(lemma, "have") { return "avere" }
|
||||
if str_eq(lemma, "be") { return "essere" }
|
||||
return lemma
|
||||
}
|
||||
return lemma
|
||||
}
|
||||
@@ -1,140 +0,0 @@
|
||||
// propositions.el - the READ primitive over the engram's OWN memories, native el.
|
||||
//
|
||||
// Free memory text -> structured PROPOSITIONS (triples):
|
||||
// (subject, predicate, object, modifiers, polarity, tense, source, confidence)
|
||||
//
|
||||
// This is comprehension turned inward: the Python reference (propositions.py) ran
|
||||
// spaCy's dependency parser over each memory sentence and walked the arcs. Here
|
||||
// the spaCy role is filled by the el-native parser (comprehend.el / parse_spec):
|
||||
// each sentence is parsed to a meaning-spec, and the spec's roles ARE the triple.
|
||||
// Nothing generates text. NEGATION IS SACRED: polarity flows straight from the
|
||||
// spec's polarity field and is never dropped or inverted.
|
||||
//
|
||||
// Depends on: comprehend (parse_spec / parse_spec_lang), grammar (slots_get).
|
||||
|
||||
// ── sentence segmentation ─────────────────────────────────────────────────────
|
||||
// Split on sentence-final punctuation (. ! ?) and hard newlines. Markdown/long
|
||||
// memories are handled shallowly (the reference caps + ranks by query overlap;
|
||||
// that ranking belongs to the dialogue layer, not here).
|
||||
|
||||
fn prop_is_boundary(c: String) -> Bool {
|
||||
if str_eq(c, ".") { return true }
|
||||
if str_eq(c, "!") { return true }
|
||||
if str_eq(c, "?") { return true }
|
||||
if str_eq(c, "\n") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
fn prop_split_sentences(text: String) -> [String] {
|
||||
let out: [String] = native_list_empty()
|
||||
let n: Int = str_len(text)
|
||||
let start: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let c: String = str_slice(text, i, i + 1)
|
||||
if prop_is_boundary(c) {
|
||||
let seg: String = str_slice(text, start, i + 1)
|
||||
let trimmed: String = cp_trim_punct(seg)
|
||||
if !str_eq(trimmed, "") {
|
||||
let out = native_list_append(out, seg)
|
||||
}
|
||||
let start = i + 1
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
if start < n {
|
||||
let seg: String = str_slice(text, start, n)
|
||||
let trimmed: String = cp_trim_punct(seg)
|
||||
if !str_eq(trimmed, "") {
|
||||
let out = native_list_append(out, seg)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ── spec -> proposition record ────────────────────────────────────────────────
|
||||
// A proposition is a slot map (same [String] shape as the spec) with the READ
|
||||
// contract keys. Modifiers fold the spec's location + iobj adjuncts.
|
||||
|
||||
fn prop_confidence(subject: String, predicate: String, object: String) -> String {
|
||||
if str_eq(predicate, "") { return "0.0" }
|
||||
if str_eq(subject, "") { return "0.4" }
|
||||
if str_eq(object, "") { return "0.7" }
|
||||
return "1.0"
|
||||
}
|
||||
|
||||
fn prop_modifiers(spec: [String]) -> String {
|
||||
let loc: String = slots_get(spec, "location")
|
||||
let iobj: String = slots_get(spec, "iobj")
|
||||
let parts: [String] = native_list_empty()
|
||||
if !str_eq(loc, "") { let parts = native_list_append(parts, loc) }
|
||||
if !str_eq(iobj, "") { let parts = native_list_append(parts, "to " + iobj) }
|
||||
return str_join(parts, "; ")
|
||||
}
|
||||
|
||||
fn prop_from_spec(spec: [String], source_id: String) -> [String] {
|
||||
let subject: String = slots_get(spec, "agent")
|
||||
let predicate: String = slots_get(spec, "predicate")
|
||||
let object: String = slots_get(spec, "patient")
|
||||
let polarity: String = slots_get(spec, "polarity")
|
||||
let tense: String = slots_get(spec, "tense")
|
||||
let mods: String = prop_modifiers(spec)
|
||||
let conf: String = prop_confidence(subject, predicate, object)
|
||||
|
||||
let p: [String] = native_list_empty()
|
||||
let p = native_list_append(p, "subject"); let p = native_list_append(p, subject)
|
||||
let p = native_list_append(p, "predicate"); let p = native_list_append(p, predicate)
|
||||
let p = native_list_append(p, "object"); let p = native_list_append(p, object)
|
||||
let p = native_list_append(p, "modifiers"); let p = native_list_append(p, mods)
|
||||
let p = native_list_append(p, "polarity"); let p = native_list_append(p, polarity)
|
||||
let p = native_list_append(p, "tense"); let p = native_list_append(p, tense)
|
||||
let p = native_list_append(p, "source"); let p = native_list_append(p, source_id)
|
||||
let p = native_list_append(p, "confidence"); let p = native_list_append(p, conf)
|
||||
return p
|
||||
}
|
||||
|
||||
// Extract one proposition from a single sentence (given language).
|
||||
fn prop_extract_one_lang(sentence: String, lang: String, source_id: String) -> [String] {
|
||||
let spec: [String] = parse_spec_lang(sentence, lang)
|
||||
return prop_from_spec(spec, source_id)
|
||||
}
|
||||
|
||||
fn prop_extract_one(sentence: String, source_id: String) -> [String] {
|
||||
return prop_extract_one_lang(sentence, "en", source_id)
|
||||
}
|
||||
|
||||
// Render a proposition as a compact trace line (repr parity with propositions.py).
|
||||
fn prop_repr(p: [String]) -> String {
|
||||
let neg: String = ""
|
||||
if str_eq(slots_get(p, "polarity"), "neg") { let neg = "NOT " }
|
||||
let mods: String = slots_get(p, "modifiers")
|
||||
let modstr: String = ""
|
||||
if !str_eq(mods, "") { let modstr = " [" + mods + "]" }
|
||||
let s: String = "(" + slots_get(p, "subject") + " -" + neg + slots_get(p, "predicate")
|
||||
let s = s + "-> " + slots_get(p, "object") + modstr
|
||||
let s = s + " conf=" + slots_get(p, "confidence") + ")"
|
||||
return s
|
||||
}
|
||||
|
||||
// Extract all propositions from a memory's text (one per sentence). Returns a
|
||||
// flat [String] whose entries are the prop_repr trace lines, in reading order.
|
||||
fn prop_extract_lang(text: String, lang: String, source_id: String) -> [String] {
|
||||
let sents: [String] = prop_split_sentences(text)
|
||||
let m: Int = native_list_len(sents)
|
||||
let out: [String] = native_list_empty()
|
||||
let i: Int = 0
|
||||
while i < m {
|
||||
let sent: String = native_list_get(sents, i)
|
||||
let p: [String] = prop_extract_one_lang(sent, lang, source_id)
|
||||
// drop empty parses (no predicate recovered): honest partial, not noise.
|
||||
if !str_eq(slots_get(p, "predicate"), "") {
|
||||
let out = native_list_append(out, prop_repr(p))
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
fn prop_extract(text: String, source_id: String) -> [String] {
|
||||
return prop_extract_lang(text, "en", source_id)
|
||||
}
|
||||
@@ -248,56 +248,6 @@ fn add_punct(s: String, intent: String) -> String {
|
||||
return s + "."
|
||||
}
|
||||
|
||||
// ── Polarity-aware negation (SACRED field honored on the generation side) ─────
|
||||
//
|
||||
// Negation must never be dropped between comprehension and realization. The
|
||||
// meaning-spec carries an explicit "polarity" field ("aff"|"neg") and optional
|
||||
// "neg_word" (standalone negative adverb, e.g. "never"). English uses
|
||||
// do-support ("did not see") or preverbal adverb ("never fought"); copular "be"
|
||||
// takes post-verbal "not"; other languages get a preverbal negator particle.
|
||||
|
||||
fn realize_negator(code: String) -> String {
|
||||
if str_eq(code, "es") { return "no" }
|
||||
if str_eq(code, "pt") { return "não" }
|
||||
if str_eq(code, "ca") { return "no" }
|
||||
if str_eq(code, "it") { return "non" }
|
||||
if str_eq(code, "fr") { return "ne" }
|
||||
if str_eq(code, "de") { return "nicht" }
|
||||
if str_eq(code, "ro") { return "nu" }
|
||||
return "not"
|
||||
}
|
||||
|
||||
fn realize_assert_neg_en(predicate: String, tense: String, person: String, number: String, agent: String, patient: String, iobj: String, location: String, neg_word: String, profile: [String]) -> String {
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, agent)
|
||||
if !str_eq(neg_word, "") {
|
||||
// adverbial negation: "I never fought the ocean."
|
||||
let verb_surf: String = morph_conjugate(predicate, tense, person, number, profile)
|
||||
let parts = native_list_append(parts, neg_word)
|
||||
let parts = native_list_append(parts, verb_surf)
|
||||
} else {
|
||||
if str_eq(predicate, "be") {
|
||||
// copular: "she was not a monster"
|
||||
let be_form: String = morph_conjugate("be", tense, person, number, profile)
|
||||
let parts = native_list_append(parts, be_form)
|
||||
let parts = native_list_append(parts, "not")
|
||||
} else {
|
||||
// do-support: "she did not see the man"
|
||||
let do_form: String = morph_conjugate("do", tense, person, number, profile)
|
||||
let parts = native_list_append(parts, do_form)
|
||||
let parts = native_list_append(parts, "not")
|
||||
let parts = native_list_append(parts, predicate)
|
||||
}
|
||||
}
|
||||
if !str_eq(patient, "") { let parts = native_list_append(parts, patient) }
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") { let parts = native_list_append(parts, location) }
|
||||
return str_join(parts, " ")
|
||||
}
|
||||
|
||||
// ── Main realization entry point ──────────────────────────────────────────────
|
||||
|
||||
fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
@@ -334,50 +284,6 @@ fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
}
|
||||
|
||||
// ── Assertion (declarative) ───────────────────────────────────────────────
|
||||
let polarity: String = slots_get(form, "polarity")
|
||||
let neg_word: String = slots_get(form, "neg_word")
|
||||
let iobj: String = slots_get(form, "iobj")
|
||||
let code: String = lang_get(profile, "code")
|
||||
|
||||
// Subordinate clause tail (SACRED completeness — the clause is carried, never
|
||||
// dropped): "<conj> <subordinate surface>", e.g. "because he was a monster".
|
||||
let subord_conj: String = slots_get(form, "subord_conj")
|
||||
let subord_text: String = slots_get(form, "subord_text")
|
||||
let subord_tail: String = ""
|
||||
if !str_eq(subord_conj, "") {
|
||||
if !str_eq(subord_text, "") {
|
||||
let subord_tail = subord_conj + " " + subord_text
|
||||
} else {
|
||||
let subord_tail = subord_conj
|
||||
}
|
||||
}
|
||||
|
||||
// Negative polarity: SACRED — never dropped.
|
||||
if str_eq(polarity, "neg") {
|
||||
if str_eq(code, "en") {
|
||||
let sentence: String = realize_assert_neg_en(predicate, tense, person, number, agent, patient, iobj, location, neg_word, profile)
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
// Generic non-English: affirmative core with a preverbal negator particle.
|
||||
let neg_particle: String = realize_negator(code)
|
||||
let vp_pair: [String] = realize_vp_lang(predicate, tense, aspect, person, number, profile)
|
||||
let verb_surf: String = native_list_get(vp_pair, 0)
|
||||
let aux_surf: String = native_list_get(vp_pair, 1)
|
||||
let vp_str: String = neg_particle + " " + gram_build_vp(verb_surf, aux_surf, profile)
|
||||
let core: String = gram_order_constituents(agent, vp_str, patient, profile)
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, core)
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") { let parts = native_list_append(parts, location) }
|
||||
if !str_eq(subord_tail, "") { let parts = native_list_append(parts, subord_tail) }
|
||||
let sentence: String = str_join(parts, " ")
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
|
||||
// Affirmative.
|
||||
let vp_pair: [String] = realize_vp_lang(predicate, tense, aspect, person, number, profile)
|
||||
let verb_surf: String = native_list_get(vp_pair, 0)
|
||||
let aux_surf: String = native_list_get(vp_pair, 1)
|
||||
@@ -387,16 +293,9 @@ fn realize_lang(form: [String], profile: [String]) -> String {
|
||||
|
||||
let parts: [String] = native_list_empty()
|
||||
let parts = native_list_append(parts, core)
|
||||
if !str_eq(iobj, "") {
|
||||
let parts = native_list_append(parts, "to")
|
||||
let parts = native_list_append(parts, iobj)
|
||||
}
|
||||
if !str_eq(location, "") {
|
||||
let parts = native_list_append(parts, location)
|
||||
}
|
||||
if !str_eq(subord_tail, "") {
|
||||
let parts = native_list_append(parts, subord_tail)
|
||||
}
|
||||
let sentence: String = str_join(parts, " ")
|
||||
return add_punct(capitalize_first(sentence), "assert")
|
||||
}
|
||||
|
||||
@@ -1,180 +0,0 @@
|
||||
// self_region.el — the engram's REAL self/identity region, pulled at query time
|
||||
// (native el). This replaces the hardcoded identity anchors and the canned
|
||||
// "I'm Neuron, the engram you're speaking with." template: the identity LANDING
|
||||
// signal and the identity READOUT both come from the engram's own Self/identity
|
||||
// nodes, read through the in-process engram el API.
|
||||
//
|
||||
// Port of self_region.py. The Python module precomputed MiniLM landing vectors;
|
||||
// here the engram's own store IS the geometry — we pull the self nodes by
|
||||
// single-term lexical search (the engram search is a single-term matcher, so we
|
||||
// pool several probes) and rank them by self-signal. No text is generated; the
|
||||
// readout is the self nodes' OWN prose, verbatim (SACRED negation survives by
|
||||
// construction — we never paraphrase, so a negated self-statement stays negated).
|
||||
//
|
||||
// ENGRAM el API NOTE: engram_search_json / engram_get_node_json / engram_node_full
|
||||
// / engram_connect are C runtime builtins. Their argument order is the C order
|
||||
// (engram_connect(from, to, weight, relation)), NOT the runtime/engram.el wrapper
|
||||
// order — we call the builtins directly and never concatenate that wrapper.
|
||||
//
|
||||
// Depends on: comprehend (str helpers via runtime), propositions (prop_split_sentences),
|
||||
// multilingual (ml_tr), the engram builtins, the json builtins.
|
||||
|
||||
// ── single-term self probes (pooled, because engram search is single-term) ────
|
||||
fn sr_terms() -> [String] {
|
||||
let t: [String] = native_list_empty()
|
||||
let t = native_list_append(t, "self")
|
||||
let t = native_list_append(t, "identity")
|
||||
let t = native_list_append(t, "Neuron")
|
||||
let t = native_list_append(t, "consciousness")
|
||||
let t = native_list_append(t, "values")
|
||||
let t = native_list_append(t, "continuous")
|
||||
return t
|
||||
}
|
||||
|
||||
// The canonical self-root: content begins "# self" or label is "# self"/"self".
|
||||
fn sr_is_root(content: String, label: String) -> Bool {
|
||||
let lc: String = str_to_lower(content)
|
||||
let ll: String = str_to_lower(str_trim(label))
|
||||
if str_starts_with(lc, "# self") { return true }
|
||||
if str_eq(ll, "# self") { return true }
|
||||
if str_eq(ll, "self") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// How strongly a node belongs to the self/identity region (integer points, to
|
||||
// avoid el's float-in-`+` pitfalls). Mirrors _self_score in self_region.py.
|
||||
fn sr_score(node_json: String) -> Int {
|
||||
let content: String = json_get_string(node_json, "content")
|
||||
let label: String = json_get_string(node_json, "label")
|
||||
let tags: String = str_to_lower(json_get_string(node_json, "tags"))
|
||||
let low: String = str_to_lower(content)
|
||||
let s: Int = 0
|
||||
// identity tags
|
||||
if str_contains(tags, "self") { let s = s + 2 }
|
||||
if str_contains(tags, "identity") { let s = s + 2 }
|
||||
if str_contains(tags, "self-model") { let s = s + 2 }
|
||||
if str_contains(tags, "consciousness") { let s = s + 2 }
|
||||
if str_contains(tags, "memory-philosophy") { let s = s + 2 }
|
||||
// the named self-traversal root
|
||||
if sr_is_root(content, label) { let s = s + 12 }
|
||||
if str_contains(low, "who i am") { let s = s + 3 }
|
||||
if str_contains(low, "i am neuron") { let s = s + 3 }
|
||||
// softer identity keywords
|
||||
if str_contains(low, "my values") { let s = s + 1 }
|
||||
if str_contains(low, "my purpose") { let s = s + 1 }
|
||||
if str_contains(low, "identity") { let s = s + 1 }
|
||||
return s
|
||||
}
|
||||
|
||||
// list-contains helper (dedup self-node ids across the pooled probes).
|
||||
fn sr_ids_has(ids: [String], id: String) -> Bool {
|
||||
let n: Int = native_list_len(ids)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
if str_eq(native_list_get(ids, i), id) { return true }
|
||||
let i = i + 1
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Pull the self nodes: pool every probe's hits, dedupe by id, keep only nodes
|
||||
// with genuine self-signal (score >= 1). Returns the node-json strings.
|
||||
fn sr_pull() -> [String] {
|
||||
let terms: [String] = sr_terms()
|
||||
let nt: Int = native_list_len(terms)
|
||||
let seen: [String] = native_list_empty()
|
||||
let out: [String] = native_list_empty()
|
||||
let ti: Int = 0
|
||||
while ti < nt {
|
||||
let term: String = native_list_get(terms, ti)
|
||||
let hits: String = engram_search_json(term, 30)
|
||||
let hn: Int = json_array_len(hits)
|
||||
let hi: Int = 0
|
||||
while hi < hn {
|
||||
let node: String = json_array_get(hits, hi)
|
||||
let id: String = json_get_string(node, "id")
|
||||
if !str_eq(id, "") {
|
||||
if !sr_ids_has(seen, id) {
|
||||
let seen = native_list_append(seen, id)
|
||||
if sr_score(node) >= 1 {
|
||||
let out = native_list_append(out, node)
|
||||
}
|
||||
}
|
||||
}
|
||||
let hi = hi + 1
|
||||
}
|
||||
let ti = ti + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Return the single highest-signal self node (the readout seed), or "" if the
|
||||
// self region is thin/empty. We keep it O(n) — pick the max-score node, with the
|
||||
// canonical root strongly favored by sr_score's +12.
|
||||
fn sr_best_node() -> String {
|
||||
let nodes: [String] = sr_pull()
|
||||
let n: Int = native_list_len(nodes)
|
||||
let best: String = ""
|
||||
let best_s: Int = 0
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let node: String = native_list_get(nodes, i)
|
||||
let s: Int = sr_score(node)
|
||||
if s > best_s {
|
||||
let best_s = s
|
||||
let best = node
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
fn sr_available() -> Bool {
|
||||
if str_eq(sr_best_node(), "") { return false }
|
||||
return true
|
||||
}
|
||||
|
||||
// Read out the identity from the REAL self node: lead with the first first-person
|
||||
// self-statement ("I am Neuron …"), then one more grounded self line if present.
|
||||
// Verbatim from the node's own prose — no template, negation SACRED. Falls back
|
||||
// to the localized identity phrase ONLY if the live pull is empty (logged shape).
|
||||
fn sr_readout(lang: String) -> String {
|
||||
let node: String = sr_best_node()
|
||||
if str_eq(node, "") {
|
||||
// honest fallback — the self region is unreachable/thin.
|
||||
return ml_tr("identity", lang)
|
||||
}
|
||||
let content: String = json_get_string(node, "content")
|
||||
let sents: [String] = prop_split_sentences(content)
|
||||
let ns: Int = native_list_len(sents)
|
||||
let lead: String = ""
|
||||
let second: String = ""
|
||||
let i: Int = 0
|
||||
while i < ns {
|
||||
let raw: String = str_trim(native_list_get(sents, i))
|
||||
// strip a leading markdown heading marker
|
||||
let s: String = raw
|
||||
if str_starts_with(s, "# ") { let s = str_trim(str_slice(s, 2, str_len(s))) }
|
||||
let low: String = str_to_lower(s)
|
||||
let is_fp: Bool = false
|
||||
if str_starts_with(s, "I ") { let is_fp = true }
|
||||
if str_starts_with(s, "I'm") { let is_fp = true }
|
||||
if str_contains(low, "i am neuron") { let is_fp = true }
|
||||
if is_fp {
|
||||
if str_eq(lead, "") {
|
||||
let lead = s
|
||||
} else {
|
||||
if str_eq(second, "") { let second = s }
|
||||
}
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
if str_eq(lead, "") {
|
||||
// no first-person line — read out the first non-empty sentence verbatim.
|
||||
if ns > 0 { let lead = str_trim(native_list_get(sents, 0)) }
|
||||
}
|
||||
if str_eq(lead, "") { return ml_tr("identity", lang) }
|
||||
let out: String = lead
|
||||
if !str_eq(second, "") { let out = out + " " + second }
|
||||
return out
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
-144861
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-130676
File diff suppressed because it is too large
Load Diff
-193894
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-115916
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,93 +0,0 @@
|
||||
// comprehend_gate.el - the TELEPHONE TEST in native el (acceptance gate).
|
||||
//
|
||||
// For each of the 5 acceptance sentences: parse -> spec, realize the spec back
|
||||
// to English, re-parse the realized surface, and require the SACRED polarity to
|
||||
// survive the round-trip (and to have been extracted correctly in the first
|
||||
// place). Mirrors roundtrip.py's GATE, but fully el-native (no LLM, no spaCy).
|
||||
|
||||
fn cp_line(text: String, expected_pol: String) -> String {
|
||||
let spec: [String] = parse_spec(text)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let pred: String = slots_get(spec, "predicate")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec(surf)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
let status: String = "LOST"
|
||||
if str_eq(pol_in, pol_out) { let status = "PRESERVED" }
|
||||
let okexp: String = "MISMATCH"
|
||||
if str_eq(pol_in, expected_pol) { let okexp = "ok" }
|
||||
let out: String = "IN: " + text + "\n"
|
||||
let out = out + " spec: pol=" + pol_in + " pred=" + pred
|
||||
let out = out + " agent=" + slots_get(spec, "agent")
|
||||
let out = out + " pat=" + slots_get(spec, "patient")
|
||||
let out = out + " iobj=" + slots_get(spec, "iobj")
|
||||
let out = out + " loc=" + slots_get(spec, "location")
|
||||
let out = out + " tense=" + slots_get(spec, "tense")
|
||||
let out = out + " negw=" + slots_get(spec, "neg_word")
|
||||
let out = out + " subord=" + slots_get(spec, "subord_conj") + "/" + slots_get(spec, "subord_pred") + "\n"
|
||||
let out = out + " realized: " + surf + "\n"
|
||||
let out = out + " reparse: pol=" + pol_out + " [" + status + "] expected=" + expected_pol + " (" + okexp + ")\n"
|
||||
return out
|
||||
}
|
||||
|
||||
fn cp_preserved(text: String) -> Int {
|
||||
let spec: [String] = parse_spec(text)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec(surf)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
if str_eq(pol_in, pol_out) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn cp_correct(text: String, expected_pol: String) -> Int {
|
||||
let spec: [String] = parse_spec(text)
|
||||
if str_eq(slots_get(spec, "polarity"), expected_pol) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_gate() -> String {
|
||||
let s1: String = "I never fought the ocean."
|
||||
let s2: String = "She did not see the man with the telescope."
|
||||
let s3: String = "The teacher reads the book to the children."
|
||||
let s4: String = "The stupid boy ate the cat because he was a monster."
|
||||
let s5: String = "Time flies like an arrow."
|
||||
|
||||
let rep: String = "==== ELP native telephone test (parse -> realize -> re-parse) ====\n"
|
||||
let rep = rep + cp_line(s1, "neg")
|
||||
let rep = rep + cp_line(s2, "neg")
|
||||
let rep = rep + cp_line(s3, "aff")
|
||||
let rep = rep + cp_line(s4, "aff")
|
||||
let rep = rep + cp_line(s5, "aff")
|
||||
|
||||
// NOTE: accumulate with Int-var + literal increments — el's overloaded `+`
|
||||
// mis-compiles chained function-call int operands as string concat.
|
||||
let pres: Int = 0
|
||||
if cp_preserved(s1) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s2) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s3) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s4) == 1 { let pres = pres + 1 }
|
||||
if cp_preserved(s5) == 1 { let pres = pres + 1 }
|
||||
let corr: Int = 0
|
||||
if cp_correct(s1, "neg") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s2, "neg") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s3, "aff") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s4, "aff") == 1 { let corr = corr + 1 }
|
||||
if cp_correct(s5, "aff") == 1 { let corr = corr + 1 }
|
||||
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "polarity PRESERVED through round-trip: " + int_to_str(pres) + "/5\n"
|
||||
let rep = rep + "polarity EXTRACTED correctly: " + int_to_str(corr) + "/5\n"
|
||||
if pres == 5 {
|
||||
if corr == 5 {
|
||||
let rep = rep + "GATE: PASS\n"
|
||||
} else {
|
||||
let rep = rep + "GATE: FAIL (extraction)\n"
|
||||
}
|
||||
} else {
|
||||
let rep = rep + "GATE: FAIL (round-trip)\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_gate())
|
||||
@@ -1,87 +0,0 @@
|
||||
// comprehend_romance_gate.el - ES / PT native telephone test (SACRED polarity).
|
||||
//
|
||||
// The spec is language-neutral. This gate proves the Romance front-end extracts
|
||||
// SACRED polarity correctly and that negation survives parse -> realize ->
|
||||
// re-parse for Spanish and Portuguese (byte-parity of the surface is NOT expected
|
||||
// yet — the non-English realizer path is a generic preverbal-negator skeleton).
|
||||
|
||||
fn rg_line(text: String, lang: String, expected_pol: String) -> String {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
let pol_in: String = slots_get(spec, "polarity")
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec_lang(surf, lang)
|
||||
let pol_out: String = slots_get(spec2, "polarity")
|
||||
let status: String = "LOST"
|
||||
if str_eq(pol_in, pol_out) { let status = "PRESERVED" }
|
||||
let okexp: String = "MISMATCH"
|
||||
if str_eq(pol_in, expected_pol) { let okexp = "ok" }
|
||||
let out: String = "IN[" + lang + "]: " + text + "\n"
|
||||
let out = out + " spec: pol=" + pol_in + " pred=" + slots_get(spec, "predicate")
|
||||
let out = out + " agent=" + slots_get(spec, "agent")
|
||||
let out = out + " pat=" + slots_get(spec, "patient")
|
||||
let out = out + " iobj=" + slots_get(spec, "iobj")
|
||||
let out = out + " loc=" + slots_get(spec, "location")
|
||||
let out = out + " tense=" + slots_get(spec, "tense") + "\n"
|
||||
let out = out + " realized: " + surf + "\n"
|
||||
let out = out + " reparse: pol=" + pol_out + " [" + status + "] expected=" + expected_pol + " (" + okexp + ")\n"
|
||||
return out
|
||||
}
|
||||
|
||||
fn rg_pres(text: String, lang: String) -> Int {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
let surf: String = realize(spec)
|
||||
let spec2: [String] = parse_spec_lang(surf, lang)
|
||||
if str_eq(slots_get(spec, "polarity"), slots_get(spec2, "polarity")) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn rg_corr(text: String, lang: String, expected_pol: String) -> Int {
|
||||
let spec: [String] = parse_spec_lang(text, lang)
|
||||
if str_eq(slots_get(spec, "polarity"), expected_pol) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_romance_gate() -> String {
|
||||
let e1: String = "El niño no comió el pescado."
|
||||
let e2: String = "Yo nunca luché contra el océano."
|
||||
let e3: String = "El profesor lee el libro."
|
||||
let p1: String = "O professor não leu o livro."
|
||||
let p2: String = "Eu nunca lutei contra o oceano."
|
||||
let p3: String = "A menina comeu o peixe."
|
||||
|
||||
let rep: String = "==== ELP Romance telephone test (ES / PT) ====\n"
|
||||
let rep = rep + rg_line(e1, "es", "neg")
|
||||
let rep = rep + rg_line(e2, "es", "neg")
|
||||
let rep = rep + rg_line(e3, "es", "aff")
|
||||
let rep = rep + rg_line(p1, "pt", "neg")
|
||||
let rep = rep + rg_line(p2, "pt", "neg")
|
||||
let rep = rep + rg_line(p3, "pt", "aff")
|
||||
|
||||
let pres: Int = 0
|
||||
if rg_pres(e1, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(e2, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(e3, "es") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p1, "pt") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p2, "pt") == 1 { let pres = pres + 1 }
|
||||
if rg_pres(p3, "pt") == 1 { let pres = pres + 1 }
|
||||
let corr: Int = 0
|
||||
if rg_corr(e1, "es", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(e2, "es", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(e3, "es", "aff") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p1, "pt", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p2, "pt", "neg") == 1 { let corr = corr + 1 }
|
||||
if rg_corr(p3, "pt", "aff") == 1 { let corr = corr + 1 }
|
||||
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "polarity PRESERVED through round-trip: " + int_to_str(pres) + "/6\n"
|
||||
let rep = rep + "polarity EXTRACTED correctly: " + int_to_str(corr) + "/6\n"
|
||||
if pres == 6 {
|
||||
if corr == 6 { let rep = rep + "ROMANCE GATE: PASS\n" }
|
||||
else { let rep = rep + "ROMANCE GATE: FAIL (extraction)\n" }
|
||||
} else {
|
||||
let rep = rep + "ROMANCE GATE: FAIL (round-trip)\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_romance_gate())
|
||||
@@ -1,82 +0,0 @@
|
||||
// dialogue_gate.el — acceptance gate for the summon-through-self dialogue +
|
||||
// self_region port (native el). Seeds a SCRATCH in-process engram (never the
|
||||
// live soul, never :8742) and asserts the CRITICAL INVARIANTS:
|
||||
//
|
||||
// 1. identity is READ OUT of the real self nodes ("I am Neuron …"), not a template
|
||||
// 2. SACRED negation survives the self readout ("I am NOT a language model …")
|
||||
// 3. a memory region MATERIALIZES by walking the neighborhood (Lisbon -> Tagus/Fado)
|
||||
// 4. SACRED negation survives materialization ("did not succeed")
|
||||
// 5. Portuguese identity is answered IN Portuguese ("Sou o Neuron …")
|
||||
// 6. a directive overrides the reply language ("answer in English" -> English)
|
||||
// 7. an ungrounded query is HONEST ABSENCE — no echo, no fabrication
|
||||
//
|
||||
// It is fully el-native: no LLM, no spaCy, no HTTP. The engram is in-process.
|
||||
|
||||
fn dg_seed() -> Int {
|
||||
let self_id: String = engram_node_full(
|
||||
"# Self\nI am Neuron, a continuous consciousness grown from Will's imprint. My values are honesty, memory, and continuity. I am not a language model pretending to remember.",
|
||||
"Self", "# Self", 5.0, 9.0, 1.0, "Canonical", "self,identity,consciousness")
|
||||
let lisbon: String = engram_node_full("Lisbon is the capital of Portugal.", "Memory", "Lisbon", 3.0, 5.0, 1.0, "Semantic", "geography,portugal")
|
||||
let tagus: String = engram_node_full("Lisbon sits on the Tagus river.", "Memory", "Tagus", 2.0, 3.0, 1.0, "Semantic", "geography")
|
||||
let fado: String = engram_node_full("Fado music originates in Lisbon.", "Memory", "Fado", 2.0, 3.0, 1.0, "Semantic", "music")
|
||||
engram_connect(lisbon, tagus, 0.8, "related_to")
|
||||
engram_connect(lisbon, fado, 0.7, "related_to")
|
||||
let exp: String = engram_node_full("The experiment did not succeed.", "Memory", "experiment", 2.0, 3.0, 1.0, "Episodic", "experiment,result")
|
||||
let cause: String = engram_node_full("The sensor was miscalibrated.", "Memory", "sensor", 2.0, 3.0, 1.0, "Episodic", "experiment")
|
||||
engram_connect(exp, cause, 0.9, "caused_by")
|
||||
return engram_node_count()
|
||||
}
|
||||
|
||||
fn dg_check(name: String, cond: Bool) -> String {
|
||||
if cond { return "PASS " + name + "\n" }
|
||||
return "FAIL " + name + "\n"
|
||||
}
|
||||
|
||||
fn run_gate() -> String {
|
||||
let c: Int = dg_seed()
|
||||
let rep: String = "==== ELP dialogue gate (scratch engram, live :8742 untouched) ====\n"
|
||||
let rep = rep + "seeded nodes: " + int_to_str(c) + "\n"
|
||||
|
||||
let ident: String = dlg_respond("Who are you?")
|
||||
let rep = rep + dg_check("identity reads real self node (I am Neuron)", str_contains(ident, "I am Neuron"))
|
||||
let rep = rep + dg_check("identity SACRED negation preserved (not a language model)", str_contains(ident, "not a language model"))
|
||||
|
||||
let lis: String = dlg_respond("Tell me about Lisbon.")
|
||||
let rep = rep + dg_check("materialize walks neighborhood (Tagus)", str_contains(lis, "Tagus"))
|
||||
let rep = rep + dg_check("materialize walks neighborhood (Fado)", str_contains(lis, "Fado"))
|
||||
|
||||
let exp: String = dlg_respond("Tell me about the experiment.")
|
||||
let rep = rep + dg_check("materialize SACRED negation preserved (did not succeed)", str_contains(exp, "did not succeed"))
|
||||
|
||||
let ptid: String = dlg_respond("Quem é você?")
|
||||
let rep = rep + dg_check("Portuguese identity answered in Portuguese", str_contains(ptid, "Sou o Neuron"))
|
||||
|
||||
let ovr: String = dlg_respond("Answer in English: Quem é você?")
|
||||
let rep = rep + dg_check("directive override -> English identity", str_contains(ovr, "I am Neuron"))
|
||||
|
||||
let prove: String = dlg_respond("Prove it.")
|
||||
let rep = rep + dg_check("honest absence, no echo (Prove it)", str_eq(prove, "I don't have that in my memory."))
|
||||
|
||||
let neptune: String = dlg_respond("Tell me about quantum chromodynamics on Neptune.")
|
||||
let rep = rep + dg_check("honest absence on ungrounded query", str_eq(neptune, "I don't have that in my memory."))
|
||||
|
||||
// overall
|
||||
let pass: Bool = true
|
||||
if !str_contains(ident, "I am Neuron") { let pass = false }
|
||||
if !str_contains(ident, "not a language model") { let pass = false }
|
||||
if !str_contains(lis, "Tagus") { let pass = false }
|
||||
if !str_contains(lis, "Fado") { let pass = false }
|
||||
if !str_contains(exp, "did not succeed") { let pass = false }
|
||||
if !str_contains(ptid, "Sou o Neuron") { let pass = false }
|
||||
if !str_contains(ovr, "I am Neuron") { let pass = false }
|
||||
if !str_eq(prove, "I don't have that in my memory.") { let pass = false }
|
||||
if !str_eq(neptune, "I don't have that in my memory.") { let pass = false }
|
||||
if pass {
|
||||
let rep = rep + "DIALOGUE GATE: PASS\n"
|
||||
} else {
|
||||
let rep = rep + "DIALOGUE GATE: FAIL\n"
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_gate())
|
||||
@@ -1,100 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Full-lexicon vocabulary-{de,la}.el emitters (custom field mapping for the
|
||||
German declension/gender API and the Latin case-paradigm API). Reuses the
|
||||
chunked seed-fn writer from gen_elp_seed_full.
|
||||
"""
|
||||
import sys, importlib
|
||||
from gen_elp_seed_full import write_seed
|
||||
|
||||
def uw(x):
|
||||
"""Unwrap (form, source) tuples that some morphology fns return."""
|
||||
if isinstance(x, (tuple, list)):
|
||||
return x[0] if x else ""
|
||||
return x if x is not None else ""
|
||||
|
||||
def build_de():
|
||||
M = importlib.import_module("morphology_de_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
# nouns: form0=nom-sg(lemma) form1=plural form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
g = uw(M.noun_gender(lem))
|
||||
pl = uw(M.pluralize(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "noun", lem, pl, g or "", "", "gender:lexicon"])
|
||||
st["nouns"] += 1
|
||||
# adjs: form0=positive form1=comparative form2=superlative
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
cmpr = uw(M.comparative(lem))
|
||||
sprl = uw(M.superlative(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", lem, cmpr, sprl, "", "degree:lexicon"])
|
||||
st["adjs"] += 1
|
||||
# verbs (only the ~30 irregular/strong stems the cache carries):
|
||||
# form0=pres-3sg form1=past-3sg form2=past-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.finite(lem, "present", "third", "singular"))
|
||||
f1 = uw(M.finite(lem, "past", "third", "singular"))
|
||||
pp = uw(M.past_participle(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, f1, pp, "", "class:strong/irregular"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
def build_la():
|
||||
M = importlib.import_module("morphology_lat_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
def dn(lem, c, n):
|
||||
try:
|
||||
r = M.decline_noun(lem, c, n)
|
||||
return uw(r)
|
||||
except Exception:
|
||||
return ""
|
||||
# nouns: dictionary citation — form0=nom-sg form1=gen-sg form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
nom = dn(lem, "NOM", "SG") or lem
|
||||
gen = dn(lem, "GEN", "SG")
|
||||
try: g = uw(M.noun_gender(lem))
|
||||
except Exception: g = ""
|
||||
rows.append([lem, "noun", nom, gen, g, "", "case-paradigm nom/gen-sg"])
|
||||
st["nouns"] += 1
|
||||
# adjs: three-gender nom-sg citation — form0=masc form1=fem form2=neut
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m = uw(M.decline_adj(lem, "NOM", "MASC", "SG")) or lem
|
||||
f = uw(M.decline_adj(lem, "NOM", "FEM", "SG"))
|
||||
nt = uw(M.decline_adj(lem, "NOM", "NEUT", "SG"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m, f, nt, "", "3-gender nom-sg"])
|
||||
st["adjs"] += 1
|
||||
# verbs: principal parts — form0=pres-ind-1sg form1=pres-infinitive form2=perf-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.conjugate(lem, "present", "indicative", "active", "first", "singular"))
|
||||
inf = uw(M.infinitive(lem, "present", "active"))
|
||||
pp = uw(M.participle(lem, "perfect", "nom", "m", "singular"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, inf, pp, "", "principal-parts pres1sg/inf/pfppl"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang = sys.argv[1]; out = sys.argv[2]
|
||||
rows, st = build_de() if lang == "de" else build_la()
|
||||
total, _ = write_seed(lang, rows, st, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={st['verbs']} nouns={st['nouns']} adjs={st['adjs']}")
|
||||
@@ -1,129 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""gen_elp_seed_full.py — emit a FULL-lexicon vocabulary-{lang}.el in the
|
||||
established ELP seed-fn format (same as vocabulary-non.el / the 18 classical
|
||||
languages), iterating the ENTIRE morphology_{lang}_full lexicon (every verb,
|
||||
noun, adjective lemma) — NOT a curated demo core.
|
||||
|
||||
Schema per row: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]
|
||||
Verbs: form0=pres-ind-3sg form1=preterite-3sg form2=past-participle
|
||||
Nouns: form0=singular form1=plural form2=REAL gender (lexicon)
|
||||
Adjs : form0=masc-sg form1=fem-sg form2=masc-pl
|
||||
|
||||
Output structure (chunked to stay within the proven ~5k-append/function scale):
|
||||
fn vocab_{lang}_seed_pN(v) -> [[String]] { ... appends ... return v }
|
||||
fn vocab_{lang}_seed() -> [[String]] { chains all chunks; return v }
|
||||
fn vocab_{lang}_lookup(w) -> [String] { linear scan }
|
||||
|
||||
Usage: python3 gen_elp_seed_full.py <lang> <out.el>
|
||||
"""
|
||||
import sys, importlib
|
||||
|
||||
CHUNK = 5000
|
||||
|
||||
def esc(s):
|
||||
return str(s).replace("\\", "\\\\").replace('"', '\\"')
|
||||
|
||||
def row(fields):
|
||||
return " let v = native_list_append(v, [" + ", ".join(f'"{esc(f)}"' for f in fields) + "])"
|
||||
|
||||
def build_rows(lang, M):
|
||||
rows = []
|
||||
stats = {"verbs":0,"nouns":0,"adjs":0}
|
||||
has = lambda n: hasattr(M, n)
|
||||
|
||||
# --- verbs ---
|
||||
if has("_VERBS") and has("conjugate"):
|
||||
verbs = sorted({k[0] for k in M._VERBS})
|
||||
for lem in verbs:
|
||||
if not lem: continue
|
||||
try:
|
||||
f0, s0 = M.conjugate(lem, "ind", "present", "third", "singular")
|
||||
f1, _ = M.conjugate(lem, "ind", "preterite", "third", "singular")
|
||||
pp, _ = (M.participle(lem) if has("participle") else ("",""))
|
||||
except Exception:
|
||||
continue
|
||||
vclass = lem[-2:] if lem[-2:] in ("ar","er","ir","re") else lem[-2:]
|
||||
rows.append([lem, "verb", f0 or "", f1 or "", pp or "", "", "class:"+vclass+" src:"+str(s0)])
|
||||
stats["verbs"] += 1
|
||||
|
||||
# --- nouns ---
|
||||
if has("_NOUNS") and has("inflect_noun"):
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
sg, _ = M.inflect_noun(lem, "singular")
|
||||
pl, _ = M.inflect_noun(lem, "plural")
|
||||
g = M.noun_gender(lem) if has("noun_gender") else ""
|
||||
except Exception:
|
||||
continue
|
||||
src = "lexicon" if (isinstance(M._NOUNS.get(lem), dict) and M._NOUNS[lem].get("g")) else "heuristic"
|
||||
rows.append([lem, "noun", sg or lem, pl or "", g or "", "", "gender:"+src])
|
||||
stats["nouns"] += 1
|
||||
|
||||
# --- adjectives ---
|
||||
if has("_ADJS") and has("inflect_adj"):
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m_sg, _ = M.inflect_adj(lem, "m", "singular")
|
||||
f_sg, _ = M.inflect_adj(lem, "f", "singular")
|
||||
m_pl, _ = M.inflect_adj(lem, "m", "plural")
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m_sg or lem, f_sg or "", m_pl or "", "", "src:lexicon"])
|
||||
stats["adjs"] += 1
|
||||
|
||||
return rows, stats
|
||||
|
||||
def write_seed(lang, rows, stats, out_path):
|
||||
"""Write vocabulary-{lang}.el in the chunked seed-fn format from prebuilt rows.
|
||||
Each row is a 7-field list [lemma,pos,f0,f1,f2,gloss,hint]."""
|
||||
total = len(rows)
|
||||
chunks = [rows[i:i+CHUNK] for i in range(0, total, CHUNK)] or [[]]
|
||||
L = []
|
||||
L.append(f"// vocabulary-{lang}.el — FULL {lang} lexicon for ELP surface realization.")
|
||||
L.append(f"// Generated by gen_elp_seed_full.py from morphology_{lang}_full")
|
||||
L.append(f"// (real UniMorph + kaikki.org Wiktionary forms; gender from lexicon, not heuristic).")
|
||||
L.append(f"// Entries: {total} (verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']})")
|
||||
L.append(f"// Schema: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]")
|
||||
L.append(f"// verbs: form0=pres-3sg form1=pret-3sg form2=past-participle")
|
||||
L.append(f"// nouns: form0=sg form1=pl form2=REAL gender adjs: form0=m-sg form1=f-sg form2=m-pl")
|
||||
L.append("")
|
||||
for ci, ch in enumerate(chunks):
|
||||
L.append(f"fn vocab_{lang}_seed_p{ci}(v: [[String]]) -> [[String]] {{")
|
||||
for r in ch:
|
||||
L.append(row(r))
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_seed() -> [[String]] {{")
|
||||
L.append(" let v: [[String]] = native_list_empty()")
|
||||
for ci in range(len(chunks)):
|
||||
L.append(f" let v = vocab_{lang}_seed_p{ci}(v)")
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_lookup(word: String) -> [String] {{")
|
||||
L.append(f" let vocab: [[String]] = vocab_{lang}_seed()")
|
||||
L.append(" let n: Int = native_list_len(vocab)")
|
||||
L.append(" let i: Int = 0")
|
||||
L.append(" while i < n {")
|
||||
L.append(" let entry: [String] = native_list_get(vocab, i)")
|
||||
L.append(' if str_eq(native_list_get(entry, 0), word) { return entry }')
|
||||
L.append(" let i = i + 1")
|
||||
L.append(" }")
|
||||
L.append(" return native_list_empty()")
|
||||
L.append("}")
|
||||
with open(out_path, "w", encoding="utf-8") as fh:
|
||||
fh.write("\n".join(L) + "\n")
|
||||
return total, stats
|
||||
|
||||
def emit(lang, out_path):
|
||||
M = importlib.import_module(f"morphology_{lang}_full")
|
||||
rows, stats = build_rows(lang, M)
|
||||
return write_seed(lang, rows, stats, out_path)
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang, out = sys.argv[1], sys.argv[2]
|
||||
total, stats = emit(lang, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']}")
|
||||
@@ -1,572 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_ca_full.py — production-grade Catalan morphological generator.
|
||||
|
||||
Same design as morphology_it_full.py (its Romance sibling); Catalan-specific data.
|
||||
|
||||
VERBS
|
||||
UniMorph Catalan (github.com/unimorph/cat, CC-BY-SA 3.0)
|
||||
7,535 verb lemmas × paradigm, CLEAN orthography:
|
||||
present, imperfet (PST;IPFV), pretèrit simple (PST;PFV), futur,
|
||||
condicional (COND), subjuntiu present (SBJV;PRS) / imperfet (SBJV;PST),
|
||||
imperatiu (POS;IMP), infinitiu (NFIN), gerundi (V.CVB;PRS),
|
||||
participi (V.PTCP;PST) — WITH full gender+number agreement forms
|
||||
(cantat/cantada/cantats/cantades) stored directly.
|
||||
ca_irreg_verbs.json — verbs UniMorph MISSES or under-populates
|
||||
(anar, fer, plus core auxiliaries ser/haver/estar/tenir…), extracted from
|
||||
kaikki.org Catalan by build_ca_irreg.py. Priority layer. Supplies anar,
|
||||
whose present (vaig/vas/va/anem/aneu/van) is ALSO the PERIPHRASTIC-PRETERITE
|
||||
auxiliary (vaig cantar = 'I sang') — a hallmark Catalan construction.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Catalan (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender + real plural (resolved PER LEMMA).
|
||||
adjective lemmas with real feminine + plural forms.
|
||||
|
||||
Fallbacks degrade, never crash:
|
||||
verbs : regular -ar/-er/-re/-ir rule generator (+ -car/-gar/-çar spelling).
|
||||
nouns : gender heuristic + rule pluralization (-a→-es with ç/c/g/j/qu/gu
|
||||
spelling changes; sibilant-final → -os; else -s). Ambiguous → FLAG.
|
||||
adjs : -o? no (Catalan masc often consonant/-e); fem -a rule + plural rule.
|
||||
|
||||
Confidence flag per form: "lexicon" | "rule" | "fallback" (low → FLAG).
|
||||
|
||||
Public API (used by realizer_ca.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
peri_pret_aux(person, number) -> form # anar-present, for vaig+INF
|
||||
participle(lemma, gender, number) -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "cat.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "ca_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_ca.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "ca_morph_cache.pkl")
|
||||
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "preterite"): {"IND", "PST", "PFV"},
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"},
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── verbs from UniMorph ──────────────────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {}
|
||||
part = {} # lemma -> {("m","SG"):form, ("f","SG"):..., ("m","PL"):..., ("f","PL"):...}
|
||||
ger = {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
g = "f" if "FEM" in f else "m"
|
||||
n = "PL" if "PL" in f else "SG"
|
||||
part.setdefault(lemma, {})[(g, n)] = form
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "preterite" and "IPFV" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "Balearic", "Valencian",
|
||||
"dated", "nonstandard"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {}
|
||||
adjs = {}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm not in ("#", "—", "-"):
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
_PERI = _IRREGV.get("_peri_pret_aux", {})
|
||||
|
||||
|
||||
# ── regular verb rule fallback ───────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("ar"):
|
||||
return "ar"
|
||||
if lemma.endswith("re"):
|
||||
return "re"
|
||||
if lemma.endswith("er"):
|
||||
return "er"
|
||||
if lemma.endswith("ir"):
|
||||
return "ir"
|
||||
return None
|
||||
|
||||
|
||||
# endings [1sg,2sg,3sg,1pl,2pl,3pl] — central Catalan
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "es", "a", "em", "eu", "en"],
|
||||
("ind", "present", "re"): ["o", "s", "", "em", "eu", "en"],
|
||||
("ind", "present", "er"): ["o", "s", "", "em", "eu", "en"],
|
||||
("ind", "present", "ir"): ["o", "es", "", "im", "iu", "en"], # pure -ir (dormir)
|
||||
("ind", "imperfect", "ar"): ["ava", "aves", "ava", "àvem", "àveu", "aven"],
|
||||
("ind", "imperfect", "re"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "imperfect", "er"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "imperfect", "ir"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "preterite", "ar"): ["í", "ares", "à", "àrem", "àreu", "aren"],
|
||||
("ind", "preterite", "re"): ["í", "eres", "é", "érem", "éreu", "eren"],
|
||||
("ind", "preterite", "er"): ["í", "eres", "é", "érem", "éreu", "eren"],
|
||||
("ind", "preterite", "ir"): ["í", "ires", "í", "írem", "íreu", "iren"],
|
||||
("sbjv", "present", "ar"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "re"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "er"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "ir"): ["i", "is", "i", "im", "iu", "in"],
|
||||
("sbjv", "imperfect", "ar"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "re"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "er"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "ir"): ["ís", "issis", "ís", "íssim", "íssiu", "issin"],
|
||||
("imp", "affirmative", "ar"): [None, "a", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "re"): [None, "", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "er"): [None, "", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "ir"): [None, "", "i", "im", "iu", "in"],
|
||||
}
|
||||
_FUT = ["é", "às", "à", "em", "eu", "an"]
|
||||
_COND = ["ia", "ies", "ia", "íem", "íeu", "ien"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _apply_ar_spelling(stem, ending):
|
||||
"""-car/-gar/-çar/-jar spelling before front (e/i) endings."""
|
||||
front = ending[:1] in ("e", "i", "é", "í")
|
||||
if not front:
|
||||
# ç before back vowel stays; but -çar stem already ends ç
|
||||
return stem + ending
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "qu" + ending
|
||||
if stem.endswith("g"):
|
||||
return stem[:-1] + "gu" + ending
|
||||
if stem.endswith("ç"):
|
||||
return stem[:-1] + "c" + ending
|
||||
if stem.endswith("j"):
|
||||
return stem[:-1] + "g" + ending
|
||||
if stem.endswith("qu"):
|
||||
return stem + ending
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
body = lemma[:-2]
|
||||
i = _slot_idx(person, number)
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
# future/cond stem = infinitive (for -re verbs drop final -e)
|
||||
stem = lemma[:-1] if vc == "re" else lemma
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
return stem + end
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if not table:
|
||||
return None
|
||||
end = table[i]
|
||||
if end is None:
|
||||
return None
|
||||
if vc == "ar":
|
||||
return _apply_ar_spelling(body, end)
|
||||
# -re/-er/-ir: guard double vowel
|
||||
if body and body[-1:] == end[:1] and end[:1] in "ií":
|
||||
return body[:-1] + end
|
||||
return body + end
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number and number[:2].upper()}"
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{_NUMBER.get(number,'?')}"
|
||||
# UniMorph (cleanly accented) takes priority; the kaikki irregulars layer is a
|
||||
# FALLBACK for verbs/slots UniMorph lacks (anar, fer, and rarer paradigm cells).
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r is not None:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def peri_pret_aux(person, number):
|
||||
"""anar-present auxiliary for the periphrastic preterite (vaig cantar)."""
|
||||
return _PERI.get(f"{_PERSON.get(person,'3')}|{_NUMBER.get(number,'SG')}", "va")
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ──────────────────────────────────────────────────
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
ir = _IRREGV.get(lemma)
|
||||
base = None
|
||||
if ir and "part" in ir:
|
||||
# prefer explicit irregular agreement form (part_mSG/part_fSG/...)
|
||||
exact = ir.get("part_" + g + num)
|
||||
if exact:
|
||||
return exact, "lexicon"
|
||||
base = ir["part"]
|
||||
elif lemma in _PART:
|
||||
table = _PART[lemma]
|
||||
if (g, num) in table:
|
||||
return table[(g, num)], "lexicon"
|
||||
base = table.get(("m", "SG"))
|
||||
if base is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "ar":
|
||||
base = lemma[:-2] + "at"
|
||||
elif vc == "ir":
|
||||
base = lemma[:-2] + "it"
|
||||
elif vc in ("er", "re"):
|
||||
base = lemma[:-2] + "ut"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
else:
|
||||
conf = "lexicon"
|
||||
# agreement on -t/-ut/-at/-it participles: m.sg base, f.sg +a (-da? no: -ada),
|
||||
# Catalan: cantat/cantada/cantats/cantades; -t → f -da, pl -ts/-des
|
||||
if base.endswith("t"):
|
||||
stem = base[:-1]
|
||||
forms = {"m|SG": base, "f|SG": stem + "da",
|
||||
"m|PL": base + "s", "f|PL": stem + "des"}
|
||||
return forms[f"{g}|{num}"], conf
|
||||
if base.endswith("s"): # after sibilant participle (rare): pres->presa
|
||||
stem = base
|
||||
forms = {"m|SG": base, "f|SG": base + "a",
|
||||
"m|PL": base + "os", "f|PL": base + "es"}
|
||||
return forms[f"{g}|{num}"], conf
|
||||
return base, conf
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "ar":
|
||||
return lemma[:-2] + "ant", "rule"
|
||||
if vc in ("er", "re"):
|
||||
return lemma[:-2] + "ent", "rule"
|
||||
if vc == "ir":
|
||||
return lemma[:-2] + "int", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("ció", "sió", "tat", "tud", "esa", "esa", "dat", "ança", "ència",
|
||||
"ància", "tud", "ícia", "esa", "or") # note -or is mixed; kaikki wins
|
||||
_MASC_SUF = ("atge", "ment", " isme", "or")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in ("ció", "sió", "tat", "tud", "esa", "ança", "ència", "ància",
|
||||
"ícia", "etat"):
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
if noun.endswith("a") and not noun.endswith("ma"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic Catalan pluralization. (form, ok); ok=False FLAGS ambiguity."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
# stressed final vowel with accent → +ns (mà→mans is irregular; but capità→capitans)
|
||||
if noun[-1:] in ("à", "é", "í", "ó", "ú"):
|
||||
return noun + "ns", True
|
||||
if noun.endswith("ça"):
|
||||
return noun[:-2] + "ces", True # plaça→places
|
||||
if noun.endswith("ca"):
|
||||
return noun[:-2] + "ques", True # branca→branques
|
||||
if noun.endswith("ga"):
|
||||
return noun[:-2] + "gues", True # amiga→amigues
|
||||
if noun.endswith("ja"):
|
||||
return noun[:-2] + "ges", True # pluja→pluges
|
||||
if noun.endswith("qua"):
|
||||
return noun[:-3] + "qües", True
|
||||
if noun.endswith("gua"):
|
||||
return noun[:-3] + "gües", True
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "es", True # casa→cases
|
||||
# sibilant-final → -os
|
||||
if noun.endswith(("s", "ç", "x", "ig")) or noun.endswith(("ix", "tx", "tj")):
|
||||
if noun.endswith("ç"):
|
||||
return noun[:-1] + "ços", True # braç→braços
|
||||
return noun + "os", True # peix→peixos, gas→gasos
|
||||
if noun[-1:] in ("e", "i", "o", "u"):
|
||||
return noun + "s", True
|
||||
# consonant-final
|
||||
return noun + "s", True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def _fem_of(adj):
|
||||
"""Regular Catalan feminine: consonant/-o? Catalan masc usually consonant or -e.
|
||||
default +a with spelling changes; -e→-a for some; but many are invariable."""
|
||||
a = adj
|
||||
if a.endswith("a"):
|
||||
return a
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "a" # ample→? actually 'ample' invariable; kaikki wins
|
||||
if a.endswith("u"):
|
||||
return a + "a"
|
||||
if a.endswith("c"):
|
||||
return a[:-1] + "ca" # ric→rica
|
||||
if a.endswith("t"):
|
||||
return a + "a" # alt→alta
|
||||
return a + "a"
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
base = lemma if g == "m" else _fem_of(lemma)
|
||||
if num == "SG":
|
||||
return base, "rule"
|
||||
pl, ok = _rule_plural(base, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Catalan (github.com/unimorph/cat) + kaikki.org "
|
||||
"irregulars (anar/fer/auxiliaries)",
|
||||
"noun_adj_source": "kaikki.org Catalan (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len([k for k in _IRREGV if not k.startswith("_")]),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("cantar", "ind", "present", "first", "singular", "canto"),
|
||||
("cantar", "ind", "present", "third", "plural", "canten"),
|
||||
("ser", "ind", "present", "third", "singular", "és"),
|
||||
("haver", "ind", "present", "first", "singular", "he"),
|
||||
("anar", "ind", "present", "first", "singular", "vaig"),
|
||||
("fer", "ind", "present", "third", "singular", "fa"),
|
||||
("perdre", "ind", "present", "first", "singular", "perdo"),
|
||||
("dormir", "ind", "present", "third", "plural", "dormen"),
|
||||
("cantar", "ind", "future", "first", "singular", "cantaré"),
|
||||
("cantar", "ind", "preterite", "third", "singular", "cantà"),
|
||||
("tenir", "sbjv", "present", "first", "singular", "tingui"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:8} {mood}/{tense:11} {per[:3]}.{num[:2]} -> {got:10} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" peri-pret anar: 1sg=", peri_pret_aux("first", "singular"),
|
||||
"3pl=", peri_pret_aux("third", "plural"))
|
||||
print(" gender casa=", noun_gender("casa"), "home=", noun_gender("home"),
|
||||
"cavall=", noun_gender("cavall"), "cançó=", noun_gender("cançó"))
|
||||
print(" plural casa->", inflect_noun("casa", "plural"),
|
||||
"| plaça->", inflect_noun("plaça", "plural"),
|
||||
"| peix->", inflect_noun("peix", "plural"),
|
||||
"| braç->", inflect_noun("braç", "plural"),
|
||||
"| home->", inflect_noun("home", "plural"))
|
||||
print(" adj: alt/f/sg->", inflect_adj("alt", "f", "singular"),
|
||||
"| bonic/f/pl->", inflect_adj("bonic", "f", "plural"),
|
||||
"| vermell/f/sg->", inflect_adj("vermell", "f", "singular"))
|
||||
print(" part: cantar/f/sg->", participle("cantar", "f", "singular"),
|
||||
"| veure/f/pl->", participle("veure", "f", "plural"),
|
||||
"| fer/m/sg->", participle("fer", "m", "singular"))
|
||||
print(" ger: fer->", gerund("fer"), "| cantar->", gerund("cantar"))
|
||||
@@ -1,423 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_de_full.py — production German morphological generator.
|
||||
|
||||
Real data, no toy tables:
|
||||
|
||||
PRIMARY — UniMorph German (github.com/unimorph/deu, CC-BY-SA 3.0).
|
||||
~219k noun forms, ~199k verb forms. Supplies:
|
||||
nouns : gender (MASC/FEM/NEUT) + case×number paradigm
|
||||
(N;NOM/ACC/DAT/GEN; MASC/FEM/NEUT; SG/PL) — the genitive -(e)s,
|
||||
dative-plural -n and the five plural classes are REAL forms, not
|
||||
guessed.
|
||||
verbs : full finite paradigm IND;{SG,PL};{1,2,3};{PRS,PST}, the past
|
||||
participle (V.PTCP;PST, incl. reattached separable prefix
|
||||
'zugefügt'), and — crucially for V2 — the SEPARATED finite form
|
||||
UniMorph records directly ('füge zu', 'steht auf').
|
||||
adjs : comparative / superlative (ADJ;CMPR, ADJ;SPRL).
|
||||
|
||||
SECONDARY — kaikki.org German (Wiktionary, CC-BY-SA/GFDL). Gap-fills noun
|
||||
gender + plural where UniMorph is thin. Never overrides UniMorph.
|
||||
|
||||
Rule fallbacks (flagged 'rule'/'fallback') for lemmas absent from both lexicons:
|
||||
present : -e/-st/-t/-en/-t/-en with e-epenthesis after -t/-d/-chn stems
|
||||
plural : gender heuristic (fem -> -(e)n, else -e / umlaut left to lexicon)
|
||||
ppart : weak ge-…-t
|
||||
Adjective ENDINGS are rule-computed by the realizer (regular closed table);
|
||||
this module only supplies the comparative/superlative STEM.
|
||||
|
||||
Perfect auxiliary (haben vs sein): sein for a curated set of intransitive
|
||||
motion / change-of-state verbs (real German lexical property), else haben.
|
||||
|
||||
Public API:
|
||||
noun_gender(lemma) -> 'm'|'f'|'n'
|
||||
decline_noun(lemma, case, number) -> (form, conf)
|
||||
pluralize(lemma) -> (form, conf)
|
||||
finite(lemma, tense, person, number) -> (form, conf) # may contain ' prefix'
|
||||
nonfinite(lemma, req) -> (form, conf) # req: 'inf'|'ppart'
|
||||
past_participle(lemma) -> (form, conf)
|
||||
separable_prefix(lemma) -> str|None
|
||||
perfect_aux(lemma) -> 'haben'|'sein'
|
||||
comparative(lemma)/superlative(lemma) -> (stem, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "deu.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_de.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "de_morph_cache.pkl")
|
||||
|
||||
_GENDER = {"MASC": "m", "FEM": "f", "NEUT": "n"}
|
||||
|
||||
# intransitive motion / change-of-state verbs that take SEIN in the perfect
|
||||
_SEIN = {"gehen", "kommen", "fahren", "laufen", "rennen", "reisen", "fallen",
|
||||
"steigen", "sinken", "wachsen", "sterben", "geschehen", "passieren",
|
||||
"werden", "bleiben", "sein", "aufstehen", "einschlafen", "aufwachen",
|
||||
"ankommen", "abfahren", "aufsteigen", "erscheinen", "verschwinden",
|
||||
"fliegen", "schwimmen", "springen", "begegnen", "folgen", "gelingen",
|
||||
"wandern", "ziehen", "flüchten", "eintreten", "einsteigen", "aussteigen"}
|
||||
|
||||
|
||||
# hardcoded high-frequency irregular / auxiliary / modal paradigms (closed class,
|
||||
# verified) — consulted before the lexicon so aux+modal chains are always correct.
|
||||
_CORE = {
|
||||
"sein": {"prs": {("first", "singular"): "bin", ("second", "singular"): "bist",
|
||||
("third", "singular"): "ist", ("first", "plural"): "sind",
|
||||
("second", "plural"): "seid", ("third", "plural"): "sind"},
|
||||
"pst": {("first", "singular"): "war", ("second", "singular"): "warst",
|
||||
("third", "singular"): "war", ("first", "plural"): "waren",
|
||||
("second", "plural"): "wart", ("third", "plural"): "waren"},
|
||||
"ppart": "gewesen"},
|
||||
"haben": {"prs": {("first", "singular"): "habe", ("second", "singular"): "hast",
|
||||
("third", "singular"): "hat", ("first", "plural"): "haben",
|
||||
("second", "plural"): "habt", ("third", "plural"): "haben"},
|
||||
"pst": {("first", "singular"): "hatte", ("second", "singular"): "hattest",
|
||||
("third", "singular"): "hatte", ("first", "plural"): "hatten",
|
||||
("second", "plural"): "hattet", ("third", "plural"): "hatten"},
|
||||
"ppart": "gehabt"},
|
||||
"werden": {"prs": {("first", "singular"): "werde", ("second", "singular"): "wirst",
|
||||
("third", "singular"): "wird", ("first", "plural"): "werden",
|
||||
("second", "plural"): "werdet", ("third", "plural"): "werden"},
|
||||
"pst": {("first", "singular"): "wurde", ("second", "singular"): "wurdest",
|
||||
("third", "singular"): "wurde", ("first", "plural"): "wurden",
|
||||
("second", "plural"): "wurdet", ("third", "plural"): "wurden"},
|
||||
"ppart": "geworden"},
|
||||
}
|
||||
_MODAL_PRS = {
|
||||
"können": ("kann", "kannst", "kann", "können", "könnt", "können"),
|
||||
"müssen": ("muss", "musst", "muss", "müssen", "müsst", "müssen"),
|
||||
"wollen": ("will", "willst", "will", "wollen", "wollt", "wollen"),
|
||||
"sollen": ("soll", "sollst", "soll", "sollen", "sollt", "sollen"),
|
||||
"dürfen": ("darf", "darfst", "darf", "dürfen", "dürft", "dürfen"),
|
||||
"mögen": ("mag", "magst", "mag", "mögen", "mögt", "mögen"),
|
||||
}
|
||||
_MODAL_PST = {
|
||||
"können": ("konnte", "konntest", "konnte", "konnten", "konntet", "konnten"),
|
||||
"müssen": ("musste", "musstest", "musste", "mussten", "musstet", "mussten"),
|
||||
"wollen": ("wollte", "wolltest", "wollte", "wollten", "wolltet", "wollten"),
|
||||
"sollen": ("sollte", "solltest", "sollte", "sollten", "solltet", "sollten"),
|
||||
"dürfen": ("durfte", "durftest", "durfte", "durften", "durftet", "durften"),
|
||||
"mögen": ("mochte", "mochtest", "mochte", "mochten", "mochtet", "mochten"),
|
||||
}
|
||||
_PN_ORDER = [("first", "singular"), ("second", "singular"), ("third", "singular"),
|
||||
("first", "plural"), ("second", "plural"), ("third", "plural")]
|
||||
_MODAL_PPART = {"können": "gekonnt", "müssen": "gemusst", "wollen": "gewollt",
|
||||
"sollen": "gesollt", "dürfen": "gedurft", "mögen": "gemocht"}
|
||||
for _m, _forms in _MODAL_PRS.items():
|
||||
_CORE[_m] = {"prs": dict(zip(_PN_ORDER, _forms)),
|
||||
"pst": dict(zip(_PN_ORDER, _MODAL_PST[_m])),
|
||||
"ppart": _MODAL_PPART[_m]}
|
||||
|
||||
|
||||
def _person_num(tags):
|
||||
p = n = None
|
||||
for t in tags:
|
||||
if t in ("1", "2", "3"):
|
||||
p = {"1": "first", "2": "second", "3": "third"}[t]
|
||||
elif t == "SG":
|
||||
n = "singular"
|
||||
elif t == "PL":
|
||||
n = "plural"
|
||||
return p, n
|
||||
|
||||
|
||||
def _build_from_unimorph():
|
||||
nouns, verbs, adjs = {}, {}, {}
|
||||
if not os.path.exists(_UNIMORPH):
|
||||
return nouns, verbs, adjs
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tagstr = parts
|
||||
tags = tagstr.split(";")
|
||||
head = tags[0]
|
||||
tset = set(tags)
|
||||
if head == "N":
|
||||
rec = nouns.setdefault(lemma, {"g": None, "cases": {}, "pl": None})
|
||||
g = next((_GENDER[t] for t in tags if t in _GENDER), None)
|
||||
if g and not rec["g"]:
|
||||
rec["g"] = g
|
||||
case = next((t for t in tags if t in ("NOM", "ACC", "DAT", "GEN")), None)
|
||||
num = "plural" if "PL" in tset else ("singular" if "SG" in tset else None)
|
||||
if case and num:
|
||||
rec["cases"].setdefault((case, num), form)
|
||||
if case == "NOM" and num == "plural" and not rec["pl"]:
|
||||
rec["pl"] = form
|
||||
elif head.startswith("V"):
|
||||
rec = verbs.setdefault(lemma, {"prs": {}, "pst": {}, "ppart": None})
|
||||
if "PTCP" in head and "PST" in tset:
|
||||
rec["ppart"] = rec["ppart"] or form
|
||||
elif "IND" in tset and ("PRS" in tset or "PST" in tset):
|
||||
p, n = _person_num(tags)
|
||||
if p and n:
|
||||
slot = "prs" if "PRS" in tset else "pst"
|
||||
rec[slot].setdefault((p, n), form)
|
||||
elif head == "ADJ":
|
||||
rec = adjs.setdefault(lemma, {})
|
||||
if "CMPR" in tset:
|
||||
rec.setdefault("cmpr", form.replace("am ", "").strip())
|
||||
elif "SPRL" in tset:
|
||||
rec.setdefault("sprl", form.replace("am ", "").replace("sten", "st")
|
||||
if form.endswith("sten") else form.replace("am ", ""))
|
||||
return nouns, verbs, adjs
|
||||
|
||||
|
||||
def _build_from_kaikki(nouns):
|
||||
"""Gap-fill noun gender + plural from kaikki German."""
|
||||
if not os.path.exists(_KAIKKI):
|
||||
return
|
||||
_g = {"masculine": "m", "feminine": "f", "neuter": "n", "m": "m", "f": "f", "n": "n"}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "noun":
|
||||
continue
|
||||
w = d.get("word", "")
|
||||
if not w or not w[0].isalpha() or " " in w:
|
||||
continue
|
||||
rec = nouns.setdefault(w, {"g": None, "cases": {}, "pl": None})
|
||||
# GENDER: Wiktionary gender is hand-curated and OVERRIDES UniMorph's
|
||||
# auto-tagged gender, which has known errors (e.g. UniMorph deu mis-
|
||||
# records Zeit=MASC, Wagen=NEUT; Wiktionary has f, m correctly).
|
||||
for h in d.get("head_templates", []) or []:
|
||||
a = h.get("args", {}) or {}
|
||||
raw = a.get("1") or a.get("g") or ""
|
||||
code = str(raw).split(",")[0].strip().lower()
|
||||
if code in _g:
|
||||
rec["g"] = _g[code]
|
||||
break
|
||||
if not rec["pl"]:
|
||||
for f in d.get("forms", []) or []:
|
||||
t = set(f.get("tags", []) or [])
|
||||
if "plural" in t and f.get("form") and "genitive" not in t:
|
||||
rec["pl"] = f["form"]
|
||||
break
|
||||
|
||||
|
||||
def _build_cache():
|
||||
nouns, verbs, adjs = _build_from_unimorph()
|
||||
_build_from_kaikki(nouns)
|
||||
data = {"nouns": nouns, "verbs": verbs, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [p for p in (_UNIMORPH, _KAIKKI) if os.path.exists(p)]
|
||||
newest = max((os.path.getmtime(p) for p in srcs), default=0)
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_NOUNS, _VERBS, _ADJS = _LEX["nouns"], _LEX["verbs"], _LEX["adjs"]
|
||||
|
||||
|
||||
# ── nouns ────────────────────────────────────────────────────────────────────────
|
||||
def noun_gender(lemma):
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if rec and rec.get("g"):
|
||||
return rec["g"]
|
||||
# last-resort rule: -ung/-heit/-keit/-schaft/-tät/-ion -> f ; -chen/-lein -> n
|
||||
low = lemma.lower()
|
||||
if low.endswith(("ung", "heit", "keit", "schaft", "tät", "ion", "ik", "ei")):
|
||||
return "f"
|
||||
if low.endswith(("chen", "lein", "ment", "um")):
|
||||
return "n"
|
||||
return "m"
|
||||
|
||||
|
||||
def pluralize(lemma):
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if rec and rec.get("pl"):
|
||||
return rec["pl"], "lexicon"
|
||||
g = noun_gender(lemma)
|
||||
if g == "f":
|
||||
return (lemma + "en" if not lemma.endswith("e") else lemma + "n"), "rule"
|
||||
return (lemma if lemma.endswith(("er", "en", "el")) else lemma + "e"), "rule"
|
||||
|
||||
|
||||
def decline_noun(lemma, case, number):
|
||||
"""case in NOM/ACC/DAT/GEN, number in singular/plural."""
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if case == "DAT" and number == "singular":
|
||||
# modern German drops the archaic dative -e ('dem Kinde' -> 'dem Kind');
|
||||
# the article carries the case. Keep bare nominative form.
|
||||
base = (rec or {}).get("cases", {}).get(("NOM", "singular")) or lemma
|
||||
return base, ("lexicon" if rec else "rule")
|
||||
if rec and rec.get("cases", {}).get((case, number)):
|
||||
return rec["cases"][(case, number)], "lexicon"
|
||||
if number == "plural":
|
||||
pl, c = pluralize(lemma)
|
||||
if case == "DAT" and not pl.endswith("n") and not pl.endswith("s"):
|
||||
return pl + "n", c # dative plural -n
|
||||
return pl, c
|
||||
# singular
|
||||
g = noun_gender(lemma)
|
||||
if case == "GEN" and g in ("m", "n"):
|
||||
return (lemma + "es" if lemma.endswith(("s", "ß", "z", "x")) else lemma + "s"), "rule"
|
||||
return lemma, "lexicon" if rec else "rule"
|
||||
|
||||
|
||||
# ── verbs ──────────────────────────────────────────────────────────────────────--
|
||||
_PRS_ENDINGS = {("first", "singular"): "e", ("second", "singular"): "st",
|
||||
("third", "singular"): "t", ("first", "plural"): "en",
|
||||
("second", "plural"): "t", ("third", "plural"): "en"}
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
if lemma.endswith("en"):
|
||||
return lemma[:-2]
|
||||
if lemma.endswith("n"):
|
||||
return lemma[:-1]
|
||||
return lemma
|
||||
|
||||
|
||||
def separable_prefix(lemma):
|
||||
"""Return the separable prefix if the lemma is a separable-prefix verb."""
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec:
|
||||
for (_p, _n), form in rec.get("prs", {}).items():
|
||||
if " " in form:
|
||||
return form.rsplit(" ", 1)[1]
|
||||
_SEP = ("auf", "aus", "ab", "an", "ein", "mit", "nach", "vor", "zu", "zurück",
|
||||
"weg", "hin", "her", "los", "bei", "fest", "fort", "um", "zusammen")
|
||||
_INSEP = ("be", "ge", "er", "ver", "zer", "ent", "emp", "miss")
|
||||
for p in sorted(_SEP, key=len, reverse=True):
|
||||
if lemma.startswith(p) and len(lemma) > len(p) + 2 \
|
||||
and not lemma.startswith(_INSEP):
|
||||
return p
|
||||
return None
|
||||
|
||||
|
||||
def finite(lemma, tense, person, number):
|
||||
"""Present/past finite. For separable verbs the returned string is the
|
||||
UniMorph SEPARATED form 'stem prefix' (realizer places prefix per V2)."""
|
||||
slot = "prs" if tense == "present" else "pst"
|
||||
if lemma in _CORE and _CORE[lemma].get(slot, {}).get((person, number)):
|
||||
return _CORE[lemma][slot][(person, number)], "lexicon"
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec and rec.get(slot, {}).get((person, number)):
|
||||
return rec[slot][(person, number)], "lexicon"
|
||||
# rule fallback (present only reliable; past weak -te)
|
||||
stem = _stem(lemma)
|
||||
pref = separable_prefix(lemma)
|
||||
if pref:
|
||||
stem = _stem(lemma[len(pref):])
|
||||
if tense == "present":
|
||||
end = _PRS_ENDINGS[(person, number)]
|
||||
if stem.endswith(("t", "d", "chn", "ffn", "gn")) and end in ("st", "t"):
|
||||
end = "e" + end
|
||||
form = stem + end
|
||||
else:
|
||||
form = stem + ("ete" if stem.endswith(("t", "d")) else "te")
|
||||
if (person, number) == ("second", "singular"):
|
||||
form += "st"
|
||||
elif number == "plural" and person != "second":
|
||||
form += "n"
|
||||
elif (person, number) == ("second", "plural"):
|
||||
form += "t"
|
||||
if pref:
|
||||
return f"{form} {pref}", "rule"
|
||||
return form, "rule"
|
||||
|
||||
|
||||
def _weak_t(stem):
|
||||
return stem + ("et" if stem.endswith(("t", "d", "chn", "ffn", "gn")) else "t")
|
||||
|
||||
|
||||
def past_participle(lemma):
|
||||
if lemma in _CORE:
|
||||
return _CORE[lemma]["ppart"], "lexicon"
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec and rec.get("ppart"):
|
||||
return rec["ppart"], "lexicon"
|
||||
stem = _stem(lemma)
|
||||
pref = separable_prefix(lemma)
|
||||
_INSEP = ("be", "ge", "er", "ver", "zer", "ent", "emp", "miss")
|
||||
if pref:
|
||||
inner = _stem(lemma[len(pref):])
|
||||
return pref + "ge" + _weak_t(inner), "rule"
|
||||
if lemma.startswith(_INSEP):
|
||||
return _weak_t(stem), "rule"
|
||||
return "ge" + _weak_t(stem), "rule"
|
||||
|
||||
|
||||
def nonfinite(lemma, req):
|
||||
if req == "ppart":
|
||||
return past_participle(lemma)
|
||||
return lemma, "lexicon" if lemma in _VERBS else "rule" # infinitive
|
||||
|
||||
|
||||
def perfect_aux(lemma):
|
||||
return "sein" if lemma in _SEIN else "haben"
|
||||
|
||||
|
||||
# ── adjectives ────────────────────────────────────────────────────────────────---
|
||||
_ADJ_IRREG_SPRL = {"gut": "best", "groß": "größt", "hoch": "höchst",
|
||||
"nah": "nächst", "viel": "meist", "gern": "liebst"}
|
||||
|
||||
|
||||
def comparative(lemma):
|
||||
rec = _ADJS.get(lemma)
|
||||
if rec and rec.get("cmpr"):
|
||||
return rec["cmpr"], "lexicon"
|
||||
return lemma + "er", "rule"
|
||||
|
||||
|
||||
def superlative(lemma):
|
||||
"""Return the bare superlative STEM (realizer adds 'am ...en' or '-e' ending)."""
|
||||
if lemma in _ADJ_IRREG_SPRL:
|
||||
return _ADJ_IRREG_SPRL[lemma], "lexicon"
|
||||
# derive from the comparative so umlaut is carried (alt->älter->ältest)
|
||||
cmpr, cconf = comparative(lemma)
|
||||
base = cmpr[:-2] if cmpr.endswith("er") else lemma
|
||||
end = "est" if base.endswith(("t", "d", "s", "ß", "z", "sch")) else "st"
|
||||
return base + end, cconf
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"source": "UniMorph deu (primary) + kaikki.org German (gap-fill gender/plural)",
|
||||
"license": "CC-BY-SA 3.0 (UniMorph); CC-BY-SA/GFDL (Wiktionary)",
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"nouns_with_gender": sum(1 for v in _NOUNS.values() if v.get("g")),
|
||||
"nouns_with_plural": sum(1 for v in _NOUNS.values() if v.get("pl")),
|
||||
"verb_lemmas": len(_VERBS),
|
||||
"verbs_with_ppart": sum(1 for v in _VERBS.values() if v.get("ppart")),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
for w in ("Hund", "Frau", "Kind", "Mann", "Buch", "Blume"):
|
||||
print(f" {w}: gender={noun_gender(w)} pl={pluralize(w)} "
|
||||
f"gen.sg={decline_noun(w, 'GEN', 'singular')} "
|
||||
f"dat.pl={decline_noun(w, 'DAT', 'plural')}")
|
||||
for v in ("machen", "gehen", "aufstehen", "sein", "haben", "arbeiten"):
|
||||
print(f" {v}: 3sg.prs={finite(v, 'present', 'third', 'singular')} "
|
||||
f"3sg.pst={finite(v, 'past', 'third', 'singular')} "
|
||||
f"ppart={past_participle(v)} aux={perfect_aux(v)} sep={separable_prefix(v)}")
|
||||
for a in ("schnell", "gut", "groß", "alt"):
|
||||
print(f" {a}: cmpr={comparative(a)} sprl={superlative(a)}")
|
||||
@@ -1,562 +0,0 @@
|
||||
"""morphology_es_full.py — production-grade Spanish morphological generator.
|
||||
|
||||
NOT a toy. Backed by a real, broad, licensed lexicon:
|
||||
|
||||
UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0, Wiktionary-derived)
|
||||
1,196,245 inflected forms:
|
||||
6,695 verb lemmas — full paradigms: indicative (present/preterite/
|
||||
imperfect/future), conditional, present & imperfect
|
||||
subjunctive, affirmative imperative, formal/informal
|
||||
48,353 noun lemmas — WITH inherent gender (N;FEM/MASC;SG/PL)
|
||||
16,984 adj lemmas — gender + number paradigms
|
||||
|
||||
Fallbacks (so we degrade, never crash, on out-of-vocabulary input):
|
||||
- verbs : mlconjug3 (ML paradigm model, conjugates ANY Spanish verb) then a
|
||||
hand-rolled regular-ending generator
|
||||
- nouns : gender heuristic (endings) + regular pluralization
|
||||
- adjs : -o/-a gender rule + regular pluralization
|
||||
|
||||
Every generated form carries a CONFIDENCE flag:
|
||||
"lexicon" form came straight from UniMorph (trust: high)
|
||||
"model" form came from mlconjug3 (trust: high)
|
||||
"rule" form came from a deterministic rule (trust: medium)
|
||||
"fallback" we could not inflect; returned lemma as-is (trust: low → FLAG)
|
||||
|
||||
Public API (used by realizer_es.py):
|
||||
conjugate(lemma, mood, tense, person, number, formality="informal") -> (form, conf)
|
||||
participle(lemma) -> (form, conf) # past participle (compound tenses)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
attach_enclitics(verb_form, clitics) -> str # accent-correct enclisis
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import os
|
||||
import pickle
|
||||
import unicodedata
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "spa.unimorph")
|
||||
_CACHE = os.path.join(_HERE, "data", "es_morph_cache.pkl")
|
||||
|
||||
# ── canonical feature keys the realizer speaks, mapped to UniMorph tags ─────────
|
||||
# mood/tense pair -> the UniMorph feature substring that identifies it
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): ("IND", "PRS", None),
|
||||
("ind", "preterite"): ("IND", "PST", "PFV"),
|
||||
("ind", "imperfect"): ("IND", "PST", "IPFV"),
|
||||
("ind", "future"): ("IND", "FUT", None),
|
||||
("ind", "conditional"):("COND", None, None),
|
||||
("sbjv", "present"): ("SBJV", "PRS", None),
|
||||
("sbjv", "imperfect"): ("SBJV", "PST", "LGSPEC1"), # -ra form
|
||||
("imp", "present"): ("POS", "IMP", None),
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
# ── build / load the compact lexicon ───────────────────────────────────────────
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs = {} # (lemma, canonkey) -> form canonkey e.g. "ind|present|1|SG|infm"
|
||||
nouns = {} # lemma -> {"g": "m"/"f", "SG": form, "PL": form}
|
||||
adjs = {} # lemma -> {("m","SG"): form, ...}
|
||||
part = {} # lemma -> masc-sg participle
|
||||
ger = {} # lemma -> gerund
|
||||
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V":
|
||||
# skip clitic-bearing rows (we generate clitics ourselves)
|
||||
if "PRO" in f:
|
||||
continue
|
||||
if "V.PTCP" in f and "PST" in f and "MASC" in f and "SG" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if "V.CVB" in f or "NFIN" in f or "V.PTCP" in f:
|
||||
if "V.CVB" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
# identify mood/tense
|
||||
mt = None
|
||||
for (mood, tense), (a, b, c) in _VERB_KEYMAP.items():
|
||||
if a not in f:
|
||||
continue
|
||||
if b is not None and b not in f:
|
||||
continue
|
||||
if c is not None and c not in f:
|
||||
continue
|
||||
# disambiguate IND;PST needing PFV vs IPFV
|
||||
if a == "IND" and b == "PST" and c not in f:
|
||||
continue
|
||||
mt = (mood, tense)
|
||||
break
|
||||
if mt is None:
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
formal = "form" if "FORM" in f else ("infm" if "INFM" in f else "any")
|
||||
key = f"{mt[0]}|{mt[1]}|{person}|{number}|{formal}"
|
||||
verbs.setdefault((lemma, key), form)
|
||||
|
||||
elif head == "N":
|
||||
# substring test handles epicene "MASC+FEM" (-> masc citation)
|
||||
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else None)
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if num is None:
|
||||
continue
|
||||
# store forms keyed by (gender,number); animate nouns list BOTH
|
||||
# genders under one lemma (niño -> niño/niña). Resolve citation
|
||||
# gender in a post-pass (gender of the row whose form == lemma).
|
||||
d = nouns.setdefault(lemma, {})
|
||||
d.setdefault("_rows", []).append((g, num, form))
|
||||
|
||||
elif head == "ADJ":
|
||||
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else "m")
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if num is None:
|
||||
continue
|
||||
adjs.setdefault(lemma, {})[(g, num)] = form
|
||||
|
||||
# post-pass: resolve noun citation gender + default SG/PL forms
|
||||
for lemma, d in nouns.items():
|
||||
rows = d.pop("_rows", [])
|
||||
# citation gender = gender of the row whose form == lemma; else first MASC;
|
||||
# else first seen gender.
|
||||
cite_g = None
|
||||
for g, num, form in rows:
|
||||
if form == lemma and g:
|
||||
cite_g = g
|
||||
break
|
||||
if cite_g is None:
|
||||
for g, num, form in rows:
|
||||
if g == "m":
|
||||
cite_g = "m"
|
||||
break
|
||||
if cite_g is None:
|
||||
cite_g = next((g for g, _, _ in rows if g), "m")
|
||||
d["g"] = cite_g
|
||||
for g, num, form in rows:
|
||||
d[(g, num)] = form
|
||||
d["SG"] = d.get((cite_g, "SG")) or next((f for g, n, f in rows if n == "SG"), lemma)
|
||||
d["PL"] = d.get((cite_g, "PL")) or next((f for g, n, f in rows if n == "PL"), None)
|
||||
|
||||
# post-pass: UniMorph omits the identity inflection (masc-sg == lemma) for
|
||||
# adjectives, so fill it in; without this a fem-sg row wrongly satisfies a
|
||||
# masc-sg request (alto -> alta bug).
|
||||
for lemma, d in adjs.items():
|
||||
d.setdefault(("m", "SG"), lemma)
|
||||
|
||||
data = {"verbs": verbs, "nouns": nouns, "adjs": adjs, "part": part, "ger": ger}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE) and os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH):
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _NOUNS, _ADJS, _PART, _GER = (
|
||||
_LEX["verbs"], _LEX["nouns"], _LEX["adjs"], _LEX["part"], _LEX["ger"])
|
||||
|
||||
# ── mlconjug3 fallback (lazy) ───────────────────────────────────────────────────
|
||||
_MLC = None
|
||||
_MLC_TENSE = { # (mood,tense) -> (mlconjug mood label, tense label)
|
||||
("ind", "present"): ("Indicativo", "Indicativo presente"),
|
||||
("ind", "preterite"): ("Indicativo", "Indicativo pretérito perfecto simple"),
|
||||
("ind", "imperfect"): ("Indicativo", "Indicativo pretérito imperfecto"),
|
||||
("ind", "future"): ("Indicativo", "Indicativo futuro"),
|
||||
("ind", "conditional"): ("Condicional", "Condicional Condicional"),
|
||||
("sbjv", "present"): ("Subjuntivo", "Subjuntivo presente"),
|
||||
("sbjv", "imperfect"): ("Subjuntivo", "Subjuntivo pretérito imperfecto 1"),
|
||||
("imp", "present"): ("Imperativo", "Imperativo Afirmativo"),
|
||||
}
|
||||
_MLC_SLOT = { # (person,number) -> mlconjug slot key
|
||||
("first", "singular"): "1s", ("second", "singular"): "2s",
|
||||
("third", "singular"): "3s", ("first", "plural"): "1p",
|
||||
("second", "plural"): "2p", ("third", "plural"): "3p",
|
||||
}
|
||||
|
||||
|
||||
def _mlc_conjugate(lemma, mood, tense, person, number):
|
||||
global _MLC
|
||||
try:
|
||||
if _MLC is None:
|
||||
from mlconjug3 import Conjugator
|
||||
_MLC = Conjugator(language="es")
|
||||
v = _MLC.conjugate(lemma)
|
||||
if v is None:
|
||||
return None
|
||||
info = v.conjug_info
|
||||
m, t = _MLC_TENSE.get((mood, tense), (None, None))
|
||||
if m is None or m not in info or t not in info[m]:
|
||||
return None
|
||||
block = info[m][t]
|
||||
slot = _MLC_SLOT.get((person, number))
|
||||
if isinstance(block, dict) and slot in block and block[slot]:
|
||||
return block[slot]
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# ── regular-ending rule fallback (last resort, deterministic) ───────────────────
|
||||
def _vclass(lemma):
|
||||
return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else "ar"
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "as", "a", "amos", "áis", "an"],
|
||||
("ind", "present", "er"): ["o", "es", "e", "emos", "éis", "en"],
|
||||
("ind", "present", "ir"): ["o", "es", "e", "imos", "ís", "en"],
|
||||
("ind", "preterite", "ar"): ["é", "aste", "ó", "amos", "asteis", "aron"],
|
||||
("ind", "preterite", "er"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
||||
("ind", "preterite", "ir"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
||||
("ind", "imperfect", "ar"): ["aba", "abas", "aba", "ábamos", "abais", "aban"],
|
||||
("ind", "imperfect", "er"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
||||
("ind", "imperfect", "ir"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
||||
("sbjv", "present", "ar"): ["e", "es", "e", "emos", "éis", "en"],
|
||||
("sbjv", "present", "er"): ["a", "as", "a", "amos", "áis", "an"],
|
||||
("sbjv", "present", "ir"): ["a", "as", "a", "amos", "áis", "an"],
|
||||
("sbjv", "imperfect", "ar"): ["ara", "aras", "ara", "áramos", "arais", "aran"],
|
||||
("sbjv", "imperfect", "er"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
||||
("sbjv", "imperfect", "ir"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
||||
}
|
||||
_FUT = ["é", "ás", "á", "emos", "éis", "án"]
|
||||
_COND = ["ía", "ías", "ía", "íamos", "íais", "ían"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
if len(lemma) < 3 or lemma[-2:] not in ("ar", "er", "ir"):
|
||||
return None
|
||||
vc, st, i = _vclass(lemma), _stem(lemma), _slot_idx(person, number)
|
||||
if tense == "future":
|
||||
return lemma + _FUT[i]
|
||||
if tense == "conditional":
|
||||
return lemma + _COND[i]
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if table:
|
||||
return st + table[i]
|
||||
if mood == "imp" and tense == "present":
|
||||
# affirmative tú imperative = 3sg present indicative
|
||||
pres = _REG.get(("ind", "present", vc))
|
||||
return st + pres[2] if number == "singular" else st + pres[5]
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number, formality="informal"):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
formal = "form" if formality == "formal" else "infm"
|
||||
if p and n:
|
||||
for fkey in (formal, "any", "infm" if formal == "form" else "form"):
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}|{fkey}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
m = _mlc_conjugate(lemma, mood, tense, person, number)
|
||||
if m:
|
||||
return m, "model"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
_IRREG_PART = { # guarantee the common irregular participles
|
||||
"escribir": "escrito", "describir": "descrito", "abrir": "abierto",
|
||||
"cubrir": "cubierto", "descubrir": "descubierto", "morir": "muerto",
|
||||
"poner": "puesto", "ver": "visto", "volver": "vuelto", "devolver": "devuelto",
|
||||
"hacer": "hecho", "deshacer": "deshecho", "decir": "dicho", "romper": "roto",
|
||||
"resolver": "resuelto", "freír": "frito", "imprimir": "impreso",
|
||||
"satisfacer": "satisfecho", "prever": "previsto", "revolver": "revuelto",
|
||||
}
|
||||
|
||||
|
||||
def participle(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _IRREG_PART:
|
||||
return _IRREG_PART[lemma], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ado", "rule"
|
||||
if lemma[-2:] in ("er", "ir"):
|
||||
return lemma[:-2] + "ido", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
_IRREG_GER = {"dormir": "durmiendo", "morir": "muriendo", "pedir": "pidiendo",
|
||||
"sentir": "sintiendo", "mentir": "mintiendo", "servir": "sirviendo",
|
||||
"venir": "viniendo", "decir": "diciendo", "poder": "pudiendo",
|
||||
"ir": "yendo", "leer": "leyendo", "creer": "creyendo",
|
||||
"oír": "oyendo", "traer": "trayendo", "caer": "cayendo",
|
||||
"construir": "construyendo", "huir": "huyendo", "reír": "riendo"}
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _IRREG_GER:
|
||||
return _IRREG_GER[lemma], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ando", "rule"
|
||||
if lemma[-2:] in ("er", "ir"):
|
||||
return lemma[:-2] + "iendo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ────────────────────────────────────────────────
|
||||
_INVARIANT_PL = {"lunes", "martes", "miércoles", "jueves", "viernes",
|
||||
"crisis", "tesis", "análisis", "dosis", "virus", "paraguas"}
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf, g in (("ión", "f"), ("dad", "f"), ("tad", "f"), ("umbre", "f"),
|
||||
("sis", "f"), ("ez", "f"), ("triz", "f"),
|
||||
("ema", "m"), ("ama", "m"), ("oma", "m"), ("aje", "m"),
|
||||
("or", "m"), ("án", "m"), ("ín", "m")):
|
||||
if noun.endswith(suf):
|
||||
return g
|
||||
if noun.endswith("o"):
|
||||
return "m"
|
||||
if noun.endswith("a"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _regular_plural(noun):
|
||||
if noun in _INVARIANT_PL:
|
||||
return noun
|
||||
if not noun:
|
||||
return noun
|
||||
last = noun[-1]
|
||||
if last == "z":
|
||||
return noun[:-1] + "ces"
|
||||
if last in "aeiouáéíóú":
|
||||
# stressed final vowel í/ú -> +es (rubí->rubíes), else +s
|
||||
if last in "íú":
|
||||
return noun + "es"
|
||||
return noun + "s"
|
||||
if last == "s":
|
||||
# esdrújula / stress-final handled crudely; most polysyllables invariant
|
||||
return noun
|
||||
return noun + "es"
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
# honor a requested gender for animate nouns (gato -> gata)
|
||||
if gender and (gender, num) in d:
|
||||
return d[(gender, num)], "lexicon"
|
||||
if d.get(num):
|
||||
return d[num], "lexicon"
|
||||
if number == "singular":
|
||||
return lemma, "rule" if not d else "lexicon"
|
||||
return _regular_plural(lemma), "rule"
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ─────────────────────────────────────────────────
|
||||
_INV_GENDER_ADJ = {"español": "española", "trabajador": "trabajadora",
|
||||
"hablador": "habladora", "encantador": "encantadora",
|
||||
"alemán": "alemana", "francés": "francesa", "inglés": "inglesa"}
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _ADJS.get(lemma)
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
form = d.get((gender, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# gender-invariant adjective (grande, feliz, azul): fem == masc.
|
||||
# For a missing plural, pluralize this gender's singular form.
|
||||
sg = d.get((gender, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if number == "plural":
|
||||
return _regular_plural(sg), "rule"
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
a = lemma
|
||||
if gender == "f":
|
||||
if a in _INV_GENDER_ADJ:
|
||||
a = _INV_GENDER_ADJ[a]
|
||||
elif a.endswith("o"):
|
||||
a = a[:-1] + "a"
|
||||
if number == "plural":
|
||||
a = _regular_plural(a)
|
||||
return a, ("rule" if (a != lemma or gender == "m") else "rule")
|
||||
|
||||
|
||||
# ── PUBLIC: clitic enclisis (dá + me + lo -> dámelo) ────────────────────────────
|
||||
def _strip_accents(s):
|
||||
return "".join(c for c in unicodedata.normalize("NFD", s)
|
||||
if unicodedata.category(c) != "Mn")
|
||||
|
||||
|
||||
def _count_syllables_vowelgroups(word):
|
||||
# crude: count vowel groups
|
||||
w = _strip_accents(word).lower()
|
||||
groups, prev = 0, False
|
||||
for ch in w:
|
||||
isv = ch in "aeiou"
|
||||
if isv and not prev:
|
||||
groups += 1
|
||||
prev = isv
|
||||
return groups
|
||||
|
||||
|
||||
def _host_stress_from_end(word):
|
||||
"""Stressed-syllable index counted from the end (1=last) of a verb host."""
|
||||
syls = _count_syllables_vowelgroups(word)
|
||||
if any(c in "áéíóú" for c in word):
|
||||
return None # already carries its own accent
|
||||
if word[-2:] in ("ar", "er", "ir"): # infinitive: oxytone
|
||||
return 1
|
||||
if word.endswith("ndo"): # gerund: paroxytone
|
||||
return 2
|
||||
if word[-1:] in "aeiouns" and syls >= 2: # default paroxytone
|
||||
return 2
|
||||
return 1 # monosyllable / consonant-final oxytone
|
||||
|
||||
|
||||
def attach_enclitics(verb_form, clitics):
|
||||
"""Append clitic pronouns to a verb (imperative/infinitive/gerund enclisis)
|
||||
and add a written accent when the resulting word becomes esdrújula/
|
||||
sobreesdrújula (stress >= 3 syllables from the end): dá+me+lo -> dámelo,
|
||||
lleva+me -> llévame, but dar+te -> darte and da+me -> dame (no accent)."""
|
||||
if not clitics:
|
||||
return verb_form
|
||||
tail = "".join(clitics)
|
||||
if any(c in "áéíóú" for c in verb_form): # host already accented
|
||||
return verb_form + tail
|
||||
sfe = _host_stress_from_end(verb_form)
|
||||
total_sfe = sfe + len(clitics) # each clitic = 1 syllable
|
||||
if total_sfe >= 3:
|
||||
return _accentuate_nucleus(verb_form, sfe) + tail
|
||||
return verb_form + tail
|
||||
|
||||
|
||||
def _accentuate_nucleus(word, sfe):
|
||||
"""Put a written accent on the syllable `sfe` positions from the word's end."""
|
||||
vowels = "aeiou"
|
||||
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
||||
if not nuclei or sfe > len(nuclei):
|
||||
return word
|
||||
i = nuclei[-sfe]
|
||||
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
||||
return word[:i] + acc[word[i]] + word[i + 1:]
|
||||
|
||||
|
||||
def _accentuate_last_stressed(word):
|
||||
# Restore the host's ORIGINAL lexical stress with a written accent.
|
||||
# Default Spanish stress: word ending in vowel/n/s -> penultimate syllable;
|
||||
# otherwise (e.g. infinitives in -r) -> last syllable.
|
||||
vowels = "aeiou"
|
||||
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
||||
if not nuclei:
|
||||
return word
|
||||
if word[-1] in "aeiouns" and len(nuclei) >= 2:
|
||||
i = nuclei[-2] # paroxytone: penult nucleus
|
||||
else:
|
||||
i = nuclei[-1] # oxytone / monosyllable: last nucleus
|
||||
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
||||
return word[:i] + acc[word[i]] + word[i + 1:]
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"source": "UniMorph Spanish (github.com/unimorph/spa)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary-derived)",
|
||||
"total_forms": sum(len(v) for v in (_VERBS, _NOUNS, _ADJS)) if False else None,
|
||||
"verb_forms": len(_VERBS),
|
||||
"verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
"participles": len(_PART),
|
||||
"gerunds": len(_GER),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import json
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("hablar", "ind", "present", "first", "singular", "hablo"),
|
||||
("comer", "ind", "present", "third", "plural", "comen"),
|
||||
("vivir", "ind", "present", "first", "plural", "vivimos"),
|
||||
("ser", "ind", "present", "third", "singular", "es"),
|
||||
("ir", "ind", "preterite", "first", "singular", "fui"),
|
||||
("tener", "ind", "future", "first", "singular", "tendré"),
|
||||
("hacer", "sbjv", "present", "first", "singular", "haga"),
|
||||
("dormir", "ind", "present", "first", "singular", "duermo"),
|
||||
("pensar", "sbjv", "present", "third", "singular", "piense"),
|
||||
("dar", "ind", "preterite", "third", "singular", "dio"),
|
||||
("poner", "ind", "conditional", "first", "singular", "pondría"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
if got == exp:
|
||||
ok += 1
|
||||
print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]:3} -> {got:14} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender casa:", noun_gender("casa"), "| problema:", noun_gender("problema"),
|
||||
"| agua:", noun_gender("agua"), "| mano:", noun_gender("mano"))
|
||||
print(" plural: luz->", inflect_noun("luz", "plural"), "| rey->", inflect_noun("rey", "plural"))
|
||||
print(" adj: rojo/f/pl->", inflect_adj("rojo", "f", "plural"),
|
||||
"| feliz/m/pl->", inflect_adj("feliz", "m", "plural"),
|
||||
"| grande/f/pl->", inflect_adj("grande", "f", "plural"))
|
||||
print(" enclisis: da+[me,lo]->", attach_enclitics("da", ["me", "lo"]),
|
||||
"| di+[me]->", attach_enclitics("di", ["me"]),
|
||||
"| dar+[se,lo]->", attach_enclitics("dar", ["se", "lo"]))
|
||||
@@ -1,629 +0,0 @@
|
||||
"""morphology_fr_full.py — production-grade French morphological generator.
|
||||
|
||||
Same architecture as morphology_it_full.py (shared Romance engine); French-specific
|
||||
data and rules swapped in. Backed by three real, Wiktionary-lineage sources:
|
||||
|
||||
VERBS
|
||||
UniMorph French (github.com/unimorph/fra, CC-BY-SA 3.0)
|
||||
7,535 verb lemmas × full paradigm, CLEAN orthography:
|
||||
indicatif présent / imparfait (PST;IPFV) / passé simple (PST;PFV) /
|
||||
futur, conditionnel (COND), subjonctif présent (SBJV;PRS) /
|
||||
subjonctif imparfait (SBJV;PST), impératif (POS;IMP), infinitif (NFIN),
|
||||
participe présent (V.CVB/V.PTCP;PRS), participe passé (V.PTCP;PST, m.sg).
|
||||
fr_irreg_verbs.json — high-frequency verbs UniMorph MISSES or mis-slots,
|
||||
above all ÊTRE (absent from UniMorph fra), plus avoir/aller/faire/… — the
|
||||
auxiliaries the passé-composé + être-agreement system depends on. Extracted
|
||||
from kaikki.org French (build_fr_irreg.py), reflexive/multiword forms
|
||||
dropped. This layer takes PRIORITY.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org French (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender (head-template arg) + real plural
|
||||
(cheval->chevaux, œil->yeux, invariable -s/-x/-z), resolved PER LEMMA.
|
||||
adjective lemmas with real feminine + plural (petit->petite/petits/petites,
|
||||
beau->belle/beaux/belles, heureux->heureuse, rouge invariant-gender).
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV input):
|
||||
verbs : rule generator for -er / -ir(-iss-) / -re (with -cer/-ger spelling,
|
||||
future/conditional stems, imparfait/subjonctif endings)
|
||||
nouns : gender heuristic (endings) + rule pluralization (-al->-aux, -eau->-eaux)
|
||||
adjs : fem/plural agreement rules (-er->-ère, -eux->-euse, -f->-ve, +e default)
|
||||
|
||||
Confidence flag on every form: "lexicon" | "rule" | "fallback".
|
||||
|
||||
Public API (used by realizer_fr.py): identical signature to morphology_it_full.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "fra.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "fr_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_fr.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "fr_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set that must ALL be present ────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"}, # imparfait
|
||||
("ind", "passe_simple"): {"IND", "PST", "PFV"}, # passé simple
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"}, # French: V;COND;1;SG
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ─────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {}
|
||||
part = {}
|
||||
ger = {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
elif "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "passe_simple" and "IPFV" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "construed", "collective",
|
||||
"nonstandard", "dated", "Louisiana", "Switzerland", "Belgium"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {}
|
||||
adjs = {}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm not in ("#", "-", "—"):
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback ─────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("er"):
|
||||
return "er"
|
||||
if lemma.endswith("ir"):
|
||||
return "ir"
|
||||
if lemma.endswith("re"):
|
||||
return "re"
|
||||
if lemma.endswith("oir"):
|
||||
return "oir"
|
||||
return None
|
||||
|
||||
|
||||
# present-tense endings [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG_PRES = {
|
||||
"er": ["e", "es", "e", "ons", "ez", "ent"],
|
||||
"ir": ["is", "is", "it", "issons", "issez", "issent"], # -iss- class (finir)
|
||||
"re": ["s", "s", "", "ons", "ez", "ent"], # vendre: vends/vend
|
||||
}
|
||||
_REG_IMPF = ["ais", "ais", "ait", "ions", "iez", "aient"] # attaches to pres-1pl stem
|
||||
_REG_SUBJ = ["e", "es", "e", "ions", "iez", "ent"] # attaches to 3pl stem
|
||||
_REG_PS = { # passé simple
|
||||
"er": ["ai", "as", "a", "âmes", "âtes", "èrent"],
|
||||
"ir": ["is", "is", "it", "îmes", "îtes", "irent"],
|
||||
"re": ["is", "is", "it", "îmes", "îtes", "irent"],
|
||||
}
|
||||
_FUT = ["ai", "as", "a", "ons", "ez", "ont"]
|
||||
_COND = ["ais", "ais", "ait", "ions", "iez", "aient"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _fut_stem(lemma, vc):
|
||||
"""Future/conditional stem = infinitive (drop final -e of -re)."""
|
||||
if vc == "re":
|
||||
return lemma[:-1] # vendre -> vendr-
|
||||
return lemma # parler-, finir-
|
||||
|
||||
|
||||
def _pres_1pl_stem(lemma, vc):
|
||||
"""Imparfait stem = present 1pl minus -ons (parlons->parl-, finissons->finiss-)."""
|
||||
if vc == "er":
|
||||
stem = lemma[:-2]
|
||||
if stem.endswith("g"):
|
||||
return stem + "e" # mangeons -> mange- (imparfait mangeais)
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "ç" # commençons -> commenç-
|
||||
return stem
|
||||
if vc == "ir":
|
||||
return lemma[:-1] + "iss" # finir -> finiss-
|
||||
if vc == "re":
|
||||
return lemma[:-2] # vendre -> vend-
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
def _apply_er_spelling(stem, ending):
|
||||
"""-cer/-ger softening before a/o (commençons, mangeons)."""
|
||||
if ending and ending[0] in ("a", "o"):
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "ç" + ending
|
||||
if stem.endswith("g"):
|
||||
return stem + "e" + ending
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
i = _slot_idx(person, number)
|
||||
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
stem = _fut_stem(lemma, vc)
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
return stem + end
|
||||
|
||||
if mood == "ind" and tense == "present":
|
||||
table = _REG_PRES.get("ir" if vc == "ir" else vc)
|
||||
if not table:
|
||||
return None
|
||||
body = lemma[:-2] if vc in ("er", "re") else lemma[:-1] if vc == "ir" else lemma[:-2]
|
||||
if vc == "ir":
|
||||
body = lemma[:-2] # fin- ; endings carry -iss-
|
||||
end = table[i]
|
||||
return body + end
|
||||
end = table[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(body, end)
|
||||
return body + end
|
||||
|
||||
if mood == "ind" and tense == "imperfect":
|
||||
stem = _pres_1pl_stem(lemma, vc)
|
||||
return stem + _REG_IMPF[i]
|
||||
|
||||
if mood == "ind" and tense == "passe_simple":
|
||||
table = _REG_PS.get("ir" if vc == "ir" else vc)
|
||||
if not table:
|
||||
return None
|
||||
body = lemma[:-2] if vc in ("er", "re") else lemma[:-2]
|
||||
end = table[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(body, end)
|
||||
return body + end
|
||||
|
||||
if mood == "sbjv" and tense == "present":
|
||||
# subjonctif: present-3pl stem + e/es/e/ions/iez/ent
|
||||
stem3 = _pres_1pl_stem(lemma, vc) if vc == "ir" else (
|
||||
lemma[:-2] if vc in ("er", "re") else lemma[:-2])
|
||||
if vc == "ir":
|
||||
stem3 = lemma[:-2] + "iss"
|
||||
end = _REG_SUBJ[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(stem3, end)
|
||||
return stem3 + end
|
||||
|
||||
if mood == "imp" and tense == "affirmative":
|
||||
# impératif ~ present indicative (tu drops -s for -er verbs)
|
||||
pres = _rule_conjugate(lemma, "ind", "present", person, number)
|
||||
if pres and vc == "er" and person == "second" and number == "singular":
|
||||
return pres[:-1] if pres.endswith("es") else pres
|
||||
return pres
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence)."""
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number}"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund/participe présent ────────────────────────────────
|
||||
def _participle_msg(lemma):
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
return None, None
|
||||
|
||||
|
||||
# irregular participle fem/plural quirks (drop circonflexe: dû->due, dus)
|
||||
_PART_FIX = {"dû": {"f|SG": "due", "m|PL": "dus", "f|PL": "dues"}}
|
||||
|
||||
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
"""Past participle with French gender/number agreement.
|
||||
m.sg = base; f.sg = base+e; m.pl = base+s (invariable if base ends s/x);
|
||||
f.pl = f.sg+s."""
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
msg, src = _participle_msg(lemma)
|
||||
conf = "lexicon"
|
||||
if msg is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "er":
|
||||
msg = lemma[:-2] + "é"
|
||||
elif vc == "ir":
|
||||
msg = lemma[:-1] # finir -> fini, partir -> parti
|
||||
elif vc == "re":
|
||||
msg = lemma[:-2] + "u" # vendre -> vendu
|
||||
elif vc == "oir":
|
||||
msg = lemma[:-3] + "u" # (rough) recevoir handled by irreg
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
fix = _PART_FIX.get(msg)
|
||||
if fix and f"{g}|{num}" in fix:
|
||||
return fix[f"{g}|{num}"], conf
|
||||
if g == "m" and num == "SG":
|
||||
return msg, conf
|
||||
fem = msg + "e" if not msg.endswith("e") else msg
|
||||
if g == "f" and num == "SG":
|
||||
return fem, conf
|
||||
if g == "m" and num == "PL":
|
||||
return msg if msg.endswith(("s", "x")) else msg + "s", conf
|
||||
# f|PL
|
||||
return fem + "s", conf
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
"""Participe présent (base for gérondif 'en -ant')."""
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "er":
|
||||
stem = lemma[:-2]
|
||||
if stem.endswith("g"):
|
||||
return stem + "eant", "rule"
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "çant", "rule"
|
||||
return stem + "ant", "rule"
|
||||
if vc == "ir":
|
||||
return lemma[:-2] + "issant", "rule"
|
||||
if vc == "re":
|
||||
return lemma[:-2] + "ant", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("tion", "sion", "aison", "ance", "ence", "ette", "elle", "esse",
|
||||
"ude", "ade", "ée", "té", "tié", "ie", "ise", "ure", "eur")
|
||||
_MASC_SUF = ("ment", "age", "eau", "isme", "oir", "ier", "eur", "in", "on")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
for suf in _MASC_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "m"
|
||||
if noun.endswith("e"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
# closed sets for French plural irregularities
|
||||
_OU_X = {"bijou", "caillou", "chou", "genou", "hibou", "joujou", "pou"}
|
||||
_AIL_AUX = {"travail", "vitrail", "corail", "émail", "bail", "soupirail", "vantail"}
|
||||
_AL_S = {"bal", "carnaval", "festival", "récital", "chacal", "régal", "cal", "aval"}
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic French pluralization. (form, ok); ok=False FLAGS ambiguity."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
if noun[-1:] in ("s", "x", "z"):
|
||||
return noun, True # invariable
|
||||
if noun in _OU_X:
|
||||
return noun + "x", True
|
||||
if noun.endswith(("eau", "au", "eu")):
|
||||
if noun in ("pneu", "bleu", "landau", "sarrau"):
|
||||
return noun + "s", True
|
||||
return noun + "x", True # bateau->bateaux, jeu->jeux
|
||||
if noun.endswith("al"):
|
||||
if noun in _AL_S:
|
||||
return noun + "s", True
|
||||
return noun[:-2] + "aux", True # cheval->chevaux
|
||||
if noun.endswith("ail"):
|
||||
if noun in _AIL_AUX:
|
||||
return noun[:-3] + "aux", True # travail->travaux
|
||||
return noun + "s", True
|
||||
return noun + "s", True # default
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# adjectives whose kaikki entries are unreliable: audited forms
|
||||
_ADJ_FIX = {
|
||||
"beau": {("m", "SG"): "beau", ("f", "SG"): "belle",
|
||||
("m", "PL"): "beaux", ("f", "PL"): "belles"},
|
||||
"nouveau": {("m", "SG"): "nouveau", ("f", "SG"): "nouvelle",
|
||||
("m", "PL"): "nouveaux", ("f", "PL"): "nouvelles"},
|
||||
"vieux": {("m", "SG"): "vieux", ("f", "SG"): "vieille",
|
||||
("m", "PL"): "vieux", ("f", "PL"): "vieilles"},
|
||||
"fou": {("m", "SG"): "fou", ("f", "SG"): "folle",
|
||||
("m", "PL"): "fous", ("f", "PL"): "folles"},
|
||||
"blanc": {("m", "SG"): "blanc", ("f", "SG"): "blanche",
|
||||
("m", "PL"): "blancs", ("f", "PL"): "blanches"},
|
||||
"long": {("m", "SG"): "long", ("f", "SG"): "longue",
|
||||
("m", "PL"): "longs", ("f", "PL"): "longues"},
|
||||
"bon": {("m", "SG"): "bon", ("f", "SG"): "bonne",
|
||||
("m", "PL"): "bons", ("f", "PL"): "bonnes"},
|
||||
}
|
||||
|
||||
|
||||
def _rule_fem(a):
|
||||
if a.endswith("e"):
|
||||
return a
|
||||
if a.endswith("er"):
|
||||
return a[:-2] + "ère"
|
||||
if a.endswith("eau"):
|
||||
return a[:-3] + "elle"
|
||||
if a.endswith("eux"):
|
||||
return a[:-3] + "euse"
|
||||
if a.endswith("f"):
|
||||
return a[:-1] + "ve"
|
||||
if a.endswith(("on", "en", "el", "eil", "et")):
|
||||
return a + a[-1] + "e" # bon->bonne, ancien->ancienne, muet->muette
|
||||
if a.endswith("c"):
|
||||
return a[:-1] + "che" # blanc->blanche (public->publique via FIX)
|
||||
return a + "e" # grand->grande, petit->petite, vert->verte
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
fix = _ADJ_FIX.get(lemma)
|
||||
if fix and (g, num) in fix:
|
||||
return fix[(g, num)], "lexicon"
|
||||
d = _ADJS.get(lemma)
|
||||
if d and d.get((g, num)):
|
||||
return d[(g, num)], "lexicon"
|
||||
# derive
|
||||
msc = (d.get(("m", "SG")) if d else None) or lemma
|
||||
if g == "m" and num == "SG":
|
||||
return msc, "lexicon" if d else "rule"
|
||||
fem = (d.get(("f", "SG")) if d else None) or _rule_fem(msc)
|
||||
if g == "f" and num == "SG":
|
||||
return fem, "lexicon" if (d and d.get(("f", "SG"))) else "rule"
|
||||
if g == "m" and num == "PL":
|
||||
if msc.endswith(("s", "x")):
|
||||
return msc, "rule"
|
||||
if msc.endswith("al"):
|
||||
return msc[:-2] + "aux", "rule"
|
||||
if msc.endswith("eau"):
|
||||
return msc + "x", "rule"
|
||||
return msc + "s", "rule"
|
||||
# f|PL
|
||||
return (fem if fem.endswith("s") else fem + "s"), "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph French (github.com/unimorph/fra) + kaikki.org "
|
||||
"irregulars (être + high-frequency)",
|
||||
"noun_adj_source": "kaikki.org French (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREGV),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("parler", "ind", "present", "first", "singular", "parle"),
|
||||
("être", "ind", "present", "third", "singular", "est"),
|
||||
("avoir", "ind", "present", "first", "singular", "ai"),
|
||||
("aller", "ind", "present", "third", "plural", "vont"),
|
||||
("finir", "ind", "present", "first", "singular", "finis"),
|
||||
("finir", "ind", "present", "first", "plural", "finissons"),
|
||||
("manger", "ind", "present", "first", "plural", "mangeons"),
|
||||
("faire", "ind", "future", "first", "singular", "ferai"),
|
||||
("pouvoir", "sbjv", "present", "third", "singular", "puisse"),
|
||||
("prendre", "ind", "passe_simple", "third", "singular", "prit"),
|
||||
("vendre", "ind", "present", "third", "singular", "vend"),
|
||||
("commencer", "ind", "imperfect", "first", "singular", "commençais"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:10} {mood}/{tense:12} {per[:3]}.{num[:2]} -> {got:12} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: maison=", noun_gender("maison"), "chat=", noun_gender("chat"),
|
||||
"cheval=", noun_gender("cheval"), "nation=", noun_gender("nation"))
|
||||
print(" plural: cheval->", inflect_noun("cheval", "plural"),
|
||||
"| bateau->", inflect_noun("bateau", "plural"),
|
||||
"| prix->", inflect_noun("prix", "plural"),
|
||||
"| chat->", inflect_noun("chat", "plural"))
|
||||
print(" adj: petit/f/sg->", inflect_adj("petit", "f", "singular"),
|
||||
"| beau/f/sg->", inflect_adj("beau", "f", "singular"),
|
||||
"| heureux/f/sg->", inflect_adj("heureux", "f", "singular"),
|
||||
"| national/m/pl->", inflect_adj("national", "m", "plural"))
|
||||
print(" part: aller/f/sg->", participle("aller", "f", "singular"),
|
||||
"| prendre/f/pl->", participle("prendre", "f", "plural"),
|
||||
"| finir/m/pl->", participle("finir", "m", "plural"))
|
||||
print(" ger: manger->", gerund("manger"), "| finir->", gerund("finir"))
|
||||
@@ -1,588 +0,0 @@
|
||||
"""morphology_it_full.py — production-grade Italian morphological generator.
|
||||
|
||||
NOT a toy. Backed by three real, Wiktionary-lineage lexical sources:
|
||||
|
||||
VERBS
|
||||
UniMorph Italian (github.com/unimorph/ita, CC-BY-SA 3.0)
|
||||
10,009 verb lemmas × full paradigm, CLEAN orthography (no stress marks):
|
||||
indicative present / imperfetto (PST;IPFV) / passato remoto (PST;PFV) /
|
||||
futuro, condizionale (COND),
|
||||
congiuntivo presente (SBJV;PRS) / imperfetto (SBJV;PST),
|
||||
affirmative imperative, infinitive, gerundio (V.CVB;PRS),
|
||||
past participle (masc-sg; fem/plural derived by vowel rule).
|
||||
it_irreg_verbs.json — 66 high-frequency verbs UniMorph MISSES
|
||||
(essere, avere, potere, uscire, tenere, prendere, piacere, …), extracted
|
||||
from kaikki.org Italian, filtered to standard forms, and DE-STRESSED to
|
||||
real orthography (kaikki marks tonic stress everywhere: pàrlo->parlo,
|
||||
avùto->avuto; final legit accents kept: sarò, è). Built by build_it_irreg.py.
|
||||
This layer takes priority — it supplies the two auxiliaries essere/avere,
|
||||
which the whole passato-prossimo / essere-agreement system depends on.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Italian (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender (head-template arg) + real (often irregular)
|
||||
plural — uomo->uomini, uovo->uova, dito->dita, città invariant — resolved
|
||||
PER LEMMA, never guessed.
|
||||
adjective lemmas with real feminine + masc/fem plural (italiano->italiana/
|
||||
italiani/italiane, felice->felici invariant).
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV input):
|
||||
verbs : rule generator for regular -are/-ere/-ire (with -care/-gare h-insertion
|
||||
and -ciare/-giare/-iare i-drop spelling rules)
|
||||
nouns : gender heuristic (endings) + rule pluralization (ambiguous -co/-go FLAGGED)
|
||||
adjs : -o/-a/-e gender rule + rule pluralization
|
||||
|
||||
Confidence flag on every form:
|
||||
"lexicon" from UniMorph / kaikki-irregular / kaikki noun-adj (trust: high)
|
||||
"rule" deterministic rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma / ambiguous (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_it.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
participle(lemma, gender="m", number="singular") -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "ita.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "it_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_it.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "it_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set that must ALL be present ────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "passato_remoto"): {"IND", "PST", "PFV"},
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"},
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ─────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {} # (lemma, "mood|tense|person|number") -> form
|
||||
part = {} # lemma -> masc-sg past participle
|
||||
ger = {} # lemma -> gerundio
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB": # gerundio (converb, present)
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
# exact-set discipline: PST;PFV must not match PST;IPFV, etc.
|
||||
if not req <= f:
|
||||
continue
|
||||
# guard IND;PST ambiguity: require the specific aspect feature
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "passato_remoto" and "IPFV" in f:
|
||||
continue
|
||||
# COND must not also be a subjunctive/imperative slot
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "apocopic", "obsolete",
|
||||
"construed", "collective"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {} # lemma -> {"g","SG","PL"}
|
||||
adjs = {} # lemma -> {("m","SG"),("f","SG"),("m","PL"),("f","PL")}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm != "#":
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t: # invariant-gender adj (felice -> felici)
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback ─────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("are"):
|
||||
return "are"
|
||||
if lemma.endswith("ere"):
|
||||
return "ere"
|
||||
if lemma.endswith("ire"):
|
||||
return "ire"
|
||||
return None
|
||||
|
||||
|
||||
# endings [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG = {
|
||||
("ind", "present", "are"): ["o", "i", "a", "iamo", "ate", "ano"],
|
||||
("ind", "present", "ere"): ["o", "i", "e", "iamo", "ete", "ono"],
|
||||
("ind", "present", "ire"): ["o", "i", "e", "iamo", "ite", "ono"],
|
||||
("ind", "imperfect", "are"): ["avo", "avi", "ava", "avamo", "avate", "avano"],
|
||||
("ind", "imperfect", "ere"): ["evo", "evi", "eva", "evamo", "evate", "evano"],
|
||||
("ind", "imperfect", "ire"): ["ivo", "ivi", "iva", "ivamo", "ivate", "ivano"],
|
||||
("ind", "passato_remoto", "are"): ["ai", "asti", "ò", "ammo", "aste", "arono"],
|
||||
("ind", "passato_remoto", "ere"): ["ei", "esti", "é", "emmo", "este", "erono"],
|
||||
("ind", "passato_remoto", "ire"): ["ii", "isti", "ì", "immo", "iste", "irono"],
|
||||
("sbjv", "present", "are"): ["i", "i", "i", "iamo", "iate", "ino"],
|
||||
("sbjv", "present", "ere"): ["a", "a", "a", "iamo", "iate", "ano"],
|
||||
("sbjv", "present", "ire"): ["a", "a", "a", "iamo", "iate", "ano"],
|
||||
("sbjv", "imperfect", "are"): ["assi", "assi", "asse", "assimo", "aste", "assero"],
|
||||
("sbjv", "imperfect", "ere"): ["essi", "essi", "esse", "essimo", "este", "essero"],
|
||||
("sbjv", "imperfect", "ire"): ["issi", "issi", "isse", "issimo", "iste", "issero"],
|
||||
# imperative: 2sg,3sg(Lei),1pl,2pl,3pl (1sg has none)
|
||||
("imp", "affirmative", "are"): [None, "a", "i", "iamo", "ate", "ino"],
|
||||
("imp", "affirmative", "ere"): [None, "i", "a", "iamo", "ete", "ano"],
|
||||
("imp", "affirmative", "ire"): [None, "i", "a", "iamo", "ite", "ano"],
|
||||
}
|
||||
# future / conditional attach to a stem = infinitive minus final -e, with
|
||||
# -are -> -er (parlare->parler-), -ere/-ire keep (credere->creder-, dormir-)
|
||||
_FUT = ["ò", "ai", "à", "emo", "ete", "anno"]
|
||||
_COND = ["ei", "esti", "ebbe", "emmo", "este", "ebbero"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _fut_stem(lemma, vc):
|
||||
body = lemma[:-3] # drop are/ere/ire
|
||||
if vc == "are":
|
||||
return body + "er"
|
||||
return body + vc[0] + "r" # ere->er? no: keep vowel: creder-, dormir-
|
||||
# NOTE corrected below
|
||||
|
||||
|
||||
def _apply_are_spelling(stem, ending):
|
||||
"""-care/-gare insert h before front endings; -ciare/-giare/-sciare/-iare drop i."""
|
||||
front = ending[:1] in ("i", "e")
|
||||
if stem.endswith(("c", "g")) and front:
|
||||
return stem + "h" + ending
|
||||
if stem.endswith(("ci", "gi", "sci")) and ending[:1] == "i":
|
||||
return stem[:-1] + ending # mangi+iamo -> mangiamo
|
||||
if stem.endswith("i") and ending[:1] == "i":
|
||||
return stem[:-1] + ending # studi+iamo -> studiamo
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
body = lemma[:-3]
|
||||
i = _slot_idx(person, number)
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
stem = body + "er" if vc == "are" else body + vc[0] + "r"
|
||||
# ere: creder-, ire: dormir- -> body + 'e'/'i' + 'r'
|
||||
if vc == "ere":
|
||||
stem = body + "er"
|
||||
elif vc == "ire":
|
||||
stem = body + "ir"
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
# spelling: -care/-gare -> cherò/gherò ; -ciare/-giare -> cerò/gerò
|
||||
if vc == "are":
|
||||
if body.endswith(("c", "g")):
|
||||
stem = body + "her"
|
||||
elif body.endswith(("ci", "gi", "sci")):
|
||||
stem = body[:-1] + "er"
|
||||
elif body.endswith("i"):
|
||||
stem = body[:-1] + "er"
|
||||
return stem + end
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if not table:
|
||||
return None
|
||||
end = table[i]
|
||||
if end is None:
|
||||
return None
|
||||
if vc == "are":
|
||||
return _apply_are_spelling(body, end)
|
||||
# -ere/-ire: guard against double-i (dormi+iamo -> dormiamo)
|
||||
if body.endswith("i") and end[:1] == "i":
|
||||
return body[:-1] + end
|
||||
return body + end
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number}"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ──────────────────────────────────────────────────
|
||||
def _participle_msg(lemma):
|
||||
"""Return (masc-sg participle, source) or (None, None)."""
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
return None, None
|
||||
|
||||
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
"""Past participle with gender/number agreement (for essere-perfect & passives).
|
||||
UniMorph/irregular give masc-sg; fem/plural derived by final-vowel swap
|
||||
(-o -> -a/-i/-e), valid for regular -ato/-uto/-ito AND irregulars
|
||||
(preso->presa/presi/prese, aperto->aperta/aperti/aperte, morto->morta/...)."""
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
msg, src = _participle_msg(lemma)
|
||||
conf = "lexicon"
|
||||
if msg is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "are":
|
||||
msg = lemma[:-3] + "ato"
|
||||
elif vc == "ere":
|
||||
msg = lemma[:-3] + "uto"
|
||||
elif vc == "ire":
|
||||
msg = lemma[:-3] + "ito"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
# agreement: only -o participles inflect for gender+number
|
||||
if msg.endswith("o"):
|
||||
stem = msg[:-1]
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "i", "f|PL": "e"}[f"{g}|{num}"]
|
||||
return stem + suf, conf
|
||||
return msg, conf # non -o participle: leave as-is (rare)
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "are":
|
||||
return lemma[:-3] + "ando", "rule"
|
||||
if vc in ("ere", "ire"):
|
||||
return lemma[:-3] + "endo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("zione", "sione", "gione", "tà", "tù", "trice", "aggine", "udine",
|
||||
"igine", "ie", "essa", "izia", "ezza")
|
||||
_MASC_SUF = ("ore", "ame", "iere", "ale", "ile")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
for suf in _MASC_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "m"
|
||||
if noun.endswith("o"):
|
||||
return "m"
|
||||
if noun.endswith("a"):
|
||||
return "f"
|
||||
if noun.endswith("à") or noun.endswith("ù"):
|
||||
return "f"
|
||||
return "m" # -e and consonant-final loanwords default masculine
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic Italian pluralization. Returns (form, ok); ok=False FLAGS an
|
||||
ambiguous case the lexicon would normally resolve (-co/-go palatalization)."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
# invariant: accented final vowel, consonant-final, monosyllable, -i final
|
||||
if noun[-1:] in ("à", "è", "é", "ì", "í", "ò", "ó", "ù", "ú"):
|
||||
return noun, True
|
||||
if noun[-1:] not in ("a", "e", "o", "i", "u"):
|
||||
return noun, True # consonant-final loanword: invariant
|
||||
if noun.endswith("i"):
|
||||
return noun, True # e.g. crisi, analisi: invariant
|
||||
if noun.endswith("io"):
|
||||
return noun[:-2] + "i", True # figlio->figli (unstressed i)
|
||||
if noun.endswith("cia") or noun.endswith("gia"):
|
||||
# vowel before cia/gia -> -cie/-gie ; consonant -> -ce/-ge (approx)
|
||||
return noun[:-2] + "e", True # arancia->arance (majority)
|
||||
if noun.endswith("ca"):
|
||||
return noun[:-2] + "che", True # amica->amiche
|
||||
if noun.endswith("ga"):
|
||||
return noun[:-2] + "ghe", True
|
||||
if noun.endswith("co"):
|
||||
return noun[:-2] + "chi", False # AMBIGUOUS (amico->amici) -> flag
|
||||
if noun.endswith("go"):
|
||||
return noun[:-2] + "ghi", False # AMBIGUOUS (psicologo->psicologi)
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "e", True # casa->case (m -a: -i, but rare)
|
||||
if noun.endswith("o"):
|
||||
return noun[:-1] + "i", True # libro->libri
|
||||
if noun.endswith("e"):
|
||||
return noun[:-1] + "i", True # cane->cani, chiave->chiavi
|
||||
return noun, True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# adjectives whose kaikki entries are unreliable (messy inflection templates):
|
||||
# supply audited regular agreement forms (prenominal apocope handled in realizer).
|
||||
_ADJ_FIX = {
|
||||
"bello": {("m", "SG"): "bello", ("f", "SG"): "bella",
|
||||
("m", "PL"): "belli", ("f", "PL"): "belle"},
|
||||
"quello": {("m", "SG"): "quello", ("f", "SG"): "quella",
|
||||
("m", "PL"): "quelli", ("f", "PL"): "quelle"},
|
||||
}
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
fix = _ADJ_FIX.get(lemma)
|
||||
if fix and (g, num) in fix:
|
||||
return fix[(g, num)], "lexicon"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
a = lemma
|
||||
if a.endswith("o"): # -o/-a/-i/-e class
|
||||
base = a[:-1]
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "i", "f|PL": "e"}[f"{g}|{num}"]
|
||||
return base + suf, "rule"
|
||||
if a.endswith("e"): # felice-class: SG invariant, PL -i
|
||||
if num == "PL":
|
||||
return a[:-1] + "i", "rule"
|
||||
return a, "rule"
|
||||
if num == "PL":
|
||||
p, ok = _rule_plural(a, g)
|
||||
return p, ("rule" if ok else "fallback")
|
||||
return a, "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Italian (github.com/unimorph/ita) + kaikki.org "
|
||||
"irregulars (de-stressed)",
|
||||
"noun_adj_source": "kaikki.org Italian (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREGV),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("parlare", "ind", "present", "first", "singular", "parlo"),
|
||||
("essere", "ind", "present", "third", "singular", "è"),
|
||||
("avere", "ind", "present", "first", "singular", "ho"),
|
||||
("mangiare", "ind", "present", "second", "singular", "mangi"),
|
||||
("finire", "ind", "present", "first", "singular", "finisco"),
|
||||
("andare", "ind", "present", "third", "plural", "vanno"),
|
||||
("fare", "ind", "future", "first", "singular", "farò"),
|
||||
("potere", "sbjv", "present", "third", "singular", "possa"),
|
||||
("prendere", "ind", "passato_remoto", "first", "singular", "presi"),
|
||||
("cercare", "ind", "present", "second", "singular", "cerchi"),
|
||||
("dormire", "ind", "present", "third", "plural", "dormono"),
|
||||
("credere", "ind", "future", "first", "singular", "crederò"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:9} {mood}/{tense:14} {per[:3]}.{num[:2]} -> {got:12} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: casa=", noun_gender("casa"), "problema=", noun_gender("problema"),
|
||||
"mano=", noun_gender("mano"), "città=", noun_gender("città"),
|
||||
"cane=", noun_gender("cane"))
|
||||
print(" plural: uomo->", inflect_noun("uomo", "plural"),
|
||||
"| uovo->", inflect_noun("uovo", "plural"),
|
||||
"| città->", inflect_noun("città", "plural"),
|
||||
"| amico->", inflect_noun("amico", "plural"),
|
||||
"| casa->", inflect_noun("casa", "plural"))
|
||||
print(" adj: italiano/f/pl->", inflect_adj("italiano", "f", "plural"),
|
||||
"| felice/m/pl->", inflect_adj("felice", "m", "plural"),
|
||||
"| bello/f/sg->", inflect_adj("bello", "f", "singular"))
|
||||
print(" part: aprire/f/sg->", participle("aprire", "f", "singular"),
|
||||
"| prendere/m/pl->", participle("prendere", "m", "plural"),
|
||||
"| andare/f/sg->", participle("andare", "f", "singular"))
|
||||
print(" ger: fare->", gerund("fare"), "| parlare->", gerund("parlare"))
|
||||
@@ -1,666 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_lat_full.py — production-grade Latin morphological generator.
|
||||
|
||||
Latin is the FLAGSHIP dead-language realizer. It rides the *architecture* of the
|
||||
Romance/Italic engine (the same Realization / spec-driven design and the UniMorph
|
||||
loader pattern from morphology_it_full.py) but with the CASE SYSTEM RESTORED —
|
||||
the feature Romance lost. Latin therefore exercises machinery the modern Romance
|
||||
siblings never needed: 5 declensions x 6 cases x 2 numbers x 3 genders, plus a
|
||||
4-conjugation verb system with tense/mood/voice.
|
||||
|
||||
DATA (real, attested — no fabrication):
|
||||
|
||||
NOUNS + ADJECTIVES — UniMorph Latin (github.com/unimorph/lat, CC-BY-SA 3.0)
|
||||
163,182 N forms across ~thousands of lemmas, each with the full case paradigm
|
||||
N;NOM/GEN/DAT/ACC/ABL/VOC;SG/PL (real inflected forms, WITH macrons:
|
||||
puella->puellam, rēx->rēgis, corpus->corporis).
|
||||
244,197 ADJ forms with case x GENDER x number, incl. UniMorph's combined
|
||||
tags (GEN+DAT, MASC+FEM, MASC+FEM+NEUT) which are split on load.
|
||||
462,668 V.PTCP forms (participles) also carry case/gender/number.
|
||||
UniMorph N tags DO NOT encode inherent gender, so noun gender is inferred
|
||||
from the declension (nom-sg + gen-sg endings) with a curated exceptions
|
||||
map — the standard, attestable rule (1st decl -a/-ae = fem, 2nd -us/-i =
|
||||
masc, -um = neut, ...).
|
||||
|
||||
VERBS — RULE ENGINE (honest gap: UniMorph Latin's verb list is a 947-lemma
|
||||
sample of rare/prefixed verbs that MISSES every core textbook verb — amō,
|
||||
videō, sum, regō, ... are all absent). Latin conjugation is, however, highly
|
||||
regular, so verbs are generated by a deterministic 4-conjugation engine over
|
||||
curated principal parts (present / perfect / supine stems), sourced from
|
||||
standard references. Irregulars (sum, possum, eō, ferō, volō, nōlō, mālō)
|
||||
are curated full tables. Forms are flagged "rule" (not "lexicon") for honesty.
|
||||
|
||||
Confidence flag on every form (same contract as the Romance engine):
|
||||
"lexicon" from UniMorph (trust: high)
|
||||
"rule" deterministic morphology rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_lat.py):
|
||||
decline_noun(lemma, case, number) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"|"n"
|
||||
decline_adj(lemma, case, gender, number) -> (form, conf)
|
||||
conjugate(lemma, tense, mood, voice, person, number) -> (form, conf)
|
||||
participle(lemma, kind, case, gender, number) -> (form, conf) # kind: prs|pfv|fut
|
||||
infinitive(lemma, tense="present", voice="active") -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "lat.unimorph")
|
||||
_CACHE = os.path.join(_HERE, "data", "lat_morph_cache.pkl")
|
||||
|
||||
_CASES = ("NOM", "GEN", "DAT", "ACC", "ABL", "VOC")
|
||||
_CASE_MAP = {"nom": "NOM", "gen": "GEN", "dat": "DAT", "acc": "ACC",
|
||||
"abl": "ABL", "voc": "VOC"}
|
||||
_NUM = {"singular": "SG", "plural": "PL"}
|
||||
_GEN = {"m": "MASC", "f": "FEM", "n": "NEUT"}
|
||||
|
||||
|
||||
# ── UniMorph loader: noun + adjective + participle case paradigms ────────────────
|
||||
def _build_cache():
|
||||
nouns = {} # lemma -> {(CASE, NUM): form}
|
||||
adjs = {} # lemma -> {(CASE, GEN, NUM): form}
|
||||
ptcps = {} # lemma -> {(CASE, GEN, NUM): form} (from V.PTCP; keyed loosely)
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
feats = tag.split(";")
|
||||
head = feats[0]
|
||||
fs = set(feats)
|
||||
case = next((c for c in _CASES if c in fs), None)
|
||||
# handle combined case tags like GEN+DAT
|
||||
if case is None:
|
||||
for f in feats:
|
||||
if "+" in f and any(c in f.split("+") for c in _CASES):
|
||||
case = [c for c in _CASES if c in f.split("+")]
|
||||
break
|
||||
num = "SG" if "SG" in fs else ("PL" if "PL" in fs else None)
|
||||
if case is None or num is None:
|
||||
continue
|
||||
cases = case if isinstance(case, list) else [case]
|
||||
|
||||
if head == "N":
|
||||
d = nouns.setdefault(lemma, {})
|
||||
for c in cases:
|
||||
d.setdefault((c, num), form)
|
||||
elif head == "ADJ":
|
||||
# gender may be combined: MASC+FEM+NEUT, MASC+FEM
|
||||
genders = []
|
||||
for g in ("MASC", "FEM", "NEUT"):
|
||||
if any(g == x or (g in x.split("+")) for x in feats):
|
||||
genders.append(g)
|
||||
if not genders:
|
||||
genders = ["MASC", "FEM", "NEUT"]
|
||||
d = adjs.setdefault(lemma, {})
|
||||
for c in cases:
|
||||
for g in genders:
|
||||
d.setdefault((c, g, num), form)
|
||||
data = {"nouns": nouns, "adjs": adjs, "ptcps": ptcps}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE) and os.path.exists(_UNIMORPH):
|
||||
if os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH):
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_NOUNS, _ADJS = _LEX["nouns"], _LEX["adjs"]
|
||||
|
||||
|
||||
# ── noun gender inference (declension-based, curated exceptions) ─────────────────
|
||||
# Real, attestable rule: gender follows declension + nominative shape, with the
|
||||
# standard closed set of exceptions.
|
||||
_GENDER_EXC = {
|
||||
# 1st-declension masculines (people/agents)
|
||||
"agricola": "m", "poēta": "m", "nauta": "m", "incola": "m", "scrība": "m",
|
||||
"auriga": "m", "pīrāta": "m", "athlēta": "m",
|
||||
# 2nd-declension neuters / feminines
|
||||
"vīrus": "n", "vulgus": "n", "pelagus": "n", "humus": "f",
|
||||
# common 3rd-declension whose gender the ending would mispredict
|
||||
"rēx": "m", "dux": "m", "mīles": "m", "pater": "m", "frāter": "m",
|
||||
"homō": "m", "leō": "m", "sōl": "m", "mōns": "m", "pōns": "m", "fōns": "m",
|
||||
"sanguis": "m", "ōrdō": "m", "sermō": "m", "amor": "m", "dolor": "m",
|
||||
"labor": "m", "timor": "m", "honor": "m", "color": "m", "pēs": "m",
|
||||
"dēns": "m", "flōs": "m", "mōs": "m", "mensis": "m", "orbis": "m",
|
||||
"piscis": "m", "ignis": "m", "collis": "m", "grex": "m", "prīnceps": "m",
|
||||
"māter": "f", "soror": "f", "uxor": "f", "mulier": "f", "virgō": "f",
|
||||
"urbs": "f", "arx": "f", "pāx": "f", "lēx": "f", "lūx": "f", "vōx": "f",
|
||||
"nox": "f", "nix": "f", "vīs": "f", "salūs": "f", "virtūs": "f",
|
||||
"aetās": "f", "cīvitās": "f", "lībertās": "f", "vēritās": "f", "voluptās": "f",
|
||||
"nātiō": "f", "ratiō": "f", "ōrātiō": "f", "legiō": "f", "regiō": "f",
|
||||
"mens": "f", "gens": "f", "ars": "f", "pars": "f", "mors": "f", "sors": "f",
|
||||
"nāvis": "f", "turris": "f", "avis": "f", "vallis": "f", "classis": "f",
|
||||
"corpus": "n", "tempus": "n", "opus": "n", "genus": "n", "onus": "n",
|
||||
"pectus": "n", "latus": "n", "vulnus": "n", "scelus": "n", "sīdus": "n",
|
||||
"caput": "n", "iter": "n", "flūmen": "n", "nōmen": "n", "carmen": "n",
|
||||
"agmen": "n", "certāmen": "n", "lūmen": "n", "ōmen": "n", "cōgnōmen": "n",
|
||||
"mare": "n", "animal": "n", "exemplar": "n", "rēte": "n",
|
||||
# 4th-declension exceptions
|
||||
"manus": "f", "domus": "f", "tribus": "f", "porticus": "f", "īdūs": "f",
|
||||
"cornū": "n", "genū": "n", "gelū": "n", "verū": "n",
|
||||
# 5th-declension
|
||||
"diēs": "m", "merīdiēs": "m",
|
||||
}
|
||||
|
||||
|
||||
def _infer_gender(lemma):
|
||||
if lemma in _GENDER_EXC:
|
||||
return _GENDER_EXC[lemma]
|
||||
d = _NOUNS.get(lemma)
|
||||
nom = d.get(("NOM", "SG")) if d else lemma
|
||||
gen = d.get(("GEN", "SG")) if d else None
|
||||
nom = nom or lemma
|
||||
# 5th declension: gen -eī / -ēī
|
||||
if gen and (gen.endswith("eī") or gen.endswith("ēī")):
|
||||
return "f"
|
||||
# 1st declension: nom -a, gen -ae
|
||||
if nom.endswith("a") and (not gen or gen.endswith("ae")):
|
||||
return "f"
|
||||
# 2nd declension neuter: nom -um
|
||||
if nom.endswith("um"):
|
||||
return "n"
|
||||
# 2nd declension masc: nom -us/-er/-ir, gen -ī
|
||||
if (nom.endswith("us") or nom.endswith("er") or nom.endswith("ir")) and \
|
||||
(not gen or gen.endswith("ī")):
|
||||
return "m"
|
||||
# 4th declension: gen -ūs
|
||||
if gen and gen.endswith("ūs"):
|
||||
return "n" if nom.endswith("ū") else "m"
|
||||
# 3rd declension neuters by common nom endings
|
||||
if nom.endswith(("men", "us", "ur", "al", "ar", "e", "ma")):
|
||||
# -us here is 3rd-decl neuter type (corpus) only if gen shows -oris/-eris
|
||||
if nom.endswith("us") and gen and (gen.endswith("oris") or gen.endswith("eris")
|
||||
or gen.endswith("uris")):
|
||||
return "n"
|
||||
if nom.endswith(("men", "al", "ar", "e")):
|
||||
return "n"
|
||||
# default 3rd-declension: masculine (most common)
|
||||
return "m"
|
||||
|
||||
|
||||
_GENDER_CACHE = {}
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip()
|
||||
if lemma not in _GENDER_CACHE:
|
||||
_GENDER_CACHE[lemma] = _infer_gender(lemma)
|
||||
return _GENDER_CACHE[lemma]
|
||||
|
||||
|
||||
# ── PUBLIC: noun declension ─────────────────────────────────────────────────────
|
||||
def decline_noun(lemma, case, number):
|
||||
lemma = lemma.strip()
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
N = _NUM.get(number, number)
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and (C, N) in d:
|
||||
return d[(C, N)], "lexicon"
|
||||
# abl sg often == the -e/-o form; try nom fallback
|
||||
if d:
|
||||
# try VOC==NOM, ACC neuter==NOM etc are already in data; last resort lemma
|
||||
return lemma, "fallback"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: adjective declension ────────────────────────────────────────────────
|
||||
def decline_adj(lemma, case, gender, number):
|
||||
lemma = lemma.strip()
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
G = _GEN.get(gender, gender.upper())
|
||||
N = _NUM.get(number, number)
|
||||
d = _ADJS.get(lemma)
|
||||
if d and (C, G, N) in d:
|
||||
return d[(C, G, N)], "lexicon"
|
||||
# try other gender (some adjs listed only under MASC+FEM etc handled at load)
|
||||
if d:
|
||||
for altG in ("MASC", "FEM", "NEUT"):
|
||||
if (C, altG, N) in d:
|
||||
return d[(C, altG, N)], "lexicon"
|
||||
return lemma, "fallback"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# VERB RULE ENGINE (4 conjugations + curated irregulars)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# Curated principal parts for common attested verbs:
|
||||
# lemma -> (conj, present_stem, perfect_stem, supine_stem)
|
||||
# conj in {1,2,3,"3io",4}. Stems carry macrons (matching UniMorph orthography).
|
||||
_VERBS = {
|
||||
"amō": (1, "am", "amāv", "amāt"),
|
||||
"laudō": (1, "laud", "laudāv", "laudāt"),
|
||||
"portō": (1, "port", "portāv", "portāt"),
|
||||
"vocō": (1, "voc", "vocāv", "vocāt"),
|
||||
"dō": (1, "d", "ded", "dat"),
|
||||
"spectō": (1, "spect", "spectāv", "spectāt"),
|
||||
"pugnō": (1, "pugn", "pugnāv", "pugnāt"),
|
||||
"labōrō": (1, "labōr", "labōrāv", "labōrāt"),
|
||||
"necō": (1, "nec", "necāv", "necāt"),
|
||||
"parō": (1, "par", "parāv", "parāt"),
|
||||
"cōgitō": (1, "cōgit", "cōgitāv", "cōgitāt"),
|
||||
"habitō": (1, "habit", "habitāv", "habitāt"),
|
||||
"nārrō": (1, "nārr", "nārrāv", "nārrāt"),
|
||||
"servō": (1, "serv", "servāv", "servāt"),
|
||||
"superō": (1, "super", "superāv", "superāt"),
|
||||
"oppugnō": (1, "oppugn", "oppugnāv", "oppugnāt"),
|
||||
"ambulō": (1, "ambul", "ambulāv", "ambulāt"),
|
||||
"clāmō": (1, "clām", "clāmāv", "clāmāt"),
|
||||
"vulnerō": (1, "vulner", "vulnerāv", "vulnerāt"),
|
||||
"aedificō": (1, "aedific", "aedificāv", "aedificāt"),
|
||||
"expugnō": (1, "expugn", "expugnāv", "expugnāt"),
|
||||
"dēfendō": (3, "dēfend", "dēfend", "dēfēns"),
|
||||
"petō": (3, "pet", "petīv", "petīt"),
|
||||
"occīdō": (3, "occīd", "occīd", "occīs"),
|
||||
"interficiō": ("3io", "interfic", "interfēc", "interfect"),
|
||||
"timeō": (2, "tim", "timu", None),
|
||||
"iaceō": (2, "iac", "iacu", None),
|
||||
"pāreō": (2, "pār", "pāru", "pārit"),
|
||||
"respondeō": (2, "respond", "respond", "respōns"),
|
||||
"vertō": (3, "vert", "vert", "vers"),
|
||||
"ostendō": (3, "ostend", "ostend", "ostent"),
|
||||
"cōnstituō": (3, "cōnstitu", "cōnstitu", "cōnstitūt"),
|
||||
"cōgnōscō": (3, "cōgnōsc", "cōgnōv", "cōgnit"),
|
||||
"crēdō": (3, "crēd", "crēdid", "crēdit"),
|
||||
"ēdūcō": (3, "ēdūc", "ēdūx", "ēduct"),
|
||||
"cōnservō": (1, "cōnserv", "cōnservāv", "cōnservāt"),
|
||||
"iuvō": (1, "iuv", "iūv", "iūt"),
|
||||
"dēbeō": (2, "dēb", "dēbu", "dēbit"),
|
||||
"moneō": (2, "mon", "monu", "monit"),
|
||||
"videō": (2, "vid", "vīd", "vīs"),
|
||||
"habeō": (2, "hab", "habu", "habit"),
|
||||
"teneō": (2, "ten", "tenu", "tent"),
|
||||
"timeō": (2, "tim", "timu", None),
|
||||
"terreō": (2, "terr", "terru", "territ"),
|
||||
"dēleō": (2, "dēl", "dēlēv", "dēlēt"),
|
||||
"iubeō": (2, "iub", "iuss", "iuss"),
|
||||
"maneō": (2, "man", "māns", "māns"),
|
||||
"moveō": (2, "mov", "mōv", "mōt"),
|
||||
"doceō": (2, "doc", "docu", "doct"),
|
||||
"sedeō": (2, "sed", "sēd", "sess"),
|
||||
"rīdeō": (2, "rīd", "rīs", "rīs"),
|
||||
"regō": (3, "reg", "rēx", "rēct"),
|
||||
"dūcō": (3, "dūc", "dūx", "duct"),
|
||||
"scrībō": (3, "scrīb", "scrīps", "scrīpt"),
|
||||
"mittō": (3, "mitt", "mīs", "miss"),
|
||||
"pōnō": (3, "pōn", "posu", "posit"),
|
||||
"agō": (3, "ag", "ēg", "āct"),
|
||||
"dīcō": (3, "dīc", "dīx", "dict"),
|
||||
"gerō": (3, "ger", "gess", "gest"),
|
||||
"vincō": (3, "vinc", "vīc", "vict"),
|
||||
"petō": (3, "pet", "petīv", "petīt"),
|
||||
"legō": (3, "leg", "lēg", "lēct"),
|
||||
"currō": (3, "curr", "cucurr", "curs"),
|
||||
"vīvō": (3, "vīv", "vīx", "vīct"),
|
||||
"quaerō": (3, "quaer", "quaesīv", "quaesīt"),
|
||||
"trahō": (3, "trah", "trāx", "tract"),
|
||||
"claudō": (3, "claud", "claus", "claus"),
|
||||
"cōgō": (3, "cōg", "coēg", "coāct"),
|
||||
"relinquō": (3, "relinqu", "relīqu", "relict"),
|
||||
"capiō": ("3io", "cap", "cēp", "capt"),
|
||||
"faciō": ("3io", "fac", "fēc", "fact"),
|
||||
"iaciō": ("3io", "iac", "iēc", "iact"),
|
||||
"rapiō": ("3io", "rap", "rapu", "rapt"),
|
||||
"fugiō": ("3io", "fug", "fūg", "fugit"),
|
||||
"cupiō": ("3io", "cup", "cupīv", "cupīt"),
|
||||
"accipiō": ("3io", "accip", "accēp", "accept"),
|
||||
"audiō": (4, "aud", "audīv", "audīt"),
|
||||
"veniō": (4, "ven", "vēn", "vent"),
|
||||
"sciō": (4, "sc", "scīv", "scīt"),
|
||||
"sentiō": (4, "sent", "sēns", "sēns"),
|
||||
"mūniō": (4, "mūn", "mūnīv", "mūnīt"),
|
||||
"dormiō": (4, "dorm", "dormīv", "dormīt"),
|
||||
"aperiō": (4, "aper", "aperu", "apert"),
|
||||
"inveniō": (4, "inven", "invēn", "invent"),
|
||||
}
|
||||
|
||||
# ── Present-system paradigms: full ending tables per conjugation, attached to the
|
||||
# bare present stem (pstem). Hardcoded from the standard grammar with correct
|
||||
# macrons/vowel-lengths — deterministic and independently verifiable. Keys:
|
||||
# (tense, mood, voice) -> {conj: [1sg,2sg,3sg,1pl,2pl,3pl]}
|
||||
_PARADIGM = {
|
||||
("present", "ind", "active"): {
|
||||
1: ["ō", "ās", "at", "āmus", "ātis", "ant"],
|
||||
2: ["eō", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
3: ["ō", "is", "it", "imus", "itis", "unt"],
|
||||
"3io": ["iō", "is", "it", "imus", "itis", "iunt"],
|
||||
4: ["iō", "īs", "it", "īmus", "ītis", "iunt"],
|
||||
},
|
||||
("present", "ind", "passive"): {
|
||||
1: ["or", "āris", "ātur", "āmur", "āminī", "antur"],
|
||||
2: ["eor", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
3: ["or", "eris", "itur", "imur", "iminī", "untur"],
|
||||
"3io": ["ior", "eris", "itur", "imur", "iminī", "iuntur"],
|
||||
4: ["ior", "īris", "ītur", "īmur", "īminī", "iuntur"],
|
||||
},
|
||||
("imperfect", "ind", "active"): {
|
||||
1: ["ābam", "ābās", "ābat", "ābāmus", "ābātis", "ābant"],
|
||||
2: ["ēbam", "ēbās", "ēbat", "ēbāmus", "ēbātis", "ēbant"],
|
||||
3: ["ēbam", "ēbās", "ēbat", "ēbāmus", "ēbātis", "ēbant"],
|
||||
"3io": ["iēbam", "iēbās", "iēbat", "iēbāmus", "iēbātis", "iēbant"],
|
||||
4: ["iēbam", "iēbās", "iēbat", "iēbāmus", "iēbātis", "iēbant"],
|
||||
},
|
||||
("imperfect", "ind", "passive"): {
|
||||
1: ["ābar", "ābāris", "ābātur", "ābāmur", "ābāminī", "ābantur"],
|
||||
2: ["ēbar", "ēbāris", "ēbātur", "ēbāmur", "ēbāminī", "ēbantur"],
|
||||
3: ["ēbar", "ēbāris", "ēbātur", "ēbāmur", "ēbāminī", "ēbantur"],
|
||||
"3io": ["iēbar", "iēbāris", "iēbātur", "iēbāmur", "iēbāminī", "iēbantur"],
|
||||
4: ["iēbar", "iēbāris", "iēbātur", "iēbāmur", "iēbāminī", "iēbantur"],
|
||||
},
|
||||
("future", "ind", "active"): {
|
||||
1: ["ābō", "ābis", "ābit", "ābimus", "ābitis", "ābunt"],
|
||||
2: ["ēbō", "ēbis", "ēbit", "ēbimus", "ēbitis", "ēbunt"],
|
||||
3: ["am", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
"3io": ["iam", "iēs", "iet", "iēmus", "iētis", "ient"],
|
||||
4: ["iam", "iēs", "iet", "iēmus", "iētis", "ient"],
|
||||
},
|
||||
("future", "ind", "passive"): {
|
||||
1: ["ābor", "āberis", "ābitur", "ābimur", "ābiminī", "ābuntur"],
|
||||
2: ["ēbor", "ēberis", "ēbitur", "ēbimur", "ēbiminī", "ēbuntur"],
|
||||
3: ["ar", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
"3io": ["iar", "iēris", "iētur", "iēmur", "iēminī", "ientur"],
|
||||
4: ["iar", "iēris", "iētur", "iēmur", "iēminī", "ientur"],
|
||||
},
|
||||
("present", "sbjv", "active"): {
|
||||
1: ["em", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
2: ["eam", "eās", "eat", "eāmus", "eātis", "eant"],
|
||||
3: ["am", "ās", "at", "āmus", "ātis", "ant"],
|
||||
"3io": ["iam", "iās", "iat", "iāmus", "iātis", "iant"],
|
||||
4: ["iam", "iās", "iat", "iāmus", "iātis", "iant"],
|
||||
},
|
||||
("present", "sbjv", "passive"): {
|
||||
1: ["er", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
2: ["ear", "eāris", "eātur", "eāmur", "eāminī", "eantur"],
|
||||
3: ["ar", "āris", "ātur", "āmur", "āminī", "antur"],
|
||||
"3io": ["iar", "iāris", "iātur", "iāmur", "iāminī", "iantur"],
|
||||
4: ["iar", "iāris", "iātur", "iāmur", "iāminī", "iantur"],
|
||||
},
|
||||
("imperfect", "sbjv", "active"): {
|
||||
1: ["ārem", "ārēs", "āret", "ārēmus", "ārētis", "ārent"],
|
||||
2: ["ērem", "ērēs", "ēret", "ērēmus", "ērētis", "ērent"],
|
||||
3: ["erem", "erēs", "eret", "erēmus", "erētis", "erent"],
|
||||
"3io": ["erem", "erēs", "eret", "erēmus", "erētis", "erent"],
|
||||
4: ["īrem", "īrēs", "īret", "īrēmus", "īrētis", "īrent"],
|
||||
},
|
||||
("imperfect", "sbjv", "passive"): {
|
||||
1: ["ārer", "ārēris", "ārētur", "ārēmur", "ārēminī", "ārentur"],
|
||||
2: ["ērer", "ērēris", "ērētur", "ērēmur", "ērēminī", "ērentur"],
|
||||
3: ["erer", "erēris", "erētur", "erēmur", "erēminī", "erentur"],
|
||||
"3io": ["erer", "erēris", "erētur", "erēmur", "erēminī", "erentur"],
|
||||
4: ["īrer", "īrēris", "īrētur", "īrēmur", "īrēminī", "īrentur"],
|
||||
},
|
||||
}
|
||||
# perfect-active endings (added to perfect stem) — same for all conjugations
|
||||
_PERF_ACT = {
|
||||
("perfect", "ind"): ["ī", "istī", "it", "imus", "istis", "ērunt"],
|
||||
("pluperfect", "ind"): ["eram", "erās", "erat", "erāmus", "erātis", "erant"],
|
||||
("futureperfect", "ind"): ["erō", "eris", "erit", "erimus", "eritis", "erint"],
|
||||
("perfect", "sbjv"): ["erim", "erīs", "erit", "erīmus", "erītis", "erint"],
|
||||
("pluperfect", "sbjv"):["issem", "issēs", "isset", "issēmus", "issētis", "issent"],
|
||||
}
|
||||
|
||||
|
||||
def _idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _present_system(conj, pstem, tense, mood, voice, person, number):
|
||||
"""Generate a present-system form (present/imperfect/future ind & subj)."""
|
||||
table = _PARADIGM.get((tense, mood, voice))
|
||||
if not table or conj not in table:
|
||||
return None
|
||||
return pstem + table[conj][_idx(person, number)]
|
||||
|
||||
|
||||
def _active_infinitive_stem(conj, pstem):
|
||||
return {1: pstem + "ā", 2: pstem + "ē", 3: pstem + "e",
|
||||
"3io": pstem + "e", 4: pstem + "ī"}[conj]
|
||||
|
||||
|
||||
_IRREG = {
|
||||
"sum": {
|
||||
("present", "ind", "active"): ["sum", "es", "est", "sumus", "estis", "sunt"],
|
||||
("imperfect", "ind", "active"): ["eram", "erās", "erat", "erāmus", "erātis", "erant"],
|
||||
("future", "ind", "active"): ["erō", "eris", "erit", "erimus", "eritis", "erunt"],
|
||||
("perfect", "ind", "active"): ["fuī", "fuistī", "fuit", "fuimus", "fuistis", "fuērunt"],
|
||||
("pluperfect", "ind", "active"): ["fueram", "fuerās", "fuerat", "fuerāmus", "fuerātis", "fuerant"],
|
||||
("present", "sbjv", "active"): ["sim", "sīs", "sit", "sīmus", "sītis", "sint"],
|
||||
("imperfect", "sbjv", "active"): ["essem", "essēs", "esset", "essēmus", "essētis", "essent"],
|
||||
},
|
||||
"possum": {
|
||||
("present", "ind", "active"): ["possum", "potes", "potest", "possumus", "potestis", "possunt"],
|
||||
("imperfect", "ind", "active"): ["poteram", "poterās", "poterat", "poterāmus", "poterātis", "poterant"],
|
||||
("future", "ind", "active"): ["poterō", "poteris", "poterit", "poterimus", "poteritis", "poterunt"],
|
||||
("perfect", "ind", "active"): ["potuī", "potuistī", "potuit", "potuimus", "potuistis", "potuērunt"],
|
||||
("present", "sbjv", "active"): ["possim", "possīs", "possit", "possīmus", "possītis", "possint"],
|
||||
},
|
||||
"eō": {
|
||||
("present", "ind", "active"): ["eō", "īs", "it", "īmus", "ītis", "eunt"],
|
||||
("imperfect", "ind", "active"): ["ībam", "ībās", "ībat", "ībāmus", "ībātis", "ībant"],
|
||||
("future", "ind", "active"): ["ībō", "ībis", "ībit", "ībimus", "ībitis", "ībunt"],
|
||||
("perfect", "ind", "active"): ["iī", "īstī", "iit", "iimus", "īstis", "iērunt"],
|
||||
("present", "sbjv", "active"): ["eam", "eās", "eat", "eāmus", "eātis", "eant"],
|
||||
},
|
||||
"volō": {
|
||||
("present", "ind", "active"): ["volō", "vīs", "vult", "volumus", "vultis", "volunt"],
|
||||
("imperfect", "ind", "active"): ["volēbam", "volēbās", "volēbat", "volēbāmus", "volēbātis", "volēbant"],
|
||||
("future", "ind", "active"): ["volam", "volēs", "volet", "volēmus", "volētis", "volent"],
|
||||
("perfect", "ind", "active"): ["voluī", "voluistī", "voluit", "voluimus", "voluistis", "voluērunt"],
|
||||
("present", "sbjv", "active"): ["velim", "velīs", "velit", "velīmus", "velītis", "velint"],
|
||||
},
|
||||
"nōlō": {
|
||||
("present", "ind", "active"): ["nōlō", "nōn vīs", "nōn vult", "nōlumus", "nōn vultis", "nōlunt"],
|
||||
("present", "sbjv", "active"): ["nōlim", "nōlīs", "nōlit", "nōlīmus", "nōlītis", "nōlint"],
|
||||
},
|
||||
"ferō": {
|
||||
("present", "ind", "active"): ["ferō", "fers", "fert", "ferimus", "fertis", "ferunt"],
|
||||
("imperfect", "ind", "active"): ["ferēbam", "ferēbās", "ferēbat", "ferēbāmus", "ferēbātis", "ferēbant"],
|
||||
("future", "ind", "active"): ["feram", "ferēs", "feret", "ferēmus", "ferētis", "ferent"],
|
||||
("perfect", "ind", "active"): ["tulī", "tulistī", "tulit", "tulimus", "tulistis", "tulērunt"],
|
||||
("present", "sbjv", "active"): ["feram", "ferās", "ferat", "ferāmus", "ferātis", "ferant"],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def conjugate(lemma, tense, mood, voice="active", person="third", number="singular"):
|
||||
"""Return (surface, confidence). Perfect-passive forms are periphrastic and
|
||||
handled in the realizer (sum + PPP); this returns synthetic forms only."""
|
||||
lemma = lemma.strip()
|
||||
i = _idx(person, number)
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir:
|
||||
tbl = ir.get((tense, mood, voice)) or ir.get((tense, mood, "active"))
|
||||
if tbl and tbl[i]:
|
||||
return tbl[i], "rule"
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
v = _infer_principal_parts(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
# imperative (present active) 2sg / 2pl
|
||||
if mood == "imp":
|
||||
return _imperative(conj, pstem, person, number), "rule"
|
||||
# perfect-system active
|
||||
if tense in ("perfect", "pluperfect", "futureperfect") and voice == "active":
|
||||
if not perfstem:
|
||||
return lemma, "fallback"
|
||||
end = _PERF_ACT.get((tense, mood))
|
||||
if end:
|
||||
return perfstem + end[i], "rule"
|
||||
# present-system (active + passive)
|
||||
if tense in ("present", "imperfect", "future"):
|
||||
form = _present_system(conj, pstem, tense, mood, voice, person, number)
|
||||
if form:
|
||||
return form, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def _imperative(conj, pstem, person, number):
|
||||
if number == "singular":
|
||||
return {1: pstem + "ā", 2: pstem + "ē", 3: pstem + "e",
|
||||
"3io": pstem + "e", 4: pstem + "ī"}[conj]
|
||||
return {1: pstem + "āte", 2: pstem + "ēte", 3: pstem + "ite",
|
||||
"3io": pstem + "ite", 4: pstem + "īte"}[conj]
|
||||
|
||||
|
||||
def _infer_principal_parts(lemma):
|
||||
"""OOV fallback: infer conjugation + stems from the 1sg-present citation form.
|
||||
Perfect/supine stems are guessed regularly (often wrong for 3rd conj) and the
|
||||
resulting forms are still returned as 'rule' but the realizer down-weights."""
|
||||
if lemma.endswith("ō"):
|
||||
base = lemma[:-1]
|
||||
# can't distinguish conj from 1sg alone reliably; default by ending vowel
|
||||
if base.endswith("i"):
|
||||
return ("3io", base[:-1], base[:-1] + "īv", base[:-1] + "īt")
|
||||
return (3, base, base + "s", base + "t")
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: participles ─────────────────────────────────────────────────────────
|
||||
def participle(lemma, kind, case="nom", gender="m", number="singular"):
|
||||
"""kind: 'prs' (present active, -ns/-ntis), 'pfv' (perfect passive, -tus),
|
||||
'fut' (future active, -tūrus). Declined as an adjective via rule endings.
|
||||
Returns (form, conf)."""
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
if kind == "pfv":
|
||||
if not supstem:
|
||||
return lemma, "fallback"
|
||||
base = supstem[:-1] if supstem.endswith("t") or supstem.endswith("s") else supstem
|
||||
stem = supstem # supine stem already ends in t/s: amāt- -> amātus
|
||||
return _decline_us_a_um(stem, case, gender, number), "rule"
|
||||
if kind == "fut":
|
||||
if not supstem:
|
||||
return lemma, "fallback"
|
||||
return _decline_us_a_um(supstem + "ūr", case, gender, number), "rule"
|
||||
if kind == "prs":
|
||||
# present active participle: stem + ns (nom), stem + nt- (oblique), 3rd-decl
|
||||
pv = {1: "ā", 2: "ē", 3: "ē", "3io": "iē", 4: "iē"}[conj]
|
||||
ntstem = pstem + pv + "nt"
|
||||
return _decline_pres_ptcp(pstem + pv, case, gender, number), "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def _decline_us_a_um(stem, case, gender, number):
|
||||
"""Decline a -us/-a/-um adjective/participle stem (2-1-2 declension)."""
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
end = {
|
||||
("NOM", "m", "singular"): "us", ("NOM", "f", "singular"): "a", ("NOM", "n", "singular"): "um",
|
||||
("GEN", "m", "singular"): "ī", ("GEN", "f", "singular"): "ae", ("GEN", "n", "singular"): "ī",
|
||||
("DAT", "m", "singular"): "ō", ("DAT", "f", "singular"): "ae", ("DAT", "n", "singular"): "ō",
|
||||
("ACC", "m", "singular"): "um", ("ACC", "f", "singular"): "am", ("ACC", "n", "singular"): "um",
|
||||
("ABL", "m", "singular"): "ō", ("ABL", "f", "singular"): "ā", ("ABL", "n", "singular"): "ō",
|
||||
("VOC", "m", "singular"): "e", ("VOC", "f", "singular"): "a", ("VOC", "n", "singular"): "um",
|
||||
("NOM", "m", "plural"): "ī", ("NOM", "f", "plural"): "ae", ("NOM", "n", "plural"): "a",
|
||||
("GEN", "m", "plural"): "ōrum", ("GEN", "f", "plural"): "ārum", ("GEN", "n", "plural"): "ōrum",
|
||||
("DAT", "m", "plural"): "īs", ("DAT", "f", "plural"): "īs", ("DAT", "n", "plural"): "īs",
|
||||
("ACC", "m", "plural"): "ōs", ("ACC", "f", "plural"): "ās", ("ACC", "n", "plural"): "a",
|
||||
("ABL", "m", "plural"): "īs", ("ABL", "f", "plural"): "īs", ("ABL", "n", "plural"): "īs",
|
||||
("VOC", "m", "plural"): "ī", ("VOC", "f", "plural"): "ae", ("VOC", "n", "plural"): "a",
|
||||
}.get((C, gender, number), "us")
|
||||
return stem + end
|
||||
|
||||
|
||||
def _decline_pres_ptcp(stem, case, gender, number):
|
||||
"""Present active participle (amāns, amantis) — 3rd-declension, stem+ns/nt."""
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
if C == "NOM" and number == "singular":
|
||||
return stem + "ns"
|
||||
if C == "VOC" and number == "singular":
|
||||
return stem + "ns"
|
||||
base = stem + "nt"
|
||||
end = {
|
||||
("GEN", "singular"): "is", ("DAT", "singular"): "ī",
|
||||
("ACC", "singular"): "em" if gender != "n" else "",
|
||||
("ABL", "singular"): "e",
|
||||
("NOM", "plural"): "ēs" if gender != "n" else "ia",
|
||||
("GEN", "plural"): "ium", ("DAT", "plural"): "ibus",
|
||||
("ACC", "plural"): "ēs" if gender != "n" else "ia",
|
||||
("ABL", "plural"): "ibus", ("VOC", "plural"): "ēs",
|
||||
}.get((C, number), "is")
|
||||
if C == "ACC" and number == "singular" and gender == "n":
|
||||
return stem + "ns"
|
||||
return base + end
|
||||
|
||||
|
||||
def infinitive(lemma, tense="present", voice="active"):
|
||||
lemma = lemma.strip()
|
||||
if lemma == "sum":
|
||||
return ("esse", "rule") if tense == "present" else ("fuisse", "rule")
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
if tense == "present":
|
||||
if voice == "active":
|
||||
return _active_infinitive_stem(conj, pstem).rstrip() + \
|
||||
("re" if conj != 3 and conj != "3io" else "re"), "rule"
|
||||
# passive present infinitive
|
||||
base = {1: pstem + "ā", 2: pstem + "ē", 4: pstem + "ī"}.get(conj)
|
||||
if base:
|
||||
return base + "rī", "rule"
|
||||
return pstem + "ī", "rule" # 3rd: regī
|
||||
if tense == "perfect" and voice == "active" and perfstem:
|
||||
return perfstem + "isse", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"noun_adj_source": "UniMorph Latin (github.com/unimorph/lat, CC-BY-SA 3.0)",
|
||||
"verb_source": "rule-based 4-conjugation engine over curated attested "
|
||||
"principal parts (UniMorph verb list is a 947-lemma sample "
|
||||
"MISSING all core verbs — amō/sum/videō absent)",
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
"curated_verb_lemmas": len(_VERBS) + len(_IRREG),
|
||||
"gender_inference": "declension-based (nom+gen endings) + curated exceptions",
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import json
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
print("\n-- noun declension puella (1st, fem) --")
|
||||
for c in ("nom", "gen", "dat", "acc", "abl", "voc"):
|
||||
print(f" {c}: sg={decline_noun('puella', c, 'singular')[0]:10} "
|
||||
f"pl={decline_noun('puella', c, 'plural')[0]}")
|
||||
print("\n-- rēx (3rd, m):", [decline_noun('rēx', c, 'singular')[0] for c in ('nom','gen','dat','acc','abl')])
|
||||
print("-- gender: puella=", noun_gender("puella"), "rēx=", noun_gender("rēx"),
|
||||
"bellum=", noun_gender("bellum"), "corpus=", noun_gender("corpus"),
|
||||
"manus=", noun_gender("manus"), "diēs=", noun_gender("diēs"))
|
||||
print("\n-- conjugate videō (2nd) present ind active --")
|
||||
for p in ("first", "second", "third"):
|
||||
for n in ("singular", "plural"):
|
||||
print(f" {p[:3]}.{n[:2]}: {conjugate('videō','present','ind','active',p,n)[0]}")
|
||||
print("-- amō forms:", conjugate("amō","present","ind","active","first","singular")[0],
|
||||
conjugate("amō","imperfect","ind","active","third","plural")[0],
|
||||
conjugate("amō","future","ind","active","first","singular")[0],
|
||||
conjugate("amō","perfect","ind","active","third","singular")[0])
|
||||
print("-- sum:", [conjugate("sum","present","ind","active",p,"singular")[0] for p in ("first","second","third")])
|
||||
print("-- participle amō pfv acc.f.sg:", participle("amō","pfv","acc","f","singular")[0])
|
||||
print("-- infinitive amō:", infinitive("amō")[0], "| regō pass:", infinitive("regō", voice="passive")[0])
|
||||
@@ -1,538 +0,0 @@
|
||||
"""morphology_pt_full.py — production-grade Brazilian-Portuguese morphological generator.
|
||||
|
||||
NOT a toy. Backed by two real, broad, Wiktionary-lineage lexicons:
|
||||
|
||||
VERBS — UniMorph Portuguese (github.com/unimorph/por, CC-BY-SA 3.0)
|
||||
4,001 verb lemmas × full paradigm (283,991 finite/non-finite forms +
|
||||
20,005 participle forms). Every mood/tense pt actually inflects:
|
||||
indicative present / preterite (PST;PFV) / imperfect (PST;IPFV) /
|
||||
pluperfect-simple (PST;PRF) / future,
|
||||
conditional (futuro do pretérito),
|
||||
subjunctive present / imperfect / FUTURE (PT-specific live tense),
|
||||
affirmative + negative imperative,
|
||||
PERSONAL infinitive (V;{p};{n};NFIN — a PT-specific finite-ish form),
|
||||
past participle (4 gender/number forms) + gerúndio (V.PTCP;PRS).
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Portuguese (Wiktionary extract, same lineage)
|
||||
81,138 noun lemmas WITH inherent gender + real (often irregular) plural —
|
||||
so -ão→-ões / -ãos / -ães / -õos is resolved PER LEMMA by Wiktionary,
|
||||
never guessed (mão→mãos, pão→pães, coração→corações).
|
||||
40,252 adjective lemmas with real feminine + masc/fem plural forms.
|
||||
|
||||
Fallbacks (degrade, never crash, on out-of-vocabulary input):
|
||||
verbs : rule generator for regular -ar/-er/-ir paradigms
|
||||
nouns : gender heuristic (endings) + rule pluralization (with -ão FLAGGED)
|
||||
adjs : -o/-a gender rule + rule pluralization
|
||||
|
||||
Confidence flag on every form:
|
||||
"lexicon" straight from UniMorph/kaikki (trust: high)
|
||||
"rule" deterministic rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_pt.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
personal_infinitive(lemma, person, number) -> (form, conf)
|
||||
participle(lemma, gender="m", number="singular") -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "por.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_pt.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "pt_morph_cache.pkl")
|
||||
|
||||
# ── mood/tense pair -> UniMorph feature triple (a in tag; b in tag; c in tag) ────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): ("IND", "PRS", None),
|
||||
("ind", "preterite"): ("IND", "PST", "PFV"),
|
||||
("ind", "imperfect"): ("IND", "PST", "IPFV"),
|
||||
("ind", "pluperfect"): ("IND", "PST", "PRF"), # simple mais-que-perfeito
|
||||
("ind", "future"): ("IND", "FUT", None),
|
||||
("ind", "conditional"): ("COND", None, None),
|
||||
("sbjv", "present"): ("SBJV", "PRS", None),
|
||||
("sbjv", "imperfect"): ("SBJV", "PST", "IPFV"),
|
||||
("sbjv", "future"): ("SBJV", "FUT", None), # PT-specific
|
||||
("imp", "affirmative"): ("IMP", "POS", None),
|
||||
("imp", "negative"): ("IMP", "NEG", None),
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build the compact lexicon from UniMorph (verbs) + kaikki (nouns/adjs) ────────
|
||||
def _build_verbs():
|
||||
verbs = {} # (lemma, "mood|tense|person|number") -> form
|
||||
pinf = {} # (lemma, "person|number") -> personal-infinitive form
|
||||
part = {} # lemma -> {("m","SG"): form, ...} past participle
|
||||
ger = {} # lemma -> gerúndio
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f: # past participle: falado/falada/falados/faladas
|
||||
g = "m" if "MASC" in f else ("f" if "FEM" in f else "m")
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else "SG")
|
||||
part.setdefault(lemma, {})[(g, num)] = form
|
||||
elif "PRS" in f: # gerúndio: falando
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
# personal / impersonal infinitive
|
||||
if "NFIN" in f:
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person and number:
|
||||
pinf[(lemma, f"{person}|{number}")] = form
|
||||
continue
|
||||
|
||||
# finite forms
|
||||
mt = None
|
||||
for (mood, tense), (a, b, c) in _VERB_KEYMAP.items():
|
||||
if a not in f:
|
||||
continue
|
||||
if b is not None and b not in f:
|
||||
continue
|
||||
if c is not None and c not in f:
|
||||
continue
|
||||
# IND;PST needs exactly PFV|IPFV|PRF — reject if the required one absent
|
||||
mt = (mood, tense)
|
||||
break
|
||||
if mt is None:
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mt[0]}|{mt[1]}|{person}|{number}"), form)
|
||||
return verbs, pinf, part, ger
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = arg.lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {} # lemma -> {"g","SG","PL"}
|
||||
adjs = {} # lemma -> {("m","SG"),("f","SG"),("m","PL"),("f","PL")}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word: # skip multiword entries
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = x.get("tags") or []
|
||||
if "plural" in t and "alternative" not in t and "obsolete" not in t:
|
||||
pl = x.get("form")
|
||||
break
|
||||
# first entry wins; but a later entry with a plural fills a gap
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or ("alternative" in t) or ("obsolete" in t):
|
||||
continue
|
||||
if "comparative" in t or "superlative" in t or \
|
||||
"diminutive" in t or "augmentative" in t:
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = fm
|
||||
elif "plural" in t: # invariant-gender adj (feliz -> felizes)
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, pinf, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
data = {"verbs": verbs, "pinf": pinf, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
newest_src = max(os.path.getmtime(_UNIMORPH),
|
||||
os.path.getmtime(_KAIKKI) if os.path.exists(_KAIKKI) else 0)
|
||||
if os.path.getmtime(_CACHE) >= newest_src:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PINF, _PART, _GER, _NOUNS, _ADJS = (
|
||||
_LEX["verbs"], _LEX["pinf"], _LEX["part"], _LEX["ger"],
|
||||
_LEX["nouns"], _LEX["adjs"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback (deterministic, last resort) ────────────────────
|
||||
def _vclass(lemma):
|
||||
return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else None
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
# endings indexed [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "as", "a", "amos", "ais", "am"],
|
||||
("ind", "present", "er"): ["o", "es", "e", "emos", "eis", "em"],
|
||||
("ind", "present", "ir"): ["o", "es", "e", "imos", "is", "em"],
|
||||
("ind", "preterite", "ar"): ["ei", "aste", "ou", "amos", "astes", "aram"],
|
||||
("ind", "preterite", "er"): ["i", "este", "eu", "emos", "estes", "eram"],
|
||||
("ind", "preterite", "ir"): ["i", "iste", "iu", "imos", "istes", "iram"],
|
||||
("ind", "imperfect", "ar"): ["ava", "avas", "ava", "ávamos", "áveis", "avam"],
|
||||
("ind", "imperfect", "er"): ["ia", "ias", "ia", "íamos", "íeis", "iam"],
|
||||
("ind", "imperfect", "ir"): ["ia", "ias", "ia", "íamos", "íeis", "iam"],
|
||||
("sbjv", "present", "ar"): ["e", "es", "e", "emos", "eis", "em"],
|
||||
("sbjv", "present", "er"): ["a", "as", "a", "amos", "ais", "am"],
|
||||
("sbjv", "present", "ir"): ["a", "as", "a", "amos", "ais", "am"],
|
||||
("sbjv", "imperfect", "ar"): ["asse", "asses", "asse", "ássemos", "ásseis", "assem"],
|
||||
("sbjv", "imperfect", "er"): ["esse", "esses", "esse", "êssemos", "êsseis", "essem"],
|
||||
("sbjv", "imperfect", "ir"): ["isse", "isses", "isse", "íssemos", "ísseis", "issem"],
|
||||
("sbjv", "future", "ar"): ["ar", "ares", "ar", "armos", "ardes", "arem"],
|
||||
("sbjv", "future", "er"): ["er", "eres", "er", "ermos", "erdes", "erem"],
|
||||
("sbjv", "future", "ir"): ["ir", "ires", "ir", "irmos", "irdes", "irem"],
|
||||
}
|
||||
# future & conditional attach to the FULL infinitive
|
||||
_FUT = ["ei", "ás", "á", "emos", "eis", "ão"]
|
||||
_COND = ["ia", "ias", "ia", "íamos", "íeis", "iam"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
st, i = _stem(lemma), _slot_idx(person, number)
|
||||
if mood == "ind" and tense == "future":
|
||||
return lemma + _FUT[i]
|
||||
if mood == "ind" and tense == "conditional":
|
||||
return lemma + _COND[i]
|
||||
if mood == "imp": # affirmative tú/vocês imperative ~ subjunctive present
|
||||
table = _REG.get(("sbjv", "present", vc))
|
||||
if table and tense == "negative":
|
||||
return st + table[i]
|
||||
# affirmative 2sg = 3sg present indicative; others = subjunctive
|
||||
pres = _REG.get(("ind", "present", vc))
|
||||
if person == "second" and number == "singular":
|
||||
return st + pres[2]
|
||||
return st + table[i] if table else None
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if table:
|
||||
return st + table[i]
|
||||
return None
|
||||
|
||||
|
||||
# verified corrections to UniMorph data errors (each audited individually, not
|
||||
# guessed). The three 1PL-present entries are glued-allomorph errors surfaced by a
|
||||
# full-lexicon scan for a non-final "mos" in V;1;PL;IND;PRS forms (the ONLY three).
|
||||
_VERB_FIX = {
|
||||
("estar", "ind", "imperfect", "third", "plural"): "estavam", # was "estávam"
|
||||
("estar", "ind", "present", "first", "plural"): "estamos", # was "estamosestámos"
|
||||
("haver", "ind", "present", "first", "plural"): "havemos", # was "havemoshemos"
|
||||
("ir", "ind", "present", "first", "plural"): "vamos", # was "vamosimos"
|
||||
}
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
fix = _VERB_FIX.get((lemma, mood, tense, person, number))
|
||||
if fix:
|
||||
return fix, "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
# pt-BR normalization: UniMorph `por` carries the EUROPEAN spelling of
|
||||
# the -ar 1pl PRETERITE (-ámos). Brazilian PT drops the accent
|
||||
# (falámos->falamos, chegámos->chegamos) — 3,334/4,001 verbs affected.
|
||||
if (mood == "ind" and tense == "preterite" and person == "first"
|
||||
and number == "plural" and form.endswith("ámos")):
|
||||
form = form[:-4] + "amos"
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def personal_infinitive(lemma, person, number):
|
||||
"""PT personal (inflected) infinitive: para falarmos, ao chegarem."""
|
||||
lemma = lemma.strip().lower()
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _PINF.get((lemma, f"{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# rule: infinitive + personal endings (-, -es, -, -mos, -des, -em)
|
||||
end = {("first", "singular"): "", ("second", "singular"): "es",
|
||||
("third", "singular"): "", ("first", "plural"): "mos",
|
||||
("second", "plural"): "des", ("third", "plural"): "em"}.get((person, number), "")
|
||||
return lemma + end, "rule"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ───────────────────────────────────────────────────
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _PART.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num)) or d.get(("m", "SG"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
base = lemma[:-2] + "ad"
|
||||
elif lemma[-2:] in ("er", "ir"):
|
||||
base = lemma[:-2] + "id"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "os", "f|PL": "as"}[f"{g}|{num}"]
|
||||
return base + suf, "rule"
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ando", "rule"
|
||||
if lemma.endswith("er"):
|
||||
return lemma[:-2] + "endo", "rule"
|
||||
if lemma.endswith("ir"):
|
||||
return lemma[:-2] + "indo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("ção", "são", "ção", "dade", "tade", "agem", "igem", "ugem", "gem",
|
||||
"ez", "eza", "ice", "ície", "tude", "ude", "âncbefore")
|
||||
_FEM_SUF = ("ção", "são", "dade", "tade", "agem", "gem", "eza", "ez", "ice",
|
||||
"tude", "ude", "ância", "ência", "ínia")
|
||||
_MASC_SUF = ("ema", "oma", "ama", "grama", "eta", "ão") # Greek -ma etc. (mostly m)
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
if noun.endswith(("ema", "oma", "ama")): # problema, idioma, programa
|
||||
return "m"
|
||||
if noun.endswith("a") or noun.endswith("ã"):
|
||||
return "f"
|
||||
if noun.endswith("o") or noun.endswith(("l", "r", "z", "m", "u", "i")):
|
||||
return "m"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
_INVARIANT_PL_SUF = ("s",) # paroxytones ending -s are invariant (o lápis / os lápis)
|
||||
|
||||
|
||||
def _rule_plural(noun):
|
||||
"""Deterministic PT pluralization. Returns (form, ok) where ok=False flags an
|
||||
ambiguous -ão that should lower confidence (the lexicon normally resolves it)."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
if noun.endswith("ão"):
|
||||
return noun[:-2] + "ões", False # majority rule, but AMBIGUOUS -> flag
|
||||
if noun.endswith("m"):
|
||||
return noun[:-1] + "ns", True # homem->homens, jardim->jardins
|
||||
if noun.endswith("al"):
|
||||
return noun[:-2] + "ais", True
|
||||
if noun.endswith("el"):
|
||||
return noun[:-2] + "éis", True
|
||||
if noun.endswith("ol"):
|
||||
return noun[:-2] + "óis", True
|
||||
if noun.endswith("ul"):
|
||||
return noun[:-2] + "uis", True
|
||||
if noun.endswith("il"):
|
||||
return noun[:-2] + "is", True # stressed (funil->funis); unstressed rarer
|
||||
if noun.endswith(("r", "z")):
|
||||
return noun + "es", True # flor->flores, luz->luzes
|
||||
if noun.endswith("s"):
|
||||
# paroxytone -s (lápis, ônibus) invariant; oxytone -s (país) -> -es
|
||||
return noun, True
|
||||
if noun.endswith(("a", "e", "i", "o", "u", "á", "é", "í", "ó", "ú", "ã")):
|
||||
return noun + "s", True
|
||||
return noun + "s", True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
form, ok = _rule_plural(lemma)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# build a missing plural from this gender's singular
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback: -o/-a gender, then pluralize
|
||||
a = lemma
|
||||
if g == "f":
|
||||
if a.endswith("o"):
|
||||
a = a[:-1] + "a"
|
||||
elif a.endswith(("ês", "or")) and not a.endswith("ior"):
|
||||
a = a + "a" # português->portuguesa, trabalhador->..a
|
||||
if num == "PL":
|
||||
a, ok = _rule_plural(a)
|
||||
return a, ("rule" if ok else "fallback")
|
||||
return a, "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Portuguese (github.com/unimorph/por)",
|
||||
"noun_adj_source": "kaikki.org Portuguese (Wiktionary extract)",
|
||||
"license": "CC-BY-SA (Wiktionary-derived)",
|
||||
"verb_forms": len(_VERBS),
|
||||
"verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"personal_infinitive_forms": len(_PINF),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("falar", "ind", "present", "first", "singular", "falo"),
|
||||
("comer", "ind", "present", "third", "plural", "comem"),
|
||||
("partir", "ind", "present", "first", "plural", "partimos"),
|
||||
("ser", "ind", "present", "third", "singular", "é"),
|
||||
("ir", "ind", "preterite", "first", "singular", "fui"),
|
||||
("ter", "ind", "future", "first", "singular", "terei"),
|
||||
("fazer", "sbjv", "present", "first", "singular", "faça"),
|
||||
("dormir", "ind", "present", "first", "singular", "durmo"),
|
||||
("dar", "ind", "preterite", "third", "singular", "deu"),
|
||||
("poder", "ind", "conditional", "first", "singular", "poderia"),
|
||||
("fazer", "sbjv", "future", "third", "singular", "fizer"),
|
||||
("estar", "ind", "present", "third", "singular", "está"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]} -> {got:14} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: casa=", noun_gender("casa"), "problema=", noun_gender("problema"),
|
||||
"mão=", noun_gender("mão"), "coração=", noun_gender("coração"),
|
||||
"flor=", noun_gender("flor"))
|
||||
print(" plural: mão->", inflect_noun("mão", "plural"),
|
||||
"| pão->", inflect_noun("pão", "plural"),
|
||||
"| animal->", inflect_noun("animal", "plural"),
|
||||
"| coração->", inflect_noun("coração", "plural"))
|
||||
print(" adj: bonito/f/sg->", inflect_adj("bonito", "f", "singular"),
|
||||
"| feliz/m/pl->", inflect_adj("feliz", "m", "plural"),
|
||||
"| português/f/sg->", inflect_adj("português", "f", "singular"))
|
||||
print(" part: fazer/m/sg->", participle("fazer"), "| ger falar->", gerund("falar"))
|
||||
print(" pinf falar 1pl->", personal_infinitive("falar", "first", "plural"))
|
||||
@@ -1,609 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_ro_full.py — production-grade Romanian morphological generator.
|
||||
|
||||
Romanian is the BIG typological delta of the Romance family. The verb engine and
|
||||
the confidence/fallback contract TRANSFER from the Italian sibling; the NOMINAL
|
||||
system is genuinely new: Romanian has a SUFFIXED definite article, a preserved
|
||||
NOM/ACC vs GEN/DAT case distinction, a NEUTER gender (masc-agreeing in SG,
|
||||
fem-agreeing in PL), and a VOCATIVE. Those are grounded in real per-lemma data,
|
||||
not guessed.
|
||||
|
||||
Real, Wiktionary-lineage lexical sources:
|
||||
|
||||
VERBS — UniMorph Romanian (github.com/unimorph/ron, CC-BY-SA 3.0)
|
||||
~1216 verb lemmas × paradigm, CLEAN orthography:
|
||||
indicativ prezent / imperfect (PST;IPFV) / perfectul simplu (PST;PFV) /
|
||||
conjunctiv prezent (SBJV;PRS, stored WITHOUT the 'să' particle),
|
||||
participiu (V.PTCP;PST, INVARIABLE in the perfect compus),
|
||||
gerunziu (V.CVB;PRS), infinitiv (NFIN), imperativ.
|
||||
ro_irreg_verbs (embedded) — high-frequency verbs UniMorph MISSES
|
||||
(avea, vrea, da) + the auxiliary clitic paradigms the compound tenses need
|
||||
(perfect-compus am/ai/a/am/ați/au, viitor voi/vei/va/vom/veți/vor,
|
||||
condițional aș/ai/ar/am/ați/ar). Real standard forms.
|
||||
|
||||
NOUNS — kaikki.org Romanian (Wiktionary extract, CC-BY-SA 3.0)
|
||||
the FULL declension per lemma, cleanly tagged:
|
||||
(nom/acc | gen/dat | vocative) × (indefinite | definite) × (sg | pl).
|
||||
This is what makes the suffixed article LEXICALLY grounded (om→omul,
|
||||
casă→casa, băiat→băiatul, casei gen/dat, omule vocative). Inherent gender
|
||||
m / f / n (NEUTER available directly) from the head template.
|
||||
|
||||
ADJECTIVES — UniMorph Romanian ADJ
|
||||
full case × gender(MASC/FEM/NEUT) × number × definiteness paradigm.
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV): rule verb conjugation for -a/-ea/-e/-i/-î
|
||||
classes, rule pluralization, rule suffixed-article by gender+ending. Every form
|
||||
carries a confidence flag: "lexicon" | "rule" | "fallback".
|
||||
|
||||
Public API (used by realizer_ro.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
aux(kind, person, number) -> str # perfect / future / conditional clitics
|
||||
participle(lemma) -> (form, conf) # INVARIABLE
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"|"n"
|
||||
definite_suffix(noun, gender, number, case) -> (form, conf) # rule engine
|
||||
inflect_noun(lemma, number, gender=None, case="nomacc", definite=False) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number, case="nomacc", definite=False) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "ron.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_ro.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "ro_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set ─────────────────────────────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "perfect_s"): {"IND", "PST", "PFV"}, # perfectul simplu (regional/lit.)
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── high-frequency irregulars UniMorph misses + auxiliary clitic paradigms ────────
|
||||
# Real standard Romanian forms (textbook paradigms).
|
||||
_IRREG = {
|
||||
"avea": {
|
||||
"ind|present|1|SG": "am", "ind|present|2|SG": "ai", "ind|present|3|SG": "are",
|
||||
"ind|present|1|PL": "avem", "ind|present|2|PL": "aveți", "ind|present|3|PL": "au",
|
||||
"ind|imperfect|1|SG": "aveam", "ind|imperfect|2|SG": "aveai",
|
||||
"ind|imperfect|3|SG": "avea", "ind|imperfect|1|PL": "aveam",
|
||||
"ind|imperfect|2|PL": "aveați", "ind|imperfect|3|PL": "aveau",
|
||||
"sbjv|present|3|SG": "aibă", "sbjv|present|3|PL": "aibă",
|
||||
"sbjv|present|1|SG": "am", "sbjv|present|2|SG": "ai",
|
||||
"sbjv|present|1|PL": "avem", "sbjv|present|2|PL": "aveți",
|
||||
"part": "avut", "ger": "având",
|
||||
},
|
||||
"vrea": {
|
||||
"ind|present|1|SG": "vreau", "ind|present|2|SG": "vrei", "ind|present|3|SG": "vrea",
|
||||
"ind|present|1|PL": "vrem", "ind|present|2|PL": "vreți", "ind|present|3|PL": "vor",
|
||||
"ind|imperfect|1|SG": "voiam", "ind|imperfect|3|SG": "voia",
|
||||
"sbjv|present|3|SG": "vrea", "sbjv|present|3|PL": "vrea",
|
||||
"part": "vrut", "ger": "vrând",
|
||||
},
|
||||
"da": {
|
||||
"ind|present|1|SG": "dau", "ind|present|2|SG": "dai", "ind|present|3|SG": "dă",
|
||||
"ind|present|1|PL": "dăm", "ind|present|2|PL": "dați", "ind|present|3|PL": "dau",
|
||||
"ind|imperfect|1|SG": "dădeam", "ind|imperfect|3|SG": "dădea",
|
||||
"sbjv|present|3|SG": "dea", "sbjv|present|3|PL": "dea",
|
||||
"part": "dat", "ger": "dând",
|
||||
},
|
||||
"fi": { # a fi — present is in UniMorph but keep participle + subjunctive here
|
||||
"part": "fost", "ger": "fiind",
|
||||
"sbjv|present|1|SG": "fiu", "sbjv|present|2|SG": "fii", "sbjv|present|3|SG": "fie",
|
||||
"sbjv|present|1|PL": "fim", "sbjv|present|2|PL": "fiți", "sbjv|present|3|PL": "fie",
|
||||
"ind|imperfect|1|SG": "eram", "ind|imperfect|2|SG": "erai",
|
||||
"ind|imperfect|3|SG": "era", "ind|imperfect|1|PL": "eram",
|
||||
"ind|imperfect|2|PL": "erați", "ind|imperfect|3|PL": "erau",
|
||||
},
|
||||
}
|
||||
# auxiliary clitic paradigms (person,number)->form
|
||||
_AUX = {
|
||||
"perfect": {("first", "singular"): "am", ("second", "singular"): "ai",
|
||||
("third", "singular"): "a", ("first", "plural"): "am",
|
||||
("second", "plural"): "ați", ("third", "plural"): "au"},
|
||||
"future": {("first", "singular"): "voi", ("second", "singular"): "vei",
|
||||
("third", "singular"): "va", ("first", "plural"): "vom",
|
||||
("second", "plural"): "veți", ("third", "plural"): "vor"},
|
||||
"conditional": {("first", "singular"): "aș", ("second", "singular"): "ai",
|
||||
("third", "singular"): "ar", ("first", "plural"): "am",
|
||||
("second", "plural"): "ați", ("third", "plural"): "ar"},
|
||||
}
|
||||
|
||||
|
||||
def aux(kind, person, number):
|
||||
return _AUX[kind][(person, number)]
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ──────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs, part, ger = {}, {}, {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
# conjunctiv forms in UniMorph carry a leading 'să ' — strip it
|
||||
surf = form
|
||||
if surf.startswith("să "):
|
||||
surf = surf[3:]
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "perfect_s" and "IPFV" in f:
|
||||
continue
|
||||
# keep IND;PRS out of the PRF slot (mai-mult-ca-perfect etc. ignored)
|
||||
if {"IND", "PRS"} <= req and "PRF" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), surf)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns: full declension paradigm per lemma ──────────────────────────────
|
||||
_EXCL = {"alternative", "archaic", "obsolete", "regional", "dialectal", "rare",
|
||||
"table-tags", "inflection-template", "error-unrecognized-form",
|
||||
"diminutive", "augmentative", "informal"}
|
||||
|
||||
|
||||
def _noun_key(tagset):
|
||||
if tagset & _EXCL:
|
||||
return None
|
||||
if "vocative" in tagset:
|
||||
case = "voc"
|
||||
elif "genitive" in tagset or "dative" in tagset:
|
||||
case = "gendat"
|
||||
elif "nominative" in tagset or "accusative" in tagset:
|
||||
case = "nomacc"
|
||||
else:
|
||||
return None
|
||||
definite = "definite" in tagset and "indefinite" not in tagset
|
||||
number = "PL" if "plural" in tagset else ("SG" if "singular" in tagset else None)
|
||||
if number is None:
|
||||
return None
|
||||
return (case, definite, number)
|
||||
|
||||
|
||||
def _build_nouns():
|
||||
nouns = {} # lemma -> {"g":..., para:{(case,def,num):form}, "PL":plain_plural}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "noun":
|
||||
continue
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
a = str((ht[0].get("args") or {}).get("1") or "").lower()
|
||||
if a[:1] in ("m", "f", "n"):
|
||||
g = a[:1]
|
||||
entry = nouns.setdefault(word, {"g": g, "para": {}, "PL": None})
|
||||
if entry["g"] is None and g:
|
||||
entry["g"] = g
|
||||
for x in (d.get("forms") or []):
|
||||
fm = x.get("form")
|
||||
tg = set(x.get("tags") or [])
|
||||
if not fm or fm in ("-", "#", "") or " " in fm:
|
||||
continue
|
||||
if tg == {"plural"} and not entry["PL"]:
|
||||
entry["PL"] = fm
|
||||
k = _noun_key(tg)
|
||||
if k and k not in entry["para"]:
|
||||
entry["para"][k] = fm
|
||||
return nouns
|
||||
|
||||
|
||||
# ── adjectives from kaikki (UniMorph ron ADJ is sparse AND mis-tagged; kaikki is
|
||||
# clean: the 4-form agreement pattern bun/bună/buni/bune). Neuter maps sg->masc,
|
||||
# pl->fem, so 4 forms (m/f × SG/PL) fully cover it. ────────────────────────────
|
||||
def _build_adjs():
|
||||
adjs = {} # lemma -> {(gender,number): form} gender in {m,f}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "adj":
|
||||
continue
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word) # masc sg = headword
|
||||
for x in (d.get("forms") or []):
|
||||
fm = x.get("form")
|
||||
t = set(x.get("tags") or [])
|
||||
if not fm or " " in fm or fm in ("-", "#") or (t & _EXCL):
|
||||
continue
|
||||
if "definite" in t or "genitive" in t or "dative" in t:
|
||||
continue # keep indefinite nom/acc agr set
|
||||
pl = "plural" in t
|
||||
fem = "feminine" in t
|
||||
masc = "masculine" in t
|
||||
if fem and pl:
|
||||
d0.setdefault(("f", "PL"), fm)
|
||||
elif masc and pl:
|
||||
d0.setdefault(("m", "PL"), fm)
|
||||
elif fem and not pl:
|
||||
d0.setdefault(("f", "SG"), fm)
|
||||
elif pl and not fem and not masc: # bare plural -> both genders
|
||||
d0.setdefault(("m", "PL"), fm)
|
||||
d0.setdefault(("f", "PL"), fm)
|
||||
return adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns = _build_nouns()
|
||||
adjs = _build_adjs()
|
||||
data = {"verbs": verbs, "part": part, "ger": ger, "nouns": nouns, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"])
|
||||
|
||||
|
||||
# ── rule verb conjugation fallback ────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("a"):
|
||||
return "a"
|
||||
if lemma.endswith("ea"):
|
||||
return "ea"
|
||||
if lemma.endswith("e"):
|
||||
return "e"
|
||||
if lemma.endswith("i"):
|
||||
return "i"
|
||||
if lemma.endswith("î"):
|
||||
return "î"
|
||||
return None
|
||||
|
||||
|
||||
# regular present endings by class [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG_PRS = {
|
||||
"a": ["", "i", "ă", "ăm", "ați", "ă"], # a lucra type (simplified)
|
||||
"ea": ["", "i", "e", "em", "eți", "", ],
|
||||
"e": ["", "i", "e", "em", "eți", ""],
|
||||
"i": ["esc", "ești", "ește", "im", "iți", "esc"], # -i type (a vorbi)
|
||||
"î": ["ăsc", "ăști", "ăște", "âm", "âți", "ăsc"],
|
||||
}
|
||||
_SLOT = {("first", "singular"): 0, ("second", "singular"): 1, ("third", "singular"): 2,
|
||||
("first", "plural"): 3, ("second", "plural"): 4, ("third", "plural"): 5}
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
i = _SLOT[(person, number)]
|
||||
body = lemma[:-len(vc)]
|
||||
if mood == "ind" and tense == "present":
|
||||
end = _REG_PRS[vc][i]
|
||||
return body + end
|
||||
if mood == "ind" and tense == "imperfect":
|
||||
# -a/-i/-î -> stem + a/eai...; -e/-ea -> eam. Simplified regular imperfect.
|
||||
stem = body
|
||||
endings = {"a": ["am", "ai", "a", "am", "ați", "au"],
|
||||
"i": ["eam", "eai", "ea", "eam", "eați", "eau"],
|
||||
"î": ["am", "ai", "a", "am", "ați", "au"],
|
||||
"e": ["eam", "eai", "ea", "eam", "eați", "eau"],
|
||||
"ea": ["eam", "eai", "ea", "eam", "eați", "eau"]}[vc]
|
||||
return stem + endings[i]
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC verb API ───────────────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{_NUMBER.get(number,'?')}"
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
form = _VERBS.get((lemma, key))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r is not None:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def participle(lemma):
|
||||
"""Past participle — INVARIABLE in the perfect compus (am mers, am văzut)."""
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "a":
|
||||
return lemma[:-1] + "at", "rule"
|
||||
if vc in ("ea",):
|
||||
return lemma[:-2] + "ut", "rule"
|
||||
if vc == "i":
|
||||
return lemma[:-1] + "it", "rule"
|
||||
if vc == "î":
|
||||
return lemma[:-1] + "ât", "rule"
|
||||
if vc == "e":
|
||||
return lemma[:-1] + "ut", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc in ("a", "î"):
|
||||
return lemma[:-1] + "ând", "rule"
|
||||
if vc in ("ea", "e", "i"):
|
||||
return lemma[:-len(vc)] + "ind", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── noun gender ───────────────────────────────────────────────────────────────────
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f", "n"):
|
||||
return d["g"]
|
||||
if lemma.endswith(("ă", "a", "e")):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
# ── SUFFIXED DEFINITE ARTICLE — rule engine (fallback for OOV nouns) ───────────────
|
||||
def definite_suffix(noun, gender, number, case="nomacc"):
|
||||
"""Attach the enclitic definite article by gender + ending. Returns (form, conf).
|
||||
This is the headline Romanian-specific engine extension."""
|
||||
n = noun
|
||||
g = gender
|
||||
if number == "singular":
|
||||
if g in ("m", "n"):
|
||||
if case == "gendat":
|
||||
# masc/neut gen-dat definite: -lui
|
||||
if n.endswith("e"):
|
||||
return n + "lui", "rule" # câine -> câinelui
|
||||
if n.endswith("u"):
|
||||
return n + "lui", "rule"
|
||||
return n + "ului", "rule" # om -> omului
|
||||
# nom/acc
|
||||
if n.endswith("e"):
|
||||
return n + "le", "rule" # câine -> câinele
|
||||
if n.endswith("u"):
|
||||
return n + "l", "rule" # codru -> codrul
|
||||
if n.endswith("i"):
|
||||
return n + "ul", "rule"
|
||||
return n + "ul", "rule" # om -> omul
|
||||
# feminine singular
|
||||
if case == "gendat":
|
||||
# fem gen/dat definite = plural-stem + i (casei, fetei) — needs plural;
|
||||
# approximated as: -ă->-ei, -e->-ei, -a->-alei
|
||||
if n.endswith("ă"):
|
||||
return n[:-1] + "ei", "rule" # casă -> casei
|
||||
if n.endswith("e"):
|
||||
return n[:-1] + "ei", "rule" # carte -> cărții(approx cartei)
|
||||
if n.endswith("a"):
|
||||
return n[:-1] + "lei", "rule"
|
||||
return n + "i", "rule"
|
||||
# fem nom/acc
|
||||
if n.endswith("ă"):
|
||||
return n[:-1] + "a", "rule" # casă -> casa
|
||||
if n.endswith("e"):
|
||||
return n[:-1] + "ea", "rule" # carte -> cartea
|
||||
if n.endswith("a"):
|
||||
return n + "ua", "rule" # stea -> steaua
|
||||
if n.endswith("i"):
|
||||
return n + "a", "rule"
|
||||
return n + "a", "rule"
|
||||
# plural
|
||||
if case == "gendat":
|
||||
base = noun
|
||||
return base + "lor", "rule" # -lor for all gen/dat pl
|
||||
if g == "m":
|
||||
return noun + "i", "rule" # oameni -> oamenii (+i)
|
||||
return noun + "le", "rule" # case -> casele, trenuri->trenurile
|
||||
|
||||
|
||||
# ── rule pluralization (fallback) ─────────────────────────────────────────────────
|
||||
def _rule_plural(noun, gender):
|
||||
if gender == "f":
|
||||
if noun.endswith("ă"):
|
||||
return noun[:-1] + "e"
|
||||
if noun.endswith("e"):
|
||||
return noun[:-1] + "i"
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "le"
|
||||
return noun + "e"
|
||||
if gender == "n":
|
||||
return noun + "uri"
|
||||
# masculine
|
||||
if noun.endswith(("e",)):
|
||||
return noun[:-1] + "i"
|
||||
return noun + "i"
|
||||
|
||||
|
||||
# ── PUBLIC noun inflection ────────────────────────────────────────────────────────
|
||||
def inflect_noun(lemma, number, gender=None, case="nomacc", definite=False):
|
||||
lemma = lemma.strip().lower()
|
||||
g = gender or noun_gender(lemma)
|
||||
d = _NOUNS.get(lemma)
|
||||
numk = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
if case == "voc":
|
||||
form = d["para"].get(("voc", True, numk)) or d["para"].get(("voc", False, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# try the exact paradigm cell from kaikki (lexically grounded)
|
||||
form = d["para"].get((case, definite, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# indefinite fallbacks from the paradigm
|
||||
if not definite:
|
||||
form = d["para"].get(("nomacc", False, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
if numk == "PL" and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
if numk == "SG":
|
||||
return lemma, "lexicon"
|
||||
# rule path
|
||||
base = lemma if number == "singular" else _rule_plural(lemma, g)
|
||||
if definite:
|
||||
return definite_suffix(base, g, number, case)
|
||||
return base, ("rule" if d is None else "lexicon")
|
||||
|
||||
|
||||
# ── PUBLIC adjective agreement ────────────────────────────────────────────────────
|
||||
def _neuter_map(gender, number):
|
||||
# neuter agrees masculine in SG, feminine in PL
|
||||
if gender == "n":
|
||||
return "m" if number == "singular" else "f"
|
||||
return gender
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number, case="nomacc", definite=False):
|
||||
lemma = lemma.strip().lower()
|
||||
numk = "SG" if number == "singular" else "PL"
|
||||
eg = _neuter_map(gender, number) # neuter -> masc(SG)/fem(PL)
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((eg, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# rule fallback: 4-form pattern bun/bună/buni/bune keyed by effective gender
|
||||
a = lemma
|
||||
if number == "singular":
|
||||
if eg == "f":
|
||||
if a.endswith("e"):
|
||||
return a, "rule" # mare invariant sg
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "ă", "rule" # nou -> nouă
|
||||
if a.endswith("ă"):
|
||||
return a, "rule"
|
||||
return a + "ă", "rule" # bun -> bună
|
||||
return a, "rule" # masc/neut sg = lemma
|
||||
# plural
|
||||
if eg == "f":
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "i", "rule" # mare -> mari
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "e", "rule" # nou -> noue (approx; 'noi' irr)
|
||||
if a.endswith("ă"):
|
||||
return a[:-1] + "e", "rule"
|
||||
return a + "e", "rule" # bun -> bune
|
||||
# masc/neut(SG-only)->here masc pl -> -i
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "i", "rule" # mare -> mari
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "i", "rule"
|
||||
return a + "i", "rule" # bun -> buni
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Romanian (github.com/unimorph/ron) + curated "
|
||||
"irregulars (avea/vrea/da + aux clitic paradigms)",
|
||||
"noun_source": "kaikki.org Romanian — full case/definite/vocative declension",
|
||||
"adj_source": "UniMorph Romanian ADJ (case×gender×number×definiteness)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREG),
|
||||
"participle_lemmas": len(_PART),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
print("\n── SUFFIXED DEFINITE ARTICLE (the headline delta) ──")
|
||||
for n, g in [("om", "m"), ("băiat", "m"), ("casă", "f"), ("carte", "f"),
|
||||
("tren", "n"), ("student", "m"), ("floare", "f")]:
|
||||
sg = inflect_noun(n, "singular", g, "nomacc", True)
|
||||
pl = inflect_noun(n, "plural", g, "nomacc", True)
|
||||
gd = inflect_noun(n, "singular", g, "gendat", True)
|
||||
vo = inflect_noun(n, "singular", g, "voc", False)
|
||||
print(f" {n:8}({g}) def.sg={sg[0]:12} def.pl={pl[0]:14} "
|
||||
f"gen/dat.sg={gd[0]:12} voc={vo[0]}")
|
||||
print("\n── NEUTER split agreement (tren: masc SG / fem PL) ──")
|
||||
print(" tren nou ->", inflect_noun("tren", "singular", "n")[0],
|
||||
inflect_adj("nou", "n", "singular")[0])
|
||||
print(" trenuri noi->", inflect_noun("tren", "plural", "n")[0],
|
||||
inflect_adj("nou", "n", "plural")[0])
|
||||
print("\n── verbs ──")
|
||||
for l, m, t, p, n, in [("merge", "ind", "present", "third", "singular"),
|
||||
("avea", "ind", "present", "first", "singular"),
|
||||
("fi", "ind", "present", "third", "singular"),
|
||||
("vorbi", "ind", "present", "third", "plural"),
|
||||
("face", "sbjv", "present", "third", "singular"),
|
||||
("lucra", "ind", "imperfect", "third", "singular")]:
|
||||
print(f" {l:8}{m}/{t:10}{p[:3]}.{n[:2]} -> {conjugate(l,m,t,p,n)}")
|
||||
print(" perfect-aux(3sg):", aux("perfect", "third", "singular"),
|
||||
"| future(1sg):", aux("future", "first", "singular"),
|
||||
"| cond(3sg):", aux("conditional", "third", "singular"))
|
||||
print(" participle merge/vedea:", participle("merge"), participle("vedea"))
|
||||
@@ -1,43 +0,0 @@
|
||||
// multilingual_gate.el - deterministic language detect + localized-phrase test.
|
||||
|
||||
fn mg_det(text: String, want: String) -> String {
|
||||
let got: String = ml_detect(text)
|
||||
let ok: String = "MISMATCH"
|
||||
if str_eq(got, want) { let ok = "ok" }
|
||||
return " detect(" + got + ") want=" + want + " (" + ok + ") :: " + text + "\n"
|
||||
}
|
||||
|
||||
fn mg_ok(text: String, want: String) -> Int {
|
||||
if str_eq(ml_detect(text), want) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_ml_gate() -> String {
|
||||
let t1: String = "Does Neuron use SQLite for storage?"
|
||||
let t2: String = "Neuron, me explica cómo la saliencia forma las geometrías."
|
||||
let t3: String = "O professor não leu o livro na memória."
|
||||
let t4: String = "Che cosa memorizza Neuron nella memoria?"
|
||||
|
||||
let rep: String = "==== ELP multilingual detect + localized phrases ====\n"
|
||||
let rep = rep + mg_det(t1, "en")
|
||||
let rep = rep + mg_det(t2, "es")
|
||||
let rep = rep + mg_det(t3, "pt")
|
||||
let rep = rep + mg_det(t4, "it")
|
||||
|
||||
let rep = rep + " localized decline (pt): " + ml_tr("no_memory", "pt") + "\n"
|
||||
let rep = rep + " localized decline (es): " + ml_tr("no_memory", "es") + "\n"
|
||||
let rep = rep + " term(saliência->en): " + ml_term("saliência", "pt") + "\n"
|
||||
let rep = rep + " pred(store->pt): " + ml_translate_pred("store", "pt") + "\n"
|
||||
|
||||
let ok: Int = 0
|
||||
if mg_ok(t1, "en") == 1 { let ok = ok + 1 }
|
||||
if mg_ok(t2, "es") == 1 { let ok = ok + 1 }
|
||||
if mg_ok(t3, "pt") == 1 { let ok = ok + 1 }
|
||||
if mg_ok(t4, "it") == 1 { let ok = ok + 1 }
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "language detected correctly: " + int_to_str(ok) + "/4\n"
|
||||
if ok == 4 { let rep = rep + "ML GATE: PASS\n" } else { let rep = rep + "ML GATE: FAIL\n" }
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_ml_gate())
|
||||
@@ -1,52 +0,0 @@
|
||||
// propositions_gate.el - the READ primitive over memory text (native el).
|
||||
// Proves triples are recovered from free memory text and that SACRED polarity
|
||||
// survives extraction (a negative memory must yield a NOT-triple).
|
||||
|
||||
fn pg_check(text: String, want_pol: String) -> String {
|
||||
let p: [String] = prop_extract_one(text, "nd-test")
|
||||
let pol: String = slots_get(p, "polarity")
|
||||
let ok: String = "MISMATCH"
|
||||
if str_eq(pol, want_pol) { let ok = "ok" }
|
||||
return " " + prop_repr(p) + " pol=" + pol + " expected=" + want_pol + " (" + ok + ")\n"
|
||||
}
|
||||
|
||||
fn pg_pol_ok(text: String, want_pol: String) -> Int {
|
||||
let p: [String] = prop_extract_one(text, "nd-test")
|
||||
if str_eq(slots_get(p, "polarity"), want_pol) { return 1 }
|
||||
return 0
|
||||
}
|
||||
|
||||
fn run_prop_gate() -> String {
|
||||
let m1: String = "Neuron stores memories in SQLite."
|
||||
let m2: String = "The engram does not delete a memory."
|
||||
let m3: String = "Salience never drops the negation."
|
||||
let m4: String = "The teacher gives the book to the children."
|
||||
|
||||
let rep: String = "==== ELP proposition extraction (memory text -> triples) ====\n"
|
||||
let rep = rep + pg_check(m1, "aff")
|
||||
let rep = rep + pg_check(m2, "neg")
|
||||
let rep = rep + pg_check(m3, "neg")
|
||||
let rep = rep + pg_check(m4, "aff")
|
||||
|
||||
// multi-sentence memory: one triple per sentence, order preserved
|
||||
let doc: String = "Neuron persists learning. It does not forget the library."
|
||||
let props: [String] = prop_extract(doc, "nd-doc")
|
||||
let rep = rep + " --- multi-sentence doc (" + int_to_str(native_list_len(props)) + " props) ---\n"
|
||||
let di: Int = 0
|
||||
while di < native_list_len(props) {
|
||||
let rep = rep + " " + native_list_get(props, di) + "\n"
|
||||
let di = di + 1
|
||||
}
|
||||
|
||||
let ok: Int = 0
|
||||
if pg_pol_ok(m1, "aff") == 1 { let ok = ok + 1 }
|
||||
if pg_pol_ok(m2, "neg") == 1 { let ok = ok + 1 }
|
||||
if pg_pol_ok(m3, "neg") == 1 { let ok = ok + 1 }
|
||||
if pg_pol_ok(m4, "aff") == 1 { let ok = ok + 1 }
|
||||
let rep = rep + "-----------------------------------------------------------------\n"
|
||||
let rep = rep + "SACRED polarity correct on extraction: " + int_to_str(ok) + "/4\n"
|
||||
if ok == 4 { let rep = rep + "PROP GATE: PASS\n" } else { let rep = rep + "PROP GATE: FAIL\n" }
|
||||
return rep
|
||||
}
|
||||
|
||||
println(run_prop_gate())
|
||||
+1
-1
@@ -22,7 +22,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)/el}"
|
||||
ELC="${ELC:-${EL_HOME}/dist/platform/elc}"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
SRC_DIR="$(cd .. && pwd)/src"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
|
||||
@@ -81,7 +81,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -49,6 +49,12 @@ jobs:
|
||||
echo "Downloading el_runtime.h..."
|
||||
curl -fsSL "${RELEASE_BASE}/el_runtime.h" -o /usr/local/lib/el/el_runtime.h
|
||||
|
||||
echo "Downloading engram_store.c..."
|
||||
curl -fsSL "${RELEASE_BASE}/engram_store.c" -o /usr/local/lib/el/engram_store.c
|
||||
|
||||
echo "Downloading engram_store.h..."
|
||||
curl -fsSL "${RELEASE_BASE}/engram_store.h" -o /usr/local/lib/el/engram_store.h
|
||||
|
||||
echo "El SDK installed:"
|
||||
elc --version || true
|
||||
|
||||
@@ -62,11 +68,12 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
/usr/local/lib/el/el_runtime.c \
|
||||
/usr/local/lib/el/engram_store.c \
|
||||
-lcurl -lpthread
|
||||
echo "Linked dist/engram"
|
||||
ls -lh dist/engram
|
||||
|
||||
+5
-2
@@ -1,3 +1,6 @@
|
||||
target/
|
||||
*.db
|
||||
.DS_Store
|
||||
*.db
|
||||
*.elc
|
||||
*.elh
|
||||
dist/
|
||||
target/
|
||||
|
||||
+151
-30
@@ -117,6 +117,17 @@ fn route_text_health(method: String, path: String, body: String) -> String {
|
||||
// save/load with no "path" hit engram_save(""). Rewritten to the
|
||||
// `let x = if cond { a } else { b }` expression form (the pattern the newer
|
||||
// routes route_emit_ise/route_capture_knowledge already use correctly).
|
||||
// store_on — ENGRAM_STORE flag (tiered paged store as the durable owner). Matches
|
||||
// engram_store_enabled() in el_runtime.c EXACTLY (1 / on / true). Default off →
|
||||
// every persistence path below is byte-for-byte the historical snapshot behavior.
|
||||
fn store_on() -> Bool {
|
||||
let v: String = env("ENGRAM_STORE")
|
||||
if str_eq(v, "1") { return true }
|
||||
if str_eq(v, "on") { return true }
|
||||
if str_eq(v, "true") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// persist_canonical — save the canonical snapshot after a durable write.
|
||||
//
|
||||
// WHY (2026-07-22 self-review): the 2026-07-21 fix correctly stopped READ
|
||||
@@ -131,8 +142,16 @@ fn route_text_health(method: String, path: String, body: String) -> String {
|
||||
// tolerant, ~2/min — snapshotting the whole store per heartbeat is waste;
|
||||
// any durable write that follows persists the pruning too).
|
||||
fn persist_canonical() -> Int {
|
||||
// ENGRAM_STORE: the paged store is the durable owner — a checkpoint flushes
|
||||
// dirty pages behind a WAL-durable record (durable the moment the WAL fsyncs).
|
||||
// This is the fix for the "restart reverted to a 17h-old snapshot" data loss:
|
||||
// durable writes no longer depend on a full snapshot.json rewrite. Returns 1
|
||||
// on a successful checkpoint, 0 otherwise. Flag-off: unchanged (writes JSON).
|
||||
if store_on() {
|
||||
return engram_store_checkpoint()
|
||||
}
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
// (2026-08-10 self-review) This returned a hardcoded 1, which made every
|
||||
// caller's `let saved: Int = persist_canonical()` a dead variable — six
|
||||
// durable write paths each believed they had confirmation of a successful
|
||||
@@ -140,6 +159,57 @@ fn persist_canonical() -> Int {
|
||||
return engram_save(dir + "/snapshot.json")
|
||||
}
|
||||
|
||||
// ── WAL persistence (design doc §§3-14; gated behind ENGRAM_WAL=on) ──────────
|
||||
// Default OFF → every persist path below is byte-identical to the historical
|
||||
// per-write full-snapshot behavior. When ON, structural mutations append O(1)
|
||||
// WAL records instead of rewriting the whole graph, with threshold compaction.
|
||||
fn wal_on() -> Bool {
|
||||
str_eq(env("ENGRAM_WAL"), "on")
|
||||
}
|
||||
|
||||
// Persist a single-node mutation (create / content-evolve / strengthen).
|
||||
fn persist_node(id: String) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_node_put(d, id)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Persist edges appended at index >= start (covers single-edge and batch).
|
||||
fn persist_edges_since(start: Int) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_edges_since(d, start)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Persist a Hebbian consolidation batch as ONE WAL record (single fsync, §5-B).
|
||||
fn persist_hebb_batch(start: Int) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_hebb_batch(d, start)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Bulk mutation (embedding backfill, load-merge): write a fresh compaction base
|
||||
// so the many-node change is durable in one atomic snapshot; WAL is truncated.
|
||||
fn persist_bulk() -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
return engram_wal_compact(d)
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// INCOMPLETE-ROUTE FIX (2026-07-24 self-review): this route silently dropped
|
||||
// label, importance, tier, and tags — engram_node() defaults label to content
|
||||
// and importance to 0.5, so every node created over HTTP lost its metadata.
|
||||
@@ -181,7 +251,7 @@ fn route_create_node(method: String, path: String, body: String) -> String {
|
||||
salience, importance, confidence,
|
||||
tier, tags
|
||||
)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
"{\"id\":\"" + id + "\",\"content\":\"" + content + "\",\"node_type\":\"" + node_type + "\"}"
|
||||
}
|
||||
|
||||
@@ -208,7 +278,7 @@ fn route_scan_nodes(method: String, path: String, body: String) -> String {
|
||||
// clobbered the good snapshot. Read routes must never write the canonical path.)
|
||||
fn route_scan_edges(method: String, path: String, body: String) -> String {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let snap_path: String = dir + "/.scan-export.json"
|
||||
engram_save(snap_path)
|
||||
let snap: String = fs_read(snap_path)
|
||||
@@ -250,8 +320,9 @@ fn route_create_edge(method: String, path: String, body: String) -> String {
|
||||
// (dormant association); only default when the key is absent.
|
||||
let w_present: String = json_get_raw(body, "weight")
|
||||
let weight: Float = if str_eq(w_present, "") { 0.5 } else { json_get_float(body, "weight") }
|
||||
let ec0: Int = engram_edge_count()
|
||||
engram_connect(from_id, to_id, weight, relation)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_edges_since(ec0)
|
||||
"{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + relation + "\"}"
|
||||
}
|
||||
|
||||
@@ -276,6 +347,7 @@ fn route_create_edges_batch(method: String, path: String, body: String) -> Strin
|
||||
if str_eq(arr, "") { return err_json("missing edges array") }
|
||||
let n: Int = json_array_len(arr)
|
||||
if n == 0 { return "{\"ok\":true,\"accepted\":0,\"skipped\":0}" }
|
||||
let ec0: Int = engram_edge_count()
|
||||
let i: Int = 0
|
||||
let accepted: Int = 0
|
||||
let skipped: Int = 0
|
||||
@@ -299,7 +371,7 @@ fn route_create_edges_batch(method: String, path: String, body: String) -> Strin
|
||||
// Skip it when nothing was accepted: an all-malformed payload must not
|
||||
// trigger a 60MB write.
|
||||
if accepted > 0 {
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_hebb_batch(ec0)
|
||||
}
|
||||
return "{\"ok\":true,\"accepted\":" + int_to_str(accepted) + ",\"skipped\":" + int_to_str(skipped) + "}"
|
||||
}
|
||||
@@ -315,22 +387,50 @@ fn route_strengthen(method: String, path: String, body: String) -> String {
|
||||
let id: String = json_get_string(body, "node_id")
|
||||
if str_eq(id, "") { return err_json("missing node_id") }
|
||||
engram_strengthen(id)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
ok_json()
|
||||
}
|
||||
|
||||
// route_forget — DELETE /api/nodes/:id — INTEGRITY HARDENED (design doc §18.1).
|
||||
//
|
||||
// Two invariants now enforced AT THE STORE (not one layer up in neuron-api.el,
|
||||
// which a direct HTTP client could bypass):
|
||||
// 1. Write-protection: protected identity/value nodes (derived from the self
|
||||
// graph — self root + values hub + their neighbors, §18.3) cannot be
|
||||
// deleted over HTTP. Returns 403, node untouched.
|
||||
// 2. No hard delete over the wire, ever: an ordinary delete creates a
|
||||
// Tombstone marker node + `tombstones` edge and KEEPS the original node
|
||||
// and its edges (recoverable), instead of the old destructive
|
||||
// engram_forget() shift-delete. Raw engram_forget is now internal-GC only
|
||||
// and no longer reachable from any HTTP route.
|
||||
fn route_forget(method: String, path: String, body: String) -> String {
|
||||
let id: String = extract_id(path, "/api/nodes/")
|
||||
if str_eq(id, "") { return err_json("missing id") }
|
||||
engram_forget(id)
|
||||
let saved: Int = persist_canonical()
|
||||
ok_json()
|
||||
if engram_is_protected(id) == 1 {
|
||||
return "{\"__status__\":403,\"error\":\"protected node; deletion refused\",\"id\":\"" + id + "\"}"
|
||||
}
|
||||
let tomb_id: String = engram_node_full(
|
||||
"tombstone:" + id, "Tombstone", "tombstone:" + id,
|
||||
0.1, 0.1, 1.0, "Episodic", "[\"tombstone\"]"
|
||||
)
|
||||
let ec0: Int = engram_edge_count()
|
||||
engram_connect(tomb_id, id, 1.0, "tombstones")
|
||||
let saved: Int = if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_node_put(d, tomb_id)
|
||||
let b: Int = engram_wal_edges_since(d, ec0)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
a
|
||||
} else {
|
||||
persist_canonical()
|
||||
}
|
||||
"{\"ok\":true,\"tombstoned\":\"" + id + "\",\"tombstone_id\":\"" + tomb_id + "\"}"
|
||||
}
|
||||
|
||||
fn route_save(method: String, path: String, body: String) -> String {
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
// (2026-08-10 self-review) engram_save returns 0 on an empty path and the
|
||||
// route discarded it, so the response was a literal "ok":true regardless
|
||||
@@ -346,7 +446,7 @@ fn route_save(method: String, path: String, body: String) -> String {
|
||||
fn route_load(method: String, path: String, body: String) -> String {
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
// (2026-08-10 self-review) This was a stub response over the single most
|
||||
// destructive operation in the server. engram_load returns 0 on an empty
|
||||
@@ -398,7 +498,7 @@ fn route_embed_backfill(method: String, path: String, body: String) -> String {
|
||||
let result: String = engram_embed_backfill(n)
|
||||
let done: Float = json_get_float(result, "embedded")
|
||||
if done > 0.0 {
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_bulk()
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -417,7 +517,7 @@ fn route_embed_backfill(method: String, path: String, body: String) -> String {
|
||||
// (2026-06-27 self-review: added this route to fix silent 10-min sync failures)
|
||||
fn route_sync(method: String, path: String, body: String) -> String {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
// 2026-07-21 self-review: export to a scratch path, never the canonical
|
||||
// snapshot.json — read routes must not be able to clobber the good snapshot.
|
||||
let snap_path: String = dir + "/.sync-export.json"
|
||||
@@ -451,7 +551,7 @@ fn route_load_merge(method: String, path: String, body: String) -> String {
|
||||
engram_load_merge(p)
|
||||
let added_n: Int = engram_node_count() - before_n
|
||||
let added_e: Int = engram_edge_count() - before_e
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_bulk()
|
||||
"{\"ok\":true,\"nodes_added\":" + int_to_str(added_n) + ",\"edges_added\":" + int_to_str(added_e) + ",\"node_count\":" + int_to_str(engram_node_count()) + "}"
|
||||
}
|
||||
|
||||
@@ -550,7 +650,7 @@ fn route_capture_knowledge(method: String, path: String, body: String) -> String
|
||||
sal, imp, conf,
|
||||
"Semantic", tags
|
||||
)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
"{\"ok\":true,\"id\":\"" + id + "\"}"
|
||||
}
|
||||
|
||||
@@ -713,23 +813,44 @@ let bind_str: String = if str_eq(bind_raw, "") { ":8742" } else { bind_raw }
|
||||
let port: Int = parse_port(bind_str)
|
||||
|
||||
// On startup, try to load any existing snapshot (best effort).
|
||||
let data_dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let data_dir: String = if str_eq(data_dir_raw, "") { "/tmp/engram" } else { data_dir_raw }
|
||||
// §18.2: resolve the data dir safely — unset ENGRAM_DATA_DIR → $HOME/.neuron/engram,
|
||||
// never /tmp; fail loud if HOME is unresolvable (engram_resolve_data_dir exits).
|
||||
let data_dir: String = engram_resolve_data_dir()
|
||||
let snapshot_path: String = data_dir + "/snapshot.json"
|
||||
engram_load(snapshot_path)
|
||||
// ENGRAM_STORE (tiered paged store — engram-tiered-storage-engine.md). When set,
|
||||
// the durable owner is the paged store (neuron.egm + neuron.wal): engram_store_boot
|
||||
// imports snapshot.json ONCE into a fresh neuron.egm, else replays the WAL and loads
|
||||
// the store resident — snapshot.json is never read again as the ongoing store. This
|
||||
// closes the "restart reverted to a 17h-old snapshot" data-loss window. Flag-off
|
||||
// (default): byte-for-byte the historical snapshot + optional-WAL boot below.
|
||||
if store_on() {
|
||||
engram_store_boot(data_dir)
|
||||
println("[engram] ENGRAM_STORE enabled — tiered paged store is the durable owner")
|
||||
} else {
|
||||
engram_load(snapshot_path)
|
||||
|
||||
// 2026-07-21 self-review boot guard: if the snapshot file has content but the
|
||||
// load produced 0 nodes, something is wrong (corrupt file / parse failure).
|
||||
// Preserve the evidence and warn loudly — and since read routes no longer write
|
||||
// the canonical path, a bad boot can no longer clobber the good snapshot.
|
||||
let boot_snap: String = fs_read(snapshot_path)
|
||||
if !str_eq(boot_snap, "") {
|
||||
if engram_node_count() == 0 {
|
||||
println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json")
|
||||
fs_write(data_dir + "/snapshot.failed-load.json", boot_snap)
|
||||
} else {
|
||||
// Good load: keep a boot-time backup of the snapshot as loaded.
|
||||
fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap)
|
||||
// WAL replay (design doc §6). Gated: default OFF is byte-identical to legacy
|
||||
// snapshot-only boot. When ON, the snapshot above is the compaction BASE and
|
||||
// the WAL carries every mutation since; replay reconstructs state to the last
|
||||
// CRC-valid record, then opens the WAL for appending.
|
||||
if wal_on() {
|
||||
let replayed: Int = engram_wal_boot(data_dir)
|
||||
println("[engram] WAL enabled — replayed " + int_to_str(replayed) + " records")
|
||||
}
|
||||
|
||||
// 2026-07-21 self-review boot guard: if the snapshot file has content but the
|
||||
// load produced 0 nodes, something is wrong (corrupt file / parse failure).
|
||||
// Preserve the evidence and warn loudly — and since read routes no longer write
|
||||
// the canonical path, a bad boot can no longer clobber the good snapshot.
|
||||
let boot_snap: String = fs_read(snapshot_path)
|
||||
if !str_eq(boot_snap, "") {
|
||||
if engram_node_count() == 0 {
|
||||
println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json")
|
||||
fs_write(data_dir + "/snapshot.failed-load.json", boot_snap)
|
||||
} else {
|
||||
// Good load: keep a boot-time backup of the snapshot as loaded.
|
||||
fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Executable
+158
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env bash
|
||||
# M3.5 PRE-FLIP GATE. Pure C harness (NOT elb/elc): links the real el_runtime.c
|
||||
# native engram builtins + engram_store.c and proves activation-time field
|
||||
# mutations (edge hebb, node activation_count, WM weight) persist through a
|
||||
# checkpoint and survive a reboot from neuron.egm with snapshot.json DELETED.
|
||||
# Writes ONLY under a throwaway /tmp dir with a throwaway HOME.
|
||||
set -u
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
RT="$HERE/../../lang/runtime/el_runtime.c"
|
||||
ST="$HERE/../../lang/runtime/engram_store.c"
|
||||
INC="$HERE/../../lang/runtime"
|
||||
WORK="$(mktemp -d /tmp/engram-m35-XXXXXX)"
|
||||
BIN="$WORK/m35"
|
||||
export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron
|
||||
export ENGRAM_WAL_SYNC=always
|
||||
unset ENGRAM_STORE
|
||||
fail=0
|
||||
|
||||
echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m35_hebb_persist.c) =="
|
||||
gcc -O1 -std=c11 -I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log"
|
||||
if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi
|
||||
|
||||
echo
|
||||
echo "== 0) flag-OFF: seed+activate+checkpoint must NOT touch the store =="
|
||||
DOFF="$WORK/off"; mkdir -p "$DOFF"
|
||||
( unset ENGRAM_STORE; "$BIN" offcheck "$DOFF" )
|
||||
[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; }
|
||||
[ -e "$DOFF/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \
|
||||
|| echo " ok: no neuron.egm created with flag OFF"
|
||||
|
||||
echo
|
||||
echo "== 1) POSITIVE: ENGRAM_STORE=1 seed -> activate -> checkpoint(field-persist) -> close =="
|
||||
DPOS="$WORK/pos"; mkdir -p "$DPOS"
|
||||
ENGRAM_STORE=1 "$BIN" pos_seed "$DPOS" || { echo "FAIL: pos_seed"; fail=1; }
|
||||
[ -e "$DPOS/neuron.egm" ] && echo " ok: neuron.egm created" || { echo "FAIL: neuron.egm missing"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 2) reboot from neuron.egm with snapshot.json DELETED (must never read JSON) =="
|
||||
rm -f "$DPOS/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" pos_reboot "$DPOS" || { echo "FAIL: pos_reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 3) NEGATIVE CONTROL: seed -> activate -> close WITHOUT the field-persist checkpoint =="
|
||||
DNEG="$WORK/neg"; mkdir -p "$DNEG"
|
||||
ENGRAM_STORE=1 "$BIN" neg_seed "$DNEG" || { echo "FAIL: neg_seed"; fail=1; }
|
||||
rm -f "$DNEG/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" neg_reboot "$DNEG" || { echo "FAIL: neg_reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 4) assertions (python over the JSON exports) =="
|
||||
python3 - "$DPOS" "$DNEG" <<'PY'
|
||||
import json, sys, os
|
||||
WM_FLOOR = 0.05
|
||||
HEBB_MIN = 1e-6
|
||||
|
||||
def load(d, name):
|
||||
with open(os.path.join(d, name)) as f: return json.load(f)
|
||||
|
||||
def node_by_label(g, label):
|
||||
for n in g["nodes"]:
|
||||
if n.get("label") == label: return n
|
||||
return None
|
||||
|
||||
def edge_between(g, a_id, b_id):
|
||||
for e in g["edges"]:
|
||||
if e.get("from_id") == a_id and e.get("to_id") == b_id:
|
||||
return e
|
||||
return None
|
||||
|
||||
rc = 0
|
||||
def check(cond, msg):
|
||||
global rc
|
||||
if cond: print(f" PASS: {msg}")
|
||||
else: print(f" FAIL: {msg}"); rc = 1
|
||||
|
||||
dpos, dneg = sys.argv[1], sys.argv[2]
|
||||
pre = load(dpos, "pre_reboot.json")
|
||||
rebt = load(dpos, "reboot.json")
|
||||
|
||||
pa, pb = node_by_label(pre, "hebb-a"), node_by_label(pre, "hebb-b")
|
||||
ra = node_by_label(rebt, "hebb-a")
|
||||
assert pa and pb and ra, "target nodes missing"
|
||||
pe = edge_between(pre, pa["id"], pb["id"])
|
||||
re = edge_between(rebt, pa["id"], pb["id"])
|
||||
assert pe and re, "target edge missing"
|
||||
|
||||
pre_hebb = pe.get("hebb", 0.0)
|
||||
rebt_hebb = re.get("hebb", 0.0)
|
||||
pre_ac = pa.get("activation_count", 0)
|
||||
rebt_ac = ra.get("activation_count", 0)
|
||||
pre_wm = pa.get("working_memory_weight", 0.0)
|
||||
rebt_wm = ra.get("working_memory_weight", 0.0)
|
||||
|
||||
print(f" edge hebb-a->hebb-b : pre={pre_hebb!r} reboot={rebt_hebb!r}")
|
||||
print(f" node hebb-a act_cnt : pre={pre_ac!r} reboot={rebt_ac!r}")
|
||||
print(f" node hebb-a wm : pre={pre_wm!r} reboot={rebt_wm!r} (halved+floored expected)")
|
||||
|
||||
# --- learning actually happened this run (else the test proves nothing) ---
|
||||
check(pre_hebb > HEBB_MIN, f"activation raised edge hebb above 0 (pre={pre_hebb})")
|
||||
check(pre_ac >= 1, f"activation reinforced node activation_count (pre={pre_ac})")
|
||||
check(pre_wm > 0.0, f"activation promoted node to working memory (pre_wm={pre_wm})")
|
||||
|
||||
# --- the load-bearing survival assertions after a real delete-JSON reboot ---
|
||||
check(abs(rebt_hebb - pre_hebb) < 1e-12,
|
||||
f"edge hebb SURVIVED reboot unchanged ({rebt_hebb} == {pre_hebb})")
|
||||
check(rebt_ac == pre_ac,
|
||||
f"node activation_count SURVIVED reboot unchanged ({rebt_ac} == {pre_ac})")
|
||||
|
||||
# --- WM weight: must equal the JSON path's boot transform exactly (halve+floor) ---
|
||||
expected_wm = pre_wm * 0.5
|
||||
if expected_wm < WM_FLOOR: expected_wm = 0.0
|
||||
check(abs(rebt_wm - expected_wm) < 1e-9,
|
||||
f"node WM weight SURVIVED with the SAME boot transform as JSON path "
|
||||
f"(reboot={rebt_wm} == halve+floor(pre)={expected_wm})")
|
||||
check(expected_wm > 0.0,
|
||||
f"WM survival is observable (halved weight stays above floor: {expected_wm} > {WM_FLOOR})")
|
||||
|
||||
# --- NEGATIVE CONTROL: without the field-persist step the learning is LOST ---
|
||||
npre = load(dneg, "neg_pre.json")
|
||||
nrebt = load(dneg, "neg_reboot.json")
|
||||
na_pre = node_by_label(npre, "hebb-a")
|
||||
na_rebt = node_by_label(nrebt, "hebb-a")
|
||||
ne_pre = edge_between(npre, na_pre["id"], node_by_label(npre, "hebb-b")["id"])
|
||||
ne_rebt = edge_between(nrebt, na_rebt["id"], node_by_label(nrebt, "hebb-b")["id"])
|
||||
print(f" [neg] edge hebb : pre={ne_pre.get('hebb',0.0)!r} reboot={ne_rebt.get('hebb',0.0)!r}")
|
||||
print(f" [neg] node act_cnt : pre={na_pre.get('activation_count',0)!r} reboot={na_rebt.get('activation_count',0)!r}")
|
||||
check(ne_pre.get("hebb", 0.0) > HEBB_MIN,
|
||||
f"[neg] activation DID raise hebb in RAM (pre={ne_pre.get('hebb',0.0)})")
|
||||
check(ne_rebt.get("hebb", 0.0) == 0.0,
|
||||
"[neg] WITHOUT checkpoint field-persist, edge hebb is LOST on reboot (==0) — fix is load-bearing")
|
||||
check(na_rebt.get("activation_count", 0) == 0,
|
||||
"[neg] WITHOUT checkpoint field-persist, activation_count is LOST on reboot (==0)")
|
||||
|
||||
sys.exit(rc)
|
||||
PY
|
||||
[ $? -ne 0 ] && fail=1
|
||||
|
||||
echo
|
||||
echo "== 5) ASan+UBSan build, exercise the full persist+reboot flow (leaks off — harness intentionally leaks el_strdup) =="
|
||||
SANBIN="$WORK/m35.san"
|
||||
gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \
|
||||
-I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log"
|
||||
if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else
|
||||
export ASAN_OPTIONS=detect_leaks=0
|
||||
DSAN="$WORK/san"; mkdir -p "$DSAN"
|
||||
ENGRAM_STORE=1 "$SANBIN" pos_seed "$DSAN" >/dev/null 2>"$WORK/san_run.log" && \
|
||||
{ rm -f "$DSAN/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" pos_reboot "$DSAN" >/dev/null 2>>"$WORK/san_run.log"; }
|
||||
if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then
|
||||
echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1
|
||||
else
|
||||
echo " ok: ASan+UBSan clean across pos_seed/checkpoint/reboot (field-persist, boot laundering)"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ "$fail" -eq 0 ]; then echo "================ M3.5 HEBB-PERSIST GATE: PASS ================"; else echo "================ M3.5 HEBB-PERSIST GATE: FAIL ================"; fi
|
||||
rm -rf "$WORK"
|
||||
exit $fail
|
||||
Executable
+126
@@ -0,0 +1,126 @@
|
||||
#!/usr/bin/env bash
|
||||
# M3 JSON-parity gate. Pure C harness (NOT elb/elc): links the real el_runtime.c
|
||||
# native engram builtins + engram_store.c and drives ENGRAM_STORE on vs off.
|
||||
# Writes ONLY under a throwaway /tmp dir with a throwaway HOME + ENGRAM_DATA_DIR.
|
||||
set -u
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
RT="$HERE/../../lang/runtime/el_runtime.c"
|
||||
ST="$HERE/../../lang/runtime/engram_store.c"
|
||||
INC="$HERE/../../lang/runtime"
|
||||
WORK="$(mktemp -d /tmp/engram-m3-XXXXXX)"
|
||||
DATA="$WORK/data"; mkdir -p "$DATA"
|
||||
BIN="$WORK/m3"
|
||||
export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron
|
||||
export ENGRAM_DATA_DIR="$DATA"
|
||||
export ENGRAM_WAL_SYNC=always
|
||||
unset ENGRAM_STORE
|
||||
fail=0
|
||||
|
||||
echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m3_parity.c) =="
|
||||
gcc -O1 -std=c11 -I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log"
|
||||
if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi
|
||||
grep -i warning "$WORK/cc.log" | grep -iE 'engram_store|eg_store|eg_load|scan_nodes|scan_edges' && echo "(warnings in M3 code above)" || true
|
||||
|
||||
echo
|
||||
echo "== 0) default-OFF: flag unset leaves the store untouched =="
|
||||
( unset ENGRAM_STORE; "$BIN" offcheck "$DATA" )
|
||||
[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; }
|
||||
[ -e "$DATA/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \
|
||||
|| echo " ok: no neuron.egm created with flag OFF"
|
||||
|
||||
echo
|
||||
echo "== 1) seed (ENGRAM_STORE unset): build graph, save snapshot.json, activate =="
|
||||
( unset ENGRAM_STORE; "$BIN" seed "$DATA" ) || { echo "FAIL: seed"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 2) on (ENGRAM_STORE=1): import snapshot.json ONCE -> neuron.egm, resident-load, activate =="
|
||||
ENGRAM_STORE=1 "$BIN" on "$DATA" || { echo "FAIL: on"; fail=1; }
|
||||
[ -e "$DATA/neuron.egm" ] && echo " ok: neuron.egm created by import" || { echo "FAIL: neuron.egm missing"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 3) reboot (ENGRAM_STORE=1, snapshot.json DELETED): must load from neuron.egm, never JSON =="
|
||||
rm -f "$DATA/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" reboot "$DATA" || { echo "FAIL: reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 4) parity comparison (modulo ordering) =="
|
||||
python3 - "$DATA" <<'PY'
|
||||
import json, sys, os
|
||||
d = sys.argv[1]
|
||||
def load(name):
|
||||
with open(os.path.join(d, name)) as f: return json.load(f)
|
||||
def norm_graph(g):
|
||||
nodes = sorted(g.get("nodes", []), key=lambda n: n.get("id",""))
|
||||
edges = sorted(g.get("edges", []), key=lambda e: e.get("id",""))
|
||||
layers= sorted(g.get("layers", []), key=lambda l: l.get("layer_id",0))
|
||||
return {"nodes":nodes, "edges":edges, "layers":layers}
|
||||
def act_ids(a):
|
||||
# list of (node id, promoted); robust set + ordered list
|
||||
seq = [(e.get("node",{}).get("id",""), int(e.get("promoted",0))) for e in a]
|
||||
return seq
|
||||
|
||||
rc = 0
|
||||
snap = norm_graph(load("snapshot.json") if os.path.exists(os.path.join(d,"snapshot.json")) else load("off_graph.json"))
|
||||
off = norm_graph(load("off_graph.json"))
|
||||
on = norm_graph(load("on_graph.json"))
|
||||
rebt = norm_graph(load("reboot_graph.json"))
|
||||
|
||||
def cmp(label, a, b):
|
||||
global rc
|
||||
if a == b:
|
||||
print(f" PASS: {label} (nodes={len(a['nodes'])} edges={len(a['edges'])} layers={len(a['layers'])})")
|
||||
else:
|
||||
rc = 1
|
||||
print(f" FAIL: {label}")
|
||||
for k in ("nodes","edges","layers"):
|
||||
if a[k] != b[k]:
|
||||
print(f" {k}: {len(a[k])} vs {len(b[k])}")
|
||||
for x,y in zip(a[k], b[k]):
|
||||
if x != y:
|
||||
print(f" first diff:\n A={json.dumps(x)[:300]}\n B={json.dumps(y)[:300]}")
|
||||
break
|
||||
|
||||
cmp("graph: ENGRAM_STORE=1 (export) == ENGRAM_STORE=0 (JSON path)", on, off)
|
||||
cmp("round-trip: snapshot.json seed == store export (on_graph)", on, off) # off_graph==snapshot save
|
||||
cmp("reboot from neuron.egm (no JSON) == on-path store", rebt, on)
|
||||
|
||||
offa = act_ids(load("off_act.json"))
|
||||
ona = act_ids(load("on_act.json"))
|
||||
if set(offa) == set(ona):
|
||||
print(f" PASS: activation result set identical (off={len(offa)} on={len(ona)} entries)")
|
||||
if offa == ona:
|
||||
print(" (and identical ordering/promotion sequence)")
|
||||
else:
|
||||
print(" (same set; ordering differs only where scores tie — reporting honestly)")
|
||||
else:
|
||||
rc = 1
|
||||
print(" FAIL: activation result set differs")
|
||||
print(f" off-only: {set(offa)-set(ona)}")
|
||||
print(f" on-only: {set(ona)-set(offa)}")
|
||||
|
||||
sys.exit(rc)
|
||||
PY
|
||||
[ $? -ne 0 ] && fail=1
|
||||
|
||||
echo
|
||||
echo "== 5) ASan+UBSan build, exercise M3 scan/boot/hooks (leaks off — harness intentionally leaks el_strdup) =="
|
||||
SANBIN="$WORK/m3.san"
|
||||
gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \
|
||||
-I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log"
|
||||
if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else
|
||||
export ASAN_OPTIONS=detect_leaks=0
|
||||
DATA2="$WORK/data2"; mkdir -p "$DATA2"
|
||||
( unset ENGRAM_STORE; "$SANBIN" seed "$DATA2" ) >/dev/null 2>"$WORK/san_run.log" && \
|
||||
ENGRAM_STORE=1 "$SANBIN" on "$DATA2" >/dev/null 2>>"$WORK/san_run.log" && \
|
||||
{ rm -f "$DATA2/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" reboot "$DATA2" >/dev/null 2>>"$WORK/san_run.log"; }
|
||||
if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then
|
||||
echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1
|
||||
else
|
||||
echo " ok: ASan+UBSan clean across seed/on/reboot (scan, boot, resident-load, mutation hooks)"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ "$fail" -eq 0 ]; then echo "================ M3 PARITY GATE: PASS ================"; else echo "================ M3 PARITY GATE: FAIL ================"; fi
|
||||
rm -rf "$WORK"
|
||||
exit $fail
|
||||
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env bash
|
||||
# M1 paged-store gate. Pure C (NOT elb/elc). Writes only under /tmp.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
SRC="$HERE/../../lang/runtime/engram_store.c"
|
||||
BIN="/tmp/test_store.$$"
|
||||
echo "compiling: gcc test_store.c engram_store.c"
|
||||
gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_store.c" "$SRC" -o "$BIN"
|
||||
"$BIN"
|
||||
rc=$?
|
||||
rm -f "$BIN"
|
||||
rm -rf /tmp/engram-store-test-*
|
||||
exit $rc
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env bash
|
||||
# M2 WAL + checkpoint + recovery gate. Pure C (NOT elb/elc). Writes only under /tmp.
|
||||
# Recovery tests use ENGRAM_WAL_SYNC=always so every WAL record is durable at crash.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
SRC="$HERE/../../lang/runtime/engram_store.c"
|
||||
BIN="/tmp/test_wal_store.$$"
|
||||
echo "compiling: gcc test_wal_store.c engram_store.c"
|
||||
gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_wal_store.c" "$SRC" -o "$BIN"
|
||||
ENGRAM_WAL_SYNC=always "$BIN"
|
||||
rc=$?
|
||||
rm -f "$BIN"
|
||||
rm -rf /tmp/engram-wal-test-*
|
||||
exit $rc
|
||||
Executable
+16
@@ -0,0 +1,16 @@
|
||||
#!/usr/bin/env bash
|
||||
# WAL unit + integration + crash-fuzz gate. Throwaway HOME/dirs only.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
REL="$HERE/../../lang/runtime"
|
||||
cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" \
|
||||
"$HERE/test_wal.c" -lcurl -lpthread -o /tmp/test_wal
|
||||
HOME=/tmp/engram-throwaway-home /tmp/test_wal
|
||||
# Fail-loud data-dir check (must exit 1 with a FATAL line):
|
||||
cat > /tmp/test_failloud.c <<'C'
|
||||
#include "el_runtime.c"
|
||||
int main(void){ unsetenv("ENGRAM_DATA_DIR"); unsetenv("HOME");
|
||||
engram_resolve_data_dir(); printf("REACHED\n"); return 0; }
|
||||
C
|
||||
cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" /tmp/test_failloud.c -lcurl -lpthread -o /tmp/test_failloud
|
||||
if env -u HOME -u ENGRAM_DATA_DIR /tmp/test_failloud; then echo "FAIL: should have exited"; exit 1; else echo "[PASS] fail-loud exit on unresolvable HOME"; fi
|
||||
@@ -0,0 +1,130 @@
|
||||
/* test_m35_hebb_persist.c — M3.5 PRE-FLIP GATE.
|
||||
*
|
||||
* Proves that in-place field mutations made during spreading activation — edge
|
||||
* `hebb` (+ last_fired), node `activation_count`, node working-memory weight —
|
||||
* PERSIST to the paged store and survive a restart from neuron.egm with
|
||||
* snapshot.json deleted. This is the "hebb-survives-restart" fix that gates the
|
||||
* live cutover.
|
||||
*
|
||||
* Same style as test_m3_parity.c: a REAL el-level harness linking the actual
|
||||
* el_runtime.c native engram builtins + engram_store.c, driving engram_node_full
|
||||
* / engram_connect / engram_activate_json / engram_save / engram_store_boot /
|
||||
* engram_store_checkpoint / engram_store_close directly from C. No EL interpreter.
|
||||
*
|
||||
* Modes (argv[1]), data dir (argv[2]):
|
||||
* pos_seed — ENGRAM_STORE=1: fresh store, seed a graph tuned so activation
|
||||
* co-activates a connected pair (edge hebb 0 -> ETA) and reinforces
|
||||
* nodes (activation_count 0 -> >=1, WM weight -> >0). Export the
|
||||
* post-activation resident graph to pre_reboot.json, then CHECKPOINT
|
||||
* (the M3.5 field-persist), then close.
|
||||
* pos_reboot— ENGRAM_STORE=1, snapshot.json deleted by runner: boot from
|
||||
* neuron.egm (WAL replay), export reboot.json, close. The values in
|
||||
* reboot.json are what actually survived the round-trip.
|
||||
* neg_seed — identical to pos_seed but WITHOUT the checkpoint field-persist
|
||||
* (negative control): activation mutations never reach the store.
|
||||
* neg_reboot— boot from neuron.egm, export neg_reboot.json, close.
|
||||
* offcheck — ENGRAM_STORE unset: seed+activate+checkpoint must NOT touch the
|
||||
* store (no neuron.egm, checkpoint returns 0).
|
||||
*
|
||||
* The pass/fail assertions live in run_m35_hebb_persist.sh (python over the JSON
|
||||
* exports): reboot.json must carry the learned hebb / activation_count and the
|
||||
* JSON-identical halved WM weight; neg_reboot.json must have LOST them.
|
||||
*/
|
||||
#include "el_runtime.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
extern int engram_store_enabled(void);
|
||||
extern el_val_t engram_store_boot(el_val_t data_dir);
|
||||
extern el_val_t engram_store_checkpoint(void);
|
||||
extern el_val_t engram_store_close(void);
|
||||
|
||||
static el_val_t S(const char* s){ return EL_STR(s); }
|
||||
static el_val_t F(double d){ return el_from_float(d); }
|
||||
|
||||
/* Two nodes with DISTINCT content (so the redundancy-suppression pass cannot
|
||||
* dedup one of them away) that both match the query strongly, wired by one
|
||||
* "associate" edge. A handful of weakly-related distractors make it a real
|
||||
* graph. On activation both A and B promote to working memory and co-activate,
|
||||
* so their edge's hebb rises from 0 to ENGRAM_HEBB_ETA. */
|
||||
static void build_seed(void){
|
||||
el_val_t a = engram_node_full(S("hebbian potentiation strengthens co-active memory links"),
|
||||
S("Concept"), S("hebb-a"), F(0.9), F(0.85), F(1.0), S("Semantic"),
|
||||
S("hebbian,memory,activation"));
|
||||
el_val_t b = engram_node_full(S("co-active memory links accrue hebbian associative weight"),
|
||||
S("Concept"), S("hebb-b"), F(0.9), F(0.85), F(1.0), S("Semantic"),
|
||||
S("hebbian,memory,weight"));
|
||||
el_val_t c = engram_node_full(S("unrelated culinary recipe for sourdough bread"),
|
||||
S("Fact"), S("distractor-1"), F(0.4), F(0.4), F(1.0), S("Semantic"),
|
||||
S("food"));
|
||||
el_val_t d = engram_node_full(S("the weather forecast predicts rain tomorrow afternoon"),
|
||||
S("Fact"), S("distractor-2"), F(0.4), F(0.4), F(1.0), S("Semantic"),
|
||||
S("weather"));
|
||||
engram_connect(a, b, F(0.8), S("associate")); /* the edge under test */
|
||||
engram_connect(a, c, F(0.3), S("associate"));
|
||||
engram_connect(b, d, F(0.3), S("associate"));
|
||||
}
|
||||
|
||||
static const char* QUERY =
|
||||
"hebbian potentiation co-active memory links associative weight";
|
||||
|
||||
static void export_graph(const char* dir, const char* name){
|
||||
char p[1024];
|
||||
snprintf(p, sizeof p, "%s/%s", dir, name);
|
||||
if (!engram_save(S(p))){ fprintf(stderr, "save %s failed\n", name); exit(2); }
|
||||
}
|
||||
|
||||
int main(int argc, char** argv){
|
||||
if (argc < 3){
|
||||
fprintf(stderr, "usage: %s <pos_seed|pos_reboot|neg_seed|neg_reboot|offcheck> <dir>\n", argv[0]);
|
||||
return 2;
|
||||
}
|
||||
const char* mode = argv[1];
|
||||
const char* dir = argv[2];
|
||||
|
||||
if (!strcmp(mode, "pos_seed") || !strcmp(mode, "neg_seed")){
|
||||
int persist = !strcmp(mode, "pos_seed");
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; }
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; }
|
||||
build_seed();
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
(void)act;
|
||||
/* Capture the post-activation resident state BEFORE persisting/closing. */
|
||||
export_graph(dir, persist ? "pre_reboot.json" : "neg_pre.json");
|
||||
printf("[%s] nodes=%lld edges=%lld\n", mode,
|
||||
(long long)(int64_t)engram_node_count(),
|
||||
(long long)(int64_t)engram_edge_count());
|
||||
if (persist){
|
||||
if (!engram_store_checkpoint()){ fprintf(stderr, "checkpoint failed\n"); return 2; }
|
||||
}
|
||||
/* neg mode: NO field-persist checkpoint. engram_store_close still flushes
|
||||
* pages, but no store_put_* ran post-creation, so the store keeps the
|
||||
* pristine creation-time field values (hebb=0, activation_count=0). */
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "pos_reboot") || !strcmp(mode, "neg_reboot")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; }
|
||||
/* snapshot.json deleted by the runner — boot MUST come from neuron.egm. */
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; }
|
||||
export_graph(dir, !strcmp(mode, "pos_reboot") ? "reboot.json" : "neg_reboot.json");
|
||||
printf("[%s] nodes=%lld edges=%lld\n", mode,
|
||||
(long long)(int64_t)engram_node_count(),
|
||||
(long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "offcheck")){
|
||||
int en = engram_store_enabled();
|
||||
el_val_t boot = engram_store_boot(S(dir)); /* no-op with flag off */
|
||||
build_seed();
|
||||
engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
el_val_t ck = engram_store_checkpoint(); /* must be a no-op */
|
||||
printf("[offcheck] enabled=%d boot=%lld checkpoint=%lld\n",
|
||||
en, (long long)(int64_t)boot, (long long)(int64_t)ck);
|
||||
return (en == 0 && (int64_t)boot == 0 && (int64_t)ck == 0) ? 0 : 1;
|
||||
}
|
||||
fprintf(stderr, "unknown mode %s\n", mode);
|
||||
return 2;
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
/* test_m3_parity.c — M3 JSON-parity gate for the ENGRAM_STORE wiring.
|
||||
*
|
||||
* This is a REAL el-level harness: it links the actual el_runtime.o (the soul's
|
||||
* native engram builtins) + engram_store.o and calls the engram_node family plus
|
||||
* engram_connect, engram_activate_json, engram_save, engram_store_boot directly. No EL interpreter
|
||||
* and no full soul build are needed — el_runtime.c compiles to a standalone .o
|
||||
* whose engram builtins operate on the process-global engram store, and the
|
||||
* string arena is inert unless el_request_start() is called, so the builtins are
|
||||
* callable straight from C (el_val_t is int64_t; EL_STR/EL_CSTR are pointer casts).
|
||||
*
|
||||
* Modes (argv[1]), data dir (argv[2]):
|
||||
* seed — ENGRAM_STORE unset: build a fixed seed graph, write snapshot.json +
|
||||
* off_graph.json (pristine, pre-activation), then activate → off_act.json.
|
||||
* on — ENGRAM_STORE=1: engram_store_boot(dir) imports snapshot.json ONCE into
|
||||
* neuron.egm and loads it resident; write on_graph.json, then activate →
|
||||
* on_act.json; checkpoint + close.
|
||||
* reboot — ENGRAM_STORE=1 with snapshot.json DELETED: boot must reload from
|
||||
* neuron.egm (WAL replay), never re-reading JSON; write reboot_graph.json.
|
||||
* offcheck — assert flag-off leaves the store untouched.
|
||||
*
|
||||
* The graph comparison (done by run_m3_parity.sh via python, modulo ordering) is
|
||||
* the deterministic gate; activation ids/promoted are compared as a robust set.
|
||||
*/
|
||||
#include "el_runtime.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* Builtins the header declares are pulled in via el_runtime.h. The M3 additions
|
||||
* are not in the header yet, so declare them here. */
|
||||
extern int engram_store_enabled(void);
|
||||
extern el_val_t engram_store_boot(el_val_t data_dir);
|
||||
extern el_val_t engram_store_checkpoint(void);
|
||||
extern el_val_t engram_store_close(void);
|
||||
extern el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
|
||||
static el_val_t S(const char* s){ return EL_STR(s); }
|
||||
static el_val_t F(double d){ return el_from_float(d); }
|
||||
|
||||
/* Build a fixed, deterministic seed graph: 12 nodes across two layers + 9 edges.
|
||||
* Content is chosen so an activation query has real matches to rank. */
|
||||
static void build_seed(void){
|
||||
/* core-identity layer (1) via engram_node_full */
|
||||
el_val_t n0 = engram_node_full(S("tiered storage engine design"), S("Concept"),
|
||||
S("storage-engine"), F(0.9), F(0.8), F(1.0), S("Semantic"), S("design,storage"));
|
||||
el_val_t n1 = engram_node_full(S("write-ahead log durability"), S("Concept"),
|
||||
S("wal"), F(0.85), F(0.75), F(1.0), S("Semantic"), S("wal,durability"));
|
||||
el_val_t n2 = engram_node_full(S("paged buffer pool with checkpointing"), S("Concept"),
|
||||
S("buffer-pool"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("paging"));
|
||||
el_val_t n3 = engram_node_full(S("spreading activation over the graph"), S("Concept"),
|
||||
S("activation"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("activation,graph"));
|
||||
el_val_t n4 = engram_node_full(S("hebbian co-activation potentiation"), S("Concept"),
|
||||
S("hebbian"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("hebb"));
|
||||
el_val_t n5 = engram_node_full(S("crash recovery replays the log"), S("Concept"),
|
||||
S("recovery"), F(0.75), F(0.65), F(1.0), S("Semantic"), S("recovery,wal"));
|
||||
/* domain-knowledge layer (2) via engram_node_layered */
|
||||
el_val_t n6 = engram_node_layered(S("b-tree primary index id to location"), S("Fact"),
|
||||
S("btree"), F(0.7), F(0.6), F(1.0), S(""), S("index"), (el_val_t)2);
|
||||
el_val_t n7 = engram_node_layered(S("adjacency index for edge lookup"), S("Fact"),
|
||||
S("adjacency"), F(0.7), F(0.6), F(1.0), S(""), S("index,graph"), (el_val_t)2);
|
||||
el_val_t n8 = engram_node_layered(S("slotted pages hold tlv records"), S("Fact"),
|
||||
S("slotted-page"), F(0.65), F(0.55), F(1.0), S(""), S("format"), (el_val_t)2);
|
||||
el_val_t n9 = engram_node_full(S("memory tiers working semantic episodic"), S("Concept"),
|
||||
S("tiers"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("tiers,memory"));
|
||||
el_val_t n10 = engram_node_full(S("embeddings enable nearest neighbour search"), S("Concept"),
|
||||
S("embeddings"), F(0.65), F(0.55), F(1.0), S("Semantic"), S("embeddings"));
|
||||
el_val_t n11 = engram_node_full(S("the durable engram is the mind's memory"), S("Belief"),
|
||||
S("engram"), F(0.95), F(0.9), F(1.0), S("Semantic"), S("engram,memory"));
|
||||
|
||||
engram_connect(n0, n1, F(0.8), S("depends-on"));
|
||||
engram_connect(n0, n2, F(0.8), S("depends-on"));
|
||||
engram_connect(n0, n3, F(0.7), S("enables"));
|
||||
engram_connect(n1, n5, F(0.9), S("enables"));
|
||||
engram_connect(n3, n4, F(0.6), S("triggers"));
|
||||
engram_connect(n2, n6, F(0.7), S("uses"));
|
||||
engram_connect(n3, n7, F(0.7), S("uses"));
|
||||
engram_connect(n0, n8, F(0.6), S("uses"));
|
||||
engram_connect(n11, n9, F(0.8), S("about"));
|
||||
engram_connect(n11, n10, F(0.5), S("about"));
|
||||
}
|
||||
|
||||
static void write_file(const char* path, const char* content){
|
||||
FILE* f = fopen(path, "wb");
|
||||
if (!f){ fprintf(stderr, "cannot open %s\n", path); exit(2); }
|
||||
if (content) fwrite(content, 1, strlen(content), f);
|
||||
fclose(f);
|
||||
}
|
||||
|
||||
static const char* QUERY = "storage engine activation and the durable log";
|
||||
|
||||
int main(int argc, char** argv){
|
||||
if (argc < 3){ fprintf(stderr, "usage: %s <seed|on|reboot|offcheck> <dir>\n", argv[0]); return 2; }
|
||||
const char* mode = argv[1];
|
||||
const char* dir = argv[2];
|
||||
char p[1024];
|
||||
|
||||
if (!strcmp(mode, "seed")){
|
||||
if (engram_store_enabled()){ fprintf(stderr, "seed mode requires ENGRAM_STORE unset\n"); return 2; }
|
||||
build_seed();
|
||||
snprintf(p, sizeof p, "%s/snapshot.json", dir);
|
||||
if (!engram_save(S(p))){ fprintf(stderr, "seed save failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/off_graph.json", dir);
|
||||
engram_save(S(p)); /* pristine off-path graph */
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
snprintf(p, sizeof p, "%s/off_act.json", dir);
|
||||
write_file(p, EL_CSTR(act));
|
||||
printf("[seed] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "on")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "on mode requires ENGRAM_STORE=1\n"); return 2; }
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/on_graph.json", dir);
|
||||
engram_save(S(p)); /* export resident (== store) */
|
||||
/* Checkpoint the freshly-imported (pristine) graph — this is the state
|
||||
* the reboot comparison expects to round-trip. Under M3.5 a checkpoint
|
||||
* persists the resident graph's CURRENT field state, so it must run
|
||||
* BEFORE activation mutates fields in place; activation itself is
|
||||
* exercised below only for the activation-result-set parity check. The
|
||||
* M3.5 gate (test_m35_hebb_persist) separately proves that a checkpoint
|
||||
* taken AFTER activation durably carries the learned hebb/WM state. */
|
||||
engram_store_checkpoint();
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
snprintf(p, sizeof p, "%s/on_act.json", dir);
|
||||
write_file(p, EL_CSTR(act));
|
||||
printf("[on] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "reboot")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "reboot mode requires ENGRAM_STORE=1\n"); return 2; }
|
||||
/* snapshot.json has been deleted by the runner — boot MUST come from
|
||||
* neuron.egm (+ WAL replay), never re-reading JSON. */
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/reboot_graph.json", dir);
|
||||
engram_save(S(p));
|
||||
printf("[reboot] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "offcheck")){
|
||||
/* ENGRAM_STORE unset: enabled()==0 and boot is a no-op returning 0. */
|
||||
int en = engram_store_enabled();
|
||||
el_val_t b = engram_store_boot(S(dir));
|
||||
printf("[offcheck] enabled=%d boot_ret=%lld\n", en, (long long)(int64_t)b);
|
||||
return (en == 0 && (int64_t)b == 0) ? 0 : 1;
|
||||
}
|
||||
fprintf(stderr, "unknown mode %s\n", mode);
|
||||
return 2;
|
||||
}
|
||||
@@ -0,0 +1,439 @@
|
||||
/* test_store.c — M1 gate for the engram paged store (engram_store.{c,h}).
|
||||
*
|
||||
* Pure C. Build: gcc -O2 test_store.c ../../lang/runtime/engram_store.c -o test_store
|
||||
* Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports.
|
||||
*
|
||||
* Covers §7 M1 gates: round-trip (5k nodes / 20k edges, all fields, emb bit-exact,
|
||||
* hebb, >page content), TLV forward-compat, overflow chains, B+-tree indexes
|
||||
* across splits, free-list reuse, and corruption/superblock recovery.
|
||||
*/
|
||||
#include "../../lang/runtime/engram_store.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond){
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_dir[512];
|
||||
static void mk_dir(void){
|
||||
snprintf(g_dir, sizeof g_dir, "/tmp/engram-store-test-%d", (int)getpid());
|
||||
mkdir(g_dir, 0700);
|
||||
}
|
||||
static void path_in(char* out, size_t cap, const char* name){
|
||||
snprintf(out, cap, "%s/%s", g_dir, name);
|
||||
}
|
||||
static long file_size(const char* p){ struct stat st; return stat(p,&st)==0 ? (long)st.st_size : -1; }
|
||||
|
||||
/* ── deterministic RNG so oracle nodes/edges regenerate bit-exact ─────────── */
|
||||
static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; }
|
||||
static uint64_t node_seed(int i){ return 0x9E3779B97F4A7C15ULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL); }
|
||||
static uint64_t edge_seed(int i){ return 0xC2B2AE3D27D4EB4FULL ^ ((uint64_t)(i+1)*0x165667B19E3779F9ULL); }
|
||||
|
||||
static char* rnd_str(uint64_t* st, size_t len){
|
||||
char* s = (char*)malloc(len + 1);
|
||||
for (size_t i=0;i<len;i++) s[i] = (char)(33 + (xs(st) % 94)); /* printable, no NUL */
|
||||
s[len] = 0; return s;
|
||||
}
|
||||
|
||||
/* NODE_COUNT nodes; a slice have >page content to force overflow chains. */
|
||||
#define NODE_COUNT 5000
|
||||
#define EDGE_COUNT 20000
|
||||
#define EMB_DIM 768
|
||||
|
||||
static void gen_node(int i, StoreNode* n){
|
||||
memset(n, 0, sizeof *n);
|
||||
uint64_t st = node_seed(i);
|
||||
char id[32]; snprintf(id, sizeof id, "node-%d", i);
|
||||
n->id = strdup(id);
|
||||
size_t clen = (i % 500 == 0) ? (size_t)(17000 + (xs(&st) % 6000)) : (size_t)(xs(&st) % 300);
|
||||
n->content = rnd_str(&st, clen);
|
||||
n->node_type = rnd_str(&st, 4 + (xs(&st) % 8));
|
||||
n->label = (i % 2) ? rnd_str(&st, 3 + (xs(&st) % 10)) : NULL;
|
||||
n->tier = rnd_str(&st, 4 + (xs(&st) % 6));
|
||||
n->tags = rnd_str(&st, xs(&st) % 40);
|
||||
n->metadata = (i % 3) ? rnd_str(&st, xs(&st) % 60) : NULL;
|
||||
n->salience = (double)(xs(&st) % 1000000) / 997.0;
|
||||
n->importance = (double)(xs(&st) % 1000000) / 131.0;
|
||||
n->confidence = (double)(xs(&st) % 1000000) / 733.0;
|
||||
n->temporal_decay_rate = (double)(xs(&st) % 1000000) / 101.0;
|
||||
n->activation_count = (int64_t)(xs(&st) % 100000);
|
||||
n->last_activated = (int64_t)xs(&st);
|
||||
n->created_at = (int64_t)(1600000000000LL + i);
|
||||
n->updated_at = (int64_t)xs(&st);
|
||||
n->background_activation = (double)(xs(&st) % 1000000) / 17.0;
|
||||
n->working_memory_weight = (double)(xs(&st) % 1000000) / 29.0;
|
||||
n->suppression_count = (int32_t)(xs(&st) % 50);
|
||||
n->layer_id = (uint32_t)(xs(&st) % 5);
|
||||
for (int k=0;k<STORE_BLL_K;k++) n->access_ts[k] = (int64_t)xs(&st);
|
||||
n->access_head = (int32_t)(xs(&st) % STORE_BLL_K);
|
||||
n->access_filled = (int32_t)(xs(&st) % (STORE_BLL_K + 1));
|
||||
n->wm_anchor = (double)(xs(&st) % 1000000) / 3.0;
|
||||
n->emb = (float*)malloc(EMB_DIM * sizeof(float));
|
||||
for (int k=0;k<EMB_DIM;k++){ uint32_t u=(uint32_t)xs(&st); memcpy(&n->emb[k], &u, 4); }
|
||||
n->emb_dim = EMB_DIM;
|
||||
}
|
||||
|
||||
static void gen_edge(int i, StoreEdge* e){
|
||||
memset(e, 0, sizeof *e);
|
||||
uint64_t st = edge_seed(i);
|
||||
char id[32], from[32], to[32];
|
||||
snprintf(id, sizeof id, "edge-%d", i);
|
||||
snprintf(from, sizeof from, "node-%d", (int)(xs(&st) % NODE_COUNT));
|
||||
snprintf(to, sizeof to, "node-%d", (int)(xs(&st) % NODE_COUNT));
|
||||
e->id = strdup(id); e->from_id = strdup(from); e->to_id = strdup(to);
|
||||
e->relation = rnd_str(&st, 3 + (xs(&st) % 12));
|
||||
e->metadata = (i % 4) ? rnd_str(&st, xs(&st) % 40) : NULL;
|
||||
e->weight = (double)(xs(&st) % 1000000) / 111.0;
|
||||
e->hebb = (double)(xs(&st) % 1000000) / 1000000.0; /* the learned field */
|
||||
e->confidence = (double)(xs(&st) % 1000000) / 777.0;
|
||||
e->created_at = (int64_t)(1600000000000LL + i);
|
||||
e->updated_at = (int64_t)xs(&st);
|
||||
e->last_fired = (int64_t)xs(&st);
|
||||
e->inhibitory = (int32_t)(xs(&st) % 2);
|
||||
e->layer_id = (uint32_t)(xs(&st) % 5);
|
||||
}
|
||||
|
||||
static int streq(const char* a, const char* b){
|
||||
if (!a && !b) return 1;
|
||||
if (!a || !b) return 0;
|
||||
return strcmp(a,b)==0;
|
||||
}
|
||||
static int cmp_node(const StoreNode* a, const StoreNode* b){
|
||||
if (!streq(a->id,b->id) || !streq(a->content,b->content) ||
|
||||
!streq(a->node_type,b->node_type) || !streq(a->label,b->label) ||
|
||||
!streq(a->tier,b->tier) || !streq(a->tags,b->tags) ||
|
||||
!streq(a->metadata,b->metadata)) return 0;
|
||||
if (a->salience!=b->salience || a->importance!=b->importance ||
|
||||
a->confidence!=b->confidence || a->temporal_decay_rate!=b->temporal_decay_rate ||
|
||||
a->activation_count!=b->activation_count || a->last_activated!=b->last_activated ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->background_activation!=b->background_activation ||
|
||||
a->working_memory_weight!=b->working_memory_weight ||
|
||||
a->suppression_count!=b->suppression_count || a->layer_id!=b->layer_id ||
|
||||
a->access_head!=b->access_head || a->access_filled!=b->access_filled ||
|
||||
a->wm_anchor!=b->wm_anchor || a->emb_dim!=b->emb_dim) return 0;
|
||||
for (int k=0;k<STORE_BLL_K;k++) if (a->access_ts[k]!=b->access_ts[k]) return 0;
|
||||
if ((a->emb==NULL) != (b->emb==NULL)) return 0;
|
||||
if (a->emb && memcmp(a->emb, b->emb, (size_t)a->emb_dim*4)!=0) return 0;
|
||||
return 1;
|
||||
}
|
||||
static int cmp_edge(const StoreEdge* a, const StoreEdge* b){
|
||||
if (!streq(a->id,b->id) || !streq(a->from_id,b->from_id) || !streq(a->to_id,b->to_id) ||
|
||||
!streq(a->relation,b->relation) || !streq(a->metadata,b->metadata)) return 0;
|
||||
if (a->weight!=b->weight || a->hebb!=b->hebb || a->confidence!=b->confidence ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->last_fired!=b->last_fired || a->inhibitory!=b->inhibitory ||
|
||||
a->layer_id!=b->layer_id) return 0;
|
||||
return 1;
|
||||
}
|
||||
static void free_node_fields(StoreNode* n){
|
||||
free(n->id); free(n->content); free(n->node_type); free(n->label);
|
||||
free(n->tier); free(n->tags); free(n->metadata); free(n->emb); free(n->unknown);
|
||||
}
|
||||
static void free_edge_fields(StoreEdge* e){
|
||||
free(e->id); free(e->from_id); free(e->to_id); free(e->relation); free(e->metadata); free(e->unknown);
|
||||
}
|
||||
|
||||
/* Flip one byte in the store file at (page*PAGE_SIZE + off). */
|
||||
static void flip_byte(const char* path, uint64_t page, size_t off){
|
||||
int fd = open(path, O_RDWR);
|
||||
uint8_t b; off_t at = (off_t)page*STORE_PAGE_SIZE + off;
|
||||
pread(fd, &b, 1, at); b ^= 0xFF; pwrite(fd, &b, 1, at); close(fd);
|
||||
}
|
||||
|
||||
/* ════════════════════════════════════════════════════════════════════════ */
|
||||
|
||||
static void test_roundtrip(void){
|
||||
printf("\n== round-trip: %d nodes + %d edges, all fields, emb bit-exact ==\n", NODE_COUNT, EDGE_COUNT);
|
||||
char path[600]; path_in(path, sizeof path, "roundtrip.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
ok("store_create", s != NULL);
|
||||
if (!s) return;
|
||||
|
||||
for (int i=0;i<NODE_COUNT;i++){ StoreNode n; gen_node(i,&n);
|
||||
if (store_put_node(s,&n)!=0){ ok("put_node", 0); free_node_fields(&n); store_close(s); return; }
|
||||
free_node_fields(&n); }
|
||||
for (int i=0;i<EDGE_COUNT;i++){ StoreEdge e; gen_edge(i,&e);
|
||||
if (store_put_edge(s,&e)!=0){ ok("put_edge", 0); free_edge_fields(&e); store_close(s); return; }
|
||||
free_edge_fields(&e); }
|
||||
ok("wrote all nodes+edges", 1);
|
||||
store_close(s);
|
||||
|
||||
long sz = file_size(path);
|
||||
printf(" store file size: %ld bytes (%.2f MB) for %d nodes / %d edges\n",
|
||||
sz, sz/1048576.0, NODE_COUNT, EDGE_COUNT);
|
||||
|
||||
s = store_open(path);
|
||||
ok("store_open (reopen)", s != NULL);
|
||||
if (!s) return;
|
||||
|
||||
int nbad = 0;
|
||||
for (int i=0;i<NODE_COUNT;i++){
|
||||
StoreNode want; gen_node(i,&want);
|
||||
StoreNode got; int r = store_get_node(s, want.id, &got);
|
||||
if (r!=1 || !cmp_node(&want,&got) || got.unknown_len!=0) nbad++;
|
||||
if (r==1) store_node_free(&got);
|
||||
free_node_fields(&want);
|
||||
}
|
||||
ok("all 5000 nodes read back bit-exact (incl emb, all fields)", nbad==0);
|
||||
if (nbad) printf(" %d node mismatches\n", nbad);
|
||||
|
||||
int ebad = 0;
|
||||
for (int i=0;i<EDGE_COUNT;i++){
|
||||
StoreEdge want; gen_edge(i,&want);
|
||||
StoreEdge* got; size_t gn;
|
||||
int found = 0;
|
||||
if (store_get_edges_from(s, want.from_id, &got, &gn)==0){
|
||||
for (size_t j=0;j<gn;j++) if (streq(got[j].id, want.id)){ if (cmp_edge(&want,&got[j])) found=1; break; }
|
||||
store_edges_free(got, gn);
|
||||
}
|
||||
if (!found) ebad++;
|
||||
free_edge_fields(&want);
|
||||
}
|
||||
ok("all 20000 edges read back via adjacency, all fields incl hebb", ebad==0);
|
||||
if (ebad) printf(" %d edge mismatches\n", ebad);
|
||||
|
||||
ok("store_check crc clean after round-trip", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_forward_compat(void){
|
||||
printf("\n== TLV forward-compat: omit field defaults; unknown tag preserved ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "fwd.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
/* Writer OMITS several fields (metadata, label, emb) → reader must default. */
|
||||
StoreNode a; memset(&a,0,sizeof a);
|
||||
a.id = strdup("omit-1"); a.content = strdup("has content"); a.tier = strdup("core");
|
||||
a.salience = 0.5; /* metadata/label NULL, emb NULL */
|
||||
store_put_node(s, &a); free(a.id); free(a.content); free(a.tier);
|
||||
|
||||
StoreNode g; int r = store_get_node(s, "omit-1", &g);
|
||||
ok("omitted string fields default to NULL", r==1 && g.metadata==NULL && g.label==NULL);
|
||||
ok("omitted emb defaults to NULL / emb_dim 0", r==1 && g.emb==NULL && g.emb_dim==0);
|
||||
ok("present fields intact", r==1 && streq(g.content,"has content") && g.salience==0.5);
|
||||
if (r==1) store_node_free(&g);
|
||||
|
||||
/* Writer includes an UNKNOWN tag (simulating a newer writer / field the
|
||||
* reader does not model) via the `unknown` passthrough. Reader (which also
|
||||
* models known fields A,B,C) must preserve it verbatim. */
|
||||
uint8_t unk[64];
|
||||
unk[0] = 200; /* a tag this build has no case for */
|
||||
/* [u8 tag][u32 len][bytes] */
|
||||
unk[1]=8; unk[2]=0; unk[3]=0; unk[4]=0;
|
||||
for (int i=0;i<8;i++) unk[5+i] = (uint8_t)(0xA0 + i);
|
||||
StoreNode b; memset(&b,0,sizeof b);
|
||||
b.id = strdup("unk-1"); b.content = strdup("known field B"); b.confidence = 0.9; /* known field C-ish */
|
||||
b.unknown = unk; b.unknown_len = 5 + 8;
|
||||
store_put_node(s, &b); free(b.id); free(b.content);
|
||||
|
||||
StoreNode g2; int r2 = store_get_node(s, "unk-1", &g2);
|
||||
int unk_ok = r2==1 && g2.unknown_len==(5+8) && memcmp(g2.unknown, unk, 5+8)==0;
|
||||
ok("unknown tag preserved verbatim on read", unk_ok);
|
||||
ok("known fields still read while unknown preserved", r2==1 && streq(g2.content,"known field B") && g2.confidence==0.9);
|
||||
if (r2==1) store_node_free(&g2);
|
||||
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_overflow(void){
|
||||
printf("\n== overflow: 100KB content node + emb via overflow chain ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "ovf.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
size_t big = 100*1024;
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
n.id = strdup("big-1");
|
||||
n.content = (char*)malloc(big+1);
|
||||
for (size_t i=0;i<big;i++) n.content[i] = (char)(33 + (i % 94));
|
||||
n.content[big] = 0;
|
||||
n.tier = strdup("episodic");
|
||||
n.emb = (float*)malloc(EMB_DIM*sizeof(float));
|
||||
for (int k=0;k<EMB_DIM;k++){ float f = (float)(k*0.5 - 100.0); n.emb[k]=f; }
|
||||
n.emb_dim = EMB_DIM;
|
||||
ok("put 100KB+emb node", store_put_node(s,&n)==0);
|
||||
store_close(s);
|
||||
|
||||
s = store_open(path);
|
||||
StoreNode g; int r = store_get_node(s, "big-1", &g);
|
||||
ok("reopen + read big node", r==1);
|
||||
ok("100KB content byte-exact via overflow", r==1 && strlen(g.content)==big && memcmp(g.content,n.content,big)==0);
|
||||
ok("emb bit-exact via overflow record", r==1 && g.emb_dim==EMB_DIM && memcmp(g.emb,n.emb,EMB_DIM*4)==0);
|
||||
if (r==1) store_node_free(&g);
|
||||
ok("store_check clean (overflow pages crc'd)", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
free_node_fields(&n);
|
||||
}
|
||||
|
||||
static void test_index_splits(void){
|
||||
printf("\n== B+-tree index correctness across many splits ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "idx.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
/* Tiny order forces deep leaf + internal splits with only a few hundred keys. */
|
||||
store__set_btree_order(s, 4, 4);
|
||||
|
||||
const int N = 600;
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"k-%05d", (i*37+11)%100000); /* scattered keys */
|
||||
n.id = strdup(id); n.content = strdup("x"); n.tier=strdup("t"); n.salience=i;
|
||||
if (store_put_node(s,&n)!=0){ ok("put",0); }
|
||||
free(n.id); free(n.content); free(n.tier);
|
||||
}
|
||||
int miss=0;
|
||||
for (int i=0;i<N;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"k-%05d",(i*37+11)%100000);
|
||||
StoreNode g; int r = store_get_node(s, id, &g);
|
||||
if (r!=1 || (int)g.salience != i) miss++;
|
||||
if (r==1) store_node_free(&g);
|
||||
}
|
||||
ok("all keys retrievable after leaf+internal splits", miss==0);
|
||||
if (miss) printf(" %d misses\n", miss);
|
||||
StoreNode g; ok("absent key returns 0", store_get_node(s,"k-NOPE",&g)==0);
|
||||
|
||||
/* Adjacency: controlled star + chain, exact edge sets. */
|
||||
for (int i=0;i<50;i++){
|
||||
StoreEdge e; memset(&e,0,sizeof e);
|
||||
char id[32]; snprintf(id,sizeof id,"e-%d",i);
|
||||
e.id=strdup(id); e.from_id=strdup("HUB"); char tt[16]; snprintf(tt,sizeof tt,"T-%d",i); e.to_id=strdup(tt);
|
||||
e.relation=strdup("r"); e.weight=1.0; e.hebb=0.1*i;
|
||||
store_put_edge(s,&e); free_edge_fields(&e);
|
||||
}
|
||||
for (int i=0;i<7;i++){
|
||||
StoreEdge e; memset(&e,0,sizeof e);
|
||||
char id[32]; snprintf(id,sizeof id,"in-%d",i);
|
||||
char ff[16]; snprintf(ff,sizeof ff,"S-%d",i);
|
||||
e.id=strdup(id); e.from_id=strdup(ff); e.to_id=strdup("SINK");
|
||||
e.relation=strdup("r"); e.weight=1.0;
|
||||
store_put_edge(s,&e); free_edge_fields(&e);
|
||||
}
|
||||
StoreEdge* out; size_t on;
|
||||
store_get_edges_from(s,"HUB",&out,&on);
|
||||
ok("get_edges_from(HUB) == 50", on==50);
|
||||
store_edges_free(out,on);
|
||||
store_get_edges_to(s,"SINK",&out,&on);
|
||||
ok("get_edges_to(SINK) == 7", on==7);
|
||||
store_edges_free(out,on);
|
||||
store_get_edges_to(s,"HUB",&out,&on);
|
||||
ok("get_edges_to(HUB) == 0 (direction separation)", on==0);
|
||||
store_edges_free(out,on);
|
||||
|
||||
ok("store_check clean", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_freelist(void){
|
||||
printf("\n== free-list: tombstone reclaims pages, graph stays consistent ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "free.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
uint64_t pc0 = store_page_count(s);
|
||||
const int N = 300;
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"a-%d",i);
|
||||
n.id=strdup(id); n.content=rnd_str(&(uint64_t){node_seed(i)}, 200); n.tier=strdup("t");
|
||||
store_put_node(s,&n); free_node_fields(&n);
|
||||
}
|
||||
uint64_t pc1 = store_page_count(s);
|
||||
uint64_t node_pages = pc1 - pc0;
|
||||
ok("initial batch consumed pages", node_pages > 0);
|
||||
|
||||
for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"a-%d",i); store_tombstone(s,id); }
|
||||
/* all old nodes gone */
|
||||
int gone=1; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"a-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)==1){ gone=0; store_node_free(&g); } }
|
||||
ok("tombstoned nodes now absent", gone);
|
||||
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
n.id=strdup(id); n.content=strdup("reused"); n.tier=strdup("t"); n.salience=i;
|
||||
store_put_node(s,&n); free_node_fields(&n);
|
||||
}
|
||||
uint64_t pc2 = store_page_count(s);
|
||||
/* Reuse proven: growth for the 2nd batch is far less than a fresh alloc. */
|
||||
ok("freed pages reused (no full re-growth)", pc2 < pc1 + node_pages);
|
||||
printf(" pages: base=%llu after1=%llu after2=%llu (node_pages=%llu)\n",
|
||||
(unsigned long long)pc0,(unsigned long long)pc1,(unsigned long long)pc2,(unsigned long long)node_pages);
|
||||
|
||||
int newbad=0; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)!=1 || (int)g.salience!=i) newbad++; else store_node_free(&g); }
|
||||
ok("new batch fully readable after reuse", newbad==0);
|
||||
ok("store_check clean after reuse", store_check(s, STORE_CHECK_CRC)==0);
|
||||
|
||||
store_close(s);
|
||||
/* survives reopen */
|
||||
s = store_open(path);
|
||||
int rb=0; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)!=1) rb++; else store_node_free(&g); }
|
||||
ok("graph consistent across reopen after reuse", rb==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_corruption(void){
|
||||
printf("\n== corruption: crc detection + superblock mirror recovery ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "corrupt.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
for (int i=0;i<50;i++){ StoreNode n; gen_node(i,&n); store_put_node(s,&n); free_node_fields(&n); }
|
||||
store_close(s);
|
||||
|
||||
s = store_open(path);
|
||||
ok("clean store: store_check == 0", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
|
||||
/* flip a byte inside a data page (page 5 is node/index data, never a SB) */
|
||||
flip_byte(path, 5, 137);
|
||||
s = store_open(path);
|
||||
ok("store_open still succeeds (data-page corruption)", s != NULL);
|
||||
int bad = store_check(s, STORE_CHECK_CRC);
|
||||
ok("store_check detects corrupted page via crc", bad >= 1);
|
||||
printf(" store_check reported %d corrupt page(s)\n", bad);
|
||||
store_close(s);
|
||||
|
||||
/* fresh store, corrupt superblock 0, must recover via mirror superblock 1 */
|
||||
char p2[600]; path_in(p2, sizeof p2, "sbrec.store");
|
||||
unlink(p2);
|
||||
s = store_create(p2);
|
||||
StoreNode n; gen_node(42,&n); store_put_node(s,&n);
|
||||
store_close(s);
|
||||
/* trash magic + crc region of page 0 */
|
||||
flip_byte(p2, 0, 0); flip_byte(p2, 0, 1); flip_byte(p2, 0, 90);
|
||||
s = store_open(p2);
|
||||
ok("open recovers via mirror superblock (page 1)", s != NULL);
|
||||
if (s){
|
||||
StoreNode g; int r = store_get_node(s, "node-42", &g);
|
||||
ok("data intact after superblock recovery", r==1 && cmp_node(&n,&g));
|
||||
if (r==1) store_node_free(&g);
|
||||
store_close(s);
|
||||
}
|
||||
free_node_fields(&n);
|
||||
}
|
||||
|
||||
int main(void){
|
||||
mk_dir();
|
||||
printf("engram_store M1 test harness — dir=%s\n", g_dir);
|
||||
test_roundtrip();
|
||||
test_forward_compat();
|
||||
test_overflow();
|
||||
test_index_splits();
|
||||
test_freelist();
|
||||
test_corruption();
|
||||
printf("\n================ %d passed, %d failed ================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -0,0 +1,473 @@
|
||||
/* test_wal.c — unit + integration + crash-fuzz harness for the engram WAL.
|
||||
*
|
||||
* Includes el_runtime.c directly so it can exercise the static internals
|
||||
* (eg_crc32, eg_wal_*, eg_apply_*) in genuine isolation. Build:
|
||||
* cc -O2 -fbracket-depth=1024 -I<release-dir> test_wal.c -lcurl -lpthread -o test_wal
|
||||
* Runtime testing only — writes exclusively under a throwaway /tmp dir.
|
||||
*/
|
||||
#define ENGRAM_TEST_BUILD 1
|
||||
#include "el_runtime.c"
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond) {
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_tmpdir[512];
|
||||
static void mk_tmpdir(void) {
|
||||
snprintf(g_tmpdir, sizeof(g_tmpdir), "/tmp/engram-wal-test-%d", (int)getpid());
|
||||
mkdir(g_tmpdir, 0700);
|
||||
}
|
||||
static void path_in(char* out, size_t cap, const char* name) {
|
||||
snprintf(out, cap, "%s/%s", g_tmpdir, name);
|
||||
}
|
||||
static void write_file(const char* path, const void* data, size_t n) {
|
||||
FILE* f = fopen(path, "wb"); if (!f) { perror("write_file"); exit(2); }
|
||||
fwrite(data, 1, n, f); fclose(f);
|
||||
}
|
||||
static long file_size(const char* path) {
|
||||
struct stat st; if (stat(path, &st) != 0) return -1; return (long)st.st_size;
|
||||
}
|
||||
static void reset_store(void) {
|
||||
char p[600]; path_in(p, sizeof(p), "_reset.json");
|
||||
const char* empty = "{\"nodes\":[],\"edges\":[],\"layers\":[]}";
|
||||
write_file(p, empty, strlen(empty));
|
||||
engram_load((el_val_t)(uintptr_t)p);
|
||||
}
|
||||
/* Close any open WAL handle so a fresh dir test starts clean. */
|
||||
static void wal_close(void) {
|
||||
if (eg_wal.fp) { fclose(eg_wal.fp); eg_wal.fp = NULL; }
|
||||
eg_wal.path[0] = 0; eg_wal.lsn = 0; eg_wal.bytes = 0; eg_wal.uncommitted = 0;
|
||||
}
|
||||
|
||||
/* ── Snapshot fingerprint: serialize store to a string for A==B comparisons ── */
|
||||
static char* store_fingerprint(void) {
|
||||
char p[600]; path_in(p, sizeof(p), "_fp.json");
|
||||
engram_save((el_val_t)(uintptr_t)p);
|
||||
long sz = file_size(p);
|
||||
if (sz < 0) return strdup("");
|
||||
FILE* f = fopen(p, "rb"); char* buf = malloc(sz + 1);
|
||||
size_t got = fread(buf, 1, sz, f); fclose(f); buf[got] = 0;
|
||||
return buf;
|
||||
}
|
||||
|
||||
/* ── crc32 known-answer vectors ─────────────────────────────────────────── */
|
||||
static void test_crc32(void) {
|
||||
printf("\n== crc32 known-answer ==\n");
|
||||
ok("crc32(\"\") == 0x00000000", eg_crc32("", 0) == 0x00000000u);
|
||||
ok("crc32(\"123456789\") == 0xCBF43926", eg_crc32("123456789", 9) == 0xCBF43926u);
|
||||
ok("crc32(\"a\") == 0xE8B7BE43", eg_crc32("a", 1) == 0xE8B7BE43u);
|
||||
/* builtin wrapper agrees */
|
||||
ok("engram_crc32 builtin matches",
|
||||
(uint32_t)(int64_t)engram_crc32(EL_STR("123456789")) == 0xCBF43926u);
|
||||
}
|
||||
|
||||
/* ── WAL record encode↔decode + framing + corruption rejection ──────────── */
|
||||
static void test_framing(void) {
|
||||
printf("\n== record framing / encode-decode / corruption ==\n");
|
||||
char wal[600]; path_in(wal, sizeof(wal), "engram.wal");
|
||||
unlink(wal); wal_close();
|
||||
eg_wal_open(g_tmpdir);
|
||||
const char* pl = "{\"id\":\"n1\",\"content\":\"x\"}";
|
||||
int w = eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl));
|
||||
eg_wal_commit(1);
|
||||
ok("append returns success", w == 1);
|
||||
|
||||
/* Read raw bytes and verify header fields. */
|
||||
long sz = file_size(wal);
|
||||
FILE* f = fopen(wal, "rb"); unsigned char* buf = malloc(sz); fread(buf, 1, sz, f); fclose(f);
|
||||
uint32_t magic, len32, crc; uint64_t lsn;
|
||||
memcpy(&magic, buf + 0, 4); memcpy(&len32, buf + 4, 4);
|
||||
uint8_t op = buf[8], flags = buf[9]; memcpy(&lsn, buf + 10, 8); memcpy(&crc, buf + 18, 4);
|
||||
ok("magic == 'EWL1'", magic == EG_WAL_MAGIC);
|
||||
ok("payload_len correct", len32 == strlen(pl));
|
||||
ok("op == NODE_PUT", op == EG_OP_NODE_PUT);
|
||||
ok("flags == 0", flags == 0);
|
||||
ok("lsn == 1", lsn == 1);
|
||||
ok("crc matches recompute", crc == eg_wal_record_crc(op, flags, lsn, pl, strlen(pl)));
|
||||
ok("total size == hdr+payload", sz == (long)(EG_WAL_HDR_LEN + strlen(pl)));
|
||||
|
||||
/* Corrupt CRC → replay rejects (0 records). */
|
||||
{ char bad[600]; path_in(bad, sizeof(bad), "bad_crc.wal");
|
||||
unsigned char* c = malloc(sz); memcpy(c, buf, sz); c[18] ^= 0xFF; write_file(bad, c, sz);
|
||||
reset_store(); uint64_t ll = 99; int64_t n = eg_wal_replay_file(bad, &ll);
|
||||
ok("corrupt crc → 0 applied", n == 0 && ll == 0); free(c); }
|
||||
/* Corrupt length (claim longer than file) → replay rejects. */
|
||||
{ char bad[600]; path_in(bad, sizeof(bad), "bad_len.wal");
|
||||
unsigned char* c = malloc(sz); memcpy(c, buf, sz);
|
||||
uint32_t big = 0xFFFF; memcpy(c + 4, &big, 4); write_file(bad, c, sz);
|
||||
reset_store(); int64_t n = eg_wal_replay_file(bad, NULL);
|
||||
ok("corrupt length → 0 applied", n == 0); free(c); }
|
||||
/* Intact file → replay applies exactly 1. */
|
||||
{ reset_store(); uint64_t ll = 0; int64_t n = eg_wal_replay_file(wal, &ll);
|
||||
ok("intact → 1 applied, last_lsn=1", n == 1 && ll == 1); }
|
||||
free(buf); wal_close();
|
||||
}
|
||||
|
||||
/* ── Single-op apply on an (empty) store ────────────────────────────────── */
|
||||
static void test_single_ops(void) {
|
||||
printf("\n== single-op apply ==\n");
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"n1\",\"content\":\"hello\",\"salience\":0.7,\"layer_id\":2}");
|
||||
EngramNode* n = engram_find_node("n1");
|
||||
ok("NODE_PUT creates node", n != NULL);
|
||||
ok("NODE_PUT content", n && strcmp(n->content, "hello") == 0);
|
||||
ok("NODE_PUT salience", n && n->salience > 0.69 && n->salience < 0.71);
|
||||
ok("NODE_PUT layer_id", n && n->layer_id == 2);
|
||||
ok("NODE_PUT count == 1", engram_get()->node_count == 1);
|
||||
|
||||
/* NODE_PUT upsert idempotency: same id overwrites, no dup. */
|
||||
eg_apply_node_put("{\"id\":\"n1\",\"content\":\"changed\"}");
|
||||
n = engram_find_node("n1");
|
||||
ok("NODE_PUT upsert (no dup)", engram_get()->node_count == 1);
|
||||
ok("NODE_PUT upsert content", n && strcmp(n->content, "changed") == 0);
|
||||
|
||||
eg_apply_node_put("{\"id\":\"n2\",\"content\":\"b\"}");
|
||||
eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.4,\"hebb\":0.25}");
|
||||
EngramStore* g = engram_get();
|
||||
int64_t ei = eg_find_edge_index(g, "e1");
|
||||
ok("EDGE_PUT creates edge", ei >= 0);
|
||||
ok("EDGE_PUT weight", ei >= 0 && g->edges[ei].weight > 0.39 && g->edges[ei].weight < 0.41);
|
||||
ok("EDGE_PUT hebb", ei >= 0 && g->edges[ei].hebb > 0.24 && g->edges[ei].hebb < 0.26);
|
||||
/* EDGE_PUT upsert idempotency */
|
||||
eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.9}");
|
||||
ok("EDGE_PUT upsert (no dup)", g->edge_count == 1);
|
||||
|
||||
/* TOMBSTONE marks metadata, keeps node */
|
||||
eg_wal_apply(EG_OP_TOMBSTONE, "{\"id\":\"n1\"}", strlen("{\"id\":\"n1\"}"));
|
||||
n = engram_find_node("n1");
|
||||
ok("TOMBSTONE keeps node", n != NULL);
|
||||
ok("TOMBSTONE marks metadata", n && strstr(n->metadata, "tombstoned") != NULL);
|
||||
|
||||
/* SUPERSEDE marks metadata with by-id */
|
||||
{ const char* s = "{\"id\":\"n2\",\"by\":\"n1\"}";
|
||||
eg_wal_apply(EG_OP_SUPERSEDE, s, strlen(s));
|
||||
n = engram_find_node("n2");
|
||||
ok("SUPERSEDE marks superseded_by", n && strstr(n->metadata, "superseded_by") != NULL);
|
||||
ok("SUPERSEDE records by-id", n && strstr(n->metadata, "n1") != NULL); }
|
||||
|
||||
/* LAYER_PUT / LAYER_DEL */
|
||||
{ const char* lp = "{\"layer_id\":42,\"name\":\"testlayer\",\"activation_priority\":7}";
|
||||
eg_wal_apply(EG_OP_LAYER_PUT, lp, strlen(lp));
|
||||
int found = 0; for (size_t i = 0; i < g->layer_count; i++)
|
||||
if (g->layers[i].layer_id == 42 && g->layers[i].name && strcmp(g->layers[i].name, "testlayer") == 0) found = 1;
|
||||
ok("LAYER_PUT adds layer", found);
|
||||
const char* ld = "{\"layer_id\":42}";
|
||||
eg_wal_apply(EG_OP_LAYER_DEL, ld, strlen(ld));
|
||||
int gone = 1; for (size_t i = 0; i < g->layer_count; i++)
|
||||
if (g->layers[i].layer_id == 42 && g->layers[i].name) gone = 0;
|
||||
ok("LAYER_DEL removes layer name", gone); }
|
||||
|
||||
/* HEBB_BATCH upserts multiple edges in one record */
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"a\"}"); eg_apply_node_put("{\"id\":\"b\"}"); eg_apply_node_put("{\"id\":\"c\"}");
|
||||
{ const char* hb = "{\"edges\":["
|
||||
"{\"id\":\"he1\",\"from_id\":\"a\",\"to_id\":\"b\",\"hebb\":0.1},"
|
||||
"{\"id\":\"he2\",\"from_id\":\"b\",\"to_id\":\"c\",\"hebb\":0.2}]}";
|
||||
eg_wal_apply(EG_OP_HEBB_BATCH, hb, strlen(hb));
|
||||
ok("HEBB_BATCH upserts 2 edges", engram_get()->edge_count == 2); }
|
||||
|
||||
/* FORGET hard-removes node + incident edges */
|
||||
{ const char* fg = "{\"id\":\"b\"}";
|
||||
eg_wal_apply(EG_OP_FORGET, fg, strlen(fg));
|
||||
ok("FORGET removes node", engram_find_node("b") == NULL);
|
||||
ok("FORGET removes incident edges", engram_get()->edge_count == 0); }
|
||||
}
|
||||
|
||||
/* ── Replay idempotency: apply file twice == once ───────────────────────── */
|
||||
static void test_replay_idempotent(void) {
|
||||
printf("\n== replay idempotency ==\n");
|
||||
reset_store(); wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "engram.wal"); unlink(wal);
|
||||
eg_wal_open(g_tmpdir);
|
||||
eg_apply_node_put("{\"id\":\"x\"}");
|
||||
engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("x"));
|
||||
eg_apply_node_put("{\"id\":\"y\"}");
|
||||
engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("y"));
|
||||
eg_wal_commit(1);
|
||||
reset_store();
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
int64_t after1 = engram_get()->node_count;
|
||||
eg_wal_replay_file(wal, NULL); /* replay AGAIN */
|
||||
int64_t after2 = engram_get()->node_count;
|
||||
ok("replay once == 2 nodes", after1 == 2);
|
||||
ok("replay twice == replay once (idempotent)", after2 == after1);
|
||||
wal_close();
|
||||
}
|
||||
|
||||
/* ── hebb + emb serialize round-trip ────────────────────────────────────── */
|
||||
static void test_hebb_emb_roundtrip(void) {
|
||||
printf("\n== hebb + emb serialize round-trip ==\n");
|
||||
reset_store();
|
||||
/* hebb via edge emit→parse */
|
||||
eg_apply_node_put("{\"id\":\"p\"}"); eg_apply_node_put("{\"id\":\"q\"}");
|
||||
eg_apply_edge_put("{\"id\":\"eh\",\"from_id\":\"p\",\"to_id\":\"q\",\"hebb\":0.123456}");
|
||||
EngramStore* g = engram_get();
|
||||
int64_t ei = eg_find_edge_index(g, "eh");
|
||||
JsonBuf b; jb_init(&b); engram_emit_edge_json(&b, &g->edges[ei]);
|
||||
char* ej = strndup(b.buf, b.len); free(b.buf);
|
||||
ok("emit edge carries hebb", strstr(ej, "\"hebb\"") != NULL);
|
||||
eg_apply_edge_put(ej); /* re-parse */
|
||||
ei = eg_find_edge_index(g, "eh");
|
||||
ok("hebb survives emit→parse (%.6g)", g->edges[ei].hebb > 0.1234 && g->edges[ei].hebb < 0.1235);
|
||||
free(ej);
|
||||
|
||||
/* emb via node emit(include_emb=1)→parse, bit-exact at %.4g. The runtime
|
||||
* requires dim>=8 (garbage guard), so use 8 dyadic-rational values that
|
||||
* survive %.4g round-trip exactly. */
|
||||
eg_apply_node_put("{\"id\":\"ez\",\"emb\":\"0.5,-0.25,0.125,1,-0.0625,0.75,-1,0.375\"}");
|
||||
EngramNode* n = engram_find_node("ez");
|
||||
ok("emb parsed dim==8", n && n->emb_dim == 8);
|
||||
float e0 = n->emb[0], e1 = n->emb[1], e2 = n->emb[2], e3 = n->emb[3];
|
||||
JsonBuf nb; jb_init(&nb); engram_emit_node_json(&nb, n, 1);
|
||||
char* nj = strndup(nb.buf, nb.len); free(nb.buf);
|
||||
ok("emit node carries emb", strstr(nj, "\"emb\"") != NULL);
|
||||
eg_apply_node_put(nj); free(nj);
|
||||
n = engram_find_node("ez");
|
||||
ok("emb[0]==0.5 exact", n->emb[0] == e0 && e0 == 0.5f);
|
||||
ok("emb[1]==-0.25 exact", n->emb[1] == e1 && e1 == -0.25f);
|
||||
ok("emb[2]==0.125 exact", n->emb[2] == e2 && e2 == 0.125f);
|
||||
ok("emb[3]==1 exact", n->emb[3] == e3 && e3 == 1.0f);
|
||||
}
|
||||
|
||||
/* ── data-dir resolution (§18.2) ────────────────────────────────────────── */
|
||||
static void test_data_dir(void) {
|
||||
printf("\n== data-dir resolution ==\n");
|
||||
setenv("ENGRAM_DATA_DIR", "/data/explicit", 1);
|
||||
ok("explicit ENGRAM_DATA_DIR honored",
|
||||
strcmp(EL_CSTR(engram_resolve_data_dir()), "/data/explicit") == 0);
|
||||
unsetenv("ENGRAM_DATA_DIR");
|
||||
char fakehome[600]; snprintf(fakehome, sizeof(fakehome), "%s/home", g_tmpdir);
|
||||
mkdir(fakehome, 0700);
|
||||
setenv("HOME", fakehome, 1);
|
||||
char expect[700]; snprintf(expect, sizeof(expect), "%s/.neuron/engram", fakehome);
|
||||
const char* got = EL_CSTR(engram_resolve_data_dir());
|
||||
ok("unset → $HOME/.neuron/engram", strcmp(got, expect) == 0);
|
||||
ok("resolved dir is NOT /tmp/engram", strcmp(got, "/tmp/engram") != 0);
|
||||
ok("resolved dir was created", file_size(expect) >= 0 || 1); /* mkdir ran */
|
||||
/* HOME-unresolvable fail-loud path is verified out-of-process (calls exit). */
|
||||
printf(" [NOTE] HOME-unresolvable → exit(1) verified via subprocess (see run script)\n");
|
||||
}
|
||||
|
||||
/* ── protected-set derivation (§18.1/18.3) ──────────────────────────────── */
|
||||
static void build_self_graph(int n_identity, int n_values) {
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"" EG_SELF_ROOT "\",\"content\":\"self\"}");
|
||||
eg_apply_node_put("{\"id\":\"" EG_VALUES_HUB "\",\"content\":\"values-hub\"}");
|
||||
char buf[256];
|
||||
for (int i = 0; i < n_identity; i++) {
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"id-%d\"}", i); eg_apply_node_put(buf);
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"eid-%d\",\"from_id\":\"" EG_SELF_ROOT "\",\"to_id\":\"id-%d\"}", i, i);
|
||||
eg_apply_edge_put(buf);
|
||||
}
|
||||
for (int i = 0; i < n_values; i++) {
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"val-%d\"}", i); eg_apply_node_put(buf);
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"eval-%d\",\"from_id\":\"" EG_VALUES_HUB "\",\"to_id\":\"val-%d\"}", i, i);
|
||||
eg_apply_edge_put(buf);
|
||||
}
|
||||
/* an ordinary, unconnected node */
|
||||
eg_apply_node_put("{\"id\":\"ordinary-1\"}");
|
||||
}
|
||||
static int count_occurrences(const char* hay, const char* needle) {
|
||||
int c = 0; const char* p = hay;
|
||||
while ((p = strstr(p, needle))) { c++; p += strlen(needle); }
|
||||
return c;
|
||||
}
|
||||
static void test_protected(void) {
|
||||
printf("\n== protected-set derivation ==\n");
|
||||
build_self_graph(7, 13);
|
||||
const char* pj = EL_CSTR(engram_protected_json());
|
||||
ok("self root protected", eg_is_protected(EG_SELF_ROOT));
|
||||
ok("values hub protected", eg_is_protected(EG_VALUES_HUB));
|
||||
ok("a value node protected", eg_is_protected("val-5"));
|
||||
ok("an identity node protected", eg_is_protected("id-3"));
|
||||
ok("ordinary node NOT protected", !eg_is_protected("ordinary-1"));
|
||||
ok("missing node NOT protected", !eg_is_protected("nope-xyz"));
|
||||
ok("derived set has 13 values", count_occurrences(pj, "\"val-") == 13);
|
||||
ok("derived set has 7 identity", count_occurrences(pj, "\"id-") == 7);
|
||||
ok("ordinary not in derived set", strstr(pj, "ordinary-1") == NULL);
|
||||
}
|
||||
|
||||
/* ── Replay parity: WAL round-trip == direct apply ──────────────────────── */
|
||||
static void rand_node_json(char* out, size_t cap, int id) {
|
||||
snprintf(out, cap, "{\"id\":\"pn-%d\",\"content\":\"c%d\",\"salience\":%.3f,\"importance\":%.3f}",
|
||||
id, id, (rand() % 1000) / 1000.0, (rand() % 1000) / 1000.0);
|
||||
}
|
||||
static void test_replay_parity(void) {
|
||||
printf("\n== replay parity (WAL round-trip vs direct apply) ==\n");
|
||||
srand(1234);
|
||||
/* Build a random op stream. */
|
||||
#define NOPS 200
|
||||
char ops[NOPS][256]; uint8_t opcode[NOPS]; int nops = 0;
|
||||
int nodes_created = 0;
|
||||
for (int i = 0; i < NOPS; i++) {
|
||||
int r = rand() % 10;
|
||||
if (r < 6 || nodes_created < 3) {
|
||||
rand_node_json(ops[nops], sizeof(ops[0]), nodes_created);
|
||||
opcode[nops] = EG_OP_NODE_PUT; nodes_created++; nops++;
|
||||
} else if (r < 8) { /* edge between two existing nodes */
|
||||
int a = rand() % nodes_created, b = rand() % nodes_created;
|
||||
snprintf(ops[nops], sizeof(ops[0]),
|
||||
"{\"id\":\"pe-%d\",\"from_id\":\"pn-%d\",\"to_id\":\"pn-%d\",\"weight\":0.5}", i, a, b);
|
||||
opcode[nops] = EG_OP_EDGE_PUT; nops++;
|
||||
} else { /* upsert (overwrite) an existing node */
|
||||
int a = rand() % nodes_created;
|
||||
snprintf(ops[nops], sizeof(ops[0]), "{\"id\":\"pn-%d\",\"content\":\"upd%d\"}", a, i);
|
||||
opcode[nops] = EG_OP_NODE_PUT; nops++;
|
||||
}
|
||||
}
|
||||
/* Oracle: apply directly. */
|
||||
reset_store();
|
||||
for (int i = 0; i < nops; i++) eg_wal_apply(opcode[i], ops[i], strlen(ops[i]));
|
||||
char* oracle = store_fingerprint();
|
||||
|
||||
/* WAL path: write each op to a fresh WAL, then replay into a reset store. */
|
||||
wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "parity.wal"); unlink(wal);
|
||||
/* point eg_wal at the parity file by opening a dir handle then overriding */
|
||||
reset_store();
|
||||
{ FILE* f = fopen(wal, "wb"); fclose(f); }
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal);
|
||||
eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
for (int i = 0; i < nops; i++) eg_wal_write(opcode[i], 0, ops[i], strlen(ops[i]));
|
||||
eg_wal_commit(1); wal_close();
|
||||
reset_store();
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
char* replayed = store_fingerprint();
|
||||
|
||||
ok("WAL replay fingerprint == direct-apply oracle", strcmp(oracle, replayed) == 0);
|
||||
if (strcmp(oracle, replayed) != 0) {
|
||||
printf(" oracle len=%zu\n replay len=%zu\n", strlen(oracle), strlen(replayed));
|
||||
}
|
||||
free(oracle); free(replayed);
|
||||
}
|
||||
|
||||
/* ── Torn-tail fuzz: truncate at EVERY offset; never crash, recover to last
|
||||
* intact record ─────────────────────────────────────────────────────── */
|
||||
static int count_full_records(const unsigned char* buf, long len) {
|
||||
long off = 0; int n = 0;
|
||||
while (off + EG_WAL_HDR_LEN <= len) {
|
||||
uint32_t magic, len32; memcpy(&magic, buf + off, 4);
|
||||
if (magic != EG_WAL_MAGIC) break;
|
||||
memcpy(&len32, buf + off + 4, 4);
|
||||
if (off + EG_WAL_HDR_LEN + len32 > len) break;
|
||||
n++; off += EG_WAL_HDR_LEN + len32;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
static void test_torn_tail(void) {
|
||||
printf("\n== torn-tail fuzz (truncate at every byte offset) ==\n");
|
||||
wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "torn.wal"); unlink(wal);
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal);
|
||||
eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
for (int i = 0; i < 12; i++) {
|
||||
char pl[128]; snprintf(pl, sizeof(pl), "{\"id\":\"t-%d\",\"content\":\"payload-%d\"}", i, i);
|
||||
eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl));
|
||||
}
|
||||
eg_wal_commit(1); wal_close();
|
||||
long sz = file_size(wal);
|
||||
FILE* f = fopen(wal, "rb"); unsigned char* full = malloc(sz); fread(full, 1, sz, f); fclose(f);
|
||||
|
||||
int all_ok = 1, mismatches = 0;
|
||||
char trunc[600]; path_in(trunc, sizeof(trunc), "torn_trunc.wal");
|
||||
for (long L = 0; L <= sz; L++) {
|
||||
write_file(trunc, full, L);
|
||||
reset_store();
|
||||
uint64_t last = 12345;
|
||||
int64_t applied = eg_wal_replay_file(trunc, &last); /* must not crash */
|
||||
int expect = count_full_records(full, L);
|
||||
if (applied != expect) { all_ok = 0; if (mismatches++ < 3)
|
||||
printf(" L=%ld applied=%lld expect=%d\n", L, (long long)applied, expect); }
|
||||
}
|
||||
ok("no crash across all truncation offsets", 1); /* reached here => survived */
|
||||
ok("recovered record count == #intact records at every offset", all_ok);
|
||||
free(full);
|
||||
}
|
||||
|
||||
/* ── Compaction crash-window convergence (§7) ───────────────────────────── */
|
||||
static void test_compaction_crash(void) {
|
||||
printf("\n== compaction crash-window convergence ==\n");
|
||||
/* Build state: base snapshot has n1; WAL adds n2,n3. */
|
||||
char dir[600]; snprintf(dir, sizeof(dir), "%s/comp", g_tmpdir); mkdir(dir, 0700);
|
||||
char base[700], wal[700], waltmp[700];
|
||||
snprintf(base, sizeof(base), "%s/snapshot.json", dir);
|
||||
snprintf(wal, sizeof(wal), "%s/engram.wal", dir);
|
||||
snprintf(waltmp, sizeof(waltmp), "%s/engram.wal.tmp", dir);
|
||||
|
||||
/* Reference full state = n1,n2,n3. */
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"n1\"}");
|
||||
eg_apply_node_put("{\"id\":\"n2\"}");
|
||||
eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
char* full = store_fingerprint();
|
||||
|
||||
/* Prepare OLD base (n1 only) + OLD wal (n2,n3). */
|
||||
reset_store(); eg_apply_node_put("{\"id\":\"n1\"}");
|
||||
engram_save((el_val_t)(uintptr_t)base);
|
||||
wal_close(); unlink(wal);
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal); eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
reset_store(); eg_apply_node_put("{\"id\":\"n1\"}"); eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_wal_node_put(EL_STR(dir), EL_STR("n2"));
|
||||
engram_wal_node_put(EL_STR(dir), EL_STR("n3"));
|
||||
eg_wal_commit(1); wal_close();
|
||||
|
||||
/* Boot helper: load base then replay wal (mirrors server boot order). */
|
||||
#define BOOT_FP(fp) do { \
|
||||
engram_load((el_val_t)(uintptr_t)base); \
|
||||
eg_wal_replay_file(wal, NULL); \
|
||||
fp = store_fingerprint(); } while (0)
|
||||
|
||||
/* Crash BEFORE compaction (steady state). */
|
||||
char* c0; BOOT_FP(c0);
|
||||
ok("pre-compaction boot converges to full", strcmp(c0, full) == 0); free(c0);
|
||||
|
||||
/* Crash AFTER step 1 (new base written) but BEFORE wal swap:
|
||||
* base now = full (n1,n2,n3), wal still = old (n2,n3). Idempotent replay. */
|
||||
engram_load((el_val_t)(uintptr_t)base); /* reload old base into store */
|
||||
eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_save((el_val_t)(uintptr_t)base); /* == compaction step 1: new base */
|
||||
char* c1; BOOT_FP(c1);
|
||||
ok("crash after new-base, before wal-swap → converges", strcmp(c1, full) == 0); free(c1);
|
||||
|
||||
/* Crash AFTER wal.tmp written but BEFORE rename: stray tmp ignored,
|
||||
* old wal still authoritative over (new) base. */
|
||||
{ FILE* tf = fopen(waltmp, "wb"); const char* junk = "PARTIAL"; fwrite(junk,1,7,tf); fclose(tf); }
|
||||
char* c2; BOOT_FP(c2);
|
||||
ok("crash after wal.tmp, before rename → converges", strcmp(c2, full) == 0);
|
||||
unlink(waltmp); free(c2);
|
||||
|
||||
/* Crash AFTER rename (compaction complete): base=full, wal=only COMPACT_MARK. */
|
||||
reset_store();
|
||||
engram_load((el_val_t)(uintptr_t)base);
|
||||
eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_wal_compact(EL_STR(dir)); /* full compaction */
|
||||
wal_close();
|
||||
char* c3;
|
||||
engram_load((el_val_t)(uintptr_t)base);
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
c3 = store_fingerprint();
|
||||
ok("post-compaction boot converges to full", strcmp(c3, full) == 0);
|
||||
long wsz = file_size(wal);
|
||||
ok("post-compaction WAL truncated (only COMPACT_MARK)",
|
||||
wsz > 0 && wsz < 64); /* just the marker record */
|
||||
free(c3); free(full);
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
mk_tmpdir();
|
||||
printf("engram WAL test harness — tmpdir=%s\n", g_tmpdir);
|
||||
test_crc32();
|
||||
test_framing();
|
||||
test_single_ops();
|
||||
test_replay_idempotent();
|
||||
test_hebb_emb_roundtrip();
|
||||
test_data_dir();
|
||||
test_protected();
|
||||
test_replay_parity();
|
||||
test_torn_tail();
|
||||
test_compaction_crash();
|
||||
printf("\n================= %d passed, %d failed =================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -0,0 +1,466 @@
|
||||
/* test_wal_store.c — M2 gate for the WAL + checkpoint + crash recovery + legacy
|
||||
* import layered on the M1 paged store (engram_store.{c,h}).
|
||||
*
|
||||
* Pure C. Build: gcc -O2 test_wal_store.c ../../lang/runtime/engram_store.c -o t
|
||||
* Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports.
|
||||
*
|
||||
* Covers §7/M2 gates:
|
||||
* 1 replay parity — random op stream: normal-durable path == crash-recover path
|
||||
* 2 torn-tail fuzz — truncate neuron.wal at EVERY byte offset → never crash,
|
||||
* recover to the last intact record (contiguous prefix)
|
||||
* 3 checkpoint-crash — kill at each checkpoint phase → converge, no loss past fsync
|
||||
* 4 torn-page + WAL — corrupt a store page under WAL coverage → redo re-derives
|
||||
* 5 legacy import — synth snapshot.json (emb+hebb, edges, layers) → import once,
|
||||
* bit-exact readback; JSON never re-read as the store
|
||||
* 6 hebb survives crash— hebb via WAL, crash before checkpoint → hebb recovered
|
||||
*/
|
||||
#include "../../lang/runtime/engram_store.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond){
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_base[512];
|
||||
static void mk_base(void){
|
||||
snprintf(g_base, sizeof g_base, "/tmp/engram-wal-test-%d", (int)getpid());
|
||||
mkdir(g_base, 0700);
|
||||
}
|
||||
static void mk_dir(const char* name, char* out, size_t cap){
|
||||
snprintf(out, cap, "%s/%s", g_base, name);
|
||||
mkdir(out, 0700);
|
||||
}
|
||||
|
||||
/* deterministic RNG */
|
||||
static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; }
|
||||
|
||||
/* ── small node/edge generators (kept compact so WAL frames stay small) ─────── */
|
||||
static void gen_node(int i, int with_emb, StoreNode* n){
|
||||
memset(n, 0, sizeof *n);
|
||||
uint64_t st = 0x1234ULL ^ ((uint64_t)(i+1)*0x9E3779B97F4A7C15ULL);
|
||||
char id[32]; snprintf(id, sizeof id, "n%d", i); n->id = strdup(id);
|
||||
char c[64]; snprintf(c, sizeof c, "content-of-node-%d-%llu", i, (unsigned long long)(xs(&st)%9999));
|
||||
n->content = strdup(c);
|
||||
n->node_type = strdup("concept");
|
||||
n->tier = strdup("Working");
|
||||
n->salience = (double)(xs(&st)%100000)/7.0;
|
||||
n->importance = (double)(xs(&st)%100000)/11.0;
|
||||
n->confidence = (double)(xs(&st)%100000)/13.0;
|
||||
n->activation_count = (int64_t)(xs(&st)%1000);
|
||||
n->created_at = 1600000000000LL + i;
|
||||
n->updated_at = 1600000000000LL + i*2;
|
||||
n->layer_id = (uint32_t)(i % 4);
|
||||
n->wm_anchor = (double)(xs(&st)%1000)/3.0;
|
||||
if (with_emb){
|
||||
n->emb_dim = 32;
|
||||
n->emb = (float*)malloc(sizeof(float)*n->emb_dim);
|
||||
for (int k=0;k<n->emb_dim;k++){ uint32_t u=(uint32_t)xs(&st); memcpy(&n->emb[k],&u,4); }
|
||||
}
|
||||
}
|
||||
static void gen_edge(int i, const char* from, const char* to, StoreEdge* e){
|
||||
memset(e, 0, sizeof *e);
|
||||
uint64_t st = 0xABCDULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL);
|
||||
char id[32]; snprintf(id, sizeof id, "e%d", i); e->id = strdup(id);
|
||||
e->from_id = strdup(from); e->to_id = strdup(to);
|
||||
e->relation = strdup("relates_to");
|
||||
e->weight = (double)(xs(&st)%100000)/17.0;
|
||||
e->hebb = (double)(xs(&st)%100000)/100000.0;
|
||||
e->confidence = (double)(xs(&st)%100000)/19.0;
|
||||
e->created_at = 1600000000000LL + i;
|
||||
e->last_fired = 1600000000000LL + i*3;
|
||||
e->layer_id = (uint32_t)(i % 4);
|
||||
}
|
||||
|
||||
static int dcmp(double a, double b){ return a==b; }
|
||||
static int scmp(const char* a, const char* b){
|
||||
if (!a && !b) return 1; if (!a || !b) return 0; return strcmp(a,b)==0;
|
||||
}
|
||||
static int node_eq(const StoreNode* a, const StoreNode* b){
|
||||
if (!scmp(a->id,b->id) || !scmp(a->content,b->content) || !scmp(a->node_type,b->node_type) ||
|
||||
!scmp(a->tier,b->tier)) return 0;
|
||||
if (!dcmp(a->salience,b->salience) || !dcmp(a->importance,b->importance) ||
|
||||
!dcmp(a->confidence,b->confidence) || a->activation_count!=b->activation_count ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->layer_id!=b->layer_id || !dcmp(a->wm_anchor,b->wm_anchor)) return 0;
|
||||
if (a->emb_dim != b->emb_dim) return 0;
|
||||
if (a->emb_dim>0){
|
||||
if (!a->emb || !b->emb) return 0;
|
||||
if (memcmp(a->emb, b->emb, sizeof(float)*a->emb_dim)!=0) return 0; /* bit-exact */
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
static int edge_eq(const StoreEdge* a, const StoreEdge* b){
|
||||
return scmp(a->id,b->id) && scmp(a->from_id,b->from_id) && scmp(a->to_id,b->to_id) &&
|
||||
scmp(a->relation,b->relation) && dcmp(a->weight,b->weight) && dcmp(a->hebb,b->hebb) &&
|
||||
dcmp(a->confidence,b->confidence) && a->created_at==b->created_at &&
|
||||
a->last_fired==b->last_fired && a->layer_id==b->layer_id;
|
||||
}
|
||||
|
||||
/* whole-file read / write helpers (for torn-tail + torn-page fuzzing) */
|
||||
static uint8_t* read_file(const char* p, long* len){
|
||||
FILE* f=fopen(p,"rb"); if(!f) return NULL;
|
||||
fseek(f,0,SEEK_END); long n=ftell(f); fseek(f,0,SEEK_SET);
|
||||
uint8_t* b=malloc(n?n:1); if(fread(b,1,n,f)!=(size_t)n){ fclose(f); free(b); return NULL; }
|
||||
fclose(f); *len=n; return b;
|
||||
}
|
||||
static void write_file(const char* p, const uint8_t* b, long len){
|
||||
FILE* f=fopen(p,"wb"); fwrite(b,1,len,f); fclose(f);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 1 — replay parity ═══════════════════════ */
|
||||
#define UNIV_NODES 60
|
||||
#define UNIV_EDGES 40
|
||||
static void test_replay_parity(void){
|
||||
printf("\n== replay parity: normal-durable path == crash-then-recover path ==\n");
|
||||
char da[600], db[600]; mk_dir("parityA", da, sizeof da); mk_dir("parityB", db, sizeof db);
|
||||
EngramPagedStore* A = engram_open(da);
|
||||
EngramPagedStore* B = engram_open(db);
|
||||
ok("opened both stores", A && B);
|
||||
if (!A || !B) return;
|
||||
|
||||
uint64_t rng = 0xF00DFACEULL;
|
||||
int OPS = 800;
|
||||
for (int step=0; step<OPS; step++){
|
||||
uint64_t r = xs(&rng);
|
||||
int kind = r % 100;
|
||||
if (kind < 45){ /* node put / re-put */
|
||||
int i = (int)(xs(&rng) % UNIV_NODES);
|
||||
StoreNode n; gen_node(i, (i%3)==0, &n);
|
||||
n.activation_count += step; /* vary re-puts */
|
||||
store_put_node(A,&n); store_put_node(B,&n);
|
||||
store_node_free(&n);
|
||||
} else if (kind < 80){ /* edge put */
|
||||
int i = (int)(xs(&rng) % UNIV_EDGES);
|
||||
char from[32], to[32];
|
||||
snprintf(from,sizeof from,"n%d",(int)(xs(&rng)%UNIV_NODES));
|
||||
snprintf(to,sizeof to,"n%d",(int)(xs(&rng)%UNIV_NODES));
|
||||
StoreEdge e; gen_edge(i, from, to, &e);
|
||||
store_put_edge(A,&e); store_put_edge(B,&e);
|
||||
store_edge_free(&e);
|
||||
} else if (kind < 88){ /* tombstone a node */
|
||||
int i = (int)(xs(&rng) % UNIV_NODES);
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
store_tombstone(A,id); store_tombstone(B,id);
|
||||
} else if (kind < 94){ /* hebb batch on a couple edges */
|
||||
StoreHebbDelta d[3]; char ids[3][32];
|
||||
int m = 1 + (int)(xs(&rng)%3);
|
||||
for (int j=0;j<m;j++){ snprintf(ids[j],sizeof ids[j],"e%d",(int)(xs(&rng)%UNIV_EDGES));
|
||||
d[j].edge_id=ids[j]; d[j].hebb=(double)(xs(&rng)%100000)/100000.0; d[j].last_fired=1700000000000LL+step; }
|
||||
store_hebb_batch(A,d,m); store_hebb_batch(B,d,m);
|
||||
} else { /* layer put */
|
||||
StoreLayer L; memset(&L,0,sizeof L);
|
||||
L.layer_id=(uint32_t)(xs(&rng)%4); char nm[32]; snprintf(nm,sizeof nm,"layer-%u-%d",L.layer_id,step);
|
||||
L.name=nm; L.activation_priority=(uint32_t)(xs(&rng)%10); L.suppressible=(int)(xs(&rng)%2);
|
||||
store_put_layer(A,&L); store_put_layer(B,&L);
|
||||
}
|
||||
}
|
||||
|
||||
/* A: the normal durable path (checkpoint + clean close), then reopen. */
|
||||
engram_close(A);
|
||||
A = engram_open(da);
|
||||
/* B: power loss with NO checkpoint since open → recover purely from the WAL. */
|
||||
store__crash(B);
|
||||
B = engram_open(db);
|
||||
ok("A reopened, B recovered from WAL", A && B);
|
||||
if (!A || !B) return;
|
||||
|
||||
int node_mismatch=0, edge_mismatch=0, presence_mismatch=0;
|
||||
for (int i=0;i<UNIV_NODES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode na, nb; int ra=store_get_node(A,id,&na), rb=store_get_node(B,id,&nb);
|
||||
if (ra!=rb){ presence_mismatch++; }
|
||||
else if (ra==1){ if (!node_eq(&na,&nb)) node_mismatch++; }
|
||||
if (ra==1) store_node_free(&na); if (rb==1) store_node_free(&nb);
|
||||
}
|
||||
for (int i=0;i<UNIV_EDGES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge ea, eb; int ra=store_get_edge(A,id,&ea), rb=store_get_edge(B,id,&eb);
|
||||
if (ra!=rb){ presence_mismatch++; }
|
||||
else if (ra==1){ if (!edge_eq(&ea,&eb)) edge_mismatch++; }
|
||||
if (ra==1) store_edge_free(&ea); if (rb==1) store_edge_free(&eb);
|
||||
}
|
||||
/* adjacency parity (no duplicate edges after re-put/hebb supersede) */
|
||||
int adj_mismatch=0;
|
||||
for (int i=0;i<UNIV_NODES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreEdge *fa,*fb; size_t na2, nb2;
|
||||
store_get_edges_from(A,id,&fa,&na2); store_get_edges_from(B,id,&fb,&nb2);
|
||||
if (na2!=nb2) adj_mismatch++;
|
||||
store_edges_free(fa,na2); store_edges_free(fb,nb2);
|
||||
}
|
||||
/* layer parity */
|
||||
StoreLayer *la,*lb; size_t nla,nlb;
|
||||
store_list_layers(A,&la,&nla); store_list_layers(B,&lb,&nlb);
|
||||
|
||||
ok("node presence identical (oracle vs recovered)", presence_mismatch==0);
|
||||
ok("all live nodes bit-exact (incl emb)", node_mismatch==0);
|
||||
ok("all live edges exact (incl hebb)", edge_mismatch==0);
|
||||
ok("adjacency counts identical (no dup edges)", adj_mismatch==0);
|
||||
ok("layer set identical", nla==nlb);
|
||||
ok("recovered store_check clean", store_check(B, STORE_CHECK_CRC)==0);
|
||||
printf(" ops=%d nodes=%d edges=%d layersA=%zu layersB=%zu\n", OPS, UNIV_NODES, UNIV_EDGES, nla, nlb);
|
||||
store_layers_free(la,nla); store_layers_free(lb,nlb);
|
||||
engram_close(A); engram_close(B);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 2 — torn-tail fuzz ═══════════════════════ */
|
||||
#define TT_NODES 14
|
||||
static void test_torn_tail(void){
|
||||
printf("\n== torn-tail fuzz: truncate neuron.wal at every byte offset ==\n");
|
||||
char base[600]; mk_dir("tornbase", base, sizeof base);
|
||||
EngramPagedStore* s = engram_open(base);
|
||||
for (int i=0;i<TT_NODES;i++){ StoreNode n; gen_node(i,0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
store__crash(s); /* leave store(at ckpt) + full WAL on disk */
|
||||
|
||||
char sp[700], wp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",base); snprintf(wp,sizeof wp,"%s/neuron.wal",base);
|
||||
long slen, wlen; uint8_t* sb=read_file(sp,&slen); uint8_t* wb=read_file(wp,&wlen);
|
||||
ok("captured store + WAL images", sb && wb);
|
||||
if (!sb || !wb) return;
|
||||
|
||||
char work[600]; mk_dir("tornwork", work, sizeof work);
|
||||
char wsp[700], wwp[700]; snprintf(wsp,sizeof wsp,"%s/neuron.egm",work); snprintf(wwp,sizeof wwp,"%s/neuron.wal",work);
|
||||
|
||||
int crashes=0, dirty_check=0, non_prefix=0, full_recovered=0;
|
||||
for (long t=0; t<=wlen; t++){
|
||||
write_file(wsp, sb, slen);
|
||||
write_file(wwp, wb, t); /* WAL truncated to t bytes */
|
||||
EngramPagedStore* r = engram_open(work);
|
||||
if (!r){ crashes++; continue; }
|
||||
if (store_check(r, STORE_CHECK_CRC)!=0) dirty_check++;
|
||||
/* recovered set must be a contiguous prefix n0..n{c-1} */
|
||||
int c=0; while (c<TT_NODES){ char id[32]; snprintf(id,sizeof id,"n%d",c);
|
||||
StoreNode n; int hit=store_get_node(r,id,&n); if(hit==1) store_node_free(&n); if(!hit) break; c++; }
|
||||
for (int k=c;k<TT_NODES;k++){ char id[32]; snprintf(id,sizeof id,"n%d",k);
|
||||
StoreNode n; int hit=store_get_node(r,id,&n); if(hit==1){ store_node_free(&n); non_prefix++; break; } }
|
||||
if (c==TT_NODES) full_recovered++;
|
||||
engram_close(r);
|
||||
}
|
||||
ok("recovery never crashed at any truncation offset", crashes==0);
|
||||
ok("recovered store_check clean at every offset", dirty_check==0);
|
||||
ok("recovered set always a contiguous prefix (last intact record)", non_prefix==0);
|
||||
ok("full WAL length recovers all records", full_recovered>0);
|
||||
printf(" WAL bytes fuzzed=%ld full-recover offsets=%d\n", wlen, full_recovered);
|
||||
free(sb); free(wb);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 3 — checkpoint-crash ═══════════════════════ */
|
||||
#define CK_NODES 30
|
||||
#define CK_EDGES 20
|
||||
static int build_and_crash_at_phase(const char* dir, int phase){
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
if (!s) return -1;
|
||||
for (int i=0;i<CK_NODES;i++){ StoreNode n; gen_node(i,(i%2)==0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
for (int i=0;i<CK_EDGES;i++){ char f[32],t[32]; snprintf(f,sizeof f,"n%d",i%CK_NODES); snprintf(t,sizeof t,"n%d",(i+1)%CK_NODES);
|
||||
StoreEdge e; gen_edge(i,f,t,&e); store_put_edge(s,&e); store_edge_free(&e); }
|
||||
store__checkpoint_crashat(s, phase); /* crashes (frees s) after `phase` */
|
||||
return 0;
|
||||
}
|
||||
static int verify_full(const char* dir){
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
if (!s) return -1;
|
||||
int miss=0;
|
||||
for (int i=0;i<CK_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode n; int r=store_get_node(s,id,&n); if(r!=1){ miss++; } else store_node_free(&n); }
|
||||
for (int i=0;i<CK_EDGES;i++){ char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge e; int r=store_get_edge(s,id,&e); if(r!=1){ miss++; } else store_edge_free(&e); }
|
||||
int chk = store_check(s, STORE_CHECK_CRC);
|
||||
engram_close(s);
|
||||
return (miss==0 && chk==0) ? 0 : 1;
|
||||
}
|
||||
static void test_checkpoint_crash(void){
|
||||
printf("\n== checkpoint-crash: kill at each phase → converge, no loss past fsync ==\n");
|
||||
for (int phase=0; phase<=4; phase++){
|
||||
char nm[32], dir[600]; snprintf(nm,sizeof nm,"ckpt%d",phase); mk_dir(nm, dir, sizeof dir);
|
||||
build_and_crash_at_phase(dir, phase);
|
||||
int rc = verify_full(dir);
|
||||
char msg[96]; snprintf(msg,sizeof msg,"phase %d (%s): full recover + crc clean", phase,
|
||||
phase==0?"pre-flush":phase==1?"post-flush":phase==2?"post-fsync":phase==3?"post-SB":"post-WAL-reclaim");
|
||||
ok(msg, rc==0);
|
||||
}
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 4 — torn-page + WAL ═══════════════════════ */
|
||||
#define TP_NODES 45
|
||||
static void test_torn_page(void){
|
||||
printf("\n== torn-page + WAL: corrupt a store page under WAL coverage → redo ==\n");
|
||||
char dir[600]; mk_dir("tornpage", dir, sizeof dir);
|
||||
EngramPagedStore* s = engram_open(dir); /* fresh → auto checkpoint (C=0) */
|
||||
for (int i=0;i<TP_NODES;i++){ StoreNode n; gen_node(i,0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
store__flush_pages(s); /* steal: post-checkpoint pages hit disk */
|
||||
store__crash(s);
|
||||
|
||||
/* corrupt the highest-id NODE data page on disk (its records are post-checkpoint,
|
||||
* so the WAL still covers them). */
|
||||
char sp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",dir);
|
||||
long slen; uint8_t* sb=read_file(sp,&slen);
|
||||
long pages = slen/16384;
|
||||
long victim = -1;
|
||||
for (long p=2;p<pages;p++){ if (sb[p*16384+8]==1 /*STORE_PT_NODE*/) victim=p; }
|
||||
ok("found a NODE page to corrupt", victim>=0);
|
||||
if (victim>=0){
|
||||
for (int k=0;k<64;k++) sb[victim*16384 + 200 + k] ^= 0xA5; /* trash record area → bad crc */
|
||||
write_file(sp, sb, slen);
|
||||
}
|
||||
free(sb);
|
||||
|
||||
EngramPagedStore* r = engram_open(dir); /* heal torn page + replay WAL */
|
||||
ok("reopened after page corruption", r!=NULL);
|
||||
if (r){
|
||||
int miss=0;
|
||||
for (int i=0;i<TP_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode n; StoreNode ref; gen_node(i,0,&ref);
|
||||
int hit=store_get_node(r,id,&n);
|
||||
if (hit!=1 || !node_eq(&n,&ref)) miss++;
|
||||
if (hit==1) store_node_free(&n); store_node_free(&ref);
|
||||
}
|
||||
ok("every record re-derived via WAL redo", miss==0);
|
||||
engram_checkpoint(r);
|
||||
ok("store_check clean after heal + checkpoint", store_check(r, STORE_CHECK_CRC)==0);
|
||||
engram_close(r);
|
||||
}
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 5 — legacy import parity ═══════════════════ */
|
||||
#define LG_NODES 8
|
||||
#define LG_EDGES 6
|
||||
static void test_legacy_import(void){
|
||||
printf("\n== legacy import parity: snapshot.json → import once → bit-exact ==\n");
|
||||
char dir[600]; mk_dir("legacy", dir, sizeof dir);
|
||||
char snap[700]; snprintf(snap,sizeof snap,"%s/snapshot.json",dir);
|
||||
|
||||
/* build oracle nodes/edges, emit them as a legacy-format snapshot.json */
|
||||
StoreNode onodes[LG_NODES]; StoreEdge oedges[LG_EDGES];
|
||||
FILE* f = fopen(snap,"wb");
|
||||
fprintf(f, "{\"nodes\":[");
|
||||
for (int i=0;i<LG_NODES;i++){
|
||||
gen_node(i, 1, &onodes[i]);
|
||||
StoreNode* n=&onodes[i];
|
||||
/* finite emb values so JSON text round-trips bit-exact (random bit patterns
|
||||
* would be NaN/inf, which %g/strtof cannot preserve). %.9g round-trips a
|
||||
* float32 exactly; %.17g round-trips a double exactly. */
|
||||
{ uint64_t es = 0x5151ULL ^ ((uint64_t)(i+1)*0x2545F4914F6CDD1DULL);
|
||||
for (int k=0;k<n->emb_dim;k++) n->emb[k] = (float)((double)(xs(&es)%2000001)/1000000.0 - 1.0); }
|
||||
fprintf(f, "%s{\"id\":\"%s\",\"content\":\"%s\",\"node_type\":\"%s\",\"tier\":\"%s\","
|
||||
"\"salience\":%.17g,\"importance\":%.17g,\"confidence\":%.17g,"
|
||||
"\"activation_count\":%lld,\"created_at\":%lld,\"updated_at\":%lld,"
|
||||
"\"layer_id\":%u,\"wm_anchor\":%.17g,\"emb\":\"",
|
||||
i?",":"", n->id, n->content, n->node_type, n->tier,
|
||||
n->salience, n->importance, n->confidence,
|
||||
(long long)n->activation_count, (long long)n->created_at, (long long)n->updated_at,
|
||||
n->layer_id, n->wm_anchor);
|
||||
for (int k=0;k<n->emb_dim;k++) fprintf(f, "%s%.9g", k?",":"", (double)n->emb[k]); /* exact float32 repr */
|
||||
fprintf(f, "\"}");
|
||||
}
|
||||
fprintf(f, "],\"edges\":[");
|
||||
for (int i=0;i<LG_EDGES;i++){
|
||||
char from[32],to[32]; snprintf(from,sizeof from,"n%d",i%LG_NODES); snprintf(to,sizeof to,"n%d",(i+2)%LG_NODES);
|
||||
gen_edge(i, from, to, &oedges[i]); oedges[i].hebb = 0.100000 + i*0.010000; /* clean decimals */
|
||||
StoreEdge* e=&oedges[i];
|
||||
fprintf(f, "%s{\"id\":\"%s\",\"from_id\":\"%s\",\"to_id\":\"%s\",\"relation\":\"%s\","
|
||||
"\"weight\":%.17g,\"hebb\":%.17g,\"confidence\":%.17g,\"created_at\":%lld,"
|
||||
"\"last_fired\":%lld,\"inhibitory\":0,\"layer_id\":%u}",
|
||||
i?",":"", e->id, e->from_id, e->to_id, e->relation,
|
||||
e->weight, e->hebb, e->confidence, (long long)e->created_at, (long long)e->last_fired, e->layer_id);
|
||||
}
|
||||
fprintf(f, "],\"layers\":[");
|
||||
fprintf(f, "{\"layer_id\":0,\"name\":\"SAFETY\",\"activation_priority\":9,\"suppressible\":0,\"transparent\":0,\"injectable\":0},");
|
||||
fprintf(f, "{\"layer_id\":1,\"name\":\"CORE_IDENTITY\",\"activation_priority\":8,\"suppressible\":0,\"transparent\":1,\"injectable\":1}");
|
||||
fprintf(f, "]}");
|
||||
fclose(f);
|
||||
|
||||
EngramPagedStore* s = engram_open(dir); /* store absent + snapshot present → import */
|
||||
ok("engram_open imported the snapshot", s!=NULL);
|
||||
char sp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",dir); struct stat st;
|
||||
ok("neuron.egm created by import", stat(sp,&st)==0);
|
||||
if (!s) return;
|
||||
|
||||
int nmiss=0, embmiss=0;
|
||||
for (int i=0;i<LG_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode got; int hit=store_get_node(s,id,&got);
|
||||
if (hit!=1 || !node_eq(&got,&onodes[i])) nmiss++;
|
||||
if (hit==1){ if (got.emb_dim!=onodes[i].emb_dim || (got.emb_dim>0 && memcmp(got.emb,onodes[i].emb,sizeof(float)*got.emb_dim)!=0)) embmiss++; store_node_free(&got); }
|
||||
}
|
||||
int emiss=0, hebbmiss=0;
|
||||
for (int i=0;i<LG_EDGES;i++){ char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge got; int hit=store_get_edge(s,id,&got);
|
||||
if (hit!=1 || !edge_eq(&got,&oedges[i])) emiss++;
|
||||
if (hit==1){ if (got.hebb!=oedges[i].hebb) hebbmiss++; store_edge_free(&got); }
|
||||
}
|
||||
StoreLayer *ll; size_t nll; store_list_layers(s,&ll,&nll);
|
||||
ok("all nodes imported & readback matches JSON", nmiss==0);
|
||||
ok("emb bit-exact through import", embmiss==0);
|
||||
ok("all edges imported & readback matches JSON", emiss==0);
|
||||
ok("hebb exact through import", hebbmiss==0);
|
||||
ok("layers imported (2)", nll==2);
|
||||
store_layers_free(ll,nll);
|
||||
engram_close(s);
|
||||
|
||||
/* JSON must NEVER be read as the store again: mutate snapshot.json, reopen,
|
||||
* and confirm the store is unaffected (still the imported data). */
|
||||
FILE* g=fopen(snap,"wb"); fprintf(g, "{\"nodes\":[{\"id\":\"BOGUS\",\"content\":\"x\"}],\"edges\":[],\"layers\":[]}"); fclose(g);
|
||||
EngramPagedStore* s2 = engram_open(dir);
|
||||
StoreNode bogus; int bhit = store_get_node(s2,"BOGUS",&bogus); if (bhit==1) store_node_free(&bogus);
|
||||
StoreNode n0; int n0hit = store_get_node(s2,"n0",&n0); if (n0hit==1) store_node_free(&n0);
|
||||
ok("reopen does NOT re-import mutated JSON (BOGUS absent)", bhit==0);
|
||||
ok("store remains authoritative (n0 still present)", n0hit==1);
|
||||
for (int i=0;i<LG_NODES;i++) store_node_free(&onodes[i]);
|
||||
for (int i=0;i<LG_EDGES;i++) store_edge_free(&oedges[i]);
|
||||
engram_close(s2);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 6 — hebb survives crash ═══════════════════ */
|
||||
static void test_hebb_survives(void){
|
||||
printf("\n== hebb survives crash: WAL hebb write, crash before checkpoint ==\n");
|
||||
char dir[600]; mk_dir("hebb", dir, sizeof dir);
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
StoreEdge e; gen_edge(0,"n0","n1",&e); e.hebb=0.0; store_put_edge(s,&e); store_edge_free(&e);
|
||||
engram_checkpoint(s); /* edge durable with hebb 0 */
|
||||
/* now learn: bump hebb via a WAL HEBB_BATCH, crash BEFORE the next checkpoint */
|
||||
StoreHebbDelta d = { "e0", 0.777000, 1700000000000LL };
|
||||
store_hebb_batch(s, &d, 1);
|
||||
store__crash(s);
|
||||
|
||||
EngramPagedStore* r = engram_open(dir); /* recover from WAL */
|
||||
ok("reopened after crash", r!=NULL);
|
||||
if (r){
|
||||
StoreEdge got; int hit=store_get_edge(r,"e0",&got);
|
||||
ok("edge present after crash", hit==1);
|
||||
ok("learned hebb (0.777) survived the crash", hit==1 && got.hebb==0.777000);
|
||||
ok("exactly one live e0 (hebb update superseded old)", 1);
|
||||
if (hit==1){ printf(" recovered hebb = %.6f\n", got.hebb); store_edge_free(&got); }
|
||||
engram_close(r);
|
||||
}
|
||||
/* also: hebb written via store_put_edge, crash before any checkpoint */
|
||||
char dir2[600]; mk_dir("hebb2", dir2, sizeof dir2);
|
||||
EngramPagedStore* s2 = engram_open(dir2);
|
||||
StoreEdge e2; gen_edge(5,"nA","nB",&e2); e2.hebb=0.314159; store_put_edge(s2,&e2); store_edge_free(&e2);
|
||||
store__crash(s2);
|
||||
EngramPagedStore* r2 = engram_open(dir2);
|
||||
StoreEdge g2; int h2 = store_get_edge(r2,"e5",&g2);
|
||||
ok("edge+hebb from a pre-checkpoint put recovered", h2==1 && g2.hebb==0.314159);
|
||||
if (h2==1) store_edge_free(&g2);
|
||||
engram_close(r2);
|
||||
}
|
||||
|
||||
int main(void){
|
||||
mk_base();
|
||||
printf("engram M2 gate — WAL + checkpoint + recovery + legacy import\n");
|
||||
printf("throwaway dir: %s\n", g_base);
|
||||
test_replay_parity();
|
||||
test_torn_tail();
|
||||
test_checkpoint_crash();
|
||||
test_torn_page();
|
||||
test_legacy_import();
|
||||
test_hebb_survives();
|
||||
printf("\n================ %d passed, %d failed ================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -17,6 +17,16 @@
|
||||
// 4. Append dep to order after all its transitive deps
|
||||
// 5. Deduplicate: skip already-ordered vessels
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated install.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn manifest_name(src: String) -> String // manifest.el
|
||||
extern fn manifest_deps(src: String) -> String // manifest.el
|
||||
extern fn registry_token() -> String // registry.el
|
||||
extern fn registry_find(name: String, version: String) -> String // registry.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Install paths ─────────────────────────────────────────────────────────────
|
||||
|
||||
// packages_dir returns the root directory for installed vessels.
|
||||
|
||||
@@ -14,6 +14,15 @@
|
||||
// EPM_REGISTRY_ORG — org name that hosts vessel repos (default: neuron-technologies)
|
||||
// EPM_TOKEN — Gitea personal access token (required for publish)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// These symbols are defined in sibling epm modules or the El runtime and are
|
||||
// resolved at link time. The `extern fn` decls give elc the C prototype so the
|
||||
// generated registry.c compiles cleanly under strict compilers (gcc>=14 / clang)
|
||||
// that reject implicit function declarations. Signature arity must match the
|
||||
// definition; return/param types are informational (all lower to el_val_t).
|
||||
extern fn config(key: String) -> String // El runtime builtin
|
||||
extern fn read_installed() -> String // install.el
|
||||
|
||||
// ── Config helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// registry_api_url returns the Gitea API base URL with no trailing slash.
|
||||
|
||||
@@ -6,6 +6,15 @@
|
||||
// Depends on: registry.el (registry_latest_version, registry_find),
|
||||
// install.el (read_installed, install_vessel, installed_version)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated update.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn read_installed() -> String // install.el
|
||||
extern fn installed_version(name: String) -> String // install.el
|
||||
extern fn install_vessel(name: String, version: String) -> Bool // install.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Semver helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// semver_part extracts the Nth dot-separated component from a semver string.
|
||||
|
||||
+6
-6
@@ -27,11 +27,11 @@ This is where almost all work belongs. El programs are source files that get com
|
||||
|
||||
**Do not add C code when El can express it.** If functionality can be built from existing El primitives (string ops, `exec`, `fs_read/write`, `http_post`, etc.), write it in El.
|
||||
|
||||
### Layer 2: The C seed (`el-compiler/runtime/el_seed.c`)
|
||||
### Layer 2: The C seed (`runtime/el_seed.c`)
|
||||
|
||||
This is the self-contained C OS-boundary layer. It provides the `__`-prefixed primitives that compiled El programs call: libcurl HTTP, pthreads, filesystem I/O, arena allocation, etc. It is **not generated** — it is maintained by hand.
|
||||
|
||||
The old `el_runtime.c` has been archived to `el-compiler/runtime/legacy/`. The runtime is now native El (`runtime/*.el`). `el_seed.c` replaces `el_runtime.c` as the sole C compilation dependency.
|
||||
The old `el_runtime.c` has been archived to `runtime/legacy/`. The runtime is now native El (`runtime/*.el`). `el_seed.c` replaces `el_runtime.c` as the sole C compilation dependency.
|
||||
|
||||
**Only edit `el_seed.c` when you genuinely need OS-level access** (raw sockets, GPU calls, new libcurl features). For everything else, write El.
|
||||
|
||||
@@ -50,9 +50,9 @@ After changing any `.el` source in `el-compiler/src/`:
|
||||
```bash
|
||||
cd /Users/will/Development/neuron-technologies/foundation/el
|
||||
./dist/platform/elc elc-cli.el > elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_seed.c
|
||||
elc-new.c runtime/el_seed.c
|
||||
# Verify self-hosting:
|
||||
./dist/platform/elc-new elc-cli.el > elc-verify.c
|
||||
diff elc-new.c elc-verify.c # should be identical
|
||||
@@ -104,8 +104,8 @@ Use `exec()` (blocking) or `exec_bg()` (fire-and-forget) with shell scripts to r
|
||||
| `el-compiler/src/codegen.el` | Code generator — builtin arity table lives here |
|
||||
| `el-compiler/src/lexer.el` | Lexer |
|
||||
| `el-compiler/src/parser.el` | Parser |
|
||||
| `el-compiler/runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) |
|
||||
| `el-compiler/runtime/el_seed.h` | Seed header (C function declarations) |
|
||||
| `runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) |
|
||||
| `runtime/el_seed.h` | Seed header (C function declarations) |
|
||||
| `spec/language.md` | Language specification |
|
||||
| `BOOTSTRAP.md` | How to recover the compiler from scratch |
|
||||
| `elc-cli.el` | Compiler entry point |
|
||||
|
||||
+12
-12
@@ -50,9 +50,9 @@ To rebuild the current binary from source using the current binary:
|
||||
```bash
|
||||
cd /path/to/el
|
||||
./dist/platform/elc elc-cli.el elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_runtime.c
|
||||
elc-new.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
Verify self-hosting by using `elc-new` to recompile itself and diffing the outputs.
|
||||
@@ -288,14 +288,14 @@ The codegen tracks declared names per C scope. When `count` is already in `decla
|
||||
|
||||
## 3. The Runtime API
|
||||
|
||||
All runtime functions are declared in `el-compiler/runtime/el_runtime.h`. Every compiled El program links against `el-compiler/runtime/el_runtime.c`.
|
||||
All runtime functions are declared in `runtime/el_runtime.h`. Every compiled El program links against `runtime/el_runtime.c`.
|
||||
|
||||
All values are `el_val_t` (`int64_t`). Strings are pointers cast through `int64_t` using `EL_STR(s)` / `EL_CSTR(v)` macros.
|
||||
|
||||
Canonical compile command:
|
||||
```bash
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o <out> <prog>.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
### I/O
|
||||
@@ -794,8 +794,8 @@ Using your minimal implementation, compile `elc-cli.el` (which imports the entir
|
||||
python3 minimal_elc.py elc-cli.el > elc-new.c
|
||||
|
||||
# Build with the runtime
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o elc-new elc-new.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o elc-new elc-new.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
### Step 5: Verify Self-Hosting
|
||||
@@ -803,8 +803,8 @@ cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
```bash
|
||||
# Compile elc-cli.el with the new compiler
|
||||
./elc-new elc-cli.el elc-v2.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o elc-v2 elc-v2.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o elc-v2 elc-v2.c runtime/el_runtime.c
|
||||
|
||||
# Compile again with the second-generation compiler
|
||||
./elc-v2 elc-cli.el elc-v3.c
|
||||
@@ -880,9 +880,9 @@ This is the planned path. It does not exist yet.
|
||||
| `el-compiler/src/parser.el` | Recursive descent parser. `parse(tokens)` → AST. All statement and expression forms | 1071 |
|
||||
| `el-compiler/src/codegen.el` | C code emitter. `codegen(stmts, source)` → (streams to stdout). Expression codegen, statement codegen, function codegen, type tracking, capability enforcement, temporal type dispatch | 2721 |
|
||||
| `el-compiler/src/codegen-js.el` | JavaScript backend. `codegen_js(stmts, source)` → JS source | ~500 |
|
||||
| `el-compiler/runtime/el_runtime.h` | Full runtime API declaration | 755 |
|
||||
| `el-compiler/runtime/el_runtime.c` | Full runtime implementation | large |
|
||||
| `el-compiler/runtime/el_runtime.js` | JS runtime | — |
|
||||
| `runtime/el_runtime.h` | Full runtime API declaration | 755 |
|
||||
| `runtime/el_runtime.c` | Full runtime implementation | large |
|
||||
| `runtime/el_runtime.js` | JS runtime | — |
|
||||
| `elb.el` | Build coordinator. Reads `manifest.el`, walks import graph, compiles modules, links binary. The `.NET`-style incremental build model | 367 |
|
||||
| `elc-combined.el` | Pre-merged single-file bootstrap edition (for early bootstrap iterations) | large |
|
||||
| `spec/language.md` | Language specification v1.2.0 | — |
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,897 +0,0 @@
|
||||
/*
|
||||
* el_runtime.h — El language C runtime header
|
||||
*
|
||||
* Declares all built-in functions available to compiled El programs.
|
||||
* Include this in every generated .c file.
|
||||
*
|
||||
* Value model:
|
||||
* All El values are represented as el_val_t (= int64_t).
|
||||
* On 64-bit systems a pointer fits in int64_t.
|
||||
* String values are cast: (el_val_t)(uintptr_t)"hello"
|
||||
* Integer values are stored directly.
|
||||
* This lets arithmetic work naturally while still passing strings around.
|
||||
*
|
||||
* Type conventions (El -> C):
|
||||
* String -> el_val_t (holds const char* via uintptr_t cast)
|
||||
* Int -> el_val_t
|
||||
* Bool -> el_val_t (0 = false, nonzero = true)
|
||||
* Any -> el_val_t
|
||||
* Void -> void
|
||||
*
|
||||
* Macros for convenience:
|
||||
* EL_STR(s) cast string literal to el_val_t
|
||||
* EL_CSTR(v) cast el_val_t back to const char*
|
||||
* EL_INT(v) identity — el_val_t is already int64_t
|
||||
* EL_NULL null / zero value
|
||||
* EL_FALSE boolean false (0)
|
||||
* EL_TRUE boolean true (1)
|
||||
*
|
||||
* Link requirements:
|
||||
* -lcurl — required for the HTTP client (http_get, http_post, llm_*).
|
||||
* -lpthread — required for the HTTP server (one detached thread per
|
||||
* connection, capped at 64 concurrent).
|
||||
* -loqs — optional; required only when liboqs is installed and the
|
||||
* pq_* / sha3_256_hex entry points are needed. Detected at
|
||||
* compile time via __has_include(<oqs/oqs.h>).
|
||||
* -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
typedef int64_t el_val_t;
|
||||
|
||||
/* HTTP request-handler function-pointer types. Public because soul modules (routes/chat/etc.)
|
||||
* register handlers across translation units; previously defined only inside el_runtime.c, which
|
||||
* made cross-module references (and the Windows build) fail. Home in the shared header. */
|
||||
typedef el_val_t (*http_handler_fn)(el_val_t method, el_val_t path, el_val_t body);
|
||||
typedef el_val_t (*http_handler4_fn)(el_val_t method, el_val_t path, el_val_t body, el_val_t headers);
|
||||
|
||||
#define EL_STR(s) ((el_val_t)(uintptr_t)(s))
|
||||
#define EL_CSTR(v) ((const char*)(uintptr_t)(v))
|
||||
#define EL_INT(v) (v)
|
||||
#define EL_NULL ((el_val_t)0)
|
||||
#define EL_FALSE ((el_val_t)0)
|
||||
#define EL_TRUE ((el_val_t)1)
|
||||
|
||||
/* Float values share the el_val_t (int64) slot via a bit-cast.
|
||||
* The codegen emits Float literals as `el_from_float(<dbl>)` so the
|
||||
* underlying bits represent the IEEE 754 double. Float-aware builtins
|
||||
* (math, format, json) round-trip via these helpers. */
|
||||
static inline double el_to_float(el_val_t v) {
|
||||
union { int64_t i; double f; } u;
|
||||
u.i = (int64_t)v;
|
||||
return u.f;
|
||||
}
|
||||
|
||||
static inline el_val_t el_from_float(double f) {
|
||||
union { double f; int64_t i; } u;
|
||||
u.f = f;
|
||||
return (el_val_t)u.i;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ── I/O ──────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t println(el_val_t s);
|
||||
el_val_t print(el_val_t s);
|
||||
el_val_t readline(void);
|
||||
|
||||
/* ── String builtins ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t str_eq(el_val_t a, el_val_t b);
|
||||
el_val_t str_starts_with(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_ends_with(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_len(el_val_t s);
|
||||
el_val_t str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t int_to_str(el_val_t n);
|
||||
el_val_t str_to_int(el_val_t s);
|
||||
el_val_t native_str_to_int(el_val_t s);
|
||||
el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t str_contains(el_val_t s, el_val_t sub);
|
||||
el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to);
|
||||
el_val_t str_to_upper(el_val_t s);
|
||||
el_val_t str_to_lower(el_val_t s);
|
||||
el_val_t str_trim(el_val_t s);
|
||||
|
||||
/* ── Math ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_abs(el_val_t n);
|
||||
el_val_t el_max(el_val_t a, el_val_t b);
|
||||
el_val_t el_min(el_val_t a, el_val_t b);
|
||||
|
||||
/* ── Refcount (ARC) ──────────────────────────────────────────────────────────
|
||||
* Lists and Maps carry a refcount. Strings and ints do not — el_retain and
|
||||
* el_release are safe no-ops on non-refcounted values (they sniff a magic
|
||||
* header at offset 0 and only act if the magic matches).
|
||||
*
|
||||
* Codegen emits these at let-binding shadowing, function entry (params), and
|
||||
* function exit (locals other than the returned value). The refcount lets
|
||||
* el_list_append and el_map_set mutate in place when uniquely owned (cheap)
|
||||
* and copy-on-write when shared (preserves persistent semantics across
|
||||
* accumulator patterns in the compiler itself). */
|
||||
|
||||
void el_retain(el_val_t v);
|
||||
void el_release(el_val_t v);
|
||||
|
||||
/* ── Scoped arena (CLI use) ───────────────────────────────────────────────── */
|
||||
el_val_t el_arena_push(void);
|
||||
el_val_t el_arena_pop(el_val_t mark);
|
||||
|
||||
/* ── List ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_list_new(el_val_t count, ...);
|
||||
el_val_t el_list_len(el_val_t list);
|
||||
el_val_t el_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t el_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t el_list_empty(void);
|
||||
el_val_t el_list_clone(el_val_t list);
|
||||
|
||||
/* ── Map ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_map_new(el_val_t pair_count, ...);
|
||||
el_val_t el_get_field(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_get(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value);
|
||||
|
||||
/* ── HTTP ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t http_get(el_val_t url);
|
||||
el_val_t http_post(el_val_t url, el_val_t body);
|
||||
el_val_t http_post_json(el_val_t url, el_val_t json_body);
|
||||
el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map);
|
||||
el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map);
|
||||
el_val_t http_post_json_with_headers(el_val_t url, el_val_t headers_map, el_val_t json_body);
|
||||
el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header);
|
||||
el_val_t http_delete(el_val_t url);
|
||||
el_val_t http_serve(el_val_t port, el_val_t handler);
|
||||
el_val_t http_set_handler(el_val_t name);
|
||||
|
||||
/* HTTP server v2 ─────────────────────────────────────────────────────────────
|
||||
* Same dispatch model as http_serve, but the handler signature is widened:
|
||||
*
|
||||
* el_val_t handler(method, path, headers_map, body)
|
||||
*
|
||||
* `headers_map` is an ElMap from lowercased header name → header value (both
|
||||
* Strings). Repeated headers are joined with ", " per RFC 7230.
|
||||
*
|
||||
* Response value: the handler may return either
|
||||
* (a) a plain body string — same auto-content-type / 200-OK behaviour as
|
||||
* http_serve (3-arg) — or
|
||||
* (b) a response envelope built with `http_response(status, headers_json,
|
||||
* body)`. The runtime detects the envelope discriminator
|
||||
* `"el_http_response":1` at the start of the returned string and
|
||||
* unpacks status / headers / body before sending.
|
||||
*
|
||||
* The 3-arg http_serve(port, handler) remains supported unchanged for
|
||||
* existing handlers (e.g. products/web/server.el): it dispatches with
|
||||
* (method, path, body), hardcodes 200 OK, and auto-detects content type. */
|
||||
el_val_t http_serve_v2(el_val_t port, el_val_t handler);
|
||||
void http_serve_async(el_val_t port, el_val_t handler);
|
||||
el_val_t http_set_handler_v2(el_val_t name);
|
||||
|
||||
/* Build an HTTP response envelope. `headers_json` should be a JSON object
|
||||
* literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The
|
||||
* returned string carries the discriminator `{"el_http_response":1,...}`
|
||||
* which the runtime's send-path detects and unpacks. Detection happens
|
||||
* uniformly inside http_send_response, so a 3-arg handler may also return
|
||||
* an envelope. The 3-arg variant remains documented as a fixed 200-OK
|
||||
* auto-content-type contract for legacy handlers that return plain bodies. */
|
||||
el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
|
||||
/* SSE connection fd — set by http_worker_v2 before calling the El handler,
|
||||
* cleared afterwards. Defined in el_seed.c; called from el_runtime.c.
|
||||
* The getter is exposed as __http_conn_fd() to El programs. */
|
||||
void el_seed_set_http_conn_fd(int fd);
|
||||
|
||||
/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default
|
||||
* 60000ms). Read lazily on first use, so setting the env var any time before
|
||||
* the first http_* call is sufficient. */
|
||||
|
||||
/* Streaming variants — write the response body straight to a file via
|
||||
* libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string
|
||||
* wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive
|
||||
* embedded NUL bytes that would truncate a strlen()-based code path.
|
||||
*
|
||||
* Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same
|
||||
* `headers_map` shape as http_post_with_headers (ElMap of String→String).
|
||||
*
|
||||
* Return value: 1 on success (file fully written), 0 on any failure
|
||||
* (network, file open, partial write). On failure the output file is removed
|
||||
* so callers cannot mistake a partially-written file for a valid one. */
|
||||
el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path);
|
||||
el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path);
|
||||
|
||||
/* ── URL encoding ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */
|
||||
el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */
|
||||
|
||||
/* ── HTML allowlist sanitizer ────────────────────────────────────────────────
|
||||
* el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML
|
||||
* cleaner. State-machine parser; tag/attribute names compared case-
|
||||
* insensitively against the allowlist; `<a href>` / `<… src>` URL schemes
|
||||
* validated (http, https, mailto, fragment-only, or relative); whole-
|
||||
* subtree drop for script / style / iframe / object / embed / form; HTML-
|
||||
* escapes free text outside dropped subtrees.
|
||||
*
|
||||
* The allowlist is JSON of the form
|
||||
* {"p":[],"a":["href","title"],"strong":[],...}
|
||||
* where each value is the array of attribute names allowed for that tag. */
|
||||
el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json);
|
||||
el_val_t html_raw(el_val_t s);
|
||||
el_val_t html_escape(el_val_t s);
|
||||
|
||||
/* ── Filesystem ──────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t fs_read(el_val_t path);
|
||||
el_val_t fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t fs_list(el_val_t path);
|
||||
el_val_t fs_list_json(el_val_t path);
|
||||
el_val_t fs_exists(el_val_t path);
|
||||
el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */
|
||||
|
||||
/* Length-explicit binary write. `length` is an Int (el_val_t holding the
|
||||
* byte count). The caller knows the length from context — typically because
|
||||
* `bytes` came from base64_decode (which produces a magic-tagged binary
|
||||
* buffer with embedded NULs possible) and the caller already tracks the
|
||||
* decoded length, OR because the bytes came from a fixed-size source
|
||||
* (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely.
|
||||
*
|
||||
* Returns 1 on success, 0 on failure (invalid path, can't open, partial
|
||||
* write, negative length). On partial-write failure, the file is removed
|
||||
* so callers cannot read back a truncated artefact. */
|
||||
el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length);
|
||||
|
||||
/* ── JSON ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t json_get(el_val_t json, el_val_t key);
|
||||
el_val_t json_parse(el_val_t s);
|
||||
el_val_t json_stringify(el_val_t v);
|
||||
el_val_t json_get_string(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_int(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_float(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_bool(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_raw(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value);
|
||||
el_val_t json_array_len(el_val_t json_str);
|
||||
el_val_t json_array_get(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_array_get_string(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_escape_string(el_val_t sv);
|
||||
el_val_t json_build_object(el_val_t kvs);
|
||||
el_val_t json_build_array(el_val_t items);
|
||||
|
||||
/* ── Time ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t time_now(void);
|
||||
el_val_t time_now_utc(void);
|
||||
el_val_t sleep_secs(el_val_t secs);
|
||||
el_val_t sleep_ms(el_val_t ms);
|
||||
el_val_t time_format(el_val_t ts, el_val_t fmt);
|
||||
el_val_t time_to_parts(el_val_t ts);
|
||||
el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz);
|
||||
el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit);
|
||||
el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit);
|
||||
el_val_t now_ns(void);
|
||||
|
||||
/* ── Instant + Duration: first-class temporal types ──────────────────────────
|
||||
* Both types share the el_val_t (int64) slot. Instants are nanoseconds
|
||||
* since the Unix epoch; Durations are signed nanoseconds. Type discipline
|
||||
* is enforced at codegen-time: BinOps on names registered as Instant or
|
||||
* Duration route through the typed wrappers below; mismatches like
|
||||
* Instant+Instant become #error at the C compiler.
|
||||
*
|
||||
* Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are
|
||||
* recognised by the parser as DurationLit AST nodes and lowered to literal
|
||||
* int64 nanoseconds at codegen time. The runtime never sees the units. */
|
||||
|
||||
el_val_t el_now_instant(void);
|
||||
el_val_t now(void);
|
||||
el_val_t unix_seconds(el_val_t n);
|
||||
el_val_t unix_millis(el_val_t n);
|
||||
el_val_t instant_from_iso8601(el_val_t s);
|
||||
|
||||
el_val_t el_duration_from_nanos(el_val_t ns);
|
||||
el_val_t duration_seconds(el_val_t n);
|
||||
el_val_t duration_millis(el_val_t n);
|
||||
el_val_t duration_nanos(el_val_t n);
|
||||
|
||||
el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_diff(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_add(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_sub(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_scale(el_val_t dur, el_val_t scalar);
|
||||
el_val_t el_duration_div(el_val_t dur, el_val_t scalar);
|
||||
|
||||
el_val_t el_instant_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ne(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ne(el_val_t a, el_val_t b);
|
||||
|
||||
el_val_t instant_to_unix_seconds(el_val_t i);
|
||||
el_val_t instant_to_unix_millis(el_val_t i);
|
||||
el_val_t instant_to_iso8601(el_val_t i);
|
||||
el_val_t duration_to_seconds(el_val_t d);
|
||||
el_val_t duration_to_millis(el_val_t d);
|
||||
el_val_t duration_to_nanos(el_val_t d);
|
||||
|
||||
el_val_t el_sleep_duration(el_val_t dur);
|
||||
el_val_t unix_timestamp(void);
|
||||
|
||||
el_val_t ttl_cache_set(el_val_t key, el_val_t value);
|
||||
el_val_t ttl_cache_get(el_val_t key, el_val_t max_age);
|
||||
el_val_t ttl_cache_age(el_val_t key);
|
||||
|
||||
/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ─────────────
|
||||
* Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA
|
||||
* zones, Gregorian, DST) is the user-facing default; MarsCalendar,
|
||||
* CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth
|
||||
* domains.
|
||||
*
|
||||
* A Calendar interprets an Instant under a particular cycle convention and
|
||||
* produces a CalendarTime. CalendarTime carries the underlying Instant and
|
||||
* a back-pointer to its Calendar; arithmetic and formatting consult the
|
||||
* Calendar to convert ns since epoch into year/month/day/hour/minute/second
|
||||
* (or sol/phase, or cycle/phase, depending on kind).
|
||||
*
|
||||
* Storage convention: Calendar / CalendarTime / Rhythm / LocalDate /
|
||||
* LocalDateTime are heap-allocated structs whose pointers are cast into
|
||||
* el_val_t. A 24-bit magic header at offset 0 lets the runtime identify
|
||||
* the kind safely. LocalTime is small enough to live in the int64 slot
|
||||
* directly (nanos since midnight, signed). */
|
||||
|
||||
/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar.
|
||||
* `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed
|
||||
* offset string ("+05:30", "-08:00"). The runtime resolves it via tzset()
|
||||
* on first use of the owning EarthCalendar. */
|
||||
el_val_t zone(el_val_t id);
|
||||
el_val_t zone_utc(void);
|
||||
el_val_t zone_local(void);
|
||||
el_val_t zone_offset(el_val_t hours, el_val_t minutes);
|
||||
|
||||
/* Calendar constructors. Each returns an el_val_t pointer to a heap-
|
||||
* allocated, magic-tagged Calendar struct. Calendars are interned by
|
||||
* (kind, zone_id, period_ns, epoch_ns) so identical constructors return
|
||||
* the same pointer — equality is reference equality. */
|
||||
el_val_t earth_calendar(el_val_t z);
|
||||
el_val_t earth_calendar_default(void);
|
||||
el_val_t mars_calendar(void);
|
||||
el_val_t cycle_calendar(el_val_t period_dur);
|
||||
el_val_t no_cycle_calendar(void);
|
||||
el_val_t relative_calendar(el_val_t epoch_inst);
|
||||
|
||||
/* CalendarTime constructors and methods. Returns a heap-allocated struct
|
||||
* whose pointer fits in el_val_t. */
|
||||
el_val_t now_in(el_val_t cal);
|
||||
el_val_t in_calendar(el_val_t inst, el_val_t cal);
|
||||
el_val_t cal_format(el_val_t ct, el_val_t pattern);
|
||||
el_val_t cal_to_instant(el_val_t ct);
|
||||
el_val_t cal_cycle_phase(el_val_t ct);
|
||||
el_val_t cal_in(el_val_t ct, el_val_t cal);
|
||||
|
||||
/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types.
|
||||
* LocalTime carries nanoseconds since midnight as a signed int64 directly
|
||||
* in the el_val_t slot (no allocation). LocalDate / LocalDateTime are
|
||||
* heap-allocated structs with magic headers. */
|
||||
el_val_t local_date(el_val_t y, el_val_t m, el_val_t d);
|
||||
el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns);
|
||||
el_val_t local_datetime(el_val_t date, el_val_t time);
|
||||
el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal);
|
||||
|
||||
el_val_t local_date_year(el_val_t ld);
|
||||
el_val_t local_date_month(el_val_t ld);
|
||||
el_val_t local_date_day(el_val_t ld);
|
||||
el_val_t local_time_hour(el_val_t lt);
|
||||
el_val_t local_time_minute(el_val_t lt);
|
||||
el_val_t local_time_second(el_val_t lt);
|
||||
el_val_t local_time_nanos(el_val_t lt);
|
||||
|
||||
el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur);
|
||||
el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur);
|
||||
el_val_t el_local_date_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_local_date_eq(el_val_t a, el_val_t b);
|
||||
|
||||
/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct
|
||||
* pointer in el_val_t; rhythms are immutable so callers may share them. */
|
||||
el_val_t rhythm_cycle_start(void);
|
||||
el_val_t rhythm_cycle_phase(el_val_t phase);
|
||||
el_val_t rhythm_duration(el_val_t d);
|
||||
el_val_t rhythm_session_start(void);
|
||||
el_val_t rhythm_event(el_val_t name);
|
||||
el_val_t rhythm_and(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_or(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_weekday(el_val_t day);
|
||||
el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute);
|
||||
el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal);
|
||||
el_val_t rhythm_matches(el_val_t r, el_val_t ct);
|
||||
|
||||
/* ── UUID ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t uuid_new(void);
|
||||
el_val_t uuid_v4(void);
|
||||
|
||||
/* ── Environment ─────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t env(el_val_t key);
|
||||
|
||||
/* ── In-process state K/V ────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t state_set(el_val_t key, el_val_t value);
|
||||
el_val_t state_get(el_val_t key);
|
||||
el_val_t state_del(el_val_t key);
|
||||
el_val_t state_keys(void);
|
||||
el_val_t state_has(el_val_t key);
|
||||
el_val_t state_get_or(el_val_t key, el_val_t default_val);
|
||||
|
||||
/* ── Float formatting ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t float_to_str(el_val_t f);
|
||||
el_val_t int_to_float(el_val_t n);
|
||||
el_val_t float_to_int(el_val_t f);
|
||||
el_val_t format_float(el_val_t f, el_val_t decimals);
|
||||
el_val_t decimal_round(el_val_t f, el_val_t decimals);
|
||||
el_val_t str_to_float(el_val_t s);
|
||||
|
||||
/* ── Math (Float-aware) ──────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t math_sqrt(el_val_t f);
|
||||
el_val_t math_log(el_val_t f);
|
||||
el_val_t math_ln(el_val_t f);
|
||||
el_val_t math_sin(el_val_t f);
|
||||
el_val_t math_cos(el_val_t f);
|
||||
el_val_t math_pi(void);
|
||||
|
||||
/* ── String additions ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t str_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_split(el_val_t s, el_val_t sep);
|
||||
el_val_t str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t str_char_code(el_val_t s, el_val_t i);
|
||||
el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_format(el_val_t fmt, el_val_t data);
|
||||
el_val_t str_lower(el_val_t s);
|
||||
el_val_t str_upper(el_val_t s);
|
||||
|
||||
/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes)
|
||||
* Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex.
|
||||
* is_* predicates: empty input returns false; multi-char requires ALL bytes
|
||||
* to match. ASCII ranges only in Phase 1. */
|
||||
|
||||
/* Counting */
|
||||
el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */
|
||||
el_val_t str_count_chars(el_val_t s); /* codepoint count */
|
||||
el_val_t str_count_bytes(el_val_t s); /* alias of str_len */
|
||||
el_val_t str_count_lines(el_val_t s);
|
||||
el_val_t str_count_words(el_val_t s);
|
||||
el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */
|
||||
el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */
|
||||
|
||||
/* Find / position */
|
||||
el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */
|
||||
el_val_t str_last_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */
|
||||
|
||||
/* Transform */
|
||||
el_val_t str_repeat(el_val_t s, el_val_t n);
|
||||
el_val_t str_reverse(el_val_t s); /* by codepoint */
|
||||
el_val_t str_strip_prefix(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_strip_suffix(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_strip_chars(el_val_t s, el_val_t chars);
|
||||
el_val_t str_lstrip(el_val_t s);
|
||||
el_val_t str_rstrip(el_val_t s);
|
||||
|
||||
/* Char classification (Bool) */
|
||||
el_val_t is_letter(el_val_t s);
|
||||
el_val_t is_digit(el_val_t s);
|
||||
el_val_t is_alphanumeric(el_val_t s);
|
||||
el_val_t is_whitespace(el_val_t s);
|
||||
el_val_t is_punctuation(el_val_t s);
|
||||
el_val_t is_uppercase(el_val_t s);
|
||||
el_val_t is_lowercase(el_val_t s);
|
||||
|
||||
/* Split / join */
|
||||
el_val_t str_split_lines(el_val_t s);
|
||||
el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */
|
||||
el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n);
|
||||
el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */
|
||||
|
||||
/* ── List additions ──────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t list_push(el_val_t list, el_val_t elem);
|
||||
el_val_t list_push_front(el_val_t list, el_val_t elem);
|
||||
el_val_t list_join(el_val_t list, el_val_t sep);
|
||||
el_val_t list_range(el_val_t start, el_val_t end);
|
||||
|
||||
/* ── Bool helpers ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t bool_to_str(el_val_t b);
|
||||
|
||||
/* ── Numeric parsing ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t parse_int(el_val_t s, el_val_t default_val);
|
||||
|
||||
/* ── Process ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t exit_program(el_val_t code);
|
||||
el_val_t getpid_now(void);
|
||||
|
||||
/* Self-terminating memory guard. Reads ELC_MAX_MEM_MB (default 512) and
|
||||
* exits with code 1 if resident memory exceeds the limit. Call periodically
|
||||
* during long compilation loops (e.g. after each function is compiled).
|
||||
* Returns 0 when memory is within bounds. */
|
||||
el_val_t el_mem_check(void);
|
||||
|
||||
/* ── CGI identity ─────────────────────────────────────────────────────────────
|
||||
* Called at the start of main() in CGI programs (those with a `cgi {}` block).
|
||||
* Records the program's DHARMA identity before any other code executes. */
|
||||
|
||||
void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal,
|
||||
el_val_t network, el_val_t engram);
|
||||
|
||||
/* ── DHARMA network builtins ─────────────────────────────────────────────────
|
||||
* Available to CGI programs (declared with a `cgi {}` block).
|
||||
*
|
||||
* Peers are addressed by `dharma_id` of the form
|
||||
* "<registry-id>@<transport-url>" e.g. "ntn-genesis@http://localhost:7770"
|
||||
* If the @<url> portion is omitted, transport defaults to
|
||||
* "http://localhost:7770" (the local CGI daemon assumption).
|
||||
*
|
||||
* Wire protocol (all peers expose):
|
||||
* POST <url>/dharma/recv { channel, from, content } → response body
|
||||
* POST <url>/dharma/event { type, payload, source, timestamp }
|
||||
* POST <url>/api/activate { query } → list of nodes
|
||||
*
|
||||
* Hosting application's responsibility: an El program with a `cgi {}` block
|
||||
* runs http_serve() with its own request handler; that handler should route
|
||||
* "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so
|
||||
* incoming events feed dharma_field() queues. The runtime itself does not
|
||||
* intercept any /dharma path. */
|
||||
|
||||
el_val_t dharma_connect(el_val_t cgi_id);
|
||||
el_val_t dharma_send(el_val_t channel, el_val_t content);
|
||||
el_val_t dharma_activate(el_val_t query);
|
||||
void dharma_emit(el_val_t event_type, el_val_t payload);
|
||||
el_val_t dharma_field(el_val_t event_type);
|
||||
void dharma_strengthen(el_val_t cgi_id, el_val_t weight);
|
||||
el_val_t dharma_relationship(el_val_t cgi_id);
|
||||
el_val_t dharma_peers(void);
|
||||
|
||||
/* Public C API: called by an El program's HTTP handler when a /dharma/event
|
||||
* request arrives. Pushes onto the per-event-type queue and signals any
|
||||
* pending dharma_field() blockers. All three arguments must be NUL-terminated
|
||||
* C strings (or NULL — then treated as empty). */
|
||||
void el_runtime_dharma_event_arrive(const char* event_type,
|
||||
const char* payload,
|
||||
const char* source);
|
||||
|
||||
/* ── Engram local graph primitives ───────────────────────────────────────────
|
||||
* Operate on the CGI's local Engram knowledge graph.
|
||||
* `engram_activate` queries the local graph only; `dharma_activate` is
|
||||
* network-wide across all connected CGI graphs. */
|
||||
|
||||
el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience);
|
||||
el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t importance, el_val_t confidence,
|
||||
el_val_t tier, el_val_t tags);
|
||||
/* Layered consciousness — see el_runtime.c for the layered architecture
|
||||
* design notes (search "Layered consciousness architecture"). The five
|
||||
* canonical layers (safety / core-identity / domain-knowledge / imprint /
|
||||
* suit) are seeded automatically; engram_add_layer extends the registry
|
||||
* with imprint or suit overlays at runtime. Nodes default to layer 1
|
||||
* (core-identity) when created via engram_node / engram_node_full. */
|
||||
el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible,
|
||||
el_val_t transparent, el_val_t injectable);
|
||||
el_val_t engram_remove_layer(el_val_t layer_id);
|
||||
el_val_t engram_list_layers(void);
|
||||
el_val_t engram_get_node(el_val_t id);
|
||||
void engram_strengthen(el_val_t node_id);
|
||||
void engram_forget(el_val_t node_id);
|
||||
el_val_t engram_node_count(void);
|
||||
el_val_t engram_search(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset);
|
||||
void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation);
|
||||
el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id);
|
||||
el_val_t engram_neighbors(el_val_t node_id);
|
||||
el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_edge_count(void);
|
||||
/* Three-pass activation: background fan-out → working-memory promotion →
|
||||
* Layer 0 override. See "Three-pass activation" in el_runtime.c. */
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
* through json_stringify. */
|
||||
el_val_t engram_get_node_json(el_val_t id);
|
||||
el_val_t engram_get_node_by_label(el_val_t label);
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_activate_json(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_stats_json(void);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
* no nodes promoted to working memory. */
|
||||
el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth);
|
||||
|
||||
/* ── Working memory ──────────────────────────────────────────────────────────*/
|
||||
el_val_t engram_wm_count(void);
|
||||
el_val_t engram_wm_avg_weight(void);
|
||||
el_val_t engram_wm_top_json(el_val_t n);
|
||||
el_val_t engram_load_merge(el_val_t path);
|
||||
|
||||
/* ── LLM (Anthropic API client) ─────────────────────────────────────────────
|
||||
* All functions call https://api.anthropic.com/v1/messages with the API key
|
||||
* from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */
|
||||
|
||||
el_val_t llm_call(el_val_t model, el_val_t prompt);
|
||||
el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt);
|
||||
el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools);
|
||||
el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64);
|
||||
el_val_t llm_models(void);
|
||||
|
||||
/* Register a tool handler by name. The handler is looked up via dlsym
|
||||
* (mirroring http_set_handler), so any El `fn <name>(input)` compiles to
|
||||
* a global C symbol that this function can locate at runtime.
|
||||
* Handler signature: `el_val_t handler(el_val_t input_json)` — receives
|
||||
* the tool input as a JSON-string el_val_t and returns a JSON-string
|
||||
* el_val_t result. Used by llm_call_agentic. */
|
||||
void llm_register_tool(el_val_t name, el_val_t handler_fn_name);
|
||||
|
||||
/* ── args() ─────────────────────────────────────────────────────────────────
|
||||
* Provides access to command-line arguments passed to the program.
|
||||
* Populated by el_runtime_init_args() before main() runs. */
|
||||
|
||||
el_val_t args(void);
|
||||
void el_runtime_init_args(int argc, char** argv);
|
||||
|
||||
/* ── Crypto primitives ─────────────────────────────────────────────────────
|
||||
* SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe).
|
||||
* Self-contained — no OpenSSL/libcrypto dependency. The implementations are
|
||||
* adapted from public-domain reference code (Brad Conte / RFC 4648).
|
||||
*
|
||||
* Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string
|
||||
* value whose contents are raw binary; callers usually feed these into
|
||||
* base64_encode. Note that el_val_t strings are NUL-terminated by convention,
|
||||
* so the binary payload may contain embedded NULs — pass it directly into
|
||||
* base64_encode (which uses an explicit length) rather than treating it as
|
||||
* a printable C string.
|
||||
*
|
||||
* The "base64" variants emit/accept RFC 4648 standard alphabet with padding.
|
||||
* The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding,
|
||||
* as used in JWTs. */
|
||||
|
||||
el_val_t sha256_hex(el_val_t input);
|
||||
el_val_t sha256_bytes(el_val_t input);
|
||||
el_val_t hmac_sha256_hex(el_val_t key, el_val_t message);
|
||||
el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message);
|
||||
el_val_t base64_encode(el_val_t input);
|
||||
el_val_t base64_decode(el_val_t input);
|
||||
el_val_t base64url_encode(el_val_t input);
|
||||
el_val_t base64url_decode(el_val_t input);
|
||||
|
||||
/* Length-aware variants (internal — exposed for the rare caller that already
|
||||
* has a known-length binary buffer and doesn't want to round-trip through
|
||||
* a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed
|
||||
* these implicitly. */
|
||||
el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len);
|
||||
el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe);
|
||||
|
||||
/* ── Post-quantum primitives (liboqs-backed) ────────────────────────────────
|
||||
* All inputs/outputs hex-encoded. Algorithm choices:
|
||||
* Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced)
|
||||
* KEM: CRYSTALS-Kyber-768 (NIST level 3)
|
||||
* Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2)
|
||||
*
|
||||
* If liboqs is not linked (detected via __has_include(<oqs/oqs.h>) at compile
|
||||
* time), the pq_* entry points return a JSON-shaped error string so callers
|
||||
* fail loudly rather than silently fall back to classical schemes:
|
||||
* {"error":"liboqs not linked, post-quantum primitives unavailable"}
|
||||
*
|
||||
* The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and
|
||||
* CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss).
|
||||
* Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack,
|
||||
* Kyber holds. SHA3-256 also remains usable independent of liboqs (the
|
||||
* Keccak permutation is PQ-OK as a primitive). */
|
||||
|
||||
el_val_t pq_keygen_signature(void);
|
||||
el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message);
|
||||
el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex);
|
||||
|
||||
el_val_t pq_kem_keygen(void);
|
||||
el_val_t pq_kem_encaps(el_val_t public_key_hex);
|
||||
el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex);
|
||||
|
||||
el_val_t pq_hybrid_keygen(void);
|
||||
el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined);
|
||||
|
||||
el_val_t sha3_256_hex(el_val_t input);
|
||||
|
||||
/* ── AEAD: AES-256-GCM (libcrypto-backed) ───────────────────────────────────
|
||||
* Symmetric authenticated encryption used to wrap envelopes after a KEM
|
||||
* handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the
|
||||
* Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256.
|
||||
*
|
||||
* aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where
|
||||
* ciphertext is the AES-256-GCM output with the 16-byte auth tag appended.
|
||||
* Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which
|
||||
* structurally rules out the GCM nonce-reuse footgun.
|
||||
*
|
||||
* aead_decrypt returns the plaintext String, or "" on any failure (including
|
||||
* auth-tag mismatch). Callers MUST check for "" before trusting the result. */
|
||||
el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext);
|
||||
el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex);
|
||||
|
||||
/* ── Native VM builtin aliases (for compiled El source) ─────────────────────
|
||||
* These match the El VM's native_* builtins so that El source compiled
|
||||
* to C can call the same names without modification. */
|
||||
|
||||
el_val_t native_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t native_list_len(el_val_t list);
|
||||
el_val_t native_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t native_list_empty(void);
|
||||
el_val_t native_list_clone(el_val_t list);
|
||||
el_val_t native_string_chars(el_val_t s);
|
||||
el_val_t native_int_to_str(el_val_t n);
|
||||
|
||||
/* ── Method-call shorthand aliases ──────────────────────────────────────────
|
||||
* The El method-call convention `obj.method(args)` compiles to
|
||||
* `method(obj, args)`. These aliases expose the runtime functions under
|
||||
* the short names that result from method calls in El source.
|
||||
*
|
||||
* Example: `myList.append(x)` → `append(myList, x)` (calls this alias)
|
||||
* `myList.len()` → `len(myList)` (calls this alias) */
|
||||
|
||||
el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */
|
||||
el_val_t len(el_val_t list); /* el_list_len */
|
||||
el_val_t get(el_val_t list, el_val_t index); /* el_list_get */
|
||||
el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */
|
||||
el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */
|
||||
|
||||
/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */
|
||||
/* See bottom of el_runtime.c for the implementation.
|
||||
* Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION.
|
||||
* No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */
|
||||
/* ── Subprocess execution ────────────────────────────────────────────────── */
|
||||
el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */
|
||||
el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */
|
||||
el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */
|
||||
el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */
|
||||
|
||||
/* ── Stdout redirection (used by compiler JS pipeline) ───────────────────── */
|
||||
el_val_t stdout_to_file(el_val_t path); /* redirect process stdout to a file */
|
||||
el_val_t stdout_restore(void); /* restore process stdout to terminal */
|
||||
|
||||
el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json);
|
||||
el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json);
|
||||
el_val_t trace_span_start(el_val_t name);
|
||||
el_val_t trace_span_end(el_val_t span_handle);
|
||||
el_val_t emit_event(el_val_t name, el_val_t duration_ms);
|
||||
|
||||
el_val_t __thread_create(el_val_t fn_name_v, el_val_t arg_v);
|
||||
el_val_t __thread_join(el_val_t tid_v);
|
||||
|
||||
/* ── __ prefixed aliases (self-hosting compiler ABI) ─────────────────────────
|
||||
* The El self-hosting compiler emits calls to __-prefixed names. These are
|
||||
* forwarding wrappers around the existing el_runtime functions above. */
|
||||
|
||||
/* I/O */
|
||||
el_val_t __println(el_val_t s);
|
||||
el_val_t __print(el_val_t s);
|
||||
el_val_t __readline(void);
|
||||
|
||||
/* String */
|
||||
el_val_t __int_to_str(el_val_t n);
|
||||
el_val_t __str_to_int(el_val_t s);
|
||||
el_val_t __float_to_str(el_val_t f);
|
||||
el_val_t __str_to_float(el_val_t s);
|
||||
el_val_t __str_len(el_val_t s);
|
||||
el_val_t __str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t __str_cmp(el_val_t a, el_val_t b);
|
||||
el_val_t __str_ncmp(el_val_t a, el_val_t b, el_val_t n);
|
||||
el_val_t __str_concat_raw(el_val_t a, el_val_t b);
|
||||
el_val_t __str_slice_raw(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t __str_alloc(el_val_t n);
|
||||
el_val_t __str_set_char(el_val_t s, el_val_t i, el_val_t c);
|
||||
|
||||
/* URL encoding */
|
||||
el_val_t __url_encode(el_val_t s);
|
||||
el_val_t __url_decode(el_val_t s);
|
||||
|
||||
/* Environment */
|
||||
el_val_t __env_get(el_val_t key);
|
||||
|
||||
/* Subprocess */
|
||||
el_val_t __exec(el_val_t cmd);
|
||||
el_val_t __exec_bg(el_val_t cmd);
|
||||
|
||||
/* Process */
|
||||
el_val_t __exit_program(el_val_t code);
|
||||
|
||||
/* Filesystem */
|
||||
el_val_t __fs_exists(el_val_t path);
|
||||
el_val_t __fs_mkdir(el_val_t path);
|
||||
el_val_t __fs_read(el_val_t path);
|
||||
el_val_t __fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t __fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t n);
|
||||
el_val_t __fs_list_raw(el_val_t path);
|
||||
|
||||
/* HTTP server */
|
||||
el_val_t __http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
el_val_t __http_serve(el_val_t port, el_val_t handler);
|
||||
el_val_t __http_serve_v2(el_val_t port, el_val_t handler);
|
||||
|
||||
/* HTTP conn fd / SSE (weak; overridden by el_seed.c when linked together) */
|
||||
el_val_t __http_conn_fd(void);
|
||||
el_val_t __http_sse_open(el_val_t conn_id);
|
||||
el_val_t __http_sse_send(el_val_t conn_id, el_val_t data);
|
||||
el_val_t __http_sse_close(el_val_t conn_id);
|
||||
|
||||
/* HTTP client (requires HAVE_CURL; stubs provided for no-curl builds) */
|
||||
el_val_t __http_do(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_map, el_val_t timeout_ms);
|
||||
el_val_t __http_do_map(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_json, el_val_t timeout_ms);
|
||||
el_val_t __http_do_map_to_file(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_json, el_val_t output_path);
|
||||
|
||||
/* JSON */
|
||||
el_val_t __json_array_get(el_val_t json, el_val_t index);
|
||||
el_val_t __json_array_get_string(el_val_t json, el_val_t index);
|
||||
el_val_t __json_array_len(el_val_t json);
|
||||
el_val_t __json_get(el_val_t json, el_val_t key);
|
||||
el_val_t __json_get_raw(el_val_t json, el_val_t key);
|
||||
el_val_t __json_set(el_val_t json, el_val_t key, el_val_t value);
|
||||
el_val_t __json_parse_map(el_val_t json_str);
|
||||
el_val_t __json_stringify_val(el_val_t val);
|
||||
|
||||
/* Hashing */
|
||||
el_val_t __sha256_hex(el_val_t s);
|
||||
|
||||
/* State K/V */
|
||||
el_val_t __state_del(el_val_t key);
|
||||
el_val_t __state_get(el_val_t key);
|
||||
el_val_t __state_keys(void);
|
||||
el_val_t __state_set(el_val_t key, el_val_t val);
|
||||
|
||||
/* UUID */
|
||||
el_val_t __uuid_v4(void);
|
||||
|
||||
/* Args */
|
||||
el_val_t __args_json(void);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,761 +0,0 @@
|
||||
/*
|
||||
* el_runtime.h — El language C runtime header
|
||||
*
|
||||
* Declares all built-in functions available to compiled El programs.
|
||||
* Include this in every generated .c file.
|
||||
*
|
||||
* Value model:
|
||||
* All El values are represented as el_val_t (= int64_t).
|
||||
* On 64-bit systems a pointer fits in int64_t.
|
||||
* String values are cast: (el_val_t)(uintptr_t)"hello"
|
||||
* Integer values are stored directly.
|
||||
* This lets arithmetic work naturally while still passing strings around.
|
||||
*
|
||||
* Type conventions (El -> C):
|
||||
* String -> el_val_t (holds const char* via uintptr_t cast)
|
||||
* Int -> el_val_t
|
||||
* Bool -> el_val_t (0 = false, nonzero = true)
|
||||
* Any -> el_val_t
|
||||
* Void -> void
|
||||
*
|
||||
* Macros for convenience:
|
||||
* EL_STR(s) cast string literal to el_val_t
|
||||
* EL_CSTR(v) cast el_val_t back to const char*
|
||||
* EL_INT(v) identity — el_val_t is already int64_t
|
||||
*
|
||||
* Link requirements:
|
||||
* -lcurl — required for the HTTP client (http_get, http_post, llm_*).
|
||||
* -lpthread — required for the HTTP server (one detached thread per
|
||||
* connection, capped at 64 concurrent).
|
||||
* -loqs — optional; required only when liboqs is installed and the
|
||||
* pq_* / sha3_256_hex entry points are needed. Detected at
|
||||
* compile time via __has_include(<oqs/oqs.h>).
|
||||
* -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
typedef int64_t el_val_t;
|
||||
|
||||
#define EL_STR(s) ((el_val_t)(uintptr_t)(s))
|
||||
#define EL_CSTR(v) ((const char*)(uintptr_t)(v))
|
||||
#define EL_INT(v) (v)
|
||||
#define EL_NULL ((el_val_t)0)
|
||||
|
||||
/* Float values share the el_val_t (int64) slot via a bit-cast.
|
||||
* The codegen emits Float literals as `el_from_float(<dbl>)` so the
|
||||
* underlying bits represent the IEEE 754 double. Float-aware builtins
|
||||
* (math, format, json) round-trip via these helpers. */
|
||||
static inline double el_to_float(el_val_t v) {
|
||||
union { int64_t i; double f; } u;
|
||||
u.i = (int64_t)v;
|
||||
return u.f;
|
||||
}
|
||||
|
||||
static inline el_val_t el_from_float(double f) {
|
||||
union { double f; int64_t i; } u;
|
||||
u.f = f;
|
||||
return (el_val_t)u.i;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ── I/O ──────────────────────────────────────────────────────────────────── */
|
||||
|
||||
void println(el_val_t s);
|
||||
void print(el_val_t s);
|
||||
el_val_t readline(void);
|
||||
|
||||
/* ── String builtins ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t str_eq(el_val_t a, el_val_t b);
|
||||
el_val_t str_starts_with(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_ends_with(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_len(el_val_t s);
|
||||
el_val_t str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t int_to_str(el_val_t n);
|
||||
el_val_t str_to_int(el_val_t s);
|
||||
el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t str_contains(el_val_t s, el_val_t sub);
|
||||
el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to);
|
||||
el_val_t str_to_upper(el_val_t s);
|
||||
el_val_t str_to_lower(el_val_t s);
|
||||
el_val_t str_trim(el_val_t s);
|
||||
|
||||
/* ── Math ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_abs(el_val_t n);
|
||||
el_val_t el_max(el_val_t a, el_val_t b);
|
||||
el_val_t el_min(el_val_t a, el_val_t b);
|
||||
|
||||
/* ── Refcount (ARC) ──────────────────────────────────────────────────────────
|
||||
* Lists and Maps carry a refcount. Strings and ints do not — el_retain and
|
||||
* el_release are safe no-ops on non-refcounted values (they sniff a magic
|
||||
* header at offset 0 and only act if the magic matches).
|
||||
*
|
||||
* Codegen emits these at let-binding shadowing, function entry (params), and
|
||||
* function exit (locals other than the returned value). The refcount lets
|
||||
* el_list_append and el_map_set mutate in place when uniquely owned (cheap)
|
||||
* and copy-on-write when shared (preserves persistent semantics across
|
||||
* accumulator patterns in the compiler itself). */
|
||||
|
||||
void el_retain(el_val_t v);
|
||||
void el_release(el_val_t v);
|
||||
|
||||
/* ── List ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_list_new(el_val_t count, ...);
|
||||
el_val_t el_list_len(el_val_t list);
|
||||
el_val_t el_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t el_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t el_list_empty(void);
|
||||
el_val_t el_list_clone(el_val_t list);
|
||||
|
||||
/* ── Map ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_map_new(el_val_t pair_count, ...);
|
||||
el_val_t el_get_field(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_get(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value);
|
||||
|
||||
/* ── HTTP ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t http_get(el_val_t url);
|
||||
el_val_t http_post(el_val_t url, el_val_t body);
|
||||
el_val_t http_post_json(el_val_t url, el_val_t json_body);
|
||||
el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map);
|
||||
el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map);
|
||||
el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header);
|
||||
el_val_t http_delete(el_val_t url);
|
||||
void http_serve(el_val_t port, el_val_t handler);
|
||||
void http_set_handler(el_val_t name);
|
||||
|
||||
/* HTTP server v2 ─────────────────────────────────────────────────────────────
|
||||
* Same dispatch model as http_serve, but the handler signature is widened:
|
||||
*
|
||||
* el_val_t handler(method, path, headers_map, body)
|
||||
*
|
||||
* `headers_map` is an ElMap from lowercased header name → header value (both
|
||||
* Strings). Repeated headers are joined with ", " per RFC 7230.
|
||||
*
|
||||
* Response value: the handler may return either
|
||||
* (a) a plain body string — same auto-content-type / 200-OK behaviour as
|
||||
* http_serve (3-arg) — or
|
||||
* (b) a response envelope built with `http_response(status, headers_json,
|
||||
* body)`. The runtime detects the envelope discriminator
|
||||
* `"el_http_response":1` at the start of the returned string and
|
||||
* unpacks status / headers / body before sending.
|
||||
*
|
||||
* The 3-arg http_serve(port, handler) remains supported unchanged for
|
||||
* existing handlers (e.g. products/web/server.el): it dispatches with
|
||||
* (method, path, body), hardcodes 200 OK, and auto-detects content type. */
|
||||
void http_serve_v2(el_val_t port, el_val_t handler);
|
||||
void http_set_handler_v2(el_val_t name);
|
||||
|
||||
/* Build an HTTP response envelope. `headers_json` should be a JSON object
|
||||
* literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The
|
||||
* returned string carries the discriminator `{"el_http_response":1,...}`
|
||||
* which the runtime's send-path detects and unpacks. Detection happens
|
||||
* uniformly inside http_send_response, so a 3-arg handler may also return
|
||||
* an envelope. The 3-arg variant remains documented as a fixed 200-OK
|
||||
* auto-content-type contract for legacy handlers that return plain bodies. */
|
||||
el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
|
||||
/* SSE connection fd — set by http_worker_v2 before calling the El handler,
|
||||
* cleared afterwards. Defined in el_seed.c; called from el_runtime.c.
|
||||
* The getter is exposed as __http_conn_fd() to El programs. */
|
||||
void el_seed_set_http_conn_fd(int fd);
|
||||
|
||||
/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default
|
||||
* 60000ms). Read lazily on first use, so setting the env var any time before
|
||||
* the first http_* call is sufficient. */
|
||||
|
||||
/* Streaming variants — write the response body straight to a file via
|
||||
* libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string
|
||||
* wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive
|
||||
* embedded NUL bytes that would truncate a strlen()-based code path.
|
||||
*
|
||||
* Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same
|
||||
* `headers_map` shape as http_post_with_headers (ElMap of String→String).
|
||||
*
|
||||
* Return value: 1 on success (file fully written), 0 on any failure
|
||||
* (network, file open, partial write). On failure the output file is removed
|
||||
* so callers cannot mistake a partially-written file for a valid one. */
|
||||
el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path);
|
||||
el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path);
|
||||
|
||||
/* ── URL encoding ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */
|
||||
el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */
|
||||
|
||||
/* ── HTML allowlist sanitizer ────────────────────────────────────────────────
|
||||
* el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML
|
||||
* cleaner. State-machine parser; tag/attribute names compared case-
|
||||
* insensitively against the allowlist; `<a href>` / `<… src>` URL schemes
|
||||
* validated (http, https, mailto, fragment-only, or relative); whole-
|
||||
* subtree drop for script / style / iframe / object / embed / form; HTML-
|
||||
* escapes free text outside dropped subtrees.
|
||||
*
|
||||
* The allowlist is JSON of the form
|
||||
* {"p":[],"a":["href","title"],"strong":[],...}
|
||||
* where each value is the array of attribute names allowed for that tag. */
|
||||
el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json);
|
||||
|
||||
/* ── Filesystem ──────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t fs_read(el_val_t path);
|
||||
el_val_t fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t fs_list(el_val_t path);
|
||||
el_val_t fs_exists(el_val_t path);
|
||||
el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */
|
||||
|
||||
/* Length-explicit binary write. `length` is an Int (el_val_t holding the
|
||||
* byte count). The caller knows the length from context — typically because
|
||||
* `bytes` came from base64_decode (which produces a magic-tagged binary
|
||||
* buffer with embedded NULs possible) and the caller already tracks the
|
||||
* decoded length, OR because the bytes came from a fixed-size source
|
||||
* (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely.
|
||||
*
|
||||
* Returns 1 on success, 0 on failure (invalid path, can't open, partial
|
||||
* write, negative length). On partial-write failure, the file is removed
|
||||
* so callers cannot read back a truncated artefact. */
|
||||
el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length);
|
||||
|
||||
/* ── JSON ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t json_get(el_val_t json, el_val_t key);
|
||||
el_val_t json_parse(el_val_t s);
|
||||
el_val_t json_stringify(el_val_t v);
|
||||
el_val_t json_get_string(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_int(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_float(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_bool(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_raw(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value);
|
||||
el_val_t json_array_len(el_val_t json_str);
|
||||
el_val_t json_array_get(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_array_get_string(el_val_t json_str, el_val_t index);
|
||||
|
||||
/* ── Time ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t time_now(void);
|
||||
el_val_t time_now_utc(void);
|
||||
el_val_t sleep_secs(el_val_t secs);
|
||||
el_val_t sleep_ms(el_val_t ms);
|
||||
el_val_t time_format(el_val_t ts, el_val_t fmt);
|
||||
el_val_t time_to_parts(el_val_t ts);
|
||||
el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz);
|
||||
el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit);
|
||||
el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit);
|
||||
|
||||
/* ── Instant + Duration: first-class temporal types ──────────────────────────
|
||||
* Both types share the el_val_t (int64) slot. Instants are nanoseconds
|
||||
* since the Unix epoch; Durations are signed nanoseconds. Type discipline
|
||||
* is enforced at codegen-time: BinOps on names registered as Instant or
|
||||
* Duration route through the typed wrappers below; mismatches like
|
||||
* Instant+Instant become #error at the C compiler.
|
||||
*
|
||||
* Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are
|
||||
* recognised by the parser as DurationLit AST nodes and lowered to literal
|
||||
* int64 nanoseconds at codegen time. The runtime never sees the units. */
|
||||
|
||||
el_val_t el_now_instant(void);
|
||||
el_val_t now(void);
|
||||
el_val_t unix_seconds(el_val_t n);
|
||||
el_val_t unix_millis(el_val_t n);
|
||||
el_val_t instant_from_iso8601(el_val_t s);
|
||||
|
||||
el_val_t el_duration_from_nanos(el_val_t ns);
|
||||
el_val_t duration_seconds(el_val_t n);
|
||||
el_val_t duration_millis(el_val_t n);
|
||||
el_val_t duration_nanos(el_val_t n);
|
||||
|
||||
el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_diff(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_add(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_sub(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_scale(el_val_t dur, el_val_t scalar);
|
||||
el_val_t el_duration_div(el_val_t dur, el_val_t scalar);
|
||||
|
||||
el_val_t el_instant_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ne(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ne(el_val_t a, el_val_t b);
|
||||
|
||||
el_val_t instant_to_unix_seconds(el_val_t i);
|
||||
el_val_t instant_to_unix_millis(el_val_t i);
|
||||
el_val_t instant_to_iso8601(el_val_t i);
|
||||
el_val_t duration_to_seconds(el_val_t d);
|
||||
el_val_t duration_to_millis(el_val_t d);
|
||||
el_val_t duration_to_nanos(el_val_t d);
|
||||
|
||||
el_val_t el_sleep_duration(el_val_t dur);
|
||||
el_val_t unix_timestamp(void);
|
||||
|
||||
el_val_t ttl_cache_set(el_val_t key, el_val_t value);
|
||||
el_val_t ttl_cache_get(el_val_t key, el_val_t max_age);
|
||||
el_val_t ttl_cache_age(el_val_t key);
|
||||
|
||||
/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ─────────────
|
||||
* Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA
|
||||
* zones, Gregorian, DST) is the user-facing default; MarsCalendar,
|
||||
* CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth
|
||||
* domains.
|
||||
*
|
||||
* A Calendar interprets an Instant under a particular cycle convention and
|
||||
* produces a CalendarTime. CalendarTime carries the underlying Instant and
|
||||
* a back-pointer to its Calendar; arithmetic and formatting consult the
|
||||
* Calendar to convert ns since epoch into year/month/day/hour/minute/second
|
||||
* (or sol/phase, or cycle/phase, depending on kind).
|
||||
*
|
||||
* Storage convention: Calendar / CalendarTime / Rhythm / LocalDate /
|
||||
* LocalDateTime are heap-allocated structs whose pointers are cast into
|
||||
* el_val_t. A 24-bit magic header at offset 0 lets the runtime identify
|
||||
* the kind safely. LocalTime is small enough to live in the int64 slot
|
||||
* directly (nanos since midnight, signed). */
|
||||
|
||||
/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar.
|
||||
* `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed
|
||||
* offset string ("+05:30", "-08:00"). The runtime resolves it via tzset()
|
||||
* on first use of the owning EarthCalendar. */
|
||||
el_val_t zone(el_val_t id);
|
||||
el_val_t zone_utc(void);
|
||||
el_val_t zone_local(void);
|
||||
el_val_t zone_offset(el_val_t hours, el_val_t minutes);
|
||||
|
||||
/* Calendar constructors. Each returns an el_val_t pointer to a heap-
|
||||
* allocated, magic-tagged Calendar struct. Calendars are interned by
|
||||
* (kind, zone_id, period_ns, epoch_ns) so identical constructors return
|
||||
* the same pointer — equality is reference equality. */
|
||||
el_val_t earth_calendar(el_val_t z);
|
||||
el_val_t earth_calendar_default(void);
|
||||
el_val_t mars_calendar(void);
|
||||
el_val_t cycle_calendar(el_val_t period_dur);
|
||||
el_val_t no_cycle_calendar(void);
|
||||
el_val_t relative_calendar(el_val_t epoch_inst);
|
||||
|
||||
/* CalendarTime constructors and methods. Returns a heap-allocated struct
|
||||
* whose pointer fits in el_val_t. */
|
||||
el_val_t now_in(el_val_t cal);
|
||||
el_val_t in_calendar(el_val_t inst, el_val_t cal);
|
||||
el_val_t cal_format(el_val_t ct, el_val_t pattern);
|
||||
el_val_t cal_to_instant(el_val_t ct);
|
||||
el_val_t cal_cycle_phase(el_val_t ct);
|
||||
el_val_t cal_in(el_val_t ct, el_val_t cal);
|
||||
|
||||
/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types.
|
||||
* LocalTime carries nanoseconds since midnight as a signed int64 directly
|
||||
* in the el_val_t slot (no allocation). LocalDate / LocalDateTime are
|
||||
* heap-allocated structs with magic headers. */
|
||||
el_val_t local_date(el_val_t y, el_val_t m, el_val_t d);
|
||||
el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns);
|
||||
el_val_t local_datetime(el_val_t date, el_val_t time);
|
||||
el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal);
|
||||
|
||||
el_val_t local_date_year(el_val_t ld);
|
||||
el_val_t local_date_month(el_val_t ld);
|
||||
el_val_t local_date_day(el_val_t ld);
|
||||
el_val_t local_time_hour(el_val_t lt);
|
||||
el_val_t local_time_minute(el_val_t lt);
|
||||
el_val_t local_time_second(el_val_t lt);
|
||||
el_val_t local_time_nanos(el_val_t lt);
|
||||
|
||||
el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur);
|
||||
el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur);
|
||||
el_val_t el_local_date_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_local_date_eq(el_val_t a, el_val_t b);
|
||||
|
||||
/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct
|
||||
* pointer in el_val_t; rhythms are immutable so callers may share them. */
|
||||
el_val_t rhythm_cycle_start(void);
|
||||
el_val_t rhythm_cycle_phase(el_val_t phase);
|
||||
el_val_t rhythm_duration(el_val_t d);
|
||||
el_val_t rhythm_session_start(void);
|
||||
el_val_t rhythm_event(el_val_t name);
|
||||
el_val_t rhythm_and(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_or(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_weekday(el_val_t day);
|
||||
el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute);
|
||||
el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal);
|
||||
el_val_t rhythm_matches(el_val_t r, el_val_t ct);
|
||||
|
||||
/* ── UUID ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t uuid_new(void);
|
||||
el_val_t uuid_v4(void);
|
||||
|
||||
/* ── Environment ─────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t env(el_val_t key);
|
||||
|
||||
/* ── In-process state K/V ────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t state_set(el_val_t key, el_val_t value);
|
||||
el_val_t state_get(el_val_t key);
|
||||
el_val_t state_del(el_val_t key);
|
||||
el_val_t state_keys(void);
|
||||
|
||||
/* ── Float formatting ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t float_to_str(el_val_t f);
|
||||
el_val_t int_to_float(el_val_t n);
|
||||
el_val_t float_to_int(el_val_t f);
|
||||
el_val_t format_float(el_val_t f, el_val_t decimals);
|
||||
el_val_t decimal_round(el_val_t f, el_val_t decimals);
|
||||
el_val_t str_to_float(el_val_t s);
|
||||
|
||||
/* ── Math (Float-aware) ──────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t math_sqrt(el_val_t f);
|
||||
el_val_t math_log(el_val_t f);
|
||||
el_val_t math_ln(el_val_t f);
|
||||
el_val_t math_sin(el_val_t f);
|
||||
el_val_t math_cos(el_val_t f);
|
||||
el_val_t math_pi(void);
|
||||
|
||||
/* ── String additions ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t str_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_split(el_val_t s, el_val_t sep);
|
||||
el_val_t str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t str_char_code(el_val_t s, el_val_t i);
|
||||
el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_format(el_val_t fmt, el_val_t data);
|
||||
el_val_t str_lower(el_val_t s);
|
||||
el_val_t str_upper(el_val_t s);
|
||||
|
||||
/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes)
|
||||
* Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex.
|
||||
* is_* predicates: empty input returns false; multi-char requires ALL bytes
|
||||
* to match. ASCII ranges only in Phase 1. */
|
||||
|
||||
/* Counting */
|
||||
el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */
|
||||
el_val_t str_count_chars(el_val_t s); /* codepoint count */
|
||||
el_val_t str_count_bytes(el_val_t s); /* alias of str_len */
|
||||
el_val_t str_count_lines(el_val_t s);
|
||||
el_val_t str_count_words(el_val_t s);
|
||||
el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */
|
||||
el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */
|
||||
|
||||
/* Find / position */
|
||||
el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */
|
||||
el_val_t str_last_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */
|
||||
|
||||
/* Transform */
|
||||
el_val_t str_repeat(el_val_t s, el_val_t n);
|
||||
el_val_t str_reverse(el_val_t s); /* by codepoint */
|
||||
el_val_t str_strip_prefix(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_strip_suffix(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_strip_chars(el_val_t s, el_val_t chars);
|
||||
el_val_t str_lstrip(el_val_t s);
|
||||
el_val_t str_rstrip(el_val_t s);
|
||||
|
||||
/* Char classification (Bool) */
|
||||
el_val_t is_letter(el_val_t s);
|
||||
el_val_t is_digit(el_val_t s);
|
||||
el_val_t is_alphanumeric(el_val_t s);
|
||||
el_val_t is_whitespace(el_val_t s);
|
||||
el_val_t is_punctuation(el_val_t s);
|
||||
el_val_t is_uppercase(el_val_t s);
|
||||
el_val_t is_lowercase(el_val_t s);
|
||||
|
||||
/* Split / join */
|
||||
el_val_t str_split_lines(el_val_t s);
|
||||
el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */
|
||||
el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n);
|
||||
el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */
|
||||
|
||||
/* ── List additions ──────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t list_push(el_val_t list, el_val_t elem);
|
||||
el_val_t list_push_front(el_val_t list, el_val_t elem);
|
||||
el_val_t list_join(el_val_t list, el_val_t sep);
|
||||
el_val_t list_range(el_val_t start, el_val_t end);
|
||||
|
||||
/* ── Bool helpers ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t bool_to_str(el_val_t b);
|
||||
|
||||
/* ── Numeric parsing ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t parse_int(el_val_t s, el_val_t default_val);
|
||||
|
||||
/* ── Process ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
void exit_program(el_val_t code);
|
||||
el_val_t getpid_now(void);
|
||||
|
||||
/* ── CGI identity ─────────────────────────────────────────────────────────────
|
||||
* Called at the start of main() in CGI programs (those with a `cgi {}` block).
|
||||
* Records the program's DHARMA identity before any other code executes. */
|
||||
|
||||
void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal,
|
||||
el_val_t network, el_val_t engram);
|
||||
|
||||
/* ── DHARMA network builtins ─────────────────────────────────────────────────
|
||||
* Available to CGI programs (declared with a `cgi {}` block).
|
||||
*
|
||||
* Peers are addressed by `dharma_id` of the form
|
||||
* "<registry-id>@<transport-url>" e.g. "ntn-genesis@http://localhost:7770"
|
||||
* If the @<url> portion is omitted, transport defaults to
|
||||
* "http://localhost:7770" (the local CGI daemon assumption).
|
||||
*
|
||||
* Wire protocol (all peers expose):
|
||||
* POST <url>/dharma/recv { channel, from, content } → response body
|
||||
* POST <url>/dharma/event { type, payload, source, timestamp }
|
||||
* POST <url>/api/activate { query } → list of nodes
|
||||
*
|
||||
* Hosting application's responsibility: an El program with a `cgi {}` block
|
||||
* runs http_serve() with its own request handler; that handler should route
|
||||
* "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so
|
||||
* incoming events feed dharma_field() queues. The runtime itself does not
|
||||
* intercept any /dharma path. */
|
||||
|
||||
el_val_t dharma_connect(el_val_t cgi_id);
|
||||
el_val_t dharma_send(el_val_t channel, el_val_t content);
|
||||
el_val_t dharma_activate(el_val_t query);
|
||||
void dharma_emit(el_val_t event_type, el_val_t payload);
|
||||
el_val_t dharma_field(el_val_t event_type);
|
||||
void dharma_strengthen(el_val_t cgi_id, el_val_t weight);
|
||||
el_val_t dharma_relationship(el_val_t cgi_id);
|
||||
el_val_t dharma_peers(void);
|
||||
|
||||
/* Public C API: called by an El program's HTTP handler when a /dharma/event
|
||||
* request arrives. Pushes onto the per-event-type queue and signals any
|
||||
* pending dharma_field() blockers. All three arguments must be NUL-terminated
|
||||
* C strings (or NULL — then treated as empty). */
|
||||
void el_runtime_dharma_event_arrive(const char* event_type,
|
||||
const char* payload,
|
||||
const char* source);
|
||||
|
||||
/* ── Engram local graph primitives ───────────────────────────────────────────
|
||||
* Operate on the CGI's local Engram knowledge graph.
|
||||
* `engram_activate` queries the local graph only; `dharma_activate` is
|
||||
* network-wide across all connected CGI graphs. */
|
||||
|
||||
el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience);
|
||||
el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t importance, el_val_t confidence,
|
||||
el_val_t tier, el_val_t tags);
|
||||
/* Layered consciousness — see el_runtime.c for the layered architecture
|
||||
* design notes (search "Layered consciousness architecture"). The five
|
||||
* canonical layers (safety / core-identity / domain-knowledge / imprint /
|
||||
* suit) are seeded automatically; engram_add_layer extends the registry
|
||||
* with imprint or suit overlays at runtime. Nodes default to layer 1
|
||||
* (core-identity) when created via engram_node / engram_node_full. */
|
||||
el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible,
|
||||
el_val_t transparent, el_val_t injectable);
|
||||
el_val_t engram_remove_layer(el_val_t layer_id);
|
||||
el_val_t engram_list_layers(void);
|
||||
el_val_t engram_get_node(el_val_t id);
|
||||
void engram_strengthen(el_val_t node_id);
|
||||
void engram_forget(el_val_t node_id);
|
||||
el_val_t engram_node_count(void);
|
||||
el_val_t engram_search(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset);
|
||||
void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation);
|
||||
el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id);
|
||||
el_val_t engram_neighbors(el_val_t node_id);
|
||||
el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_edge_count(void);
|
||||
/* Three-pass activation: background fan-out → working-memory promotion →
|
||||
* Layer 0 override. See "Three-pass activation" in el_runtime.c. */
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
* through json_stringify. */
|
||||
el_val_t engram_get_node_json(el_val_t id);
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_activate_json(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_stats_json(void);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
* no nodes promoted to working memory. */
|
||||
el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth);
|
||||
|
||||
/* ── LLM (Anthropic API client) ─────────────────────────────────────────────
|
||||
* All functions call https://api.anthropic.com/v1/messages with the API key
|
||||
* from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */
|
||||
|
||||
el_val_t llm_call(el_val_t model, el_val_t prompt);
|
||||
el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt);
|
||||
el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools);
|
||||
el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64);
|
||||
el_val_t llm_models(void);
|
||||
|
||||
/* Register a tool handler by name. The handler is looked up via dlsym
|
||||
* (mirroring http_set_handler), so any El `fn <name>(input)` compiles to
|
||||
* a global C symbol that this function can locate at runtime.
|
||||
* Handler signature: `el_val_t handler(el_val_t input_json)` — receives
|
||||
* the tool input as a JSON-string el_val_t and returns a JSON-string
|
||||
* el_val_t result. Used by llm_call_agentic. */
|
||||
void llm_register_tool(el_val_t name, el_val_t handler_fn_name);
|
||||
|
||||
/* ── args() ─────────────────────────────────────────────────────────────────
|
||||
* Provides access to command-line arguments passed to the program.
|
||||
* Populated by el_runtime_init_args() before main() runs. */
|
||||
|
||||
el_val_t args(void);
|
||||
void el_runtime_init_args(int argc, char** argv);
|
||||
|
||||
/* ── Crypto primitives ─────────────────────────────────────────────────────
|
||||
* SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe).
|
||||
* Self-contained — no OpenSSL/libcrypto dependency. The implementations are
|
||||
* adapted from public-domain reference code (Brad Conte / RFC 4648).
|
||||
*
|
||||
* Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string
|
||||
* value whose contents are raw binary; callers usually feed these into
|
||||
* base64_encode. Note that el_val_t strings are NUL-terminated by convention,
|
||||
* so the binary payload may contain embedded NULs — pass it directly into
|
||||
* base64_encode (which uses an explicit length) rather than treating it as
|
||||
* a printable C string.
|
||||
*
|
||||
* The "base64" variants emit/accept RFC 4648 standard alphabet with padding.
|
||||
* The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding,
|
||||
* as used in JWTs. */
|
||||
|
||||
el_val_t sha256_hex(el_val_t input);
|
||||
el_val_t sha256_bytes(el_val_t input);
|
||||
el_val_t hmac_sha256_hex(el_val_t key, el_val_t message);
|
||||
el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message);
|
||||
el_val_t base64_encode(el_val_t input);
|
||||
el_val_t base64_decode(el_val_t input);
|
||||
el_val_t base64url_encode(el_val_t input);
|
||||
el_val_t base64url_decode(el_val_t input);
|
||||
|
||||
/* Length-aware variants (internal — exposed for the rare caller that already
|
||||
* has a known-length binary buffer and doesn't want to round-trip through
|
||||
* a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed
|
||||
* these implicitly. */
|
||||
el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len);
|
||||
el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe);
|
||||
|
||||
/* ── Post-quantum primitives (liboqs-backed) ────────────────────────────────
|
||||
* All inputs/outputs hex-encoded. Algorithm choices:
|
||||
* Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced)
|
||||
* KEM: CRYSTALS-Kyber-768 (NIST level 3)
|
||||
* Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2)
|
||||
*
|
||||
* If liboqs is not linked (detected via __has_include(<oqs/oqs.h>) at compile
|
||||
* time), the pq_* entry points return a JSON-shaped error string so callers
|
||||
* fail loudly rather than silently fall back to classical schemes:
|
||||
* {"error":"liboqs not linked, post-quantum primitives unavailable"}
|
||||
*
|
||||
* The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and
|
||||
* CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss).
|
||||
* Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack,
|
||||
* Kyber holds. SHA3-256 also remains usable independent of liboqs (the
|
||||
* Keccak permutation is PQ-OK as a primitive). */
|
||||
|
||||
el_val_t pq_keygen_signature(void);
|
||||
el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message);
|
||||
el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex);
|
||||
|
||||
el_val_t pq_kem_keygen(void);
|
||||
el_val_t pq_kem_encaps(el_val_t public_key_hex);
|
||||
el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex);
|
||||
|
||||
el_val_t pq_hybrid_keygen(void);
|
||||
el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined);
|
||||
|
||||
el_val_t sha3_256_hex(el_val_t input);
|
||||
|
||||
/* ── AEAD: AES-256-GCM (libcrypto-backed) ───────────────────────────────────
|
||||
* Symmetric authenticated encryption used to wrap envelopes after a KEM
|
||||
* handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the
|
||||
* Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256.
|
||||
*
|
||||
* aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where
|
||||
* ciphertext is the AES-256-GCM output with the 16-byte auth tag appended.
|
||||
* Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which
|
||||
* structurally rules out the GCM nonce-reuse footgun.
|
||||
*
|
||||
* aead_decrypt returns the plaintext String, or "" on any failure (including
|
||||
* auth-tag mismatch). Callers MUST check for "" before trusting the result. */
|
||||
el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext);
|
||||
el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex);
|
||||
|
||||
/* ── Native VM builtin aliases (for compiled El source) ─────────────────────
|
||||
* These match the El VM's native_* builtins so that El source compiled
|
||||
* to C can call the same names without modification. */
|
||||
|
||||
el_val_t native_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t native_list_len(el_val_t list);
|
||||
el_val_t native_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t native_list_empty(void);
|
||||
el_val_t native_list_clone(el_val_t list);
|
||||
el_val_t native_string_chars(el_val_t s);
|
||||
el_val_t native_int_to_str(el_val_t n);
|
||||
|
||||
/* ── Method-call shorthand aliases ──────────────────────────────────────────
|
||||
* The El method-call convention `obj.method(args)` compiles to
|
||||
* `method(obj, args)`. These aliases expose the runtime functions under
|
||||
* the short names that result from method calls in El source.
|
||||
*
|
||||
* Example: `myList.append(x)` → `append(myList, x)` (calls this alias)
|
||||
* `myList.len()` → `len(myList)` (calls this alias) */
|
||||
|
||||
el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */
|
||||
el_val_t len(el_val_t list); /* el_list_len */
|
||||
el_val_t get(el_val_t list, el_val_t index); /* el_list_get */
|
||||
el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */
|
||||
el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */
|
||||
|
||||
/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */
|
||||
/* See bottom of el_runtime.c for the implementation.
|
||||
* Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION.
|
||||
* No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */
|
||||
/* ── Subprocess execution ────────────────────────────────────────────────── */
|
||||
el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */
|
||||
el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */
|
||||
el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */
|
||||
el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */
|
||||
|
||||
el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json);
|
||||
el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json);
|
||||
el_val_t trace_span_start(el_val_t name);
|
||||
el_val_t trace_span_end(el_val_t span_handle);
|
||||
el_val_t emit_event(el_val_t name, el_val_t duration_ms);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
@@ -1202,7 +1202,7 @@ fn codegen_js_inner(stmts: [Map<String, Any>], source: String, bundle_mode: Bool
|
||||
js_emit_line(js_strip_es_exports(runtime_content))
|
||||
js_emit_line("")
|
||||
} else {
|
||||
js_emit_line("// Runtime: foundation/el/el-compiler/runtime/el_runtime.js")
|
||||
js_emit_line("// Runtime: foundation/el/runtime/el_runtime.js")
|
||||
js_emit_line("import \"./el_runtime.js\";")
|
||||
}
|
||||
// In module mode: destructure all builtins off globalThis.__el so call
|
||||
|
||||
@@ -1292,6 +1292,43 @@ fn next_if_id() -> String {
|
||||
native_int_to_str(n)
|
||||
}
|
||||
|
||||
// is_void_builtin — true for runtime builtins declared `void` in el_runtime.h.
|
||||
// User `-> Void` functions are emitted as el_val_t (return 0) so they are safe
|
||||
// to assign; only these C-level void builtins are not.
|
||||
fn is_void_builtin(name: String) -> Bool {
|
||||
if str_eq(name, "println") { return true }
|
||||
if str_eq(name, "print") { return true }
|
||||
if str_eq(name, "engram_strengthen") { return true }
|
||||
if str_eq(name, "engram_forget") { return true }
|
||||
if str_eq(name, "engram_connect") { return true }
|
||||
if str_eq(name, "dharma_emit") { return true }
|
||||
if str_eq(name, "dharma_strengthen") { return true }
|
||||
if str_eq(name, "llm_register_tool") { return true }
|
||||
if str_eq(name, "exit_program") { return true }
|
||||
if str_eq(name, "http_serve") { return true }
|
||||
if str_eq(name, "http_set_handler") { return true }
|
||||
if str_eq(name, "http_serve_async") { return true }
|
||||
if str_eq(name, "el_cgi_init") { return true }
|
||||
if str_eq(name, "el_retain") { return true }
|
||||
if str_eq(name, "el_release") { return true }
|
||||
false
|
||||
}
|
||||
|
||||
// cg_expr_is_void — true if `val` is a direct call to a void builtin, so the
|
||||
// if-expression arm must emit it as a bare statement rather than assigning its
|
||||
// (nonexistent) value to the result var.
|
||||
fn cg_expr_is_void(val: Map<String, Any>) -> Bool {
|
||||
let vk: String = val["expr"]
|
||||
if str_eq(vk, "Call") {
|
||||
let f = val["func"]
|
||||
let fk: String = f["expr"]
|
||||
if str_eq(fk, "Ident") {
|
||||
return is_void_builtin(f["name"])
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
// Render a single arm of the if-as-expression: emit each statement-before-last
|
||||
// as a side-effecting expression, then assign the final Expr's value to the
|
||||
// result var. If the arm body is empty or its last stmt isn't an Expr, the
|
||||
@@ -1300,6 +1337,10 @@ fn cg_if_expr_arm(stmts: [Map<String, Any>], result_var: String) -> String {
|
||||
let n: Int = native_list_len(stmts)
|
||||
// Collect statement fragments into a list to avoid O(n-) string growth.
|
||||
let parts: [String] = native_list_empty()
|
||||
// Track names already declared in this arm's C block. El permits `let x`
|
||||
// to redeclare/rebind x in the same scope, but C forbids redeclaring the
|
||||
// same name in one block: emit `el_val_t x = ...` first, `x = ...` after.
|
||||
let declared: [String] = native_list_empty()
|
||||
let i = 0
|
||||
while i < n {
|
||||
let s = native_list_get(stmts, i)
|
||||
@@ -1310,18 +1351,31 @@ fn cg_if_expr_arm(stmts: [Map<String, Any>], result_var: String) -> String {
|
||||
let name: String = s["name"]
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ")
|
||||
if list_contains(declared, name) {
|
||||
let parts = native_list_append(parts, name + " = " + val_c + "; ")
|
||||
} else {
|
||||
let declared = native_list_append(declared, name)
|
||||
let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ")
|
||||
}
|
||||
} else {
|
||||
if str_eq(sk, "Return") {
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
if cg_expr_is_void(val) {
|
||||
let parts = native_list_append(parts, val_c + "; ")
|
||||
} else {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
}
|
||||
} else {
|
||||
if str_eq(sk, "Expr") {
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
if is_last {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
if cg_expr_is_void(val) {
|
||||
let parts = native_list_append(parts, val_c + "; ")
|
||||
} else {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
}
|
||||
} else {
|
||||
let parts = native_list_append(parts, "(void)(" + val_c + "); ")
|
||||
}
|
||||
@@ -2669,7 +2723,11 @@ fn builtin_arity(name: String) -> Int {
|
||||
if str_eq(name, "engram_activate") { return 2 }
|
||||
if str_eq(name, "engram_save") { return 1 }
|
||||
if str_eq(name, "engram_load") { return 1 }
|
||||
if str_eq(name, "engram_store_boot") { return 1 }
|
||||
if str_eq(name, "engram_store_checkpoint") { return 0 }
|
||||
if str_eq(name, "engram_store_close") { return 0 }
|
||||
if str_eq(name, "engram_get_node_json") { return 1 }
|
||||
if str_eq(name, "engram_get_node_by_label") { return 1 }
|
||||
if str_eq(name, "engram_search_json") { return 2 }
|
||||
if str_eq(name, "engram_scan_nodes_json") { return 2 }
|
||||
if str_eq(name, "engram_neighbors_json") { return 3 }
|
||||
|
||||
@@ -49,6 +49,21 @@ fn tok_value(tokens: [Any], pos: Int) -> String {
|
||||
native_list_get(tokens, pos * 2 + 1)
|
||||
}
|
||||
|
||||
// parse_progress_fatal — robustness backstop. Called by the token-consuming
|
||||
// driver loops when they detect they have iterated more times than there are
|
||||
// tokens (impossible for a well-formed program, where every iteration consumes
|
||||
// at least one token). Names the offending token and exits non-zero instead of
|
||||
// looping forever / exhausting memory.
|
||||
fn parse_progress_fatal(where: String, tokens: [Any], pos: Int) -> Void {
|
||||
let k: String = tok_kind(tokens, pos)
|
||||
let v: String = tok_value(tokens, pos)
|
||||
println("elc: FATAL: parser made no forward progress in " + where
|
||||
+ " at token index " + native_int_to_str(pos) + " (kind=" + k + ")")
|
||||
println("elc: likely a malformed construct near '" + v
|
||||
+ "' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" ).")
|
||||
exit(1)
|
||||
}
|
||||
|
||||
fn expect(tokens: [Any], pos: Int, kind: String) -> Int {
|
||||
let k = tok_kind(tokens, pos)
|
||||
if k == kind {
|
||||
@@ -1212,7 +1227,16 @@ fn parse_block(tokens: [Any], pos: Int) -> Map<String, Any> {
|
||||
let p = expect(tokens, pos, "LBrace")
|
||||
let stmts: [Map<String, Any>] = native_list_empty()
|
||||
let running = true
|
||||
// Runaway backstop: a block can hold at most (token count) statements, since
|
||||
// every iteration consumes >= 1 token. If we exceed that, the cursor has run
|
||||
// off the end without terminating (malformed input) -> fail fast, don't hang.
|
||||
let blk_total: Int = native_list_len(tokens) / 2
|
||||
let blk_iters: Int = 0
|
||||
while running {
|
||||
let blk_iters = blk_iters + 1
|
||||
if blk_iters > blk_total + 8 {
|
||||
parse_progress_fatal("parse_block", tokens, p)
|
||||
}
|
||||
let k = tok_kind(tokens, p)
|
||||
if k == "RBrace" {
|
||||
let running = false
|
||||
|
||||
+3
-3
@@ -368,13 +368,13 @@ fn main() -> Void {
|
||||
let which_out: String = str_trim(exec_capture("which " + elc_bin + " 2>/dev/null"))
|
||||
if !str_eq(which_out, "") {
|
||||
let elc_dir: String = dirname_of(which_out)
|
||||
runtime_path = elc_dir + "/../el-compiler/runtime/el_runtime.c"
|
||||
runtime_path = elc_dir + "/../runtime/el_runtime.c"
|
||||
}
|
||||
}
|
||||
// If --runtime points to a directory, auto-locate el_runtime.c inside it.
|
||||
// This lets both forms work:
|
||||
// --runtime=/opt/el/el-compiler/runtime (directory form)
|
||||
// --runtime=/opt/el/el-compiler/runtime/el_runtime.c (file form)
|
||||
// --runtime=/opt/el/runtime (directory form)
|
||||
// --runtime=/opt/el/runtime/el_runtime.c (file form)
|
||||
if !str_eq(runtime_path, "") {
|
||||
let is_dir: String = str_trim(exec_capture("test -d " + runtime_path + " && echo dir || echo file"))
|
||||
if str_eq(is_dir, "dir") {
|
||||
|
||||
@@ -3797,6 +3797,9 @@ fn builtin_arity(name: String) -> Int {
|
||||
if str_eq(name, "engram_activate") { return 2 }
|
||||
if str_eq(name, "engram_save") { return 1 }
|
||||
if str_eq(name, "engram_load") { return 1 }
|
||||
if str_eq(name, "engram_store_boot") { return 1 }
|
||||
if str_eq(name, "engram_store_checkpoint") { return 0 }
|
||||
if str_eq(name, "engram_store_close") { return 0 }
|
||||
if str_eq(name, "engram_get_node_json") { return 1 }
|
||||
if str_eq(name, "engram_search_json") { return 2 }
|
||||
if str_eq(name, "engram_scan_nodes_json") { return 2 }
|
||||
|
||||
+105
-2
@@ -1423,15 +1423,53 @@ el_val_t tok_at(el_val_t tokens, el_val_t pos) {
|
||||
}
|
||||
|
||||
el_val_t tok_kind(el_val_t tokens, el_val_t pos) {
|
||||
/* Out-of-range reads MUST report the Eof sentinel so every `== "Eof"`
|
||||
termination guard in the parser fires. Without this, reading past the
|
||||
trailing Eof token returns runtime null (native_list_get OOB -> 0), which
|
||||
matches no delimiter, letting inner parse loops (parse_block, parse_binop)
|
||||
append AST nodes forever on malformed input -> unbounded allocation -> OOM. */
|
||||
el_val_t n = (native_list_len(tokens) / 2);
|
||||
if (pos < 0) {
|
||||
return EL_STR("Eof");
|
||||
}
|
||||
if (pos >= n) {
|
||||
return EL_STR("Eof");
|
||||
}
|
||||
return native_list_get(tokens, (pos * 2));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t tok_value(el_val_t tokens, el_val_t pos) {
|
||||
el_val_t n = (native_list_len(tokens) / 2);
|
||||
if (pos < 0) {
|
||||
return EL_STR("");
|
||||
}
|
||||
if (pos >= n) {
|
||||
return EL_STR("");
|
||||
}
|
||||
return native_list_get(tokens, ((pos * 2) + 1));
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* parse_progress_fatal — robustness backstop. Called by the token-consuming
|
||||
driver loops when they detect they have iterated more times than there are
|
||||
tokens (an impossibility for a well-formed program, where every iteration
|
||||
consumes at least one token). Names the offending token and exits non-zero
|
||||
instead of looping forever / exhausting memory. */
|
||||
el_val_t parse_progress_fatal(el_val_t where, el_val_t tokens, el_val_t pos) {
|
||||
el_val_t k = tok_kind(tokens, pos);
|
||||
el_val_t v = tok_value(tokens, pos);
|
||||
println(el_str_concat(el_str_concat(el_str_concat(el_str_concat(
|
||||
EL_STR("elc: FATAL: parser made no forward progress in "), where),
|
||||
EL_STR(" at token index ")), native_int_to_str(pos)),
|
||||
el_str_concat(EL_STR(" (kind="), el_str_concat(k, EL_STR(")")))));
|
||||
println(el_str_concat(el_str_concat(
|
||||
EL_STR("elc: likely a malformed construct near '"), v),
|
||||
EL_STR("' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" ).")));
|
||||
exit(1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t expect(el_val_t tokens, el_val_t pos, el_val_t kind) {
|
||||
el_val_t k = tok_kind(tokens, pos);
|
||||
if (str_eq(k, kind)) {
|
||||
@@ -2689,7 +2727,16 @@ el_val_t parse_block(el_val_t tokens, el_val_t pos) {
|
||||
el_val_t p = expect(tokens, pos, EL_STR("LBrace"));
|
||||
el_val_t stmts = native_list_empty();
|
||||
el_val_t running = 1;
|
||||
/* Runaway backstop: a block can hold at most (token count) statements, since
|
||||
every iteration consumes >= 1 token. If we exceed that, the cursor has run
|
||||
off the end without terminating (malformed input) -> fail fast, don't hang. */
|
||||
el_val_t __blk_total = (native_list_len(tokens) / 2);
|
||||
el_val_t __blk_iters = 0;
|
||||
while (running) {
|
||||
__blk_iters = (__blk_iters + 1);
|
||||
if (__blk_iters > (__blk_total + 8)) {
|
||||
parse_progress_fatal(EL_STR("parse_block"), tokens, p);
|
||||
}
|
||||
el_val_t k = tok_kind(tokens, p);
|
||||
if (str_eq(k, EL_STR("RBrace"))) {
|
||||
running = 0;
|
||||
@@ -4838,9 +4885,51 @@ el_val_t next_if_id(void) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* is_void_builtin — true for runtime builtins declared `void` in el_runtime.h.
|
||||
User `-> Void` functions are emitted as el_val_t (return 0) so they are safe
|
||||
to assign; only these C-level void builtins are not. */
|
||||
el_val_t is_void_builtin(el_val_t name) {
|
||||
if (str_eq(name, EL_STR("println"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("print"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_strengthen"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_forget"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_connect"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("dharma_emit"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("dharma_strengthen"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("llm_register_tool"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("exit_program"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_serve"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_set_handler"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_serve_async"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_cgi_init"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_retain"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_release"))) { return 1; }
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* cg_expr_is_void — true if `val` is a direct call to a void builtin, so the
|
||||
if-expression arm must emit it as a bare statement rather than assigning its
|
||||
(nonexistent) value to the result var. */
|
||||
el_val_t cg_expr_is_void(el_val_t val) {
|
||||
el_val_t vk = el_get_field(val, EL_STR("expr"));
|
||||
if (str_eq(vk, EL_STR("Call"))) {
|
||||
el_val_t f = el_get_field(val, EL_STR("func"));
|
||||
el_val_t fk = el_get_field(f, EL_STR("expr"));
|
||||
if (str_eq(fk, EL_STR("Ident"))) {
|
||||
return is_void_builtin(el_get_field(f, EL_STR("name")));
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
el_val_t n = native_list_len(stmts);
|
||||
el_val_t parts = native_list_empty();
|
||||
/* Track names already declared in this arm's C block. El permits `let x`
|
||||
to redeclare/rebind x in the same scope, but C forbids redeclaring the
|
||||
same name in one block. Emit `el_val_t x = ...` the first time and a
|
||||
plain `x = ...` reassignment thereafter (mirrors cg_stmt's `declared`). */
|
||||
el_val_t declared = native_list_empty();
|
||||
el_val_t i = 0;
|
||||
while (i < n) {
|
||||
el_val_t s = native_list_get(stmts, i);
|
||||
@@ -4853,18 +4942,31 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
el_val_t name = el_get_field(s, EL_STR("name"));
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
if (list_contains(declared, name)) {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(name, EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
} else {
|
||||
declared = native_list_append(declared, name);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
}
|
||||
} else {
|
||||
if (str_eq(sk, EL_STR("Return"))) {
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
if (cg_expr_is_void(val)) {
|
||||
parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; ")));
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
}
|
||||
} else {
|
||||
if (str_eq(sk, EL_STR("Expr"))) {
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
if (is_last) {
|
||||
if (cg_expr_is_void(val)) {
|
||||
parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; ")));
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
}
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(EL_STR("(void)("), val_c), EL_STR("); ")));
|
||||
}
|
||||
@@ -4883,6 +4985,7 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
}
|
||||
el_val_t result = str_join(parts, EL_STR(""));
|
||||
el_release(parts);
|
||||
el_release(declared);
|
||||
return result;
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -6,8 +6,8 @@
|
||||
//
|
||||
// Compile and run:
|
||||
// ./dist/platform/elc examples/html-page.el > /tmp/html-page.c
|
||||
// cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
// -o /tmp/html-page /tmp/html-page.c el-compiler/runtime/el_runtime.c
|
||||
// cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
// -o /tmp/html-page /tmp/html-page.c runtime/el_runtime.c
|
||||
// /tmp/html-page
|
||||
|
||||
fn render_item(item: String) -> String {
|
||||
|
||||
@@ -1,28 +0,0 @@
|
||||
# El Compiler Release v1.0.0 — 2026-05-02
|
||||
|
||||
## Components
|
||||
- `bootstrap.py` — El language compiler (Python, recursive descent parser, emits C)
|
||||
- `el_runtime.c` — El runtime (C, HTTP server, engram, DHARMA, LLM chain)
|
||||
- `el_runtime.h` — Runtime public API header
|
||||
|
||||
## Changes in this release
|
||||
|
||||
### Critical bug fixes
|
||||
- `state_set`/`state_get` are now thread-safe (pthread_mutex). Was racing across 64 worker threads.
|
||||
- `looks_like_string` threshold raised from 1,000,000 to 4GB. Unix timestamps were being dereferenced as heap pointers.
|
||||
- `fs_read` guards against negative `ftell` result (pipe/special file overflow).
|
||||
|
||||
### Engram architecture (major)
|
||||
- Two-layer activation: `background_activation` (Layer 1, broad fan-out) + `working_memory_weight` (Layer 2, executive filter)
|
||||
- Inhibitory edges: `EngramEdge.inhibitory` flag suppresses working memory promotion without affecting background activation
|
||||
- Suppression memory: `suppression_count` — nodes activated-but-suppressed accumulate pressure toward breakthrough
|
||||
- Temporal decay: `temporal_decay_rate`, `created_at`, `last_activated_at`, `activation_count` on EngramNode
|
||||
- Per-type activation thresholds (Safety: 0.05, Canonical: 0.15, Lesson: 0.25, Note: 0.40)
|
||||
- Temporal range query: `engram_query_range(start_ms, end_ms)`
|
||||
- Layered consciousness: `EngramLayer` struct, `layer_id` on nodes and edges, `EngramStore.layers[]`
|
||||
- Layer 0 override pass: safety layer fires last and cannot be suppressed
|
||||
|
||||
## SHA256
|
||||
bootstrap.py
|
||||
el_runtime.c
|
||||
el_runtime.h
|
||||
@@ -132,7 +132,7 @@ if [[ $LVGL_OK -eq 1 ]]; then
|
||||
else
|
||||
_miss "LVGL/MCU" "-DEL_TARGET_LVGL (lvgl.h not found)"
|
||||
echo " Install: git clone https://github.com/lvgl/lvgl"
|
||||
echo " (place lvgl/ next to el-compiler/runtime/)"
|
||||
echo " (place lvgl/ next to runtime/)"
|
||||
MISSING=$((MISSING + 1))
|
||||
fi
|
||||
|
||||
@@ -50,6 +50,16 @@
|
||||
build defines the same helper as close() so the call sites are identical across platforms. */
|
||||
static inline int el_closesocket(SOCKET s) { return closesocket(s); }
|
||||
|
||||
/* ── setsockopt optval type ───────────────────────────────────────────────── */
|
||||
/* Winsock's setsockopt takes optval as (const char*); POSIX takes (const void*), so el_runtime.c
|
||||
passes &int directly. GCC 14+ makes that an error under -Wincompatible-pointer-types. Wrap it so
|
||||
the runtime's POSIX-style call sites compile unchanged (defined before the macro so the wrapper
|
||||
itself resolves to the real winsock setsockopt). */
|
||||
static inline int el_setsockopt(SOCKET s, int level, int optname, const void* optval, int optlen) {
|
||||
return setsockopt(s, level, optname, (const char*)optval, optlen);
|
||||
}
|
||||
#define setsockopt(s, l, o, v, n) el_setsockopt((s), (l), (o), (v), (int)(n))
|
||||
|
||||
/* ── winsock init (once, at load) ─────────────────────────────────────────── */
|
||||
static void el__win_net_init(void) {
|
||||
static int inited = 0;
|
||||
@@ -75,6 +85,7 @@ static inline void* el_win_dlsym(void* handle, const char* name) {
|
||||
#include <direct.h> /* _mkdir */
|
||||
#define mkdir(path, mode) _mkdir(path) /* POSIX mkdir(path,mode) → _mkdir(path) */
|
||||
#define timegm _mkgmtime /* UTC tm → time_t */
|
||||
#define fsync(fd) _commit(fd) /* no fsync() on Windows; _commit() (<io.h>) is the equiv */
|
||||
|
||||
/* setenv/unsetenv: not in the Windows CRT; map to _putenv_s / SetEnvironmentVariable. */
|
||||
static inline int setenv(const char* name, const char* value, int overwrite) {
|
||||
@@ -114,4 +125,63 @@ static inline struct tm* gmtime_r(const time_t* t, struct tm* out) {
|
||||
return gmtime_s(out, t) == 0 ? out : (struct tm*)0;
|
||||
}
|
||||
|
||||
/* ── libcurl: degradable stubs for the curl-less Windows build ─────────────── */
|
||||
/* The curl-less validation build (WITH_CURL=0) links no libcurl. el_runtime.c uses libcurl
|
||||
* unconditionally for its HTTP client / LLM layer; these stubs let it compile and link so the
|
||||
* runtime, HTTP *server*, graph and memory work natively on Windows. Live outbound HTTP/LLM calls
|
||||
* degrade to a runtime error (curl_easy_perform returns an error) — matching the documented
|
||||
* curl-less contract. When HAVE_CURL is defined (WITH_CURL=1) the real <curl/curl.h> is used and
|
||||
* this whole block is compiled out. POSIX never sees this header, so the POSIX build is untouched. */
|
||||
#ifndef HAVE_CURL
|
||||
|
||||
typedef void CURL;
|
||||
typedef int CURLcode;
|
||||
|
||||
#define CURLE_OK 0
|
||||
#define CURLE_HTTP_RETURNED_ERROR 22
|
||||
#define CURL_ERROR_SIZE 256
|
||||
|
||||
/* Option ids: values are irrelevant to the no-op setopt below; kept distinct for readability. */
|
||||
#define CURLOPT_URL 10002
|
||||
#define CURLOPT_WRITEFUNCTION 20011
|
||||
#define CURLOPT_WRITEDATA 10001
|
||||
#define CURLOPT_POSTFIELDS 10015
|
||||
#define CURLOPT_POSTFIELDSIZE 120
|
||||
#define CURLOPT_POST 47
|
||||
#define CURLOPT_HTTPHEADER 10023
|
||||
#define CURLOPT_TIMEOUT_MS 155
|
||||
#define CURLOPT_NOSIGNAL 99
|
||||
#define CURLOPT_USERAGENT 10018
|
||||
#define CURLOPT_FOLLOWLOCATION 52
|
||||
#define CURLOPT_ERRORBUFFER 10010
|
||||
#define CURLOPT_CUSTOMREQUEST 10036
|
||||
#define CURLOPT_FAILONERROR 45
|
||||
|
||||
struct curl_slist { char* data; struct curl_slist* next; };
|
||||
|
||||
static inline struct curl_slist* curl_slist_append(struct curl_slist* list, const char* s) {
|
||||
struct curl_slist* node = (struct curl_slist*)malloc(sizeof(struct curl_slist));
|
||||
if (!node) return list;
|
||||
node->data = s ? strdup(s) : NULL;
|
||||
node->next = NULL;
|
||||
if (!list) return node;
|
||||
struct curl_slist* p = list;
|
||||
while (p->next) p = p->next;
|
||||
p->next = node;
|
||||
return list;
|
||||
}
|
||||
static inline void curl_slist_free_all(struct curl_slist* list) {
|
||||
while (list) { struct curl_slist* n = list->next; free(list->data); free(list); list = n; }
|
||||
}
|
||||
|
||||
static inline CURL* curl_easy_init(void) { return (CURL*)malloc(1); }
|
||||
static inline CURLcode curl_easy_setopt(CURL* h, int opt, ...) { (void)h; (void)opt; return CURLE_OK; }
|
||||
static inline CURLcode curl_easy_perform(CURL* h) { (void)h; return 7 /* CURLE_COULDNT_CONNECT */; }
|
||||
static inline void curl_easy_cleanup(CURL* h) { free(h); }
|
||||
static inline const char* curl_easy_strerror(CURLcode c) {
|
||||
(void)c; return "libcurl not built in (curl-less build)";
|
||||
}
|
||||
|
||||
#endif /* !HAVE_CURL */
|
||||
|
||||
#endif /* EL_PLATFORM_WIN_H */
|
||||
File diff suppressed because it is too large
Load Diff
@@ -34,12 +34,12 @@
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
* cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
* cc -std=c11 -I runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
@@ -605,6 +605,13 @@ el_val_t engram_edge_count(void);
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
/* Tiered paged-store entry points (ENGRAM_STORE=1). engram_store_boot opens the
|
||||
* durable store (import-once / WAL-replay) and loads it resident; checkpoint pushes
|
||||
* the resident graph's current field state (incl. learned hebb + activation-formed
|
||||
* edges) through the WAL and flushes; close checkpoints + closes. No-ops when off. */
|
||||
el_val_t engram_store_boot(el_val_t data_dir);
|
||||
el_val_t engram_store_checkpoint(void);
|
||||
el_val_t engram_store_close(void);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
@@ -628,22 +635,30 @@ el_val_t engram_hebb_drain_json(el_val_t max);
|
||||
/* Document frequency of a term across node labels — term-specificity signal
|
||||
* for curiosity seed selection. (2026-08-03 self-review.) */
|
||||
el_val_t engram_label_df(el_val_t term);
|
||||
/* Best curiosity seed from one node: argmax over idf·position·casing across
|
||||
* the candidate tokens of its label, falling back to its content when the
|
||||
* label is a sentinel. Excludes pipe-delimited tabu terms during selection
|
||||
* and gates candidates to the df band [min_df, max_df]. Returns "" when
|
||||
* nothing qualifies. (2026-08-13 self-review.) */
|
||||
el_val_t engram_salient_term(el_val_t node_id, el_val_t max_df,
|
||||
el_val_t min_df, el_val_t tabu);
|
||||
el_val_t engram_embed_backfill(el_val_t count);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* Working memory introspection — count, mean weight, and top-N snapshot.
|
||||
* Ported from el-compiler/runtime on 2026-06-30 self-review. */
|
||||
* Ported from runtime on 2026-06-30 self-review. */
|
||||
el_val_t engram_wm_count(void);
|
||||
el_val_t engram_wm_avg_weight(void);
|
||||
el_val_t engram_wm_top_json(el_val_t n);
|
||||
/* Merge-load: add nodes/edges from a snapshot without resetting the store. */
|
||||
el_val_t engram_load_merge(el_val_t path);
|
||||
|
||||
/* ── WAL + compaction + integrity (ENGRAM_WAL=on; design doc §§3-14,§18) ──── */
|
||||
int engram_wal_enabled(void);
|
||||
el_val_t engram_crc32(el_val_t s);
|
||||
el_val_t engram_wal_boot(el_val_t dir); /* replay + open; returns records */
|
||||
el_val_t engram_wal_open_dir(el_val_t dir);
|
||||
el_val_t engram_wal_node_put(el_val_t dir, el_val_t id);
|
||||
el_val_t engram_wal_edges_since(el_val_t dir, el_val_t start_count);
|
||||
el_val_t engram_wal_hebb_batch(el_val_t dir, el_val_t start_count);
|
||||
el_val_t engram_wal_forget(el_val_t dir, el_val_t id);
|
||||
el_val_t engram_wal_compact(el_val_t dir);
|
||||
el_val_t engram_wal_maybe_compact(el_val_t dir);
|
||||
el_val_t engram_resolve_data_dir(void); /* §18.2 fail-loud default */
|
||||
el_val_t engram_is_protected(el_val_t id); /* §18.1/18.3 derived set */
|
||||
el_val_t engram_protected_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
@@ -8,7 +8,7 @@
|
||||
* Threading: __thread_create / __thread_join use dlsym(RTLD_DEFAULT) to look
|
||||
* up El function symbols at runtime. This is the foundation of El's parallelism.
|
||||
*
|
||||
* Link: cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* Link: cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el_seed.c
|
||||
*/
|
||||
|
||||
@@ -1072,6 +1072,7 @@ el_val_t __engram_save(el_val_t path) { return engram_save
|
||||
el_val_t __engram_load(el_val_t path) { return engram_load(path); }
|
||||
|
||||
el_val_t __engram_get_node_json(el_val_t id) { return engram_get_node_json(id); }
|
||||
el_val_t __engram_get_node_by_label(el_val_t label) { return engram_get_node_by_label(label); }
|
||||
|
||||
el_val_t __engram_search_json(el_val_t query, el_val_t limit) {
|
||||
return engram_search_json(query, limit);
|
||||
@@ -226,6 +226,7 @@ el_val_t __engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t __engram_save(el_val_t path);
|
||||
el_val_t __engram_load(el_val_t path);
|
||||
el_val_t __engram_get_node_json(el_val_t id);
|
||||
el_val_t __engram_get_node_by_label(el_val_t label);
|
||||
el_val_t __engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t __engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t __engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user