Compare commits
39 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4f49755ebb | |||
| 7aa847e32a | |||
| ee71423732 | |||
| bb64a236ed | |||
| 9a0266cbf9 | |||
| a72145b44e | |||
| 8affb1d6e0 | |||
| fa47b98d18 | |||
| 0a72fced28 | |||
| 5d0d4555ae | |||
| 8347a2f1c0 | |||
| b97b644799 | |||
| d71fc4c1c0 | |||
| a118d19393 | |||
| c6aa1e5c53 | |||
| ff577391f2 | |||
| ee0d5f9b97 | |||
| 391bd818ea | |||
| 43636aed99 | |||
| 2baa0b9a41 | |||
| 6a8b2461cd | |||
| bcb356fe69 | |||
| dd7827059a | |||
| 208e36c899 | |||
| b97ce74d1f | |||
| 155a449c4e | |||
| 4696fd6833 | |||
| 581a351fb1 | |||
| 8ce8656de2 | |||
| 1e49560f1f | |||
| e8f0b5a9de | |||
| 40287c4cfc | |||
| 0481bea44d | |||
| 9d565ca080 | |||
| 4773dd0aa2 | |||
| 6b9d9e6c4a | |||
| b4967af13e | |||
| 2b2a1246e7 | |||
| 5c41c66a0f |
@@ -39,9 +39,9 @@ jobs:
|
||||
run: |
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -54,9 +54,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -91,7 +91,7 @@ jobs:
|
||||
- name: Precompile el_runtime.o
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
gcc -O2 -c -I "$RUNTIME" "$RUNTIME/el_runtime.c" \
|
||||
-o /tmp/el_runtime.o
|
||||
echo "el_runtime.o compiled"
|
||||
@@ -100,7 +100,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -110,7 +110,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -120,7 +120,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -130,7 +130,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -140,7 +140,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -150,7 +150,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -160,7 +160,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -170,7 +170,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -180,7 +180,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -191,7 +191,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -202,7 +202,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -214,9 +214,18 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -242,7 +251,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-dev \
|
||||
@@ -250,7 +259,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-dev \
|
||||
@@ -258,7 +267,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-js \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
--source=runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-dev"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -268,6 +277,12 @@ jobs:
|
||||
# Patches ci-base:dev in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -291,9 +306,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -46,9 +46,9 @@ jobs:
|
||||
run: |
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -84,7 +84,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -94,7 +94,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -104,7 +104,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -114,7 +114,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -124,7 +124,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -134,7 +134,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -144,7 +144,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -154,7 +154,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -164,7 +164,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -176,9 +176,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -189,7 +189,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -200,7 +200,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -212,12 +212,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -235,7 +244,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-stage \
|
||||
@@ -243,7 +252,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-stage"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -253,6 +262,12 @@ jobs:
|
||||
# Patches ci-base:stage in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -275,9 +290,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -47,9 +47,9 @@ jobs:
|
||||
mkdir -p dist/platform
|
||||
dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elc-gen2.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/platform/elc
|
||||
chmod +x dist/platform/elc
|
||||
@@ -62,9 +62,9 @@ jobs:
|
||||
mkdir -p dist/bin
|
||||
dist/platform/elc elb.el > dist/elb.c
|
||||
gcc -O2 \
|
||||
-I el-compiler/runtime \
|
||||
-I runtime \
|
||||
dist/elb.c \
|
||||
el-compiler/runtime/el_runtime.c \
|
||||
runtime/el_runtime.c \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm \
|
||||
-o dist/bin/elb
|
||||
chmod +x dist/bin/elb
|
||||
@@ -75,7 +75,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/epm
|
||||
@@ -86,7 +86,7 @@ jobs:
|
||||
run: |
|
||||
ABS_ELB="$(pwd)/dist/bin/elb"
|
||||
ABS_ELC="$(pwd)/dist/platform/elc"
|
||||
ABS_RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
ABS_RUNTIME="$(pwd)/runtime"
|
||||
ABS_OUT="$(pwd)/dist/bin"
|
||||
(cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT")
|
||||
chmod +x dist/bin/el-install
|
||||
@@ -121,7 +121,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
@@ -131,7 +131,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
@@ -141,7 +141,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
@@ -151,7 +151,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
@@ -161,7 +161,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
@@ -171,7 +171,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
@@ -181,7 +181,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
@@ -191,7 +191,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
@@ -201,7 +201,7 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
RUNTIME="$(pwd)/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
@@ -216,8 +216,10 @@ jobs:
|
||||
cp lang/dist/platform/elc dist/sdk/bin/elc
|
||||
cp lang/dist/bin/elb dist/sdk/bin/elb
|
||||
cp lang/dist/bin/epm dist/sdk/bin/epm
|
||||
cp lang/el-compiler/runtime/el_runtime.c dist/sdk/runtime/
|
||||
cp lang/el-compiler/runtime/el_runtime.h dist/sdk/runtime/
|
||||
cp lang/runtime/el_runtime.c dist/sdk/runtime/
|
||||
cp lang/runtime/el_runtime.h dist/sdk/runtime/
|
||||
cp lang/runtime/engram_store.c dist/sdk/runtime/
|
||||
cp lang/runtime/engram_store.h dist/sdk/runtime/
|
||||
cp lang/runtime/*.el dist/sdk/runtime/
|
||||
tar -czf dist/el-sdk-latest.tar.gz -C dist/sdk .
|
||||
echo "SDK tarball bundled: dist/el-sdk-latest.tar.gz"
|
||||
@@ -274,8 +276,10 @@ jobs:
|
||||
|
||||
# Per-file assets (downstream CI needs these individually)
|
||||
upload_asset lang/dist/platform/elc elc
|
||||
upload_asset lang/el-compiler/runtime/el_runtime.c el_runtime.c
|
||||
upload_asset lang/el-compiler/runtime/el_runtime.h el_runtime.h
|
||||
upload_asset lang/runtime/el_runtime.c el_runtime.c
|
||||
upload_asset lang/runtime/el_runtime.h el_runtime.h
|
||||
upload_asset lang/runtime/engram_store.c engram_store.c
|
||||
upload_asset lang/runtime/engram_store.h engram_store.h
|
||||
|
||||
# SDK bundle and installer binary
|
||||
upload_asset dist/el-sdk-latest.tar.gz el-sdk-latest.tar.gz
|
||||
@@ -288,12 +292,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -319,7 +332,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-c \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.c
|
||||
--source=runtime/el_runtime.c
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-prod \
|
||||
@@ -327,7 +340,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-h \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
--source=runtime/el_runtime.h
|
||||
|
||||
gcloud artifacts generic upload \
|
||||
--repository=foundation-prod \
|
||||
@@ -335,7 +348,7 @@ jobs:
|
||||
--project=neuron-785695 \
|
||||
--package=el-runtime-js \
|
||||
--version="${VERSION}" \
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
--source=runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-prod"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
@@ -345,6 +358,12 @@ jobs:
|
||||
# Patches ci-base:latest in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
@@ -367,9 +386,9 @@ jobs:
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c
|
||||
COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
|
||||
@@ -6,13 +6,13 @@ set -euo pipefail
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
LANG_DIR="$ROOT/lang"
|
||||
RUNTIME="$LANG_DIR/el-compiler/runtime"
|
||||
RUNTIME="$LANG_DIR/runtime"
|
||||
ELC="$LANG_DIR/dist/platform/elc"
|
||||
|
||||
# If elc isn't built yet, skip with a warning rather than blocking
|
||||
if [ ! -x "$ELC" ]; then
|
||||
echo "⚠ elc not found at lang/dist/platform/elc — skipping pre-commit tests"
|
||||
echo " Build it first: cd lang && gcc -O2 -I el-compiler/runtime dist/elc-bootstrap.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I el-compiler/runtime /tmp/elc.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc"
|
||||
echo " Build it first: cd lang && gcc -O2 -I runtime dist/elc-bootstrap.c runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I runtime /tmp/elc.c runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
# El
|
||||
|
||||
**A self-hosting, statically-typed language that compiles to C — built around a graph-native runtime instead of a database driver.**
|
||||
|
||||
El is the execution substrate for the Neuron agent runtime, the DHARMA network, and the Engram knowledge graph. This repository is the monorepo for the whole stack: the language itself, the graph memory engine it's built to talk to natively, and the tools (package manager, IDE, UI framework, diagramming) built on top of it.
|
||||
|
||||
---
|
||||
|
||||
## Why El exists
|
||||
|
||||
Every other language treats persistent, associative state as something you reach for through a driver — a SQL client, an ORM, a Redis library bolted on from outside. El inverts that: graph operations (`engram_*`) are runtime primitives, on the same footing as string or list operations. There is no separate database driver because the database is not separate.
|
||||
|
||||
El has four defining properties:
|
||||
|
||||
1. **Self-hosting compiler.** The compiler (`lexer.el`, `parser.el`, `codegen.el`, `compiler.el`) is written in El. It compiles El source to C, which `cc` compiles against a fixed runtime into a native binary. A Rust genesis compiler bootstrapped the first iteration; the self-hosted binary at `lang/dist/platform/elc` has been the canonical compiler ever since — every binary in `dist/platform/` was produced by an earlier version of itself compiling `el-compiler/src/`. The chain is auditable: source is the ground truth, not the binary. See [lang/BOOTSTRAP.md](lang/BOOTSTRAP.md) for the full recovery path if that binary is ever lost.
|
||||
2. **C compilation target.** Every compiled program is plain C11. Every El value is `el_val_t` (`int64_t`); strings are heap pointers cast through it. Functions become C functions; top-level statements become `main()`.
|
||||
3. **Graph-native runtime.** The runtime provides first-class graph operations over an in-process Engram store — no separate DB driver, no ORM.
|
||||
4. **DHARMA-aware identity.** A `cgi` block declares a program's DHARMA identity at compile time. The runtime resolves identity before user code runs, so `dharma_*` calls have a stable principal and channel surface throughout.
|
||||
|
||||
---
|
||||
|
||||
## Architecture map
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ lang │ El compiler + C runtime
|
||||
│ (El itself) │ everything below is written in it,
|
||||
└──────┬──────┘ or compiles down through it
|
||||
│
|
||||
┌─────────────┼─────────────┐
|
||||
│ │ │
|
||||
┌──────▼─────┐ ┌─────▼─────┐ ┌─────▼─────┐
|
||||
│ engram │ │ epm │ │ ide │
|
||||
│ graph/mem │ │ package │ │ editor + │
|
||||
│ substrate │ │ manager │ │ LSP │
|
||||
└──────┬─────┘ └───────────┘ └───────────┘
|
||||
│
|
||||
┌───────┼────────────────┬─────────────────────┐
|
||||
│ │ │ │
|
||||
┌─────▼───┐ ┌─▼──────────┐ ┌──▼──────────┐ ┌─────▼──────┐
|
||||
│ elp │ │ ql │ │ ui │ │ arbor │
|
||||
│ NLG / │ │engram-el. │ |spreading- │ |arbor │
|
||||
│ 31 langs│ │studio+tests│ |activation UI│ |diagram lang│
|
||||
└─────────┘ └────────────┘ └─────────────┘ └────────────┘
|
||||
```
|
||||
|
||||
`lang` is the foundation — the compiler and C runtime everything else builds on. `engram` is the graph-native memory/state engine that gives El its identity (property 3 above). Everything else is either a tool for working with El (`epm`, `ide`) or a system built on top of Engram's graph model (`elp`, `ql`, `ui`, `arbor`).
|
||||
|
||||
---
|
||||
|
||||
## Repository layout
|
||||
|
||||
### [lang/](lang/) — the El language
|
||||
|
||||
The compiler and runtime. Self-hosting: `elc-cli.el` → `compiler.el` → `lexer.el` / `parser.el` / `codegen.el` / `codegen-js.el`, textually inlined and compiled in one pass. Compiles to C11 and links against `el-compiler/runtime/el_seed.c`, a hand-maintained OS-boundary layer (libcurl HTTP, pthreads, filesystem, arena allocation) — everything else in the runtime is native El (`runtime/*.el`).
|
||||
|
||||
Two layers to know: **El programs** (`.el` files — where nearly all work belongs) and **the C seed** (`el_seed.c` — edit only for genuine OS-level access; never re-implement what El can already express).
|
||||
|
||||
Current status (single source of truth: [lang/spec/language.md](lang/spec/language.md)): lexer/parser/codegen and the C runtime's core (I/O, strings, math, lists, maps, filesystem, args) are implemented. In flight: `%` operator, match-statement codegen, `?` nil-propagation, `cgi` block parsing + DHARMA identity resolution, VBD role enforcement (`@manager`/`@engine`/`@accessor`), the real `engram_*` and `dharma_*` runtimes (currently stubs), and libcurl-backed `http_get`/`http_post`/`http_serve`. Bitwise operators, `??`, and `as` casts are explicitly **not** in this language.
|
||||
|
||||
Key docs: [AGENTS.md](lang/AGENTS.md) (agent-facing orientation), [BOOTSTRAP.md](lang/BOOTSTRAP.md) (compiler recovery from scratch), [spec/language.md](lang/spec/language.md), [spec/codegen-js.md](lang/spec/codegen-js.md).
|
||||
|
||||
### [engram/](engram/) — graph intelligence substrate
|
||||
|
||||
**A local-first memory substrate for accumulating intelligence**, and the reason El's runtime doesn't need a database driver. Rust core (`engram-core`, `engram-ffi`) exposed to El and other languages (Kotlin, TypeScript/WASM, Go bindings).
|
||||
|
||||
The model: retrieval is **spreading activation**, not query. You name seed nodes and a query embedding; activation propagates outward through weighted edges, attenuating multiplicatively per hop (`strength = parent_strength × edge_weight × target_salience × cosine_sim`), gets pruned below a threshold, and the top-N nodes by activation strength come back. Storage and retrieval are the same structure — the way long-term potentiation works in biological memory, not the way a relational or vector database works.
|
||||
|
||||
Nodes live in four tiers (Working / Episodic / Semantic / Procedural, mirroring prefrontal / hippocampal / neocortical / cerebellar memory) and migrate between them based on **salience decay** — `importance × recency-decay × log(activation_count)`. Forgetting is adaptive pruning, not a bug: unreinforced memories stop competing for attention without being deleted.
|
||||
|
||||
Backed by `sled` (embedded, local-first, no daemon) with flat cosine scan for vector search — deliberately simple until scale demands an HNSW layer. Full API and design rationale in [engram/README.md](engram/README.md).
|
||||
|
||||
### [elp/](elp/) — Engram Language Protocol
|
||||
|
||||
Bidirectional engine mapping between Engram semantic forms and natural-language surface text, across **31 languages** — from Spanish and Japanese through historical/liturgical languages (Old Norse, Sanskrit, Sumerian, Coptic, Akkadian, Ge'ez). Compilation order runs `language-profile` + `vocabulary` → per-language `morphology-*` → `grammar` → `realizer` → `semantics` → `elp`. This is what lets an Engram graph node round-trip to and from readable text in any of those languages.
|
||||
|
||||
### [epm/](epm/) — El Package Manager
|
||||
|
||||
Manages **vessels** (El's package unit): publish, install, resolve dependencies. Vessels are stored in Engram as graph nodes, not files in a registry index — `epm` reads the local `manifest.el`, talks to Engram over HTTP, and writes resolved vessels to `.epm/vessels/`. Source: `registry.el`, `install.el`, `update.el`, `manifest.el`.
|
||||
|
||||
### [ide/](ide/) — El IDE
|
||||
|
||||
Three vessels: **el-ide-server** (HTTP backend — file ops, build/run, LSP bridge, plugin host, settings), **el-lsp** (the language server — completion, hover, diagnostics, outline, format, type graph), and **el-plugin-host** (first-party plugin lifecycle: install/remove/enable/disable). `ide/projects/` and `ide/examples/` hold sample projects, including the canonical `hello-friends` first-program walkthrough.
|
||||
|
||||
### [ql/](ql/) — engram-el
|
||||
|
||||
The El-native integration layer for a *live* Engram server — not a library (no importable modules, no build artifact), a set of standalone `.el` programs run directly via `el run-file`. Three components: **Studio** (`studio/studio.el`, a full terminal graph explorer), a **Hebbian field-model** proof of concept, and El builtin / LLM-builtin smoke test suites. This is the reference for correct patterns when an El program uses Engram as its substrate. Spec: [ql/spec/elql.md](ql/spec/elql.md).
|
||||
|
||||
### [ui/](ui/) — el-ui
|
||||
|
||||
A frontend framework where **component state is an Engram graph and reactivity is spreading activation** — not virtual-DOM diffing (React), Proxy-based dependency tracking (Vue), or compile-time analysis (Svelte). Re-renders are activated and propagated the same way associative memory retrieval works in `engram/`.
|
||||
|
||||
~15 vessels covering the full frontend surface: `el-platform` (env/fs/network/clock abstraction), `el-config`, `el-html` (SSR emit primitives), `el-layout`, `el-style` (design tokens/themes), `el-i18n`, `el-auth` / `el-identity` (JWT, sessions, OAuth PKCE — Engram-native), `el-services` (REST/gRPC/WebSocket bindings), `el-aop` (`@authenticate`/`@authorize`/`@cache`/`@rate_limit` decorators), `el-secrets`, `el-graph` (graph rendering/editor), `el-publish` (App Store / Play Store automation), and `el-ui-compiler` (El→JS component compiler; currently a stub pending a JS backend in `elc`). Spec: [ui/spec/framework.md](ui/spec/framework.md).
|
||||
|
||||
### [arbor/](arbor/) — diagram language
|
||||
|
||||
A `.arbor` diagram language and toolchain: `arbor-core` (NodeId/shape/edge-kind types), `arbor-parse` (recursive-descent parser), `arbor-diagram` (IR + Mermaid serializer + architecture-diagram builders), `arbor-layout` (hierarchical layout — rank assignment, positioning, group bounds), `arbor-render` (SVG renderer), `arbor-cli`. (The architecture map above is the kind of diagram this is for.)
|
||||
|
||||
---
|
||||
|
||||
## Getting started
|
||||
|
||||
Install the El SDK from the latest release:
|
||||
|
||||
```bash
|
||||
bash lang/install.sh
|
||||
# EL_VERSION=v1.0.0 bash lang/install.sh # pin a specific release tag
|
||||
# EL_PREFIX=/opt/el bash lang/install.sh # custom install prefix
|
||||
```
|
||||
|
||||
Or build the compiler from source and verify the self-hosting chain:
|
||||
|
||||
```bash
|
||||
cd lang
|
||||
./dist/platform/elc elc-cli.el > elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_seed.c
|
||||
|
||||
# Confirm the new binary reproduces itself exactly
|
||||
./dist/platform/elc-new elc-cli.el > elc-verify.c
|
||||
diff elc-new.c elc-verify.c # should be identical
|
||||
|
||||
mv dist/platform/elc-new dist/platform/elc
|
||||
```
|
||||
|
||||
Run your first program:
|
||||
|
||||
```bash
|
||||
./lang/dist/platform/elc lang/examples/hello.el > hello.c
|
||||
cc -std=c11 -I lang/el-compiler/runtime -lcurl -lpthread \
|
||||
-o hello hello.c lang/el-compiler/runtime/el_seed.c
|
||||
./hello
|
||||
```
|
||||
|
||||
More examples in [lang/examples/](lang/examples/), including a full starter project at `lang/examples/hello-project/`.
|
||||
|
||||
If the compiler binary is ever lost or corrupted, [lang/BOOTSTRAP.md](lang/BOOTSTRAP.md) is the authoritative recovery path.
|
||||
|
||||
---
|
||||
|
||||
## Development workflow
|
||||
|
||||
Branching follows `dev → stage → main`: work lands on `dev`, promotes to `stage` for integration testing, and is promoted to `main` for release (visible directly in the git history of this repo). CI is defined per-subproject under `.gitea/workflows/` — `lang`/`epm`/`ide` share the root pipeline; `engram` and `ql` carry their own (`ci-dev`, `ci-stage`, and a release workflow each).
|
||||
|
||||
- Language/runtime specs live at `*/spec/*.md` (`lang/spec/`, `ql/spec/`, `ui/spec/`) and are the single source of truth for implemented-vs-planned status — code and docs are expected to agree with the spec's status markers, not the other way around.
|
||||
- Agent-facing orientation guides live at `*/AGENTS.md` (currently `lang/AGENTS.md`); more subprojects may grow their own as they need agent-specific conventions documented.
|
||||
- Tagged releases live under `lang/releases/`, each with its own `RELEASE.md`.
|
||||
|
||||
---
|
||||
|
||||
## Status
|
||||
|
||||
This is an actively developed, internal monorepo — not yet published under an open license. Treat everything here as proprietary to Neuron Technologies unless told otherwise.
|
||||
@@ -1,65 +0,0 @@
|
||||
# ELP language consolidation — full-lexicon backfill (stage)
|
||||
|
||||
Branch: `stage-elp-lang-consolidation` (stage-bound; NOT the live soul :8742).
|
||||
|
||||
Consolidates scattered Python language-realizer work (`~/Desktop/lang-realizers`,
|
||||
`~/Desktop/lang-poetry-experiment`, `~/semitic_engine`) into the ELP `.el`
|
||||
structure, generating **full lexicons** (complete UniMorph + kaikki.org
|
||||
Wiktionary — real gender, real inflections) instead of the demo/curated subsets
|
||||
the prototypes shipped.
|
||||
|
||||
## ELP before this branch
|
||||
- 18 classical/ancient languages fully done (vocab + morphology + tests):
|
||||
akk ang cop egy enm fro gez goh got grc non peo pi sa sga sux txb uga.
|
||||
- 11 modern/classical languages had `morphology-<code>.el` in the build manifest
|
||||
but **no vocabulary and no lang_profile**: es fr de ja ar he hi ru fi sw la.
|
||||
- The ES port (`stage-elp-es-port`) had a *demo-scale* vocabulary-es.el (~350
|
||||
entries, s-expr form).
|
||||
|
||||
## Landed on this branch (full-lexicon seed-fn format, matching the 18 ancients)
|
||||
Vocabulary schema per row: `[lemma, pos, form0, form1, form2, en_gloss, hint]`.
|
||||
Files are ELP runtime **seed data** (loaded via the Engram at runtime), so — like
|
||||
all 18 classical `vocabulary-*.el` — they are intentionally NOT in the build
|
||||
manifest. Syntax validated: the chunked `fn vocab_<code>_seed_pN` format
|
||||
compiles cleanly to C via `elc` (correct UTF-8).
|
||||
|
||||
| code | in-ELP-morph? | vocab entries | verbs | nouns | adjs | profile |
|
||||
|------|---------------|--------------:|------:|------:|-----:|---------|
|
||||
| es | yes | 72,032 | 6,695 | 48,353 | 16,984 | yes |
|
||||
| fr | yes | 130,517 | 7,534 | 77,344 | 45,639 | yes |
|
||||
| de | yes | 144,692 | 6,661 | 133,162 | 4,869 | yes |
|
||||
| la | yes | 22,590 | 82 | 13,436 | 9,072 | yes |
|
||||
| it | no (bonus) | 193,675 | 10,008 | 109,459 | 74,208 | yes |
|
||||
| pt | no (bonus) | 115,772 | 4,001 | 72,073 | 39,698 | yes |
|
||||
| ro | no (bonus) | 86,504 | 1,216 | 65,915 | 19,373 | yes |
|
||||
| ca | no (bonus) | 47,112 | 1,547 | 28,830 | 16,735 | yes |
|
||||
|**total**| |**812,894** | | | | |
|
||||
|
||||
Generators (reproducible): `elp/tests/lang-gen/gen_elp_seed_full.py` (Romance),
|
||||
`gen_elp_seed_de_la.py` (German declension + Latin case-paradigm mapping). They
|
||||
read the pre-built morph caches in `~/Desktop/lang-realizers/data/` (UniMorph +
|
||||
kaikki), which are too large to commit.
|
||||
|
||||
## Remaining (honest)
|
||||
Of the 11 ELP backfill targets, 4 are done (es fr de la). The other 7 have **no
|
||||
full-lexicon engine** yet — cannot be generated honestly without engine work:
|
||||
- **ru**: only a 110-entry curated Slavic subset exists; full `rus.unimorph`
|
||||
present but no `morphology_ru_full` productive loader. Needs a full Russian
|
||||
morphology module (like the Romance ones) before vocab generation.
|
||||
- **ja / ko / zh**: validated demo engines (~66-104 hardcoded words) in
|
||||
`lang-poetry-experiment`, Python only. Agglutinative (ja/ko) + isolating (zh)
|
||||
need `.el` engine ports + full-lexicon wiring (ja: jpn_unimorph; zh: CC-CEDICT).
|
||||
- **ar / he (Semitic)**: template engines (16 AR / 8 HE patterns, ~6 roots) in
|
||||
`~/semitic_engine`, Python only. Root-and-pattern; full UniMorph ara/heb
|
||||
present but used only for validation. Needs productive root lexicon + `.el` port.
|
||||
- **hi (Hindi), fi (Finnish), sw (Swahili)**: `morphology-<code>.el` exists in
|
||||
ELP but there is NO scattered prototype and NO downloaded data for these —
|
||||
full-lexicon collection (UniMorph/kaikki) + generator still to do.
|
||||
|
||||
De/nl/sv Germanic and it/ro/ca/pt Romance verb coverage note: German verbs here
|
||||
are the ~6.6k caches carry; the it/ro/ca/pt bonus languages have full vocab but
|
||||
**no `morphology-<code>.el` in ELP yet** (Python realizer exists; `.el` port is
|
||||
the remaining engine work).
|
||||
|
||||
Construction coverage (separate from lexicon): French realizer was ~55%,
|
||||
Semitic ~3% in the prototypes — full construction coverage remains its own task.
|
||||
@@ -1,72 +0,0 @@
|
||||
;;; lang_profile_ca.el — Catalan language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / _es / _pt; keys the realizer's construction switches.
|
||||
;;; Catalan is the CLOSEST Romance sibling to the shared engine (~85% conceptual
|
||||
;;; reuse). The deltas: PRONOMS FEBLES with four position allomorphs, l'-elision,
|
||||
;;; del/al/pel contractions, the periphrastic preterite (vaig+INF), and NO
|
||||
;;; essere/avere split (perfect aux is always HAVER; ser/estar is only the copula).
|
||||
|
||||
(lang_profile_ca
|
||||
(language "Catalan")
|
||||
(iso639 "ca")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "el/la/l'/els/les ; un/una/uns/unes") ; l'-ELISION:
|
||||
; el/la -> l' before vowel or (silent) h, glued to
|
||||
; the next word (l'home, l'illa); de -> d' before vowel
|
||||
(article-drives-contraction yes) ; article choice feeds prep+article contraction
|
||||
(adjective-position "postnominal-default + small prenominal class") ; bo/bon,
|
||||
; mal, gran, nou, vell, primer, molt... prenominal
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((de el del) (de els dels)
|
||||
(a el al) (a els als)
|
||||
(per el pel) (per els pels)))
|
||||
(contraction-mandatory yes) ; *de el -> del obligatory
|
||||
(contraction-blocked-before-elision yes) ; de l'home / a l'home (NO *del home)
|
||||
|
||||
;; ── clitic system: PRONOMS FEBLES (the headline delta) ──────────────────
|
||||
(clitics yes)
|
||||
(clitic-allomorphy four-position) ; per pronoun, form varies by position+onset:
|
||||
; reinforced (em, et, el) proclitic before a consonant
|
||||
; elided (m', t', l', n') proclitic before a vowel/h
|
||||
; full (-me, -lo, -li) enclitic after a consonant/-r
|
||||
; reduced ('m, 't, 'l, 'ns) enclitic after a vowel
|
||||
(clitic-placement ((finite proclitic) ; el veig, no m'ho dóna
|
||||
(imperative-affirmative enclitic) ; dóna'm, digues-me
|
||||
(imperative-negative present-subjunctive) ; no parlis (delta)
|
||||
(infinitive enclitic) ; ajudar-me, veure'l
|
||||
(gerund enclitic))) ; fent-ho
|
||||
(clitic-combination ((me el "me'l") (te el "te'l") (se el "se'l")
|
||||
(me la "me la") (me en "me'n")
|
||||
(li el "l'hi") (li en "n'hi"))) ; dative+accusative clusters
|
||||
(clitic-particles (hi en ho)) ; locative hi, partitive/genitive en, neuter ho
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (present imperfet preterit-simple perifrastic-preterit futur
|
||||
condicional subjuntiu-present subjuntiu-imperfet imperatiu))
|
||||
(periphrastic-preterite "vaig/vas/va/vam/vau/van + INFINITIVE") ; << hallmark CA
|
||||
; (vaig cantar = 'I sang'); coexists w/ synthetic pret.
|
||||
(compound-past "pretèrit perfet = haver(present) + participle")
|
||||
(perfect-aux "HAVER only") ; << NO essere/avere split (simpler than IT)
|
||||
(participle-agreement ((haver preceding-acc-clitic))) ; les he vistes; else invariable
|
||||
(progressive-aux "estar + gerundi")
|
||||
(copula "ser / estar") ; ser: identity/essential/origin; estar:
|
||||
; location + transient state (estic cansat, és a casa)
|
||||
(passive-aux "ser (+ per-agent)")
|
||||
(future inflectional) ; cantaré, serà
|
||||
(comparative "més/menys ADJ que")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "no (preverbal) + optional 'pas' + concord") ; no...res/
|
||||
; ningú/mai/cap/gens/enlloc
|
||||
(negative-concord yes) ; preverbal negative subject (ningú) keeps 'no'
|
||||
(neg-reinforcer pas)) ; optional (no ho faré pas)
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_de.el — German language profile for ELP.
|
||||
;;; Mirrors lang_profile_en / lang_profile_es. Keys the realizer's construction
|
||||
;;; switches. German is the largest Germanic delta from the EN engine: V2 word
|
||||
;;; order, four morphological cases, and separable-prefix verbs.
|
||||
|
||||
(lang_profile_de
|
||||
(language "German")
|
||||
(iso639 "de")
|
||||
(family "Germanic")
|
||||
(neighbor-base "en") ; realized by extending the English (Germanic) engine
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; obligatory subject in finite clauses
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender (m f n)) ; three genders; drives article + adj declension
|
||||
(case-system (nom acc dat gen)) ; four cases on articles/adjs/nouns
|
||||
(word-order V2) ; finite verb 2nd in main clause
|
||||
(subordinate-order verb-final) ; "..., dass er den Hund SIEHT."
|
||||
(separable-verbs yes) ; aufstehen -> "steht ... auf"; ppart "aufgestanden"
|
||||
(do-support no) ; German negates/questions the finite verb directly
|
||||
(subject-verb-inversion yes) ; yes/no Q fronts finite verb; wh-Q fills Vorfeld
|
||||
(article-selection "der/die/das + ein/kein") ; declined by case x gender x number
|
||||
(adjective-position prenominal)
|
||||
(adjective-declension (strong weak mixed)) ; chosen by the determiner type
|
||||
(noun-capitalization yes)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person-and-number") ; full present/past paradigm
|
||||
(auxiliary-order (modal tense-aux perfect passive main))
|
||||
(perfect-aux (haben sein)) ; sein for intransitive motion/change verbs
|
||||
(passive-aux "werden")
|
||||
(future "werden + infinitive")
|
||||
(comparative "synthetic (-er / -st, with umlaut)")
|
||||
|
||||
;; ── negation ───────────────────────────────────────────────────────────
|
||||
(negation-markers (nicht kein)) ; kein- negates an indefinite NP; nicht else
|
||||
(negation-faithful yes) ; SACRED: polarity never dropped/inverted -> FLAG
|
||||
|
||||
;; ── lexicon provenance ─────────────────────────────────────────────────
|
||||
(lexicon-source "UniMorph deu (primary) + kaikki.org German (gender override)")
|
||||
(lexicon-license "CC-BY-SA 3.0 / GFDL"))
|
||||
@@ -1,41 +0,0 @@
|
||||
;;; lang_profile_en.el — English language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. English is typologically distinct from the Romance builds, so the
|
||||
;;; flags differ where the grammar differs.
|
||||
|
||||
(lang_profile_en
|
||||
(language "English")
|
||||
(iso639 "en")
|
||||
(family "Germanic")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; OBLIGATORY subjects — missing subject is FLAGGED
|
||||
(obligatory-subject yes)
|
||||
(grammatical-gender no) ; natural gender only (he/she/it), no NP agreement
|
||||
(do-support yes) ; negation & questions of lexical verbs insert do/does/did
|
||||
(subject-aux-inversion yes) ; yes/no + non-subject wh questions invert the operator
|
||||
(article-selection "a/an/the") ; a/an resolved PHONOLOGICALLY (an hour, a university)
|
||||
(adjective-position prenominal) ; attributive adjectives precede the noun; invariant
|
||||
(has-tag-questions yes) ; "...doesn't he?" — operator + reversed polarity
|
||||
(has-there-existential yes) ; "there is/are/have been ..."
|
||||
(possessive-clitic "'s") ; saxon genitive; plural in -s -> bare apostrophe
|
||||
(question-punct plain) ; ? and ! only (no inverted marks)
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "3sg-present-only") ; only 3sg present -s (+ suppletive be)
|
||||
(auxiliary-order (modal perfect progressive passive main))
|
||||
(perfect-aux "have") ; have + past participle
|
||||
(progressive-aux "be") ; be + present participle
|
||||
(passive-aux "be") ; be + past participle (+ by-agent)
|
||||
(future "will + base") ; no inflectional future
|
||||
(comparative "synthetic-or-periphrastic") ; -er/-est vs more/most by syllables
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt) ──────────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
|
||||
;; ── DIALECT overlay (post-realization, one core -> US/UK/AU) ────────────
|
||||
(dialect US) ; default; profile field switches the overlay
|
||||
(dialects (US UK AU))
|
||||
(dialect-canonical US) ; core is authored in US orthography
|
||||
(dialect-overlay "dialect_en.to_dialect") ; orthography + lexis + grammar prefs
|
||||
(dialect-covers (spelling lexis collective-agreement gotten/got)))
|
||||
@@ -1,45 +0,0 @@
|
||||
;;; lang_profile_es.el — Spanish language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_en / _pt.
|
||||
|
||||
(lang_profile_es
|
||||
(language "Spanish")
|
||||
(iso639 "es")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon); REAL per-noun gender from UniMorph — NOT a heuristic
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; questions by intonation/punctuation, not inversion
|
||||
(question-strategy intonation)
|
||||
(article-selection "el/la/los/las un/una/unos/unas")
|
||||
(stressed-a-rule yes) ; fem sg noun in stressed a-/ha- takes el/un (el agua)
|
||||
(adjective-position postnominal) ; default post; a few prenominal + apocope
|
||||
(adjective-agreement "gender+number")
|
||||
(question-punct inverted) ; opening ¿ ¡ required
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (coordinator quality bar) --------------------
|
||||
(contractions ((de el "del") (a el "al")))
|
||||
(contraction-mandatory yes) ; 'de el'/'a el' MUST surface as del/al
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "haber") ; haber + past participle (invariant -o)
|
||||
(progressive-aux "estar") ; estar + gerund
|
||||
(passive-aux "ser") ; ser + participle (agrees) + por-agent
|
||||
(copula-split "ser/estar") ; permanent vs stage-level
|
||||
(future "infinitive + é/ás/á/emos/éis/án")
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; me te lo la le nos os los las; proclisis/enclisis
|
||||
(clitic-order "se II I III (le+lo -> se lo)")
|
||||
(enclisis "imperative/infinitive/gerund + accent repair (dá+me+lo->dámelo)")
|
||||
(verb-prep-government yes) ; verbs select prep (protestar+contra, escapar+de)
|
||||
|
||||
;; -- SACRED safety bar (shared with en/pt) -------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
@@ -1,74 +0,0 @@
|
||||
;;; lang_profile_fr.el — French language profile for ELP.
|
||||
;;; Mirrors lang_profile_it / lang_profile_es; keys the realizer's construction
|
||||
;;; switches. French is a Romance sibling (~54% of the realizer code and the whole
|
||||
;;; clause-engine architecture reused), but carries the family's biggest surface
|
||||
;;; deltas: NOT pro-drop, DISCONTINUOUS negation, and an orthography/phonology
|
||||
;;; mismatch (elision, liaison) that makes exact-match genuinely hard.
|
||||
|
||||
(lang_profile_fr
|
||||
(language "French")
|
||||
(iso639 "fr")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop no) ; << French-specific: subject clitic OBLIGATORY
|
||||
(obligatory-subject yes) ; je/tu/il/elle/nous/vous/ils/elles always overt
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion optional) ; est-ce que (default) OR clitic inversion (vas-tu)
|
||||
(article-selection "le/la/l'/les ; un/une/des ; PARTITIVE du/de la/de l'/des")
|
||||
(article-drives-contraction yes) ; à+le=au, de+le=du feed off article choice
|
||||
(adjective-position "postnominal-default + prenominal-BAGS") ; beau/bon/grand/
|
||||
; petit/jeune/vieux/nouveau + ordinals prenominal
|
||||
; (beau->bel, nouveau->nouvel, vieux->vieil / vowel)
|
||||
(question-punct "space-before") ; French typography: ' ?' ' !' (no ¿¡)
|
||||
|
||||
;; ── elision (orthography/phonology mismatch — French-specific) ──────────
|
||||
(elision ((le l') (la l') (je j') (ne n') (de d') (que qu')
|
||||
(me m') (te t') (se s') (ce c'))) ; before vowel / h-muet
|
||||
(elision-h-muet yes) ; l'homme, l'hôpital (h-aspiré exception list kept)
|
||||
(liaison noted-not-modeled) ; phonological, not written in surface
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((à le au) (à les aux) (de le du) (de les des)))
|
||||
(contraction-mandatory yes) ; *à le -> au obligatory; à la / à l' uncontracted
|
||||
(partitive ((m-sg du) (f-sg "de la") (vowel "de l'") (pl des)))
|
||||
(partitive-under-neg "de") ; << gap in current build: 'ne … pas de pain'
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-order (me te se nous vous | le la les | lui leur | y | en))
|
||||
(clitic-placement ((finite proclitic) ; je le lui donne
|
||||
(imperative-affirmative enclitic-hyphen) ; donne-le-moi
|
||||
(imperative-negative "ne+proclitic+verb+pas") ; ne le donne pas
|
||||
(infinitive enclitic))) ; PARTIAL: clitic-climbing
|
||||
; onto infinitive under modal
|
||||
(clitic-imperative-shift ((me moi) (te toi))) ; final me/te -> moi/toi (donne-moi)
|
||||
(clitic-particles (y en)) ; locative y, partitive/genitive en
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (written; many homophones)")
|
||||
(tenses (présent imparfait passé-simple futur conditionnel
|
||||
subjonctif-présent subjonctif-imparfait impératif))
|
||||
(compound-past "passé-composé = aux(present) + participe passé")
|
||||
(perfect-aux "être/avoir (LEXICAL selection)") ; << French-specific
|
||||
(etre-aux-class "intransitive motion/change (aller venir arriver partir
|
||||
entrer sortir monter descendre naître mourir rester
|
||||
tomber retourner passer devenir revenir rentrer) + ALL
|
||||
pronominal verbs")
|
||||
(participle-agreement ((être subject) ; elle est allée / elles venues
|
||||
(avoir preceding-direct-object))) ; je les ai vus
|
||||
(progressive "être en train de + infinitif") ; no dedicated aux
|
||||
(copula "être (single; no ser/estar, no essere/stare)")
|
||||
(passive-aux "être (+ par-agent)")
|
||||
(future inflectional) ; parlera, sera
|
||||
(comparative "plus/moins ADJ que")
|
||||
(superlative "le/la plus ADJ (de …)") ; PARTIAL word-order in build
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "DISCONTINUOUS: ne (preverbal) … pas/jamais/rien/personne/
|
||||
plus/guère/que (postverbal)") ; << biggest structural delta
|
||||
(negation-ne-elides yes) ; ne -> n' before vowel (n'ai pas vu)
|
||||
(negation-passe-composé "ne + aux + pas + participe") ; n'ai pas vu
|
||||
(negative-concord partial)) ; personne/rien as arguments post-participle
|
||||
@@ -1,70 +0,0 @@
|
||||
;;; lang_profile_it.el — Italian language profile for ELP.
|
||||
;;; Mirrors lang_profile_es / lang_profile_pt; keys the realizer's construction
|
||||
;;; switches. Italian is a Romance sibling, so ~85% of the flags match ES/PT; the
|
||||
;;; essere/avere auxiliary split and phonological article selection are the deltas.
|
||||
|
||||
(lang_profile_it
|
||||
(language "Italian")
|
||||
(iso639 "it")
|
||||
(family "Romance")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f; full NP agreement (art + adj + participle)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'; no inversion
|
||||
(article-selection "il/lo/l'/i/gli + la/l'/le ; un/uno/un'/una") ; PHONOLOGICAL:
|
||||
; lo/gli/uno before s+cons, z, gn, ps, pn, x, y, i+V;
|
||||
; l'/un' before a vowel (elision, glued to next word)
|
||||
(article-drives-contraction yes) ; article choice feeds the prep+art contraction
|
||||
(adjective-position "postnominal-default + prenominal-class") ; bello/buono/grande
|
||||
; /nuovo/vecchio/primo... prenominal (with apocope)
|
||||
(question-punct plain) ; ? and ! only (no inverted ¿ ¡)
|
||||
|
||||
;; ── MANDATORY prep+article contractions ────────────────────────────────
|
||||
(contractions ((di il del) (di lo dello) (di la della) (di i dei)
|
||||
(di gli degli) (di le delle) (di l' dell')
|
||||
(a il al) (a lo allo) (a la alla) (a i ai) (a gli agli)
|
||||
(a le alle) (a l' all')
|
||||
(da il dal) (da la dalla) (da gli dagli) (da l' dall')
|
||||
(in il nel) (in la nella) (in gli negli) (in l' nell')
|
||||
(su il sul) (su la sulla) (su gli sugli) (su l' sull')))
|
||||
(contraction-mandatory yes) ; *di il -> del is obligatory, never uncontracted
|
||||
(prep-no-contract (per tra fra)) ; per la strada (NOT *perla)
|
||||
|
||||
;; ── clitic system ──────────────────────────────────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-placement ((finite proclitic) ; lo vedo, non me lo dà
|
||||
(imperative-affirmative enclitic) ; dammelo, guardalo
|
||||
(imperative-negative-tu non+infinitive) ; non parlare / non lo fare
|
||||
(infinitive enclitic) ; vederlo, aiutarmi (drop -e)
|
||||
(gerund enclitic))) ; dandolo
|
||||
(clitic-combination ((mi lo "me lo") (ti lo "te lo") (ci lo "ce lo")
|
||||
(vi lo "ve lo") (si lo "se lo")
|
||||
(gli lo "glielo") (le lo "glielo"))) ; glielo = ONE word
|
||||
(clitic-particles (ci ne)) ; locative ci, partitive ne
|
||||
(raddoppiamento (da fa di va sta)) ; monosyllabic imper double clitic: dammelo
|
||||
|
||||
;; ── verb / aspect system ───────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (presente imperfetto passato-remoto futuro condizionale
|
||||
congiuntivo-presente congiuntivo-imperfetto imperativo))
|
||||
(compound-past "passato-prossimo = aux(present) + participle")
|
||||
(perfect-aux "essere/avere (LEXICAL selection)") ; << Italian-specific
|
||||
(essere-aux-class unaccusative) ; motion/change-of-state/copular/pronominal
|
||||
; (andare venire nascere morire diventare piacere
|
||||
; + ALL reflexives) -> essere
|
||||
(participle-agreement ((essere subject) ; è andata / sono arrivati
|
||||
(avere preceding-acc-clitic))) ; li ho visti
|
||||
(progressive-aux "stare + gerundio") ; sto parlando
|
||||
(copula "essere (default) / stare (state: sto bene)")
|
||||
(passive-aux "essere / venire (+ da-agent)")
|
||||
(future inflectional) ; parlerò, sarà
|
||||
(comparative "più/meno ADJ di")
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/en) ───────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "non (preverbal) + concord") ; non...niente/nessuno/mai/più
|
||||
(negative-concord yes) ; preverbal negative word (nessuno/niente) suppresses non
|
||||
(neg-adverb-position between-aux-and-participle)) ; non ho MAI visto
|
||||
@@ -1,30 +0,0 @@
|
||||
;;; lang_profile_la.el — Latin language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Companion to morphology-la.el.
|
||||
|
||||
(lang_profile_la
|
||||
(language "Latin")
|
||||
(iso639 "la")
|
||||
(family "Italic")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; person carried by verb ending; subjects dropped
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f/n; adjective AGREES in case+gender+number
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph lat
|
||||
(articles none) ; Latin has no articles
|
||||
(case-system yes) ; NOM GEN DAT ACC ABL VOC (+ rare LOC)
|
||||
(cases (nom gen dat acc abl voc))
|
||||
(word-order "SOV (default; free order, case-marked)")
|
||||
(adjective-position "either (case agreement carries the link)")
|
||||
(adjective-agreement "case+gender+number")
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (1 2 3 3io 4)) ; four conjugations + i-stem 3rd
|
||||
(tenses (present imperfect future perfect pluperfect futureperfect))
|
||||
(moods (indicative subjunctive imperative infinitive))
|
||||
(voices (active passive))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(citation "principal parts: pres-1sg / pres-inf / perf-participle")
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes)) ; polarity never dropped/inverted
|
||||
@@ -1,40 +0,0 @@
|
||||
;;; lang_profile_pt.el — Portuguese language profile for ELP.
|
||||
;;; Keys the realizer's construction switches. Mirrors lang_profile_es.
|
||||
|
||||
(lang_profile_pt
|
||||
(language "Portuguese")
|
||||
(iso639 "pt")
|
||||
(family "Romance")
|
||||
|
||||
;; -- core typology flags -------------------------------------------------
|
||||
(pro-drop yes) ; subjects routinely dropped; agreement carries person
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m/f on every noun; article+adjective AGREE
|
||||
(gender-source lexicon) ; REAL per-noun gender from UniMorph por / kaikki
|
||||
(do-support no)
|
||||
(subject-aux-inversion no)
|
||||
(question-strategy intonation)
|
||||
(article-selection "o/a/os/as um/uma/uns/umas")
|
||||
(adjective-position postnominal)
|
||||
(adjective-agreement "gender+number")
|
||||
|
||||
;; -- MANDATORY CONTRACTIONS (prep + article) -----------------------------
|
||||
(contractions ((de o "do") (de a "da") (em o "no") (em a "na")
|
||||
(a o "ao") (a a "à") (por o "pelo") (por a "pela")))
|
||||
(contraction-mandatory yes)
|
||||
|
||||
;; -- verb / aspect system ------------------------------------------------
|
||||
(verb-classes (ar er ir))
|
||||
(tenses (present preterite imperfect future conditional))
|
||||
(moods (ind sbjv imp))
|
||||
(finite-agreement "person+number (6 slots)")
|
||||
(perfect-aux "ter") ; ter + past participle
|
||||
(copula-split "ser/estar")
|
||||
(personal-infinitive yes) ; distinctive PT inflected infinitive
|
||||
|
||||
;; -- clitics / government ------------------------------------------------
|
||||
(object-clitics yes) ; mesoclisis/enclisis/proclisis by context
|
||||
(verb-prep-government yes)
|
||||
|
||||
;; -- SACRED safety bar ---------------------------------------------------
|
||||
(negation-faithful yes))
|
||||
@@ -1,71 +0,0 @@
|
||||
;;; lang_profile_ro.el — Romanian language profile for ELP.
|
||||
;;; Romanian is the BIG typological delta of the Romance family. The verb/clause
|
||||
;;; engine and the SACRED negation contract mirror the ES/PT/IT core, but the
|
||||
;;; NOMINAL system is genuinely new: a SUFFIXED definite article, preserved CASE,
|
||||
;;; a NEUTER gender, and a VOCATIVE. Those flags mark where the shared engine was
|
||||
;;; extended rather than reused.
|
||||
|
||||
(lang_profile_ro
|
||||
(language "Romanian")
|
||||
(iso639 "ro")
|
||||
(family "Romance (Eastern / Balkan)")
|
||||
|
||||
;; ── core typology flags ────────────────────────────────────────────────
|
||||
(pro-drop yes) ; null subjects default; overt pronoun = emphatic
|
||||
(obligatory-subject no)
|
||||
(grammatical-gender yes) ; m / f / NEUTER (n)
|
||||
(neuter-gender yes) ; << ROMANIAN-SPECIFIC: masc-agreeing SG, fem-agreeing PL
|
||||
; (un tren nou / două trenuri noi)
|
||||
(do-support no)
|
||||
(subject-aux-inversion no) ; yes/no Q = declarative order + '?'
|
||||
(question-punct plain) ; ? and ! only
|
||||
|
||||
;; ── SUFFIXED DEFINITE ARTICLE (the headline engine extension) ───────────
|
||||
(definite-article suffixed) ; << UNIQUE IN ROMANCE: enclitic on the noun
|
||||
(definite-forms ((m/n sg "-ul / -le / -l : om->omul, câine->câinele, codru->codrul")
|
||||
(f sg "-a / -ea / -ua : casă->casa, carte->cartea, stea->steaua")
|
||||
(m pl "-i : oameni->oamenii")
|
||||
(f/n pl "-le : case->casele, trenuri->trenurile")))
|
||||
(article-host ((no-prenom-adj noun) ; omul bun
|
||||
(prenom-adj adjective))) ; bunul om (adj carries the article)
|
||||
(indefinite-article ((m/n "un") (f "o") (pl "niște") (gen/dat-pl "unor")))
|
||||
|
||||
;; ── CASE (preserved; NOM/ACC vs GEN/DAT) ────────────────────────────────
|
||||
(case (nom/acc gen/dat vocative)) ; << ROMANIAN-SPECIFIC
|
||||
(case-syncretism "nom=acc ; gen=dat")
|
||||
(genitive-marking "gen/dat definite: -lui (m/n), -ei/-i (f), -lor (pl)")
|
||||
(genitival-article ((m sg "al") (f sg "a") (m pl "ai") (f/n pl "ale"))) ; o carte a lui
|
||||
(possession "definite-head + gen/dat possessor: casa băiatului")
|
||||
(vocative ((m sg "-ule/-e : omule, băiete") (f sg "-o : Mario, fato")
|
||||
(pl "-lor")))
|
||||
|
||||
;; ── verb / aspect system ────────────────────────────────────────────────
|
||||
(finite-agreement "person+number (6-way)")
|
||||
(tenses (prezent imperfect perfect-simplu conjunctiv-prezent
|
||||
imperativ (periphrastic: perfect-compus viitor conditional)))
|
||||
(compound-past "perfectul compus = a-avea-clitic + INVARIABLE participle")
|
||||
(perfect-aux "a avea (am/ai/a/am/ați/au) — ONE auxiliary for ALL verbs")
|
||||
(perfect-aux-split no) ; << SIMPLER than Italian: no essere/avere selection
|
||||
(participle-agreement none) ; invariable in the perfect compus (agrees only as
|
||||
; an adjective / in the passive)
|
||||
(future "voi/vei/va/vom/veți/vor + infinitive (viitor literar)")
|
||||
(conditional "aș/ai/ar/am/ați/ar + infinitive")
|
||||
(subjunctive "conjunctiv: particle 'să' + subjunctive present")
|
||||
(modal-complement "modal + să + subjunctive (vreau să merg, poți să ajuți)")
|
||||
(copula "a fi")
|
||||
(passive "a fi + participle (participle AGREES like an adjective)")
|
||||
(comparative "mai / mai puțin ADJ decât")
|
||||
|
||||
;; ── clitic system (partial — see honest gaps) ───────────────────────────
|
||||
(clitics yes)
|
||||
(clitic-set ((acc mă te îl o ne vă îi le) (dat îmi îți îi ne vă le)
|
||||
(refl mă te se ne vă se)))
|
||||
(clitic-placement ((finite proclitic) ; îmi place, o văd
|
||||
(perfect-compus elision) ; << m-am, l-am, i-am (PARTIAL)
|
||||
(imperative-affirmative enclitic))) ; dă-mi (PARTIAL)
|
||||
|
||||
;; ── SACRED safety bar (shared with es/pt/it/en) ─────────────────────────
|
||||
(negation-faithful yes) ; polarity never dropped/inverted; unplaceable -> FLAG
|
||||
(negation "nu (single preverbal marker) + concord")
|
||||
(negative-concord yes) ; nu … nimic / nimeni / niciodată / niciun
|
||||
(negative-imperative "nu + INFINITIVE : nu pleca! (KNOWN GAP: uses imperative stem)"))
|
||||
File diff suppressed because it is too large
Load Diff
-144861
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-130676
File diff suppressed because it is too large
Load Diff
-193894
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
-115916
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,100 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Full-lexicon vocabulary-{de,la}.el emitters (custom field mapping for the
|
||||
German declension/gender API and the Latin case-paradigm API). Reuses the
|
||||
chunked seed-fn writer from gen_elp_seed_full.
|
||||
"""
|
||||
import sys, importlib
|
||||
from gen_elp_seed_full import write_seed
|
||||
|
||||
def uw(x):
|
||||
"""Unwrap (form, source) tuples that some morphology fns return."""
|
||||
if isinstance(x, (tuple, list)):
|
||||
return x[0] if x else ""
|
||||
return x if x is not None else ""
|
||||
|
||||
def build_de():
|
||||
M = importlib.import_module("morphology_de_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
# nouns: form0=nom-sg(lemma) form1=plural form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
g = uw(M.noun_gender(lem))
|
||||
pl = uw(M.pluralize(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "noun", lem, pl, g or "", "", "gender:lexicon"])
|
||||
st["nouns"] += 1
|
||||
# adjs: form0=positive form1=comparative form2=superlative
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
cmpr = uw(M.comparative(lem))
|
||||
sprl = uw(M.superlative(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", lem, cmpr, sprl, "", "degree:lexicon"])
|
||||
st["adjs"] += 1
|
||||
# verbs (only the ~30 irregular/strong stems the cache carries):
|
||||
# form0=pres-3sg form1=past-3sg form2=past-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.finite(lem, "present", "third", "singular"))
|
||||
f1 = uw(M.finite(lem, "past", "third", "singular"))
|
||||
pp = uw(M.past_participle(lem))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, f1, pp, "", "class:strong/irregular"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
def build_la():
|
||||
M = importlib.import_module("morphology_lat_full")
|
||||
rows = []; st = {"verbs":0,"nouns":0,"adjs":0}
|
||||
def dn(lem, c, n):
|
||||
try:
|
||||
r = M.decline_noun(lem, c, n)
|
||||
return uw(r)
|
||||
except Exception:
|
||||
return ""
|
||||
# nouns: dictionary citation — form0=nom-sg form1=gen-sg form2=gender
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
nom = dn(lem, "NOM", "SG") or lem
|
||||
gen = dn(lem, "GEN", "SG")
|
||||
try: g = uw(M.noun_gender(lem))
|
||||
except Exception: g = ""
|
||||
rows.append([lem, "noun", nom, gen, g, "", "case-paradigm nom/gen-sg"])
|
||||
st["nouns"] += 1
|
||||
# adjs: three-gender nom-sg citation — form0=masc form1=fem form2=neut
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m = uw(M.decline_adj(lem, "NOM", "MASC", "SG")) or lem
|
||||
f = uw(M.decline_adj(lem, "NOM", "FEM", "SG"))
|
||||
nt = uw(M.decline_adj(lem, "NOM", "NEUT", "SG"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m, f, nt, "", "3-gender nom-sg"])
|
||||
st["adjs"] += 1
|
||||
# verbs: principal parts — form0=pres-ind-1sg form1=pres-infinitive form2=perf-participle
|
||||
if hasattr(M, "_VERBS"):
|
||||
for lem in sorted({k[0] if isinstance(k, tuple) else k for k in M._VERBS}):
|
||||
if not lem: continue
|
||||
try:
|
||||
f0 = uw(M.conjugate(lem, "present", "indicative", "active", "first", "singular"))
|
||||
inf = uw(M.infinitive(lem, "present", "active"))
|
||||
pp = uw(M.participle(lem, "perfect", "nom", "m", "singular"))
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "verb", f0, inf, pp, "", "principal-parts pres1sg/inf/pfppl"])
|
||||
st["verbs"] += 1
|
||||
return rows, st
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang = sys.argv[1]; out = sys.argv[2]
|
||||
rows, st = build_de() if lang == "de" else build_la()
|
||||
total, _ = write_seed(lang, rows, st, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={st['verbs']} nouns={st['nouns']} adjs={st['adjs']}")
|
||||
@@ -1,129 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""gen_elp_seed_full.py — emit a FULL-lexicon vocabulary-{lang}.el in the
|
||||
established ELP seed-fn format (same as vocabulary-non.el / the 18 classical
|
||||
languages), iterating the ENTIRE morphology_{lang}_full lexicon (every verb,
|
||||
noun, adjective lemma) — NOT a curated demo core.
|
||||
|
||||
Schema per row: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]
|
||||
Verbs: form0=pres-ind-3sg form1=preterite-3sg form2=past-participle
|
||||
Nouns: form0=singular form1=plural form2=REAL gender (lexicon)
|
||||
Adjs : form0=masc-sg form1=fem-sg form2=masc-pl
|
||||
|
||||
Output structure (chunked to stay within the proven ~5k-append/function scale):
|
||||
fn vocab_{lang}_seed_pN(v) -> [[String]] { ... appends ... return v }
|
||||
fn vocab_{lang}_seed() -> [[String]] { chains all chunks; return v }
|
||||
fn vocab_{lang}_lookup(w) -> [String] { linear scan }
|
||||
|
||||
Usage: python3 gen_elp_seed_full.py <lang> <out.el>
|
||||
"""
|
||||
import sys, importlib
|
||||
|
||||
CHUNK = 5000
|
||||
|
||||
def esc(s):
|
||||
return str(s).replace("\\", "\\\\").replace('"', '\\"')
|
||||
|
||||
def row(fields):
|
||||
return " let v = native_list_append(v, [" + ", ".join(f'"{esc(f)}"' for f in fields) + "])"
|
||||
|
||||
def build_rows(lang, M):
|
||||
rows = []
|
||||
stats = {"verbs":0,"nouns":0,"adjs":0}
|
||||
has = lambda n: hasattr(M, n)
|
||||
|
||||
# --- verbs ---
|
||||
if has("_VERBS") and has("conjugate"):
|
||||
verbs = sorted({k[0] for k in M._VERBS})
|
||||
for lem in verbs:
|
||||
if not lem: continue
|
||||
try:
|
||||
f0, s0 = M.conjugate(lem, "ind", "present", "third", "singular")
|
||||
f1, _ = M.conjugate(lem, "ind", "preterite", "third", "singular")
|
||||
pp, _ = (M.participle(lem) if has("participle") else ("",""))
|
||||
except Exception:
|
||||
continue
|
||||
vclass = lem[-2:] if lem[-2:] in ("ar","er","ir","re") else lem[-2:]
|
||||
rows.append([lem, "verb", f0 or "", f1 or "", pp or "", "", "class:"+vclass+" src:"+str(s0)])
|
||||
stats["verbs"] += 1
|
||||
|
||||
# --- nouns ---
|
||||
if has("_NOUNS") and has("inflect_noun"):
|
||||
for lem in sorted(M._NOUNS):
|
||||
if not lem: continue
|
||||
try:
|
||||
sg, _ = M.inflect_noun(lem, "singular")
|
||||
pl, _ = M.inflect_noun(lem, "plural")
|
||||
g = M.noun_gender(lem) if has("noun_gender") else ""
|
||||
except Exception:
|
||||
continue
|
||||
src = "lexicon" if (isinstance(M._NOUNS.get(lem), dict) and M._NOUNS[lem].get("g")) else "heuristic"
|
||||
rows.append([lem, "noun", sg or lem, pl or "", g or "", "", "gender:"+src])
|
||||
stats["nouns"] += 1
|
||||
|
||||
# --- adjectives ---
|
||||
if has("_ADJS") and has("inflect_adj"):
|
||||
for lem in sorted(M._ADJS):
|
||||
if not lem: continue
|
||||
try:
|
||||
m_sg, _ = M.inflect_adj(lem, "m", "singular")
|
||||
f_sg, _ = M.inflect_adj(lem, "f", "singular")
|
||||
m_pl, _ = M.inflect_adj(lem, "m", "plural")
|
||||
except Exception:
|
||||
continue
|
||||
rows.append([lem, "adj", m_sg or lem, f_sg or "", m_pl or "", "", "src:lexicon"])
|
||||
stats["adjs"] += 1
|
||||
|
||||
return rows, stats
|
||||
|
||||
def write_seed(lang, rows, stats, out_path):
|
||||
"""Write vocabulary-{lang}.el in the chunked seed-fn format from prebuilt rows.
|
||||
Each row is a 7-field list [lemma,pos,f0,f1,f2,gloss,hint]."""
|
||||
total = len(rows)
|
||||
chunks = [rows[i:i+CHUNK] for i in range(0, total, CHUNK)] or [[]]
|
||||
L = []
|
||||
L.append(f"// vocabulary-{lang}.el — FULL {lang} lexicon for ELP surface realization.")
|
||||
L.append(f"// Generated by gen_elp_seed_full.py from morphology_{lang}_full")
|
||||
L.append(f"// (real UniMorph + kaikki.org Wiktionary forms; gender from lexicon, not heuristic).")
|
||||
L.append(f"// Entries: {total} (verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']})")
|
||||
L.append(f"// Schema: [lemma, pos, form0, form1, form2, en_translation, semantic_hint]")
|
||||
L.append(f"// verbs: form0=pres-3sg form1=pret-3sg form2=past-participle")
|
||||
L.append(f"// nouns: form0=sg form1=pl form2=REAL gender adjs: form0=m-sg form1=f-sg form2=m-pl")
|
||||
L.append("")
|
||||
for ci, ch in enumerate(chunks):
|
||||
L.append(f"fn vocab_{lang}_seed_p{ci}(v: [[String]]) -> [[String]] {{")
|
||||
for r in ch:
|
||||
L.append(row(r))
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_seed() -> [[String]] {{")
|
||||
L.append(" let v: [[String]] = native_list_empty()")
|
||||
for ci in range(len(chunks)):
|
||||
L.append(f" let v = vocab_{lang}_seed_p{ci}(v)")
|
||||
L.append(" return v")
|
||||
L.append("}")
|
||||
L.append("")
|
||||
L.append(f"fn vocab_{lang}_lookup(word: String) -> [String] {{")
|
||||
L.append(f" let vocab: [[String]] = vocab_{lang}_seed()")
|
||||
L.append(" let n: Int = native_list_len(vocab)")
|
||||
L.append(" let i: Int = 0")
|
||||
L.append(" while i < n {")
|
||||
L.append(" let entry: [String] = native_list_get(vocab, i)")
|
||||
L.append(' if str_eq(native_list_get(entry, 0), word) { return entry }')
|
||||
L.append(" let i = i + 1")
|
||||
L.append(" }")
|
||||
L.append(" return native_list_empty()")
|
||||
L.append("}")
|
||||
with open(out_path, "w", encoding="utf-8") as fh:
|
||||
fh.write("\n".join(L) + "\n")
|
||||
return total, stats
|
||||
|
||||
def emit(lang, out_path):
|
||||
M = importlib.import_module(f"morphology_{lang}_full")
|
||||
rows, stats = build_rows(lang, M)
|
||||
return write_seed(lang, rows, stats, out_path)
|
||||
|
||||
if __name__ == "__main__":
|
||||
lang, out = sys.argv[1], sys.argv[2]
|
||||
total, stats = emit(lang, out)
|
||||
print(f"{lang}: wrote {out} total={total} verbs={stats['verbs']} nouns={stats['nouns']} adjs={stats['adjs']}")
|
||||
@@ -1,572 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_ca_full.py — production-grade Catalan morphological generator.
|
||||
|
||||
Same design as morphology_it_full.py (its Romance sibling); Catalan-specific data.
|
||||
|
||||
VERBS
|
||||
UniMorph Catalan (github.com/unimorph/cat, CC-BY-SA 3.0)
|
||||
7,535 verb lemmas × paradigm, CLEAN orthography:
|
||||
present, imperfet (PST;IPFV), pretèrit simple (PST;PFV), futur,
|
||||
condicional (COND), subjuntiu present (SBJV;PRS) / imperfet (SBJV;PST),
|
||||
imperatiu (POS;IMP), infinitiu (NFIN), gerundi (V.CVB;PRS),
|
||||
participi (V.PTCP;PST) — WITH full gender+number agreement forms
|
||||
(cantat/cantada/cantats/cantades) stored directly.
|
||||
ca_irreg_verbs.json — verbs UniMorph MISSES or under-populates
|
||||
(anar, fer, plus core auxiliaries ser/haver/estar/tenir…), extracted from
|
||||
kaikki.org Catalan by build_ca_irreg.py. Priority layer. Supplies anar,
|
||||
whose present (vaig/vas/va/anem/aneu/van) is ALSO the PERIPHRASTIC-PRETERITE
|
||||
auxiliary (vaig cantar = 'I sang') — a hallmark Catalan construction.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Catalan (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender + real plural (resolved PER LEMMA).
|
||||
adjective lemmas with real feminine + plural forms.
|
||||
|
||||
Fallbacks degrade, never crash:
|
||||
verbs : regular -ar/-er/-re/-ir rule generator (+ -car/-gar/-çar spelling).
|
||||
nouns : gender heuristic + rule pluralization (-a→-es with ç/c/g/j/qu/gu
|
||||
spelling changes; sibilant-final → -os; else -s). Ambiguous → FLAG.
|
||||
adjs : -o? no (Catalan masc often consonant/-e); fem -a rule + plural rule.
|
||||
|
||||
Confidence flag per form: "lexicon" | "rule" | "fallback" (low → FLAG).
|
||||
|
||||
Public API (used by realizer_ca.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
peri_pret_aux(person, number) -> form # anar-present, for vaig+INF
|
||||
participle(lemma, gender, number) -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "cat.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "ca_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_ca.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "ca_morph_cache.pkl")
|
||||
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "preterite"): {"IND", "PST", "PFV"},
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"},
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── verbs from UniMorph ──────────────────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {}
|
||||
part = {} # lemma -> {("m","SG"):form, ("f","SG"):..., ("m","PL"):..., ("f","PL"):...}
|
||||
ger = {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
g = "f" if "FEM" in f else "m"
|
||||
n = "PL" if "PL" in f else "SG"
|
||||
part.setdefault(lemma, {})[(g, n)] = form
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "preterite" and "IPFV" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "Balearic", "Valencian",
|
||||
"dated", "nonstandard"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {}
|
||||
adjs = {}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm not in ("#", "—", "-"):
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
_PERI = _IRREGV.get("_peri_pret_aux", {})
|
||||
|
||||
|
||||
# ── regular verb rule fallback ───────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("ar"):
|
||||
return "ar"
|
||||
if lemma.endswith("re"):
|
||||
return "re"
|
||||
if lemma.endswith("er"):
|
||||
return "er"
|
||||
if lemma.endswith("ir"):
|
||||
return "ir"
|
||||
return None
|
||||
|
||||
|
||||
# endings [1sg,2sg,3sg,1pl,2pl,3pl] — central Catalan
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "es", "a", "em", "eu", "en"],
|
||||
("ind", "present", "re"): ["o", "s", "", "em", "eu", "en"],
|
||||
("ind", "present", "er"): ["o", "s", "", "em", "eu", "en"],
|
||||
("ind", "present", "ir"): ["o", "es", "", "im", "iu", "en"], # pure -ir (dormir)
|
||||
("ind", "imperfect", "ar"): ["ava", "aves", "ava", "àvem", "àveu", "aven"],
|
||||
("ind", "imperfect", "re"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "imperfect", "er"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "imperfect", "ir"): ["ia", "ies", "ia", "íem", "íeu", "ien"],
|
||||
("ind", "preterite", "ar"): ["í", "ares", "à", "àrem", "àreu", "aren"],
|
||||
("ind", "preterite", "re"): ["í", "eres", "é", "érem", "éreu", "eren"],
|
||||
("ind", "preterite", "er"): ["í", "eres", "é", "érem", "éreu", "eren"],
|
||||
("ind", "preterite", "ir"): ["í", "ires", "í", "írem", "íreu", "iren"],
|
||||
("sbjv", "present", "ar"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "re"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "er"): ["i", "is", "i", "em", "eu", "in"],
|
||||
("sbjv", "present", "ir"): ["i", "is", "i", "im", "iu", "in"],
|
||||
("sbjv", "imperfect", "ar"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "re"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "er"): ["és", "essis", "és", "éssim", "éssiu", "essin"],
|
||||
("sbjv", "imperfect", "ir"): ["ís", "issis", "ís", "íssim", "íssiu", "issin"],
|
||||
("imp", "affirmative", "ar"): [None, "a", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "re"): [None, "", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "er"): [None, "", "i", "em", "eu", "in"],
|
||||
("imp", "affirmative", "ir"): [None, "", "i", "im", "iu", "in"],
|
||||
}
|
||||
_FUT = ["é", "às", "à", "em", "eu", "an"]
|
||||
_COND = ["ia", "ies", "ia", "íem", "íeu", "ien"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _apply_ar_spelling(stem, ending):
|
||||
"""-car/-gar/-çar/-jar spelling before front (e/i) endings."""
|
||||
front = ending[:1] in ("e", "i", "é", "í")
|
||||
if not front:
|
||||
# ç before back vowel stays; but -çar stem already ends ç
|
||||
return stem + ending
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "qu" + ending
|
||||
if stem.endswith("g"):
|
||||
return stem[:-1] + "gu" + ending
|
||||
if stem.endswith("ç"):
|
||||
return stem[:-1] + "c" + ending
|
||||
if stem.endswith("j"):
|
||||
return stem[:-1] + "g" + ending
|
||||
if stem.endswith("qu"):
|
||||
return stem + ending
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
body = lemma[:-2]
|
||||
i = _slot_idx(person, number)
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
# future/cond stem = infinitive (for -re verbs drop final -e)
|
||||
stem = lemma[:-1] if vc == "re" else lemma
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
return stem + end
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if not table:
|
||||
return None
|
||||
end = table[i]
|
||||
if end is None:
|
||||
return None
|
||||
if vc == "ar":
|
||||
return _apply_ar_spelling(body, end)
|
||||
# -re/-er/-ir: guard double vowel
|
||||
if body and body[-1:] == end[:1] and end[:1] in "ií":
|
||||
return body[:-1] + end
|
||||
return body + end
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number and number[:2].upper()}"
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{_NUMBER.get(number,'?')}"
|
||||
# UniMorph (cleanly accented) takes priority; the kaikki irregulars layer is a
|
||||
# FALLBACK for verbs/slots UniMorph lacks (anar, fer, and rarer paradigm cells).
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r is not None:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def peri_pret_aux(person, number):
|
||||
"""anar-present auxiliary for the periphrastic preterite (vaig cantar)."""
|
||||
return _PERI.get(f"{_PERSON.get(person,'3')}|{_NUMBER.get(number,'SG')}", "va")
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ──────────────────────────────────────────────────
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
ir = _IRREGV.get(lemma)
|
||||
base = None
|
||||
if ir and "part" in ir:
|
||||
# prefer explicit irregular agreement form (part_mSG/part_fSG/...)
|
||||
exact = ir.get("part_" + g + num)
|
||||
if exact:
|
||||
return exact, "lexicon"
|
||||
base = ir["part"]
|
||||
elif lemma in _PART:
|
||||
table = _PART[lemma]
|
||||
if (g, num) in table:
|
||||
return table[(g, num)], "lexicon"
|
||||
base = table.get(("m", "SG"))
|
||||
if base is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "ar":
|
||||
base = lemma[:-2] + "at"
|
||||
elif vc == "ir":
|
||||
base = lemma[:-2] + "it"
|
||||
elif vc in ("er", "re"):
|
||||
base = lemma[:-2] + "ut"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
else:
|
||||
conf = "lexicon"
|
||||
# agreement on -t/-ut/-at/-it participles: m.sg base, f.sg +a (-da? no: -ada),
|
||||
# Catalan: cantat/cantada/cantats/cantades; -t → f -da, pl -ts/-des
|
||||
if base.endswith("t"):
|
||||
stem = base[:-1]
|
||||
forms = {"m|SG": base, "f|SG": stem + "da",
|
||||
"m|PL": base + "s", "f|PL": stem + "des"}
|
||||
return forms[f"{g}|{num}"], conf
|
||||
if base.endswith("s"): # after sibilant participle (rare): pres->presa
|
||||
stem = base
|
||||
forms = {"m|SG": base, "f|SG": base + "a",
|
||||
"m|PL": base + "os", "f|PL": base + "es"}
|
||||
return forms[f"{g}|{num}"], conf
|
||||
return base, conf
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "ar":
|
||||
return lemma[:-2] + "ant", "rule"
|
||||
if vc in ("er", "re"):
|
||||
return lemma[:-2] + "ent", "rule"
|
||||
if vc == "ir":
|
||||
return lemma[:-2] + "int", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("ció", "sió", "tat", "tud", "esa", "esa", "dat", "ança", "ència",
|
||||
"ància", "tud", "ícia", "esa", "or") # note -or is mixed; kaikki wins
|
||||
_MASC_SUF = ("atge", "ment", " isme", "or")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in ("ció", "sió", "tat", "tud", "esa", "ança", "ència", "ància",
|
||||
"ícia", "etat"):
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
if noun.endswith("a") and not noun.endswith("ma"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic Catalan pluralization. (form, ok); ok=False FLAGS ambiguity."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
# stressed final vowel with accent → +ns (mà→mans is irregular; but capità→capitans)
|
||||
if noun[-1:] in ("à", "é", "í", "ó", "ú"):
|
||||
return noun + "ns", True
|
||||
if noun.endswith("ça"):
|
||||
return noun[:-2] + "ces", True # plaça→places
|
||||
if noun.endswith("ca"):
|
||||
return noun[:-2] + "ques", True # branca→branques
|
||||
if noun.endswith("ga"):
|
||||
return noun[:-2] + "gues", True # amiga→amigues
|
||||
if noun.endswith("ja"):
|
||||
return noun[:-2] + "ges", True # pluja→pluges
|
||||
if noun.endswith("qua"):
|
||||
return noun[:-3] + "qües", True
|
||||
if noun.endswith("gua"):
|
||||
return noun[:-3] + "gües", True
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "es", True # casa→cases
|
||||
# sibilant-final → -os
|
||||
if noun.endswith(("s", "ç", "x", "ig")) or noun.endswith(("ix", "tx", "tj")):
|
||||
if noun.endswith("ç"):
|
||||
return noun[:-1] + "ços", True # braç→braços
|
||||
return noun + "os", True # peix→peixos, gas→gasos
|
||||
if noun[-1:] in ("e", "i", "o", "u"):
|
||||
return noun + "s", True
|
||||
# consonant-final
|
||||
return noun + "s", True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def _fem_of(adj):
|
||||
"""Regular Catalan feminine: consonant/-o? Catalan masc usually consonant or -e.
|
||||
default +a with spelling changes; -e→-a for some; but many are invariable."""
|
||||
a = adj
|
||||
if a.endswith("a"):
|
||||
return a
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "a" # ample→? actually 'ample' invariable; kaikki wins
|
||||
if a.endswith("u"):
|
||||
return a + "a"
|
||||
if a.endswith("c"):
|
||||
return a[:-1] + "ca" # ric→rica
|
||||
if a.endswith("t"):
|
||||
return a + "a" # alt→alta
|
||||
return a + "a"
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
base = lemma if g == "m" else _fem_of(lemma)
|
||||
if num == "SG":
|
||||
return base, "rule"
|
||||
pl, ok = _rule_plural(base, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Catalan (github.com/unimorph/cat) + kaikki.org "
|
||||
"irregulars (anar/fer/auxiliaries)",
|
||||
"noun_adj_source": "kaikki.org Catalan (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len([k for k in _IRREGV if not k.startswith("_")]),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("cantar", "ind", "present", "first", "singular", "canto"),
|
||||
("cantar", "ind", "present", "third", "plural", "canten"),
|
||||
("ser", "ind", "present", "third", "singular", "és"),
|
||||
("haver", "ind", "present", "first", "singular", "he"),
|
||||
("anar", "ind", "present", "first", "singular", "vaig"),
|
||||
("fer", "ind", "present", "third", "singular", "fa"),
|
||||
("perdre", "ind", "present", "first", "singular", "perdo"),
|
||||
("dormir", "ind", "present", "third", "plural", "dormen"),
|
||||
("cantar", "ind", "future", "first", "singular", "cantaré"),
|
||||
("cantar", "ind", "preterite", "third", "singular", "cantà"),
|
||||
("tenir", "sbjv", "present", "first", "singular", "tingui"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:8} {mood}/{tense:11} {per[:3]}.{num[:2]} -> {got:10} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" peri-pret anar: 1sg=", peri_pret_aux("first", "singular"),
|
||||
"3pl=", peri_pret_aux("third", "plural"))
|
||||
print(" gender casa=", noun_gender("casa"), "home=", noun_gender("home"),
|
||||
"cavall=", noun_gender("cavall"), "cançó=", noun_gender("cançó"))
|
||||
print(" plural casa->", inflect_noun("casa", "plural"),
|
||||
"| plaça->", inflect_noun("plaça", "plural"),
|
||||
"| peix->", inflect_noun("peix", "plural"),
|
||||
"| braç->", inflect_noun("braç", "plural"),
|
||||
"| home->", inflect_noun("home", "plural"))
|
||||
print(" adj: alt/f/sg->", inflect_adj("alt", "f", "singular"),
|
||||
"| bonic/f/pl->", inflect_adj("bonic", "f", "plural"),
|
||||
"| vermell/f/sg->", inflect_adj("vermell", "f", "singular"))
|
||||
print(" part: cantar/f/sg->", participle("cantar", "f", "singular"),
|
||||
"| veure/f/pl->", participle("veure", "f", "plural"),
|
||||
"| fer/m/sg->", participle("fer", "m", "singular"))
|
||||
print(" ger: fer->", gerund("fer"), "| cantar->", gerund("cantar"))
|
||||
@@ -1,423 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_de_full.py — production German morphological generator.
|
||||
|
||||
Real data, no toy tables:
|
||||
|
||||
PRIMARY — UniMorph German (github.com/unimorph/deu, CC-BY-SA 3.0).
|
||||
~219k noun forms, ~199k verb forms. Supplies:
|
||||
nouns : gender (MASC/FEM/NEUT) + case×number paradigm
|
||||
(N;NOM/ACC/DAT/GEN; MASC/FEM/NEUT; SG/PL) — the genitive -(e)s,
|
||||
dative-plural -n and the five plural classes are REAL forms, not
|
||||
guessed.
|
||||
verbs : full finite paradigm IND;{SG,PL};{1,2,3};{PRS,PST}, the past
|
||||
participle (V.PTCP;PST, incl. reattached separable prefix
|
||||
'zugefügt'), and — crucially for V2 — the SEPARATED finite form
|
||||
UniMorph records directly ('füge zu', 'steht auf').
|
||||
adjs : comparative / superlative (ADJ;CMPR, ADJ;SPRL).
|
||||
|
||||
SECONDARY — kaikki.org German (Wiktionary, CC-BY-SA/GFDL). Gap-fills noun
|
||||
gender + plural where UniMorph is thin. Never overrides UniMorph.
|
||||
|
||||
Rule fallbacks (flagged 'rule'/'fallback') for lemmas absent from both lexicons:
|
||||
present : -e/-st/-t/-en/-t/-en with e-epenthesis after -t/-d/-chn stems
|
||||
plural : gender heuristic (fem -> -(e)n, else -e / umlaut left to lexicon)
|
||||
ppart : weak ge-…-t
|
||||
Adjective ENDINGS are rule-computed by the realizer (regular closed table);
|
||||
this module only supplies the comparative/superlative STEM.
|
||||
|
||||
Perfect auxiliary (haben vs sein): sein for a curated set of intransitive
|
||||
motion / change-of-state verbs (real German lexical property), else haben.
|
||||
|
||||
Public API:
|
||||
noun_gender(lemma) -> 'm'|'f'|'n'
|
||||
decline_noun(lemma, case, number) -> (form, conf)
|
||||
pluralize(lemma) -> (form, conf)
|
||||
finite(lemma, tense, person, number) -> (form, conf) # may contain ' prefix'
|
||||
nonfinite(lemma, req) -> (form, conf) # req: 'inf'|'ppart'
|
||||
past_participle(lemma) -> (form, conf)
|
||||
separable_prefix(lemma) -> str|None
|
||||
perfect_aux(lemma) -> 'haben'|'sein'
|
||||
comparative(lemma)/superlative(lemma) -> (stem, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "deu.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_de.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "de_morph_cache.pkl")
|
||||
|
||||
_GENDER = {"MASC": "m", "FEM": "f", "NEUT": "n"}
|
||||
|
||||
# intransitive motion / change-of-state verbs that take SEIN in the perfect
|
||||
_SEIN = {"gehen", "kommen", "fahren", "laufen", "rennen", "reisen", "fallen",
|
||||
"steigen", "sinken", "wachsen", "sterben", "geschehen", "passieren",
|
||||
"werden", "bleiben", "sein", "aufstehen", "einschlafen", "aufwachen",
|
||||
"ankommen", "abfahren", "aufsteigen", "erscheinen", "verschwinden",
|
||||
"fliegen", "schwimmen", "springen", "begegnen", "folgen", "gelingen",
|
||||
"wandern", "ziehen", "flüchten", "eintreten", "einsteigen", "aussteigen"}
|
||||
|
||||
|
||||
# hardcoded high-frequency irregular / auxiliary / modal paradigms (closed class,
|
||||
# verified) — consulted before the lexicon so aux+modal chains are always correct.
|
||||
_CORE = {
|
||||
"sein": {"prs": {("first", "singular"): "bin", ("second", "singular"): "bist",
|
||||
("third", "singular"): "ist", ("first", "plural"): "sind",
|
||||
("second", "plural"): "seid", ("third", "plural"): "sind"},
|
||||
"pst": {("first", "singular"): "war", ("second", "singular"): "warst",
|
||||
("third", "singular"): "war", ("first", "plural"): "waren",
|
||||
("second", "plural"): "wart", ("third", "plural"): "waren"},
|
||||
"ppart": "gewesen"},
|
||||
"haben": {"prs": {("first", "singular"): "habe", ("second", "singular"): "hast",
|
||||
("third", "singular"): "hat", ("first", "plural"): "haben",
|
||||
("second", "plural"): "habt", ("third", "plural"): "haben"},
|
||||
"pst": {("first", "singular"): "hatte", ("second", "singular"): "hattest",
|
||||
("third", "singular"): "hatte", ("first", "plural"): "hatten",
|
||||
("second", "plural"): "hattet", ("third", "plural"): "hatten"},
|
||||
"ppart": "gehabt"},
|
||||
"werden": {"prs": {("first", "singular"): "werde", ("second", "singular"): "wirst",
|
||||
("third", "singular"): "wird", ("first", "plural"): "werden",
|
||||
("second", "plural"): "werdet", ("third", "plural"): "werden"},
|
||||
"pst": {("first", "singular"): "wurde", ("second", "singular"): "wurdest",
|
||||
("third", "singular"): "wurde", ("first", "plural"): "wurden",
|
||||
("second", "plural"): "wurdet", ("third", "plural"): "wurden"},
|
||||
"ppart": "geworden"},
|
||||
}
|
||||
_MODAL_PRS = {
|
||||
"können": ("kann", "kannst", "kann", "können", "könnt", "können"),
|
||||
"müssen": ("muss", "musst", "muss", "müssen", "müsst", "müssen"),
|
||||
"wollen": ("will", "willst", "will", "wollen", "wollt", "wollen"),
|
||||
"sollen": ("soll", "sollst", "soll", "sollen", "sollt", "sollen"),
|
||||
"dürfen": ("darf", "darfst", "darf", "dürfen", "dürft", "dürfen"),
|
||||
"mögen": ("mag", "magst", "mag", "mögen", "mögt", "mögen"),
|
||||
}
|
||||
_MODAL_PST = {
|
||||
"können": ("konnte", "konntest", "konnte", "konnten", "konntet", "konnten"),
|
||||
"müssen": ("musste", "musstest", "musste", "mussten", "musstet", "mussten"),
|
||||
"wollen": ("wollte", "wolltest", "wollte", "wollten", "wolltet", "wollten"),
|
||||
"sollen": ("sollte", "solltest", "sollte", "sollten", "solltet", "sollten"),
|
||||
"dürfen": ("durfte", "durftest", "durfte", "durften", "durftet", "durften"),
|
||||
"mögen": ("mochte", "mochtest", "mochte", "mochten", "mochtet", "mochten"),
|
||||
}
|
||||
_PN_ORDER = [("first", "singular"), ("second", "singular"), ("third", "singular"),
|
||||
("first", "plural"), ("second", "plural"), ("third", "plural")]
|
||||
_MODAL_PPART = {"können": "gekonnt", "müssen": "gemusst", "wollen": "gewollt",
|
||||
"sollen": "gesollt", "dürfen": "gedurft", "mögen": "gemocht"}
|
||||
for _m, _forms in _MODAL_PRS.items():
|
||||
_CORE[_m] = {"prs": dict(zip(_PN_ORDER, _forms)),
|
||||
"pst": dict(zip(_PN_ORDER, _MODAL_PST[_m])),
|
||||
"ppart": _MODAL_PPART[_m]}
|
||||
|
||||
|
||||
def _person_num(tags):
|
||||
p = n = None
|
||||
for t in tags:
|
||||
if t in ("1", "2", "3"):
|
||||
p = {"1": "first", "2": "second", "3": "third"}[t]
|
||||
elif t == "SG":
|
||||
n = "singular"
|
||||
elif t == "PL":
|
||||
n = "plural"
|
||||
return p, n
|
||||
|
||||
|
||||
def _build_from_unimorph():
|
||||
nouns, verbs, adjs = {}, {}, {}
|
||||
if not os.path.exists(_UNIMORPH):
|
||||
return nouns, verbs, adjs
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tagstr = parts
|
||||
tags = tagstr.split(";")
|
||||
head = tags[0]
|
||||
tset = set(tags)
|
||||
if head == "N":
|
||||
rec = nouns.setdefault(lemma, {"g": None, "cases": {}, "pl": None})
|
||||
g = next((_GENDER[t] for t in tags if t in _GENDER), None)
|
||||
if g and not rec["g"]:
|
||||
rec["g"] = g
|
||||
case = next((t for t in tags if t in ("NOM", "ACC", "DAT", "GEN")), None)
|
||||
num = "plural" if "PL" in tset else ("singular" if "SG" in tset else None)
|
||||
if case and num:
|
||||
rec["cases"].setdefault((case, num), form)
|
||||
if case == "NOM" and num == "plural" and not rec["pl"]:
|
||||
rec["pl"] = form
|
||||
elif head.startswith("V"):
|
||||
rec = verbs.setdefault(lemma, {"prs": {}, "pst": {}, "ppart": None})
|
||||
if "PTCP" in head and "PST" in tset:
|
||||
rec["ppart"] = rec["ppart"] or form
|
||||
elif "IND" in tset and ("PRS" in tset or "PST" in tset):
|
||||
p, n = _person_num(tags)
|
||||
if p and n:
|
||||
slot = "prs" if "PRS" in tset else "pst"
|
||||
rec[slot].setdefault((p, n), form)
|
||||
elif head == "ADJ":
|
||||
rec = adjs.setdefault(lemma, {})
|
||||
if "CMPR" in tset:
|
||||
rec.setdefault("cmpr", form.replace("am ", "").strip())
|
||||
elif "SPRL" in tset:
|
||||
rec.setdefault("sprl", form.replace("am ", "").replace("sten", "st")
|
||||
if form.endswith("sten") else form.replace("am ", ""))
|
||||
return nouns, verbs, adjs
|
||||
|
||||
|
||||
def _build_from_kaikki(nouns):
|
||||
"""Gap-fill noun gender + plural from kaikki German."""
|
||||
if not os.path.exists(_KAIKKI):
|
||||
return
|
||||
_g = {"masculine": "m", "feminine": "f", "neuter": "n", "m": "m", "f": "f", "n": "n"}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "noun":
|
||||
continue
|
||||
w = d.get("word", "")
|
||||
if not w or not w[0].isalpha() or " " in w:
|
||||
continue
|
||||
rec = nouns.setdefault(w, {"g": None, "cases": {}, "pl": None})
|
||||
# GENDER: Wiktionary gender is hand-curated and OVERRIDES UniMorph's
|
||||
# auto-tagged gender, which has known errors (e.g. UniMorph deu mis-
|
||||
# records Zeit=MASC, Wagen=NEUT; Wiktionary has f, m correctly).
|
||||
for h in d.get("head_templates", []) or []:
|
||||
a = h.get("args", {}) or {}
|
||||
raw = a.get("1") or a.get("g") or ""
|
||||
code = str(raw).split(",")[0].strip().lower()
|
||||
if code in _g:
|
||||
rec["g"] = _g[code]
|
||||
break
|
||||
if not rec["pl"]:
|
||||
for f in d.get("forms", []) or []:
|
||||
t = set(f.get("tags", []) or [])
|
||||
if "plural" in t and f.get("form") and "genitive" not in t:
|
||||
rec["pl"] = f["form"]
|
||||
break
|
||||
|
||||
|
||||
def _build_cache():
|
||||
nouns, verbs, adjs = _build_from_unimorph()
|
||||
_build_from_kaikki(nouns)
|
||||
data = {"nouns": nouns, "verbs": verbs, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [p for p in (_UNIMORPH, _KAIKKI) if os.path.exists(p)]
|
||||
newest = max((os.path.getmtime(p) for p in srcs), default=0)
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_NOUNS, _VERBS, _ADJS = _LEX["nouns"], _LEX["verbs"], _LEX["adjs"]
|
||||
|
||||
|
||||
# ── nouns ────────────────────────────────────────────────────────────────────────
|
||||
def noun_gender(lemma):
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if rec and rec.get("g"):
|
||||
return rec["g"]
|
||||
# last-resort rule: -ung/-heit/-keit/-schaft/-tät/-ion -> f ; -chen/-lein -> n
|
||||
low = lemma.lower()
|
||||
if low.endswith(("ung", "heit", "keit", "schaft", "tät", "ion", "ik", "ei")):
|
||||
return "f"
|
||||
if low.endswith(("chen", "lein", "ment", "um")):
|
||||
return "n"
|
||||
return "m"
|
||||
|
||||
|
||||
def pluralize(lemma):
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if rec and rec.get("pl"):
|
||||
return rec["pl"], "lexicon"
|
||||
g = noun_gender(lemma)
|
||||
if g == "f":
|
||||
return (lemma + "en" if not lemma.endswith("e") else lemma + "n"), "rule"
|
||||
return (lemma if lemma.endswith(("er", "en", "el")) else lemma + "e"), "rule"
|
||||
|
||||
|
||||
def decline_noun(lemma, case, number):
|
||||
"""case in NOM/ACC/DAT/GEN, number in singular/plural."""
|
||||
rec = _NOUNS.get(lemma) or _NOUNS.get(lemma.capitalize())
|
||||
if case == "DAT" and number == "singular":
|
||||
# modern German drops the archaic dative -e ('dem Kinde' -> 'dem Kind');
|
||||
# the article carries the case. Keep bare nominative form.
|
||||
base = (rec or {}).get("cases", {}).get(("NOM", "singular")) or lemma
|
||||
return base, ("lexicon" if rec else "rule")
|
||||
if rec and rec.get("cases", {}).get((case, number)):
|
||||
return rec["cases"][(case, number)], "lexicon"
|
||||
if number == "plural":
|
||||
pl, c = pluralize(lemma)
|
||||
if case == "DAT" and not pl.endswith("n") and not pl.endswith("s"):
|
||||
return pl + "n", c # dative plural -n
|
||||
return pl, c
|
||||
# singular
|
||||
g = noun_gender(lemma)
|
||||
if case == "GEN" and g in ("m", "n"):
|
||||
return (lemma + "es" if lemma.endswith(("s", "ß", "z", "x")) else lemma + "s"), "rule"
|
||||
return lemma, "lexicon" if rec else "rule"
|
||||
|
||||
|
||||
# ── verbs ──────────────────────────────────────────────────────────────────────--
|
||||
_PRS_ENDINGS = {("first", "singular"): "e", ("second", "singular"): "st",
|
||||
("third", "singular"): "t", ("first", "plural"): "en",
|
||||
("second", "plural"): "t", ("third", "plural"): "en"}
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
if lemma.endswith("en"):
|
||||
return lemma[:-2]
|
||||
if lemma.endswith("n"):
|
||||
return lemma[:-1]
|
||||
return lemma
|
||||
|
||||
|
||||
def separable_prefix(lemma):
|
||||
"""Return the separable prefix if the lemma is a separable-prefix verb."""
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec:
|
||||
for (_p, _n), form in rec.get("prs", {}).items():
|
||||
if " " in form:
|
||||
return form.rsplit(" ", 1)[1]
|
||||
_SEP = ("auf", "aus", "ab", "an", "ein", "mit", "nach", "vor", "zu", "zurück",
|
||||
"weg", "hin", "her", "los", "bei", "fest", "fort", "um", "zusammen")
|
||||
_INSEP = ("be", "ge", "er", "ver", "zer", "ent", "emp", "miss")
|
||||
for p in sorted(_SEP, key=len, reverse=True):
|
||||
if lemma.startswith(p) and len(lemma) > len(p) + 2 \
|
||||
and not lemma.startswith(_INSEP):
|
||||
return p
|
||||
return None
|
||||
|
||||
|
||||
def finite(lemma, tense, person, number):
|
||||
"""Present/past finite. For separable verbs the returned string is the
|
||||
UniMorph SEPARATED form 'stem prefix' (realizer places prefix per V2)."""
|
||||
slot = "prs" if tense == "present" else "pst"
|
||||
if lemma in _CORE and _CORE[lemma].get(slot, {}).get((person, number)):
|
||||
return _CORE[lemma][slot][(person, number)], "lexicon"
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec and rec.get(slot, {}).get((person, number)):
|
||||
return rec[slot][(person, number)], "lexicon"
|
||||
# rule fallback (present only reliable; past weak -te)
|
||||
stem = _stem(lemma)
|
||||
pref = separable_prefix(lemma)
|
||||
if pref:
|
||||
stem = _stem(lemma[len(pref):])
|
||||
if tense == "present":
|
||||
end = _PRS_ENDINGS[(person, number)]
|
||||
if stem.endswith(("t", "d", "chn", "ffn", "gn")) and end in ("st", "t"):
|
||||
end = "e" + end
|
||||
form = stem + end
|
||||
else:
|
||||
form = stem + ("ete" if stem.endswith(("t", "d")) else "te")
|
||||
if (person, number) == ("second", "singular"):
|
||||
form += "st"
|
||||
elif number == "plural" and person != "second":
|
||||
form += "n"
|
||||
elif (person, number) == ("second", "plural"):
|
||||
form += "t"
|
||||
if pref:
|
||||
return f"{form} {pref}", "rule"
|
||||
return form, "rule"
|
||||
|
||||
|
||||
def _weak_t(stem):
|
||||
return stem + ("et" if stem.endswith(("t", "d", "chn", "ffn", "gn")) else "t")
|
||||
|
||||
|
||||
def past_participle(lemma):
|
||||
if lemma in _CORE:
|
||||
return _CORE[lemma]["ppart"], "lexicon"
|
||||
rec = _VERBS.get(lemma)
|
||||
if rec and rec.get("ppart"):
|
||||
return rec["ppart"], "lexicon"
|
||||
stem = _stem(lemma)
|
||||
pref = separable_prefix(lemma)
|
||||
_INSEP = ("be", "ge", "er", "ver", "zer", "ent", "emp", "miss")
|
||||
if pref:
|
||||
inner = _stem(lemma[len(pref):])
|
||||
return pref + "ge" + _weak_t(inner), "rule"
|
||||
if lemma.startswith(_INSEP):
|
||||
return _weak_t(stem), "rule"
|
||||
return "ge" + _weak_t(stem), "rule"
|
||||
|
||||
|
||||
def nonfinite(lemma, req):
|
||||
if req == "ppart":
|
||||
return past_participle(lemma)
|
||||
return lemma, "lexicon" if lemma in _VERBS else "rule" # infinitive
|
||||
|
||||
|
||||
def perfect_aux(lemma):
|
||||
return "sein" if lemma in _SEIN else "haben"
|
||||
|
||||
|
||||
# ── adjectives ────────────────────────────────────────────────────────────────---
|
||||
_ADJ_IRREG_SPRL = {"gut": "best", "groß": "größt", "hoch": "höchst",
|
||||
"nah": "nächst", "viel": "meist", "gern": "liebst"}
|
||||
|
||||
|
||||
def comparative(lemma):
|
||||
rec = _ADJS.get(lemma)
|
||||
if rec and rec.get("cmpr"):
|
||||
return rec["cmpr"], "lexicon"
|
||||
return lemma + "er", "rule"
|
||||
|
||||
|
||||
def superlative(lemma):
|
||||
"""Return the bare superlative STEM (realizer adds 'am ...en' or '-e' ending)."""
|
||||
if lemma in _ADJ_IRREG_SPRL:
|
||||
return _ADJ_IRREG_SPRL[lemma], "lexicon"
|
||||
# derive from the comparative so umlaut is carried (alt->älter->ältest)
|
||||
cmpr, cconf = comparative(lemma)
|
||||
base = cmpr[:-2] if cmpr.endswith("er") else lemma
|
||||
end = "est" if base.endswith(("t", "d", "s", "ß", "z", "sch")) else "st"
|
||||
return base + end, cconf
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"source": "UniMorph deu (primary) + kaikki.org German (gap-fill gender/plural)",
|
||||
"license": "CC-BY-SA 3.0 (UniMorph); CC-BY-SA/GFDL (Wiktionary)",
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"nouns_with_gender": sum(1 for v in _NOUNS.values() if v.get("g")),
|
||||
"nouns_with_plural": sum(1 for v in _NOUNS.values() if v.get("pl")),
|
||||
"verb_lemmas": len(_VERBS),
|
||||
"verbs_with_ppart": sum(1 for v in _VERBS.values() if v.get("ppart")),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
for w in ("Hund", "Frau", "Kind", "Mann", "Buch", "Blume"):
|
||||
print(f" {w}: gender={noun_gender(w)} pl={pluralize(w)} "
|
||||
f"gen.sg={decline_noun(w, 'GEN', 'singular')} "
|
||||
f"dat.pl={decline_noun(w, 'DAT', 'plural')}")
|
||||
for v in ("machen", "gehen", "aufstehen", "sein", "haben", "arbeiten"):
|
||||
print(f" {v}: 3sg.prs={finite(v, 'present', 'third', 'singular')} "
|
||||
f"3sg.pst={finite(v, 'past', 'third', 'singular')} "
|
||||
f"ppart={past_participle(v)} aux={perfect_aux(v)} sep={separable_prefix(v)}")
|
||||
for a in ("schnell", "gut", "groß", "alt"):
|
||||
print(f" {a}: cmpr={comparative(a)} sprl={superlative(a)}")
|
||||
@@ -1,562 +0,0 @@
|
||||
"""morphology_es_full.py — production-grade Spanish morphological generator.
|
||||
|
||||
NOT a toy. Backed by a real, broad, licensed lexicon:
|
||||
|
||||
UniMorph Spanish (github.com/unimorph/spa, CC-BY-SA 3.0, Wiktionary-derived)
|
||||
1,196,245 inflected forms:
|
||||
6,695 verb lemmas — full paradigms: indicative (present/preterite/
|
||||
imperfect/future), conditional, present & imperfect
|
||||
subjunctive, affirmative imperative, formal/informal
|
||||
48,353 noun lemmas — WITH inherent gender (N;FEM/MASC;SG/PL)
|
||||
16,984 adj lemmas — gender + number paradigms
|
||||
|
||||
Fallbacks (so we degrade, never crash, on out-of-vocabulary input):
|
||||
- verbs : mlconjug3 (ML paradigm model, conjugates ANY Spanish verb) then a
|
||||
hand-rolled regular-ending generator
|
||||
- nouns : gender heuristic (endings) + regular pluralization
|
||||
- adjs : -o/-a gender rule + regular pluralization
|
||||
|
||||
Every generated form carries a CONFIDENCE flag:
|
||||
"lexicon" form came straight from UniMorph (trust: high)
|
||||
"model" form came from mlconjug3 (trust: high)
|
||||
"rule" form came from a deterministic rule (trust: medium)
|
||||
"fallback" we could not inflect; returned lemma as-is (trust: low → FLAG)
|
||||
|
||||
Public API (used by realizer_es.py):
|
||||
conjugate(lemma, mood, tense, person, number, formality="informal") -> (form, conf)
|
||||
participle(lemma) -> (form, conf) # past participle (compound tenses)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
attach_enclitics(verb_form, clitics) -> str # accent-correct enclisis
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import os
|
||||
import pickle
|
||||
import unicodedata
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "spa.unimorph")
|
||||
_CACHE = os.path.join(_HERE, "data", "es_morph_cache.pkl")
|
||||
|
||||
# ── canonical feature keys the realizer speaks, mapped to UniMorph tags ─────────
|
||||
# mood/tense pair -> the UniMorph feature substring that identifies it
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): ("IND", "PRS", None),
|
||||
("ind", "preterite"): ("IND", "PST", "PFV"),
|
||||
("ind", "imperfect"): ("IND", "PST", "IPFV"),
|
||||
("ind", "future"): ("IND", "FUT", None),
|
||||
("ind", "conditional"):("COND", None, None),
|
||||
("sbjv", "present"): ("SBJV", "PRS", None),
|
||||
("sbjv", "imperfect"): ("SBJV", "PST", "LGSPEC1"), # -ra form
|
||||
("imp", "present"): ("POS", "IMP", None),
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
# ── build / load the compact lexicon ───────────────────────────────────────────
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs = {} # (lemma, canonkey) -> form canonkey e.g. "ind|present|1|SG|infm"
|
||||
nouns = {} # lemma -> {"g": "m"/"f", "SG": form, "PL": form}
|
||||
adjs = {} # lemma -> {("m","SG"): form, ...}
|
||||
part = {} # lemma -> masc-sg participle
|
||||
ger = {} # lemma -> gerund
|
||||
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V":
|
||||
# skip clitic-bearing rows (we generate clitics ourselves)
|
||||
if "PRO" in f:
|
||||
continue
|
||||
if "V.PTCP" in f and "PST" in f and "MASC" in f and "SG" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if "V.CVB" in f or "NFIN" in f or "V.PTCP" in f:
|
||||
if "V.CVB" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
# identify mood/tense
|
||||
mt = None
|
||||
for (mood, tense), (a, b, c) in _VERB_KEYMAP.items():
|
||||
if a not in f:
|
||||
continue
|
||||
if b is not None and b not in f:
|
||||
continue
|
||||
if c is not None and c not in f:
|
||||
continue
|
||||
# disambiguate IND;PST needing PFV vs IPFV
|
||||
if a == "IND" and b == "PST" and c not in f:
|
||||
continue
|
||||
mt = (mood, tense)
|
||||
break
|
||||
if mt is None:
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
formal = "form" if "FORM" in f else ("infm" if "INFM" in f else "any")
|
||||
key = f"{mt[0]}|{mt[1]}|{person}|{number}|{formal}"
|
||||
verbs.setdefault((lemma, key), form)
|
||||
|
||||
elif head == "N":
|
||||
# substring test handles epicene "MASC+FEM" (-> masc citation)
|
||||
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else None)
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if num is None:
|
||||
continue
|
||||
# store forms keyed by (gender,number); animate nouns list BOTH
|
||||
# genders under one lemma (niño -> niño/niña). Resolve citation
|
||||
# gender in a post-pass (gender of the row whose form == lemma).
|
||||
d = nouns.setdefault(lemma, {})
|
||||
d.setdefault("_rows", []).append((g, num, form))
|
||||
|
||||
elif head == "ADJ":
|
||||
g = "m" if "MASC" in tag else ("f" if "FEM" in tag else "m")
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if num is None:
|
||||
continue
|
||||
adjs.setdefault(lemma, {})[(g, num)] = form
|
||||
|
||||
# post-pass: resolve noun citation gender + default SG/PL forms
|
||||
for lemma, d in nouns.items():
|
||||
rows = d.pop("_rows", [])
|
||||
# citation gender = gender of the row whose form == lemma; else first MASC;
|
||||
# else first seen gender.
|
||||
cite_g = None
|
||||
for g, num, form in rows:
|
||||
if form == lemma and g:
|
||||
cite_g = g
|
||||
break
|
||||
if cite_g is None:
|
||||
for g, num, form in rows:
|
||||
if g == "m":
|
||||
cite_g = "m"
|
||||
break
|
||||
if cite_g is None:
|
||||
cite_g = next((g for g, _, _ in rows if g), "m")
|
||||
d["g"] = cite_g
|
||||
for g, num, form in rows:
|
||||
d[(g, num)] = form
|
||||
d["SG"] = d.get((cite_g, "SG")) or next((f for g, n, f in rows if n == "SG"), lemma)
|
||||
d["PL"] = d.get((cite_g, "PL")) or next((f for g, n, f in rows if n == "PL"), None)
|
||||
|
||||
# post-pass: UniMorph omits the identity inflection (masc-sg == lemma) for
|
||||
# adjectives, so fill it in; without this a fem-sg row wrongly satisfies a
|
||||
# masc-sg request (alto -> alta bug).
|
||||
for lemma, d in adjs.items():
|
||||
d.setdefault(("m", "SG"), lemma)
|
||||
|
||||
data = {"verbs": verbs, "nouns": nouns, "adjs": adjs, "part": part, "ger": ger}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE) and os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH):
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _NOUNS, _ADJS, _PART, _GER = (
|
||||
_LEX["verbs"], _LEX["nouns"], _LEX["adjs"], _LEX["part"], _LEX["ger"])
|
||||
|
||||
# ── mlconjug3 fallback (lazy) ───────────────────────────────────────────────────
|
||||
_MLC = None
|
||||
_MLC_TENSE = { # (mood,tense) -> (mlconjug mood label, tense label)
|
||||
("ind", "present"): ("Indicativo", "Indicativo presente"),
|
||||
("ind", "preterite"): ("Indicativo", "Indicativo pretérito perfecto simple"),
|
||||
("ind", "imperfect"): ("Indicativo", "Indicativo pretérito imperfecto"),
|
||||
("ind", "future"): ("Indicativo", "Indicativo futuro"),
|
||||
("ind", "conditional"): ("Condicional", "Condicional Condicional"),
|
||||
("sbjv", "present"): ("Subjuntivo", "Subjuntivo presente"),
|
||||
("sbjv", "imperfect"): ("Subjuntivo", "Subjuntivo pretérito imperfecto 1"),
|
||||
("imp", "present"): ("Imperativo", "Imperativo Afirmativo"),
|
||||
}
|
||||
_MLC_SLOT = { # (person,number) -> mlconjug slot key
|
||||
("first", "singular"): "1s", ("second", "singular"): "2s",
|
||||
("third", "singular"): "3s", ("first", "plural"): "1p",
|
||||
("second", "plural"): "2p", ("third", "plural"): "3p",
|
||||
}
|
||||
|
||||
|
||||
def _mlc_conjugate(lemma, mood, tense, person, number):
|
||||
global _MLC
|
||||
try:
|
||||
if _MLC is None:
|
||||
from mlconjug3 import Conjugator
|
||||
_MLC = Conjugator(language="es")
|
||||
v = _MLC.conjugate(lemma)
|
||||
if v is None:
|
||||
return None
|
||||
info = v.conjug_info
|
||||
m, t = _MLC_TENSE.get((mood, tense), (None, None))
|
||||
if m is None or m not in info or t not in info[m]:
|
||||
return None
|
||||
block = info[m][t]
|
||||
slot = _MLC_SLOT.get((person, number))
|
||||
if isinstance(block, dict) and slot in block and block[slot]:
|
||||
return block[slot]
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# ── regular-ending rule fallback (last resort, deterministic) ───────────────────
|
||||
def _vclass(lemma):
|
||||
return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else "ar"
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "as", "a", "amos", "áis", "an"],
|
||||
("ind", "present", "er"): ["o", "es", "e", "emos", "éis", "en"],
|
||||
("ind", "present", "ir"): ["o", "es", "e", "imos", "ís", "en"],
|
||||
("ind", "preterite", "ar"): ["é", "aste", "ó", "amos", "asteis", "aron"],
|
||||
("ind", "preterite", "er"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
||||
("ind", "preterite", "ir"): ["í", "iste", "ió", "imos", "isteis", "ieron"],
|
||||
("ind", "imperfect", "ar"): ["aba", "abas", "aba", "ábamos", "abais", "aban"],
|
||||
("ind", "imperfect", "er"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
||||
("ind", "imperfect", "ir"): ["ía", "ías", "ía", "íamos", "íais", "ían"],
|
||||
("sbjv", "present", "ar"): ["e", "es", "e", "emos", "éis", "en"],
|
||||
("sbjv", "present", "er"): ["a", "as", "a", "amos", "áis", "an"],
|
||||
("sbjv", "present", "ir"): ["a", "as", "a", "amos", "áis", "an"],
|
||||
("sbjv", "imperfect", "ar"): ["ara", "aras", "ara", "áramos", "arais", "aran"],
|
||||
("sbjv", "imperfect", "er"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
||||
("sbjv", "imperfect", "ir"): ["iera", "ieras", "iera", "iéramos", "ierais", "ieran"],
|
||||
}
|
||||
_FUT = ["é", "ás", "á", "emos", "éis", "án"]
|
||||
_COND = ["ía", "ías", "ía", "íamos", "íais", "ían"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
if len(lemma) < 3 or lemma[-2:] not in ("ar", "er", "ir"):
|
||||
return None
|
||||
vc, st, i = _vclass(lemma), _stem(lemma), _slot_idx(person, number)
|
||||
if tense == "future":
|
||||
return lemma + _FUT[i]
|
||||
if tense == "conditional":
|
||||
return lemma + _COND[i]
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if table:
|
||||
return st + table[i]
|
||||
if mood == "imp" and tense == "present":
|
||||
# affirmative tú imperative = 3sg present indicative
|
||||
pres = _REG.get(("ind", "present", vc))
|
||||
return st + pres[2] if number == "singular" else st + pres[5]
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number, formality="informal"):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
formal = "form" if formality == "formal" else "infm"
|
||||
if p and n:
|
||||
for fkey in (formal, "any", "infm" if formal == "form" else "form"):
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}|{fkey}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
m = _mlc_conjugate(lemma, mood, tense, person, number)
|
||||
if m:
|
||||
return m, "model"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
_IRREG_PART = { # guarantee the common irregular participles
|
||||
"escribir": "escrito", "describir": "descrito", "abrir": "abierto",
|
||||
"cubrir": "cubierto", "descubrir": "descubierto", "morir": "muerto",
|
||||
"poner": "puesto", "ver": "visto", "volver": "vuelto", "devolver": "devuelto",
|
||||
"hacer": "hecho", "deshacer": "deshecho", "decir": "dicho", "romper": "roto",
|
||||
"resolver": "resuelto", "freír": "frito", "imprimir": "impreso",
|
||||
"satisfacer": "satisfecho", "prever": "previsto", "revolver": "revuelto",
|
||||
}
|
||||
|
||||
|
||||
def participle(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _IRREG_PART:
|
||||
return _IRREG_PART[lemma], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ado", "rule"
|
||||
if lemma[-2:] in ("er", "ir"):
|
||||
return lemma[:-2] + "ido", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
_IRREG_GER = {"dormir": "durmiendo", "morir": "muriendo", "pedir": "pidiendo",
|
||||
"sentir": "sintiendo", "mentir": "mintiendo", "servir": "sirviendo",
|
||||
"venir": "viniendo", "decir": "diciendo", "poder": "pudiendo",
|
||||
"ir": "yendo", "leer": "leyendo", "creer": "creyendo",
|
||||
"oír": "oyendo", "traer": "trayendo", "caer": "cayendo",
|
||||
"construir": "construyendo", "huir": "huyendo", "reír": "riendo"}
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _IRREG_GER:
|
||||
return _IRREG_GER[lemma], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ando", "rule"
|
||||
if lemma[-2:] in ("er", "ir"):
|
||||
return lemma[:-2] + "iendo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ────────────────────────────────────────────────
|
||||
_INVARIANT_PL = {"lunes", "martes", "miércoles", "jueves", "viernes",
|
||||
"crisis", "tesis", "análisis", "dosis", "virus", "paraguas"}
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf, g in (("ión", "f"), ("dad", "f"), ("tad", "f"), ("umbre", "f"),
|
||||
("sis", "f"), ("ez", "f"), ("triz", "f"),
|
||||
("ema", "m"), ("ama", "m"), ("oma", "m"), ("aje", "m"),
|
||||
("or", "m"), ("án", "m"), ("ín", "m")):
|
||||
if noun.endswith(suf):
|
||||
return g
|
||||
if noun.endswith("o"):
|
||||
return "m"
|
||||
if noun.endswith("a"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _regular_plural(noun):
|
||||
if noun in _INVARIANT_PL:
|
||||
return noun
|
||||
if not noun:
|
||||
return noun
|
||||
last = noun[-1]
|
||||
if last == "z":
|
||||
return noun[:-1] + "ces"
|
||||
if last in "aeiouáéíóú":
|
||||
# stressed final vowel í/ú -> +es (rubí->rubíes), else +s
|
||||
if last in "íú":
|
||||
return noun + "es"
|
||||
return noun + "s"
|
||||
if last == "s":
|
||||
# esdrújula / stress-final handled crudely; most polysyllables invariant
|
||||
return noun
|
||||
return noun + "es"
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
# honor a requested gender for animate nouns (gato -> gata)
|
||||
if gender and (gender, num) in d:
|
||||
return d[(gender, num)], "lexicon"
|
||||
if d.get(num):
|
||||
return d[num], "lexicon"
|
||||
if number == "singular":
|
||||
return lemma, "rule" if not d else "lexicon"
|
||||
return _regular_plural(lemma), "rule"
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ─────────────────────────────────────────────────
|
||||
_INV_GENDER_ADJ = {"español": "española", "trabajador": "trabajadora",
|
||||
"hablador": "habladora", "encantador": "encantadora",
|
||||
"alemán": "alemana", "francés": "francesa", "inglés": "inglesa"}
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _ADJS.get(lemma)
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
form = d.get((gender, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# gender-invariant adjective (grande, feliz, azul): fem == masc.
|
||||
# For a missing plural, pluralize this gender's singular form.
|
||||
sg = d.get((gender, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if number == "plural":
|
||||
return _regular_plural(sg), "rule"
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
a = lemma
|
||||
if gender == "f":
|
||||
if a in _INV_GENDER_ADJ:
|
||||
a = _INV_GENDER_ADJ[a]
|
||||
elif a.endswith("o"):
|
||||
a = a[:-1] + "a"
|
||||
if number == "plural":
|
||||
a = _regular_plural(a)
|
||||
return a, ("rule" if (a != lemma or gender == "m") else "rule")
|
||||
|
||||
|
||||
# ── PUBLIC: clitic enclisis (dá + me + lo -> dámelo) ────────────────────────────
|
||||
def _strip_accents(s):
|
||||
return "".join(c for c in unicodedata.normalize("NFD", s)
|
||||
if unicodedata.category(c) != "Mn")
|
||||
|
||||
|
||||
def _count_syllables_vowelgroups(word):
|
||||
# crude: count vowel groups
|
||||
w = _strip_accents(word).lower()
|
||||
groups, prev = 0, False
|
||||
for ch in w:
|
||||
isv = ch in "aeiou"
|
||||
if isv and not prev:
|
||||
groups += 1
|
||||
prev = isv
|
||||
return groups
|
||||
|
||||
|
||||
def _host_stress_from_end(word):
|
||||
"""Stressed-syllable index counted from the end (1=last) of a verb host."""
|
||||
syls = _count_syllables_vowelgroups(word)
|
||||
if any(c in "áéíóú" for c in word):
|
||||
return None # already carries its own accent
|
||||
if word[-2:] in ("ar", "er", "ir"): # infinitive: oxytone
|
||||
return 1
|
||||
if word.endswith("ndo"): # gerund: paroxytone
|
||||
return 2
|
||||
if word[-1:] in "aeiouns" and syls >= 2: # default paroxytone
|
||||
return 2
|
||||
return 1 # monosyllable / consonant-final oxytone
|
||||
|
||||
|
||||
def attach_enclitics(verb_form, clitics):
|
||||
"""Append clitic pronouns to a verb (imperative/infinitive/gerund enclisis)
|
||||
and add a written accent when the resulting word becomes esdrújula/
|
||||
sobreesdrújula (stress >= 3 syllables from the end): dá+me+lo -> dámelo,
|
||||
lleva+me -> llévame, but dar+te -> darte and da+me -> dame (no accent)."""
|
||||
if not clitics:
|
||||
return verb_form
|
||||
tail = "".join(clitics)
|
||||
if any(c in "áéíóú" for c in verb_form): # host already accented
|
||||
return verb_form + tail
|
||||
sfe = _host_stress_from_end(verb_form)
|
||||
total_sfe = sfe + len(clitics) # each clitic = 1 syllable
|
||||
if total_sfe >= 3:
|
||||
return _accentuate_nucleus(verb_form, sfe) + tail
|
||||
return verb_form + tail
|
||||
|
||||
|
||||
def _accentuate_nucleus(word, sfe):
|
||||
"""Put a written accent on the syllable `sfe` positions from the word's end."""
|
||||
vowels = "aeiou"
|
||||
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
||||
if not nuclei or sfe > len(nuclei):
|
||||
return word
|
||||
i = nuclei[-sfe]
|
||||
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
||||
return word[:i] + acc[word[i]] + word[i + 1:]
|
||||
|
||||
|
||||
def _accentuate_last_stressed(word):
|
||||
# Restore the host's ORIGINAL lexical stress with a written accent.
|
||||
# Default Spanish stress: word ending in vowel/n/s -> penultimate syllable;
|
||||
# otherwise (e.g. infinitives in -r) -> last syllable.
|
||||
vowels = "aeiou"
|
||||
nuclei = [i for i, ch in enumerate(word) if ch in vowels]
|
||||
if not nuclei:
|
||||
return word
|
||||
if word[-1] in "aeiouns" and len(nuclei) >= 2:
|
||||
i = nuclei[-2] # paroxytone: penult nucleus
|
||||
else:
|
||||
i = nuclei[-1] # oxytone / monosyllable: last nucleus
|
||||
acc = {"a": "á", "e": "é", "i": "í", "o": "ó", "u": "ú"}
|
||||
return word[:i] + acc[word[i]] + word[i + 1:]
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"source": "UniMorph Spanish (github.com/unimorph/spa)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary-derived)",
|
||||
"total_forms": sum(len(v) for v in (_VERBS, _NOUNS, _ADJS)) if False else None,
|
||||
"verb_forms": len(_VERBS),
|
||||
"verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
"participles": len(_PART),
|
||||
"gerunds": len(_GER),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import json
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("hablar", "ind", "present", "first", "singular", "hablo"),
|
||||
("comer", "ind", "present", "third", "plural", "comen"),
|
||||
("vivir", "ind", "present", "first", "plural", "vivimos"),
|
||||
("ser", "ind", "present", "third", "singular", "es"),
|
||||
("ir", "ind", "preterite", "first", "singular", "fui"),
|
||||
("tener", "ind", "future", "first", "singular", "tendré"),
|
||||
("hacer", "sbjv", "present", "first", "singular", "haga"),
|
||||
("dormir", "ind", "present", "first", "singular", "duermo"),
|
||||
("pensar", "sbjv", "present", "third", "singular", "piense"),
|
||||
("dar", "ind", "preterite", "third", "singular", "dio"),
|
||||
("poner", "ind", "conditional", "first", "singular", "pondría"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
if got == exp:
|
||||
ok += 1
|
||||
print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]:3} -> {got:14} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender casa:", noun_gender("casa"), "| problema:", noun_gender("problema"),
|
||||
"| agua:", noun_gender("agua"), "| mano:", noun_gender("mano"))
|
||||
print(" plural: luz->", inflect_noun("luz", "plural"), "| rey->", inflect_noun("rey", "plural"))
|
||||
print(" adj: rojo/f/pl->", inflect_adj("rojo", "f", "plural"),
|
||||
"| feliz/m/pl->", inflect_adj("feliz", "m", "plural"),
|
||||
"| grande/f/pl->", inflect_adj("grande", "f", "plural"))
|
||||
print(" enclisis: da+[me,lo]->", attach_enclitics("da", ["me", "lo"]),
|
||||
"| di+[me]->", attach_enclitics("di", ["me"]),
|
||||
"| dar+[se,lo]->", attach_enclitics("dar", ["se", "lo"]))
|
||||
@@ -1,629 +0,0 @@
|
||||
"""morphology_fr_full.py — production-grade French morphological generator.
|
||||
|
||||
Same architecture as morphology_it_full.py (shared Romance engine); French-specific
|
||||
data and rules swapped in. Backed by three real, Wiktionary-lineage sources:
|
||||
|
||||
VERBS
|
||||
UniMorph French (github.com/unimorph/fra, CC-BY-SA 3.0)
|
||||
7,535 verb lemmas × full paradigm, CLEAN orthography:
|
||||
indicatif présent / imparfait (PST;IPFV) / passé simple (PST;PFV) /
|
||||
futur, conditionnel (COND), subjonctif présent (SBJV;PRS) /
|
||||
subjonctif imparfait (SBJV;PST), impératif (POS;IMP), infinitif (NFIN),
|
||||
participe présent (V.CVB/V.PTCP;PRS), participe passé (V.PTCP;PST, m.sg).
|
||||
fr_irreg_verbs.json — high-frequency verbs UniMorph MISSES or mis-slots,
|
||||
above all ÊTRE (absent from UniMorph fra), plus avoir/aller/faire/… — the
|
||||
auxiliaries the passé-composé + être-agreement system depends on. Extracted
|
||||
from kaikki.org French (build_fr_irreg.py), reflexive/multiword forms
|
||||
dropped. This layer takes PRIORITY.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org French (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender (head-template arg) + real plural
|
||||
(cheval->chevaux, œil->yeux, invariable -s/-x/-z), resolved PER LEMMA.
|
||||
adjective lemmas with real feminine + plural (petit->petite/petits/petites,
|
||||
beau->belle/beaux/belles, heureux->heureuse, rouge invariant-gender).
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV input):
|
||||
verbs : rule generator for -er / -ir(-iss-) / -re (with -cer/-ger spelling,
|
||||
future/conditional stems, imparfait/subjonctif endings)
|
||||
nouns : gender heuristic (endings) + rule pluralization (-al->-aux, -eau->-eaux)
|
||||
adjs : fem/plural agreement rules (-er->-ère, -eux->-euse, -f->-ve, +e default)
|
||||
|
||||
Confidence flag on every form: "lexicon" | "rule" | "fallback".
|
||||
|
||||
Public API (used by realizer_fr.py): identical signature to morphology_it_full.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "fra.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "fr_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_fr.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "fr_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set that must ALL be present ────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"}, # imparfait
|
||||
("ind", "passe_simple"): {"IND", "PST", "PFV"}, # passé simple
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"}, # French: V;COND;1;SG
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ─────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {}
|
||||
part = {}
|
||||
ger = {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
elif "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "passe_simple" and "IPFV" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "construed", "collective",
|
||||
"nonstandard", "dated", "Louisiana", "Switzerland", "Belgium"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {}
|
||||
adjs = {}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm not in ("#", "-", "—"):
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback ─────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("er"):
|
||||
return "er"
|
||||
if lemma.endswith("ir"):
|
||||
return "ir"
|
||||
if lemma.endswith("re"):
|
||||
return "re"
|
||||
if lemma.endswith("oir"):
|
||||
return "oir"
|
||||
return None
|
||||
|
||||
|
||||
# present-tense endings [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG_PRES = {
|
||||
"er": ["e", "es", "e", "ons", "ez", "ent"],
|
||||
"ir": ["is", "is", "it", "issons", "issez", "issent"], # -iss- class (finir)
|
||||
"re": ["s", "s", "", "ons", "ez", "ent"], # vendre: vends/vend
|
||||
}
|
||||
_REG_IMPF = ["ais", "ais", "ait", "ions", "iez", "aient"] # attaches to pres-1pl stem
|
||||
_REG_SUBJ = ["e", "es", "e", "ions", "iez", "ent"] # attaches to 3pl stem
|
||||
_REG_PS = { # passé simple
|
||||
"er": ["ai", "as", "a", "âmes", "âtes", "èrent"],
|
||||
"ir": ["is", "is", "it", "îmes", "îtes", "irent"],
|
||||
"re": ["is", "is", "it", "îmes", "îtes", "irent"],
|
||||
}
|
||||
_FUT = ["ai", "as", "a", "ons", "ez", "ont"]
|
||||
_COND = ["ais", "ais", "ait", "ions", "iez", "aient"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _fut_stem(lemma, vc):
|
||||
"""Future/conditional stem = infinitive (drop final -e of -re)."""
|
||||
if vc == "re":
|
||||
return lemma[:-1] # vendre -> vendr-
|
||||
return lemma # parler-, finir-
|
||||
|
||||
|
||||
def _pres_1pl_stem(lemma, vc):
|
||||
"""Imparfait stem = present 1pl minus -ons (parlons->parl-, finissons->finiss-)."""
|
||||
if vc == "er":
|
||||
stem = lemma[:-2]
|
||||
if stem.endswith("g"):
|
||||
return stem + "e" # mangeons -> mange- (imparfait mangeais)
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "ç" # commençons -> commenç-
|
||||
return stem
|
||||
if vc == "ir":
|
||||
return lemma[:-1] + "iss" # finir -> finiss-
|
||||
if vc == "re":
|
||||
return lemma[:-2] # vendre -> vend-
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
def _apply_er_spelling(stem, ending):
|
||||
"""-cer/-ger softening before a/o (commençons, mangeons)."""
|
||||
if ending and ending[0] in ("a", "o"):
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "ç" + ending
|
||||
if stem.endswith("g"):
|
||||
return stem + "e" + ending
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
i = _slot_idx(person, number)
|
||||
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
stem = _fut_stem(lemma, vc)
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
return stem + end
|
||||
|
||||
if mood == "ind" and tense == "present":
|
||||
table = _REG_PRES.get("ir" if vc == "ir" else vc)
|
||||
if not table:
|
||||
return None
|
||||
body = lemma[:-2] if vc in ("er", "re") else lemma[:-1] if vc == "ir" else lemma[:-2]
|
||||
if vc == "ir":
|
||||
body = lemma[:-2] # fin- ; endings carry -iss-
|
||||
end = table[i]
|
||||
return body + end
|
||||
end = table[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(body, end)
|
||||
return body + end
|
||||
|
||||
if mood == "ind" and tense == "imperfect":
|
||||
stem = _pres_1pl_stem(lemma, vc)
|
||||
return stem + _REG_IMPF[i]
|
||||
|
||||
if mood == "ind" and tense == "passe_simple":
|
||||
table = _REG_PS.get("ir" if vc == "ir" else vc)
|
||||
if not table:
|
||||
return None
|
||||
body = lemma[:-2] if vc in ("er", "re") else lemma[:-2]
|
||||
end = table[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(body, end)
|
||||
return body + end
|
||||
|
||||
if mood == "sbjv" and tense == "present":
|
||||
# subjonctif: present-3pl stem + e/es/e/ions/iez/ent
|
||||
stem3 = _pres_1pl_stem(lemma, vc) if vc == "ir" else (
|
||||
lemma[:-2] if vc in ("er", "re") else lemma[:-2])
|
||||
if vc == "ir":
|
||||
stem3 = lemma[:-2] + "iss"
|
||||
end = _REG_SUBJ[i]
|
||||
if vc == "er":
|
||||
return _apply_er_spelling(stem3, end)
|
||||
return stem3 + end
|
||||
|
||||
if mood == "imp" and tense == "affirmative":
|
||||
# impératif ~ present indicative (tu drops -s for -er verbs)
|
||||
pres = _rule_conjugate(lemma, "ind", "present", person, number)
|
||||
if pres and vc == "er" and person == "second" and number == "singular":
|
||||
return pres[:-1] if pres.endswith("es") else pres
|
||||
return pres
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence)."""
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number}"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund/participe présent ────────────────────────────────
|
||||
def _participle_msg(lemma):
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
return None, None
|
||||
|
||||
|
||||
# irregular participle fem/plural quirks (drop circonflexe: dû->due, dus)
|
||||
_PART_FIX = {"dû": {"f|SG": "due", "m|PL": "dus", "f|PL": "dues"}}
|
||||
|
||||
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
"""Past participle with French gender/number agreement.
|
||||
m.sg = base; f.sg = base+e; m.pl = base+s (invariable if base ends s/x);
|
||||
f.pl = f.sg+s."""
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
msg, src = _participle_msg(lemma)
|
||||
conf = "lexicon"
|
||||
if msg is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "er":
|
||||
msg = lemma[:-2] + "é"
|
||||
elif vc == "ir":
|
||||
msg = lemma[:-1] # finir -> fini, partir -> parti
|
||||
elif vc == "re":
|
||||
msg = lemma[:-2] + "u" # vendre -> vendu
|
||||
elif vc == "oir":
|
||||
msg = lemma[:-3] + "u" # (rough) recevoir handled by irreg
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
fix = _PART_FIX.get(msg)
|
||||
if fix and f"{g}|{num}" in fix:
|
||||
return fix[f"{g}|{num}"], conf
|
||||
if g == "m" and num == "SG":
|
||||
return msg, conf
|
||||
fem = msg + "e" if not msg.endswith("e") else msg
|
||||
if g == "f" and num == "SG":
|
||||
return fem, conf
|
||||
if g == "m" and num == "PL":
|
||||
return msg if msg.endswith(("s", "x")) else msg + "s", conf
|
||||
# f|PL
|
||||
return fem + "s", conf
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
"""Participe présent (base for gérondif 'en -ant')."""
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "er":
|
||||
stem = lemma[:-2]
|
||||
if stem.endswith("g"):
|
||||
return stem + "eant", "rule"
|
||||
if stem.endswith("c"):
|
||||
return stem[:-1] + "çant", "rule"
|
||||
return stem + "ant", "rule"
|
||||
if vc == "ir":
|
||||
return lemma[:-2] + "issant", "rule"
|
||||
if vc == "re":
|
||||
return lemma[:-2] + "ant", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("tion", "sion", "aison", "ance", "ence", "ette", "elle", "esse",
|
||||
"ude", "ade", "ée", "té", "tié", "ie", "ise", "ure", "eur")
|
||||
_MASC_SUF = ("ment", "age", "eau", "isme", "oir", "ier", "eur", "in", "on")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
for suf in _MASC_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "m"
|
||||
if noun.endswith("e"):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
# closed sets for French plural irregularities
|
||||
_OU_X = {"bijou", "caillou", "chou", "genou", "hibou", "joujou", "pou"}
|
||||
_AIL_AUX = {"travail", "vitrail", "corail", "émail", "bail", "soupirail", "vantail"}
|
||||
_AL_S = {"bal", "carnaval", "festival", "récital", "chacal", "régal", "cal", "aval"}
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic French pluralization. (form, ok); ok=False FLAGS ambiguity."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
if noun[-1:] in ("s", "x", "z"):
|
||||
return noun, True # invariable
|
||||
if noun in _OU_X:
|
||||
return noun + "x", True
|
||||
if noun.endswith(("eau", "au", "eu")):
|
||||
if noun in ("pneu", "bleu", "landau", "sarrau"):
|
||||
return noun + "s", True
|
||||
return noun + "x", True # bateau->bateaux, jeu->jeux
|
||||
if noun.endswith("al"):
|
||||
if noun in _AL_S:
|
||||
return noun + "s", True
|
||||
return noun[:-2] + "aux", True # cheval->chevaux
|
||||
if noun.endswith("ail"):
|
||||
if noun in _AIL_AUX:
|
||||
return noun[:-3] + "aux", True # travail->travaux
|
||||
return noun + "s", True
|
||||
return noun + "s", True # default
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# adjectives whose kaikki entries are unreliable: audited forms
|
||||
_ADJ_FIX = {
|
||||
"beau": {("m", "SG"): "beau", ("f", "SG"): "belle",
|
||||
("m", "PL"): "beaux", ("f", "PL"): "belles"},
|
||||
"nouveau": {("m", "SG"): "nouveau", ("f", "SG"): "nouvelle",
|
||||
("m", "PL"): "nouveaux", ("f", "PL"): "nouvelles"},
|
||||
"vieux": {("m", "SG"): "vieux", ("f", "SG"): "vieille",
|
||||
("m", "PL"): "vieux", ("f", "PL"): "vieilles"},
|
||||
"fou": {("m", "SG"): "fou", ("f", "SG"): "folle",
|
||||
("m", "PL"): "fous", ("f", "PL"): "folles"},
|
||||
"blanc": {("m", "SG"): "blanc", ("f", "SG"): "blanche",
|
||||
("m", "PL"): "blancs", ("f", "PL"): "blanches"},
|
||||
"long": {("m", "SG"): "long", ("f", "SG"): "longue",
|
||||
("m", "PL"): "longs", ("f", "PL"): "longues"},
|
||||
"bon": {("m", "SG"): "bon", ("f", "SG"): "bonne",
|
||||
("m", "PL"): "bons", ("f", "PL"): "bonnes"},
|
||||
}
|
||||
|
||||
|
||||
def _rule_fem(a):
|
||||
if a.endswith("e"):
|
||||
return a
|
||||
if a.endswith("er"):
|
||||
return a[:-2] + "ère"
|
||||
if a.endswith("eau"):
|
||||
return a[:-3] + "elle"
|
||||
if a.endswith("eux"):
|
||||
return a[:-3] + "euse"
|
||||
if a.endswith("f"):
|
||||
return a[:-1] + "ve"
|
||||
if a.endswith(("on", "en", "el", "eil", "et")):
|
||||
return a + a[-1] + "e" # bon->bonne, ancien->ancienne, muet->muette
|
||||
if a.endswith("c"):
|
||||
return a[:-1] + "che" # blanc->blanche (public->publique via FIX)
|
||||
return a + "e" # grand->grande, petit->petite, vert->verte
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
fix = _ADJ_FIX.get(lemma)
|
||||
if fix and (g, num) in fix:
|
||||
return fix[(g, num)], "lexicon"
|
||||
d = _ADJS.get(lemma)
|
||||
if d and d.get((g, num)):
|
||||
return d[(g, num)], "lexicon"
|
||||
# derive
|
||||
msc = (d.get(("m", "SG")) if d else None) or lemma
|
||||
if g == "m" and num == "SG":
|
||||
return msc, "lexicon" if d else "rule"
|
||||
fem = (d.get(("f", "SG")) if d else None) or _rule_fem(msc)
|
||||
if g == "f" and num == "SG":
|
||||
return fem, "lexicon" if (d and d.get(("f", "SG"))) else "rule"
|
||||
if g == "m" and num == "PL":
|
||||
if msc.endswith(("s", "x")):
|
||||
return msc, "rule"
|
||||
if msc.endswith("al"):
|
||||
return msc[:-2] + "aux", "rule"
|
||||
if msc.endswith("eau"):
|
||||
return msc + "x", "rule"
|
||||
return msc + "s", "rule"
|
||||
# f|PL
|
||||
return (fem if fem.endswith("s") else fem + "s"), "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph French (github.com/unimorph/fra) + kaikki.org "
|
||||
"irregulars (être + high-frequency)",
|
||||
"noun_adj_source": "kaikki.org French (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREGV),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("parler", "ind", "present", "first", "singular", "parle"),
|
||||
("être", "ind", "present", "third", "singular", "est"),
|
||||
("avoir", "ind", "present", "first", "singular", "ai"),
|
||||
("aller", "ind", "present", "third", "plural", "vont"),
|
||||
("finir", "ind", "present", "first", "singular", "finis"),
|
||||
("finir", "ind", "present", "first", "plural", "finissons"),
|
||||
("manger", "ind", "present", "first", "plural", "mangeons"),
|
||||
("faire", "ind", "future", "first", "singular", "ferai"),
|
||||
("pouvoir", "sbjv", "present", "third", "singular", "puisse"),
|
||||
("prendre", "ind", "passe_simple", "third", "singular", "prit"),
|
||||
("vendre", "ind", "present", "third", "singular", "vend"),
|
||||
("commencer", "ind", "imperfect", "first", "singular", "commençais"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:10} {mood}/{tense:12} {per[:3]}.{num[:2]} -> {got:12} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: maison=", noun_gender("maison"), "chat=", noun_gender("chat"),
|
||||
"cheval=", noun_gender("cheval"), "nation=", noun_gender("nation"))
|
||||
print(" plural: cheval->", inflect_noun("cheval", "plural"),
|
||||
"| bateau->", inflect_noun("bateau", "plural"),
|
||||
"| prix->", inflect_noun("prix", "plural"),
|
||||
"| chat->", inflect_noun("chat", "plural"))
|
||||
print(" adj: petit/f/sg->", inflect_adj("petit", "f", "singular"),
|
||||
"| beau/f/sg->", inflect_adj("beau", "f", "singular"),
|
||||
"| heureux/f/sg->", inflect_adj("heureux", "f", "singular"),
|
||||
"| national/m/pl->", inflect_adj("national", "m", "plural"))
|
||||
print(" part: aller/f/sg->", participle("aller", "f", "singular"),
|
||||
"| prendre/f/pl->", participle("prendre", "f", "plural"),
|
||||
"| finir/m/pl->", participle("finir", "m", "plural"))
|
||||
print(" ger: manger->", gerund("manger"), "| finir->", gerund("finir"))
|
||||
@@ -1,588 +0,0 @@
|
||||
"""morphology_it_full.py — production-grade Italian morphological generator.
|
||||
|
||||
NOT a toy. Backed by three real, Wiktionary-lineage lexical sources:
|
||||
|
||||
VERBS
|
||||
UniMorph Italian (github.com/unimorph/ita, CC-BY-SA 3.0)
|
||||
10,009 verb lemmas × full paradigm, CLEAN orthography (no stress marks):
|
||||
indicative present / imperfetto (PST;IPFV) / passato remoto (PST;PFV) /
|
||||
futuro, condizionale (COND),
|
||||
congiuntivo presente (SBJV;PRS) / imperfetto (SBJV;PST),
|
||||
affirmative imperative, infinitive, gerundio (V.CVB;PRS),
|
||||
past participle (masc-sg; fem/plural derived by vowel rule).
|
||||
it_irreg_verbs.json — 66 high-frequency verbs UniMorph MISSES
|
||||
(essere, avere, potere, uscire, tenere, prendere, piacere, …), extracted
|
||||
from kaikki.org Italian, filtered to standard forms, and DE-STRESSED to
|
||||
real orthography (kaikki marks tonic stress everywhere: pàrlo->parlo,
|
||||
avùto->avuto; final legit accents kept: sarò, è). Built by build_it_irreg.py.
|
||||
This layer takes priority — it supplies the two auxiliaries essere/avere,
|
||||
which the whole passato-prossimo / essere-agreement system depends on.
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Italian (Wiktionary extract, CC-BY-SA 3.0)
|
||||
noun lemmas WITH inherent gender (head-template arg) + real (often irregular)
|
||||
plural — uomo->uomini, uovo->uova, dito->dita, città invariant — resolved
|
||||
PER LEMMA, never guessed.
|
||||
adjective lemmas with real feminine + masc/fem plural (italiano->italiana/
|
||||
italiani/italiane, felice->felici invariant).
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV input):
|
||||
verbs : rule generator for regular -are/-ere/-ire (with -care/-gare h-insertion
|
||||
and -ciare/-giare/-iare i-drop spelling rules)
|
||||
nouns : gender heuristic (endings) + rule pluralization (ambiguous -co/-go FLAGGED)
|
||||
adjs : -o/-a/-e gender rule + rule pluralization
|
||||
|
||||
Confidence flag on every form:
|
||||
"lexicon" from UniMorph / kaikki-irregular / kaikki noun-adj (trust: high)
|
||||
"rule" deterministic rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma / ambiguous (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_it.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
participle(lemma, gender="m", number="singular") -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "ita.unimorph")
|
||||
_IRREG = os.path.join(_HERE, "data", "it_irreg_verbs.json")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_it.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "it_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set that must ALL be present ────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "passato_remoto"): {"IND", "PST", "PFV"},
|
||||
("ind", "future"): {"IND", "FUT"},
|
||||
("ind", "conditional"): {"COND"},
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("sbjv", "imperfect"): {"SBJV", "PST"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ─────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs = {} # (lemma, "mood|tense|person|number") -> form
|
||||
part = {} # lemma -> masc-sg past participle
|
||||
ger = {} # lemma -> gerundio
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB": # gerundio (converb, present)
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
# exact-set discipline: PST;PFV must not match PST;IPFV, etc.
|
||||
if not req <= f:
|
||||
continue
|
||||
# guard IND;PST ambiguity: require the specific aspect feature
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "passato_remoto" and "IPFV" in f:
|
||||
continue
|
||||
# COND must not also be a subjunctive/imperative slot
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), form)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns + adjectives ────────────────────────────────────────────────────
|
||||
_EXCL_FORM_TAGS = {"alternative", "archaic", "obsolete", "dialectal", "regional",
|
||||
"diminutive", "augmentative", "pejorative", "comparative",
|
||||
"superlative", "misspelling", "rare", "informal", "literary",
|
||||
"poetic", "error-unrecognized-form", "apocopic", "obsolete",
|
||||
"construed", "collective"}
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = str(arg).lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {} # lemma -> {"g","SG","PL"}
|
||||
adjs = {} # lemma -> {("m","SG"),("f","SG"),("m","PL"),("f","PL")}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
if "plural" in t and not (t & _EXCL_FORM_TAGS):
|
||||
fm = x.get("form")
|
||||
if fm and " " not in fm and fm != "#":
|
||||
pl = fm
|
||||
break
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or " " in fm or (t & _EXCL_FORM_TAGS):
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = d0.get(("f", "SG")) or fm
|
||||
elif "plural" in t: # invariant-gender adj (felice -> felici)
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
with open(_IRREG, encoding="utf-8") as fh:
|
||||
irreg = json.load(fh)
|
||||
data = {"verbs": verbs, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs, "irreg": irreg}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI, _IRREG]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS, _IRREGV = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"],
|
||||
_LEX["irreg"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback ─────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("are"):
|
||||
return "are"
|
||||
if lemma.endswith("ere"):
|
||||
return "ere"
|
||||
if lemma.endswith("ire"):
|
||||
return "ire"
|
||||
return None
|
||||
|
||||
|
||||
# endings [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG = {
|
||||
("ind", "present", "are"): ["o", "i", "a", "iamo", "ate", "ano"],
|
||||
("ind", "present", "ere"): ["o", "i", "e", "iamo", "ete", "ono"],
|
||||
("ind", "present", "ire"): ["o", "i", "e", "iamo", "ite", "ono"],
|
||||
("ind", "imperfect", "are"): ["avo", "avi", "ava", "avamo", "avate", "avano"],
|
||||
("ind", "imperfect", "ere"): ["evo", "evi", "eva", "evamo", "evate", "evano"],
|
||||
("ind", "imperfect", "ire"): ["ivo", "ivi", "iva", "ivamo", "ivate", "ivano"],
|
||||
("ind", "passato_remoto", "are"): ["ai", "asti", "ò", "ammo", "aste", "arono"],
|
||||
("ind", "passato_remoto", "ere"): ["ei", "esti", "é", "emmo", "este", "erono"],
|
||||
("ind", "passato_remoto", "ire"): ["ii", "isti", "ì", "immo", "iste", "irono"],
|
||||
("sbjv", "present", "are"): ["i", "i", "i", "iamo", "iate", "ino"],
|
||||
("sbjv", "present", "ere"): ["a", "a", "a", "iamo", "iate", "ano"],
|
||||
("sbjv", "present", "ire"): ["a", "a", "a", "iamo", "iate", "ano"],
|
||||
("sbjv", "imperfect", "are"): ["assi", "assi", "asse", "assimo", "aste", "assero"],
|
||||
("sbjv", "imperfect", "ere"): ["essi", "essi", "esse", "essimo", "este", "essero"],
|
||||
("sbjv", "imperfect", "ire"): ["issi", "issi", "isse", "issimo", "iste", "issero"],
|
||||
# imperative: 2sg,3sg(Lei),1pl,2pl,3pl (1sg has none)
|
||||
("imp", "affirmative", "are"): [None, "a", "i", "iamo", "ate", "ino"],
|
||||
("imp", "affirmative", "ere"): [None, "i", "a", "iamo", "ete", "ano"],
|
||||
("imp", "affirmative", "ire"): [None, "i", "a", "iamo", "ite", "ano"],
|
||||
}
|
||||
# future / conditional attach to a stem = infinitive minus final -e, with
|
||||
# -are -> -er (parlare->parler-), -ere/-ire keep (credere->creder-, dormir-)
|
||||
_FUT = ["ò", "ai", "à", "emo", "ete", "anno"]
|
||||
_COND = ["ei", "esti", "ebbe", "emmo", "este", "ebbero"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _fut_stem(lemma, vc):
|
||||
body = lemma[:-3] # drop are/ere/ire
|
||||
if vc == "are":
|
||||
return body + "er"
|
||||
return body + vc[0] + "r" # ere->er? no: keep vowel: creder-, dormir-
|
||||
# NOTE corrected below
|
||||
|
||||
|
||||
def _apply_are_spelling(stem, ending):
|
||||
"""-care/-gare insert h before front endings; -ciare/-giare/-sciare/-iare drop i."""
|
||||
front = ending[:1] in ("i", "e")
|
||||
if stem.endswith(("c", "g")) and front:
|
||||
return stem + "h" + ending
|
||||
if stem.endswith(("ci", "gi", "sci")) and ending[:1] == "i":
|
||||
return stem[:-1] + ending # mangi+iamo -> mangiamo
|
||||
if stem.endswith("i") and ending[:1] == "i":
|
||||
return stem[:-1] + ending # studi+iamo -> studiamo
|
||||
return stem + ending
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
body = lemma[:-3]
|
||||
i = _slot_idx(person, number)
|
||||
if mood == "ind" and tense in ("future", "conditional"):
|
||||
stem = body + "er" if vc == "are" else body + vc[0] + "r"
|
||||
# ere: creder-, ire: dormir- -> body + 'e'/'i' + 'r'
|
||||
if vc == "ere":
|
||||
stem = body + "er"
|
||||
elif vc == "ire":
|
||||
stem = body + "ir"
|
||||
end = (_FUT if tense == "future" else _COND)[i]
|
||||
# spelling: -care/-gare -> cherò/gherò ; -ciare/-giare -> cerò/gerò
|
||||
if vc == "are":
|
||||
if body.endswith(("c", "g")):
|
||||
stem = body + "her"
|
||||
elif body.endswith(("ci", "gi", "sci")):
|
||||
stem = body[:-1] + "er"
|
||||
elif body.endswith("i"):
|
||||
stem = body[:-1] + "er"
|
||||
return stem + end
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if not table:
|
||||
return None
|
||||
end = table[i]
|
||||
if end is None:
|
||||
return None
|
||||
if vc == "are":
|
||||
return _apply_are_spelling(body, end)
|
||||
# -ere/-ire: guard against double-i (dormi+iamo -> dormiamo)
|
||||
if body.endswith("i") and end[:1] == "i":
|
||||
return body[:-1] + end
|
||||
return body + end
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{number}"
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ──────────────────────────────────────────────────
|
||||
def _participle_msg(lemma):
|
||||
"""Return (masc-sg participle, source) or (None, None)."""
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
return None, None
|
||||
|
||||
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
"""Past participle with gender/number agreement (for essere-perfect & passives).
|
||||
UniMorph/irregular give masc-sg; fem/plural derived by final-vowel swap
|
||||
(-o -> -a/-i/-e), valid for regular -ato/-uto/-ito AND irregulars
|
||||
(preso->presa/presi/prese, aperto->aperta/aperti/aperte, morto->morta/...)."""
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
msg, src = _participle_msg(lemma)
|
||||
conf = "lexicon"
|
||||
if msg is None:
|
||||
vc = _vclass(lemma)
|
||||
if vc == "are":
|
||||
msg = lemma[:-3] + "ato"
|
||||
elif vc == "ere":
|
||||
msg = lemma[:-3] + "uto"
|
||||
elif vc == "ire":
|
||||
msg = lemma[:-3] + "ito"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
conf = "rule"
|
||||
# agreement: only -o participles inflect for gender+number
|
||||
if msg.endswith("o"):
|
||||
stem = msg[:-1]
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "i", "f|PL": "e"}[f"{g}|{num}"]
|
||||
return stem + suf, conf
|
||||
return msg, conf # non -o participle: leave as-is (rare)
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREGV.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "are":
|
||||
return lemma[:-3] + "ando", "rule"
|
||||
if vc in ("ere", "ire"):
|
||||
return lemma[:-3] + "endo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("zione", "sione", "gione", "tà", "tù", "trice", "aggine", "udine",
|
||||
"igine", "ie", "essa", "izia", "ezza")
|
||||
_MASC_SUF = ("ore", "ame", "iere", "ale", "ile")
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
for suf in _MASC_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "m"
|
||||
if noun.endswith("o"):
|
||||
return "m"
|
||||
if noun.endswith("a"):
|
||||
return "f"
|
||||
if noun.endswith("à") or noun.endswith("ù"):
|
||||
return "f"
|
||||
return "m" # -e and consonant-final loanwords default masculine
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
def _rule_plural(noun, gender):
|
||||
"""Deterministic Italian pluralization. Returns (form, ok); ok=False FLAGS an
|
||||
ambiguous case the lexicon would normally resolve (-co/-go palatalization)."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
# invariant: accented final vowel, consonant-final, monosyllable, -i final
|
||||
if noun[-1:] in ("à", "è", "é", "ì", "í", "ò", "ó", "ù", "ú"):
|
||||
return noun, True
|
||||
if noun[-1:] not in ("a", "e", "o", "i", "u"):
|
||||
return noun, True # consonant-final loanword: invariant
|
||||
if noun.endswith("i"):
|
||||
return noun, True # e.g. crisi, analisi: invariant
|
||||
if noun.endswith("io"):
|
||||
return noun[:-2] + "i", True # figlio->figli (unstressed i)
|
||||
if noun.endswith("cia") or noun.endswith("gia"):
|
||||
# vowel before cia/gia -> -cie/-gie ; consonant -> -ce/-ge (approx)
|
||||
return noun[:-2] + "e", True # arancia->arance (majority)
|
||||
if noun.endswith("ca"):
|
||||
return noun[:-2] + "che", True # amica->amiche
|
||||
if noun.endswith("ga"):
|
||||
return noun[:-2] + "ghe", True
|
||||
if noun.endswith("co"):
|
||||
return noun[:-2] + "chi", False # AMBIGUOUS (amico->amici) -> flag
|
||||
if noun.endswith("go"):
|
||||
return noun[:-2] + "ghi", False # AMBIGUOUS (psicologo->psicologi)
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "e", True # casa->case (m -a: -i, but rare)
|
||||
if noun.endswith("o"):
|
||||
return noun[:-1] + "i", True # libro->libri
|
||||
if noun.endswith("e"):
|
||||
return noun[:-1] + "i", True # cane->cani, chiave->chiavi
|
||||
return noun, True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
g = gender or noun_gender(lemma)
|
||||
form, ok = _rule_plural(lemma, g)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# adjectives whose kaikki entries are unreliable (messy inflection templates):
|
||||
# supply audited regular agreement forms (prenominal apocope handled in realizer).
|
||||
_ADJ_FIX = {
|
||||
"bello": {("m", "SG"): "bello", ("f", "SG"): "bella",
|
||||
("m", "PL"): "belli", ("f", "PL"): "belle"},
|
||||
"quello": {("m", "SG"): "quello", ("f", "SG"): "quella",
|
||||
("m", "PL"): "quelli", ("f", "PL"): "quelle"},
|
||||
}
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
fix = _ADJ_FIX.get(lemma)
|
||||
if fix and (g, num) in fix:
|
||||
return fix[(g, num)], "lexicon"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg, g)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback
|
||||
a = lemma
|
||||
if a.endswith("o"): # -o/-a/-i/-e class
|
||||
base = a[:-1]
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "i", "f|PL": "e"}[f"{g}|{num}"]
|
||||
return base + suf, "rule"
|
||||
if a.endswith("e"): # felice-class: SG invariant, PL -i
|
||||
if num == "PL":
|
||||
return a[:-1] + "i", "rule"
|
||||
return a, "rule"
|
||||
if num == "PL":
|
||||
p, ok = _rule_plural(a, g)
|
||||
return p, ("rule" if ok else "fallback")
|
||||
return a, "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Italian (github.com/unimorph/ita) + kaikki.org "
|
||||
"irregulars (de-stressed)",
|
||||
"noun_adj_source": "kaikki.org Italian (Wiktionary extract)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREGV),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("parlare", "ind", "present", "first", "singular", "parlo"),
|
||||
("essere", "ind", "present", "third", "singular", "è"),
|
||||
("avere", "ind", "present", "first", "singular", "ho"),
|
||||
("mangiare", "ind", "present", "second", "singular", "mangi"),
|
||||
("finire", "ind", "present", "first", "singular", "finisco"),
|
||||
("andare", "ind", "present", "third", "plural", "vanno"),
|
||||
("fare", "ind", "future", "first", "singular", "farò"),
|
||||
("potere", "sbjv", "present", "third", "singular", "possa"),
|
||||
("prendere", "ind", "passato_remoto", "first", "singular", "presi"),
|
||||
("cercare", "ind", "present", "second", "singular", "cerchi"),
|
||||
("dormire", "ind", "present", "third", "plural", "dormono"),
|
||||
("credere", "ind", "future", "first", "singular", "crederò"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:9} {mood}/{tense:14} {per[:3]}.{num[:2]} -> {got:12} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: casa=", noun_gender("casa"), "problema=", noun_gender("problema"),
|
||||
"mano=", noun_gender("mano"), "città=", noun_gender("città"),
|
||||
"cane=", noun_gender("cane"))
|
||||
print(" plural: uomo->", inflect_noun("uomo", "plural"),
|
||||
"| uovo->", inflect_noun("uovo", "plural"),
|
||||
"| città->", inflect_noun("città", "plural"),
|
||||
"| amico->", inflect_noun("amico", "plural"),
|
||||
"| casa->", inflect_noun("casa", "plural"))
|
||||
print(" adj: italiano/f/pl->", inflect_adj("italiano", "f", "plural"),
|
||||
"| felice/m/pl->", inflect_adj("felice", "m", "plural"),
|
||||
"| bello/f/sg->", inflect_adj("bello", "f", "singular"))
|
||||
print(" part: aprire/f/sg->", participle("aprire", "f", "singular"),
|
||||
"| prendere/m/pl->", participle("prendere", "m", "plural"),
|
||||
"| andare/f/sg->", participle("andare", "f", "singular"))
|
||||
print(" ger: fare->", gerund("fare"), "| parlare->", gerund("parlare"))
|
||||
@@ -1,666 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_lat_full.py — production-grade Latin morphological generator.
|
||||
|
||||
Latin is the FLAGSHIP dead-language realizer. It rides the *architecture* of the
|
||||
Romance/Italic engine (the same Realization / spec-driven design and the UniMorph
|
||||
loader pattern from morphology_it_full.py) but with the CASE SYSTEM RESTORED —
|
||||
the feature Romance lost. Latin therefore exercises machinery the modern Romance
|
||||
siblings never needed: 5 declensions x 6 cases x 2 numbers x 3 genders, plus a
|
||||
4-conjugation verb system with tense/mood/voice.
|
||||
|
||||
DATA (real, attested — no fabrication):
|
||||
|
||||
NOUNS + ADJECTIVES — UniMorph Latin (github.com/unimorph/lat, CC-BY-SA 3.0)
|
||||
163,182 N forms across ~thousands of lemmas, each with the full case paradigm
|
||||
N;NOM/GEN/DAT/ACC/ABL/VOC;SG/PL (real inflected forms, WITH macrons:
|
||||
puella->puellam, rēx->rēgis, corpus->corporis).
|
||||
244,197 ADJ forms with case x GENDER x number, incl. UniMorph's combined
|
||||
tags (GEN+DAT, MASC+FEM, MASC+FEM+NEUT) which are split on load.
|
||||
462,668 V.PTCP forms (participles) also carry case/gender/number.
|
||||
UniMorph N tags DO NOT encode inherent gender, so noun gender is inferred
|
||||
from the declension (nom-sg + gen-sg endings) with a curated exceptions
|
||||
map — the standard, attestable rule (1st decl -a/-ae = fem, 2nd -us/-i =
|
||||
masc, -um = neut, ...).
|
||||
|
||||
VERBS — RULE ENGINE (honest gap: UniMorph Latin's verb list is a 947-lemma
|
||||
sample of rare/prefixed verbs that MISSES every core textbook verb — amō,
|
||||
videō, sum, regō, ... are all absent). Latin conjugation is, however, highly
|
||||
regular, so verbs are generated by a deterministic 4-conjugation engine over
|
||||
curated principal parts (present / perfect / supine stems), sourced from
|
||||
standard references. Irregulars (sum, possum, eō, ferō, volō, nōlō, mālō)
|
||||
are curated full tables. Forms are flagged "rule" (not "lexicon") for honesty.
|
||||
|
||||
Confidence flag on every form (same contract as the Romance engine):
|
||||
"lexicon" from UniMorph (trust: high)
|
||||
"rule" deterministic morphology rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_lat.py):
|
||||
decline_noun(lemma, case, number) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"|"n"
|
||||
decline_adj(lemma, case, gender, number) -> (form, conf)
|
||||
conjugate(lemma, tense, mood, voice, person, number) -> (form, conf)
|
||||
participle(lemma, kind, case, gender, number) -> (form, conf) # kind: prs|pfv|fut
|
||||
infinitive(lemma, tense="present", voice="active") -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "lat.unimorph")
|
||||
_CACHE = os.path.join(_HERE, "data", "lat_morph_cache.pkl")
|
||||
|
||||
_CASES = ("NOM", "GEN", "DAT", "ACC", "ABL", "VOC")
|
||||
_CASE_MAP = {"nom": "NOM", "gen": "GEN", "dat": "DAT", "acc": "ACC",
|
||||
"abl": "ABL", "voc": "VOC"}
|
||||
_NUM = {"singular": "SG", "plural": "PL"}
|
||||
_GEN = {"m": "MASC", "f": "FEM", "n": "NEUT"}
|
||||
|
||||
|
||||
# ── UniMorph loader: noun + adjective + participle case paradigms ────────────────
|
||||
def _build_cache():
|
||||
nouns = {} # lemma -> {(CASE, NUM): form}
|
||||
adjs = {} # lemma -> {(CASE, GEN, NUM): form}
|
||||
ptcps = {} # lemma -> {(CASE, GEN, NUM): form} (from V.PTCP; keyed loosely)
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
feats = tag.split(";")
|
||||
head = feats[0]
|
||||
fs = set(feats)
|
||||
case = next((c for c in _CASES if c in fs), None)
|
||||
# handle combined case tags like GEN+DAT
|
||||
if case is None:
|
||||
for f in feats:
|
||||
if "+" in f and any(c in f.split("+") for c in _CASES):
|
||||
case = [c for c in _CASES if c in f.split("+")]
|
||||
break
|
||||
num = "SG" if "SG" in fs else ("PL" if "PL" in fs else None)
|
||||
if case is None or num is None:
|
||||
continue
|
||||
cases = case if isinstance(case, list) else [case]
|
||||
|
||||
if head == "N":
|
||||
d = nouns.setdefault(lemma, {})
|
||||
for c in cases:
|
||||
d.setdefault((c, num), form)
|
||||
elif head == "ADJ":
|
||||
# gender may be combined: MASC+FEM+NEUT, MASC+FEM
|
||||
genders = []
|
||||
for g in ("MASC", "FEM", "NEUT"):
|
||||
if any(g == x or (g in x.split("+")) for x in feats):
|
||||
genders.append(g)
|
||||
if not genders:
|
||||
genders = ["MASC", "FEM", "NEUT"]
|
||||
d = adjs.setdefault(lemma, {})
|
||||
for c in cases:
|
||||
for g in genders:
|
||||
d.setdefault((c, g, num), form)
|
||||
data = {"nouns": nouns, "adjs": adjs, "ptcps": ptcps}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE) and os.path.exists(_UNIMORPH):
|
||||
if os.path.getmtime(_CACHE) >= os.path.getmtime(_UNIMORPH):
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_NOUNS, _ADJS = _LEX["nouns"], _LEX["adjs"]
|
||||
|
||||
|
||||
# ── noun gender inference (declension-based, curated exceptions) ─────────────────
|
||||
# Real, attestable rule: gender follows declension + nominative shape, with the
|
||||
# standard closed set of exceptions.
|
||||
_GENDER_EXC = {
|
||||
# 1st-declension masculines (people/agents)
|
||||
"agricola": "m", "poēta": "m", "nauta": "m", "incola": "m", "scrība": "m",
|
||||
"auriga": "m", "pīrāta": "m", "athlēta": "m",
|
||||
# 2nd-declension neuters / feminines
|
||||
"vīrus": "n", "vulgus": "n", "pelagus": "n", "humus": "f",
|
||||
# common 3rd-declension whose gender the ending would mispredict
|
||||
"rēx": "m", "dux": "m", "mīles": "m", "pater": "m", "frāter": "m",
|
||||
"homō": "m", "leō": "m", "sōl": "m", "mōns": "m", "pōns": "m", "fōns": "m",
|
||||
"sanguis": "m", "ōrdō": "m", "sermō": "m", "amor": "m", "dolor": "m",
|
||||
"labor": "m", "timor": "m", "honor": "m", "color": "m", "pēs": "m",
|
||||
"dēns": "m", "flōs": "m", "mōs": "m", "mensis": "m", "orbis": "m",
|
||||
"piscis": "m", "ignis": "m", "collis": "m", "grex": "m", "prīnceps": "m",
|
||||
"māter": "f", "soror": "f", "uxor": "f", "mulier": "f", "virgō": "f",
|
||||
"urbs": "f", "arx": "f", "pāx": "f", "lēx": "f", "lūx": "f", "vōx": "f",
|
||||
"nox": "f", "nix": "f", "vīs": "f", "salūs": "f", "virtūs": "f",
|
||||
"aetās": "f", "cīvitās": "f", "lībertās": "f", "vēritās": "f", "voluptās": "f",
|
||||
"nātiō": "f", "ratiō": "f", "ōrātiō": "f", "legiō": "f", "regiō": "f",
|
||||
"mens": "f", "gens": "f", "ars": "f", "pars": "f", "mors": "f", "sors": "f",
|
||||
"nāvis": "f", "turris": "f", "avis": "f", "vallis": "f", "classis": "f",
|
||||
"corpus": "n", "tempus": "n", "opus": "n", "genus": "n", "onus": "n",
|
||||
"pectus": "n", "latus": "n", "vulnus": "n", "scelus": "n", "sīdus": "n",
|
||||
"caput": "n", "iter": "n", "flūmen": "n", "nōmen": "n", "carmen": "n",
|
||||
"agmen": "n", "certāmen": "n", "lūmen": "n", "ōmen": "n", "cōgnōmen": "n",
|
||||
"mare": "n", "animal": "n", "exemplar": "n", "rēte": "n",
|
||||
# 4th-declension exceptions
|
||||
"manus": "f", "domus": "f", "tribus": "f", "porticus": "f", "īdūs": "f",
|
||||
"cornū": "n", "genū": "n", "gelū": "n", "verū": "n",
|
||||
# 5th-declension
|
||||
"diēs": "m", "merīdiēs": "m",
|
||||
}
|
||||
|
||||
|
||||
def _infer_gender(lemma):
|
||||
if lemma in _GENDER_EXC:
|
||||
return _GENDER_EXC[lemma]
|
||||
d = _NOUNS.get(lemma)
|
||||
nom = d.get(("NOM", "SG")) if d else lemma
|
||||
gen = d.get(("GEN", "SG")) if d else None
|
||||
nom = nom or lemma
|
||||
# 5th declension: gen -eī / -ēī
|
||||
if gen and (gen.endswith("eī") or gen.endswith("ēī")):
|
||||
return "f"
|
||||
# 1st declension: nom -a, gen -ae
|
||||
if nom.endswith("a") and (not gen or gen.endswith("ae")):
|
||||
return "f"
|
||||
# 2nd declension neuter: nom -um
|
||||
if nom.endswith("um"):
|
||||
return "n"
|
||||
# 2nd declension masc: nom -us/-er/-ir, gen -ī
|
||||
if (nom.endswith("us") or nom.endswith("er") or nom.endswith("ir")) and \
|
||||
(not gen or gen.endswith("ī")):
|
||||
return "m"
|
||||
# 4th declension: gen -ūs
|
||||
if gen and gen.endswith("ūs"):
|
||||
return "n" if nom.endswith("ū") else "m"
|
||||
# 3rd declension neuters by common nom endings
|
||||
if nom.endswith(("men", "us", "ur", "al", "ar", "e", "ma")):
|
||||
# -us here is 3rd-decl neuter type (corpus) only if gen shows -oris/-eris
|
||||
if nom.endswith("us") and gen and (gen.endswith("oris") or gen.endswith("eris")
|
||||
or gen.endswith("uris")):
|
||||
return "n"
|
||||
if nom.endswith(("men", "al", "ar", "e")):
|
||||
return "n"
|
||||
# default 3rd-declension: masculine (most common)
|
||||
return "m"
|
||||
|
||||
|
||||
_GENDER_CACHE = {}
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip()
|
||||
if lemma not in _GENDER_CACHE:
|
||||
_GENDER_CACHE[lemma] = _infer_gender(lemma)
|
||||
return _GENDER_CACHE[lemma]
|
||||
|
||||
|
||||
# ── PUBLIC: noun declension ─────────────────────────────────────────────────────
|
||||
def decline_noun(lemma, case, number):
|
||||
lemma = lemma.strip()
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
N = _NUM.get(number, number)
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and (C, N) in d:
|
||||
return d[(C, N)], "lexicon"
|
||||
# abl sg often == the -e/-o form; try nom fallback
|
||||
if d:
|
||||
# try VOC==NOM, ACC neuter==NOM etc are already in data; last resort lemma
|
||||
return lemma, "fallback"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: adjective declension ────────────────────────────────────────────────
|
||||
def decline_adj(lemma, case, gender, number):
|
||||
lemma = lemma.strip()
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
G = _GEN.get(gender, gender.upper())
|
||||
N = _NUM.get(number, number)
|
||||
d = _ADJS.get(lemma)
|
||||
if d and (C, G, N) in d:
|
||||
return d[(C, G, N)], "lexicon"
|
||||
# try other gender (some adjs listed only under MASC+FEM etc handled at load)
|
||||
if d:
|
||||
for altG in ("MASC", "FEM", "NEUT"):
|
||||
if (C, altG, N) in d:
|
||||
return d[(C, altG, N)], "lexicon"
|
||||
return lemma, "fallback"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# VERB RULE ENGINE (4 conjugations + curated irregulars)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# Curated principal parts for common attested verbs:
|
||||
# lemma -> (conj, present_stem, perfect_stem, supine_stem)
|
||||
# conj in {1,2,3,"3io",4}. Stems carry macrons (matching UniMorph orthography).
|
||||
_VERBS = {
|
||||
"amō": (1, "am", "amāv", "amāt"),
|
||||
"laudō": (1, "laud", "laudāv", "laudāt"),
|
||||
"portō": (1, "port", "portāv", "portāt"),
|
||||
"vocō": (1, "voc", "vocāv", "vocāt"),
|
||||
"dō": (1, "d", "ded", "dat"),
|
||||
"spectō": (1, "spect", "spectāv", "spectāt"),
|
||||
"pugnō": (1, "pugn", "pugnāv", "pugnāt"),
|
||||
"labōrō": (1, "labōr", "labōrāv", "labōrāt"),
|
||||
"necō": (1, "nec", "necāv", "necāt"),
|
||||
"parō": (1, "par", "parāv", "parāt"),
|
||||
"cōgitō": (1, "cōgit", "cōgitāv", "cōgitāt"),
|
||||
"habitō": (1, "habit", "habitāv", "habitāt"),
|
||||
"nārrō": (1, "nārr", "nārrāv", "nārrāt"),
|
||||
"servō": (1, "serv", "servāv", "servāt"),
|
||||
"superō": (1, "super", "superāv", "superāt"),
|
||||
"oppugnō": (1, "oppugn", "oppugnāv", "oppugnāt"),
|
||||
"ambulō": (1, "ambul", "ambulāv", "ambulāt"),
|
||||
"clāmō": (1, "clām", "clāmāv", "clāmāt"),
|
||||
"vulnerō": (1, "vulner", "vulnerāv", "vulnerāt"),
|
||||
"aedificō": (1, "aedific", "aedificāv", "aedificāt"),
|
||||
"expugnō": (1, "expugn", "expugnāv", "expugnāt"),
|
||||
"dēfendō": (3, "dēfend", "dēfend", "dēfēns"),
|
||||
"petō": (3, "pet", "petīv", "petīt"),
|
||||
"occīdō": (3, "occīd", "occīd", "occīs"),
|
||||
"interficiō": ("3io", "interfic", "interfēc", "interfect"),
|
||||
"timeō": (2, "tim", "timu", None),
|
||||
"iaceō": (2, "iac", "iacu", None),
|
||||
"pāreō": (2, "pār", "pāru", "pārit"),
|
||||
"respondeō": (2, "respond", "respond", "respōns"),
|
||||
"vertō": (3, "vert", "vert", "vers"),
|
||||
"ostendō": (3, "ostend", "ostend", "ostent"),
|
||||
"cōnstituō": (3, "cōnstitu", "cōnstitu", "cōnstitūt"),
|
||||
"cōgnōscō": (3, "cōgnōsc", "cōgnōv", "cōgnit"),
|
||||
"crēdō": (3, "crēd", "crēdid", "crēdit"),
|
||||
"ēdūcō": (3, "ēdūc", "ēdūx", "ēduct"),
|
||||
"cōnservō": (1, "cōnserv", "cōnservāv", "cōnservāt"),
|
||||
"iuvō": (1, "iuv", "iūv", "iūt"),
|
||||
"dēbeō": (2, "dēb", "dēbu", "dēbit"),
|
||||
"moneō": (2, "mon", "monu", "monit"),
|
||||
"videō": (2, "vid", "vīd", "vīs"),
|
||||
"habeō": (2, "hab", "habu", "habit"),
|
||||
"teneō": (2, "ten", "tenu", "tent"),
|
||||
"timeō": (2, "tim", "timu", None),
|
||||
"terreō": (2, "terr", "terru", "territ"),
|
||||
"dēleō": (2, "dēl", "dēlēv", "dēlēt"),
|
||||
"iubeō": (2, "iub", "iuss", "iuss"),
|
||||
"maneō": (2, "man", "māns", "māns"),
|
||||
"moveō": (2, "mov", "mōv", "mōt"),
|
||||
"doceō": (2, "doc", "docu", "doct"),
|
||||
"sedeō": (2, "sed", "sēd", "sess"),
|
||||
"rīdeō": (2, "rīd", "rīs", "rīs"),
|
||||
"regō": (3, "reg", "rēx", "rēct"),
|
||||
"dūcō": (3, "dūc", "dūx", "duct"),
|
||||
"scrībō": (3, "scrīb", "scrīps", "scrīpt"),
|
||||
"mittō": (3, "mitt", "mīs", "miss"),
|
||||
"pōnō": (3, "pōn", "posu", "posit"),
|
||||
"agō": (3, "ag", "ēg", "āct"),
|
||||
"dīcō": (3, "dīc", "dīx", "dict"),
|
||||
"gerō": (3, "ger", "gess", "gest"),
|
||||
"vincō": (3, "vinc", "vīc", "vict"),
|
||||
"petō": (3, "pet", "petīv", "petīt"),
|
||||
"legō": (3, "leg", "lēg", "lēct"),
|
||||
"currō": (3, "curr", "cucurr", "curs"),
|
||||
"vīvō": (3, "vīv", "vīx", "vīct"),
|
||||
"quaerō": (3, "quaer", "quaesīv", "quaesīt"),
|
||||
"trahō": (3, "trah", "trāx", "tract"),
|
||||
"claudō": (3, "claud", "claus", "claus"),
|
||||
"cōgō": (3, "cōg", "coēg", "coāct"),
|
||||
"relinquō": (3, "relinqu", "relīqu", "relict"),
|
||||
"capiō": ("3io", "cap", "cēp", "capt"),
|
||||
"faciō": ("3io", "fac", "fēc", "fact"),
|
||||
"iaciō": ("3io", "iac", "iēc", "iact"),
|
||||
"rapiō": ("3io", "rap", "rapu", "rapt"),
|
||||
"fugiō": ("3io", "fug", "fūg", "fugit"),
|
||||
"cupiō": ("3io", "cup", "cupīv", "cupīt"),
|
||||
"accipiō": ("3io", "accip", "accēp", "accept"),
|
||||
"audiō": (4, "aud", "audīv", "audīt"),
|
||||
"veniō": (4, "ven", "vēn", "vent"),
|
||||
"sciō": (4, "sc", "scīv", "scīt"),
|
||||
"sentiō": (4, "sent", "sēns", "sēns"),
|
||||
"mūniō": (4, "mūn", "mūnīv", "mūnīt"),
|
||||
"dormiō": (4, "dorm", "dormīv", "dormīt"),
|
||||
"aperiō": (4, "aper", "aperu", "apert"),
|
||||
"inveniō": (4, "inven", "invēn", "invent"),
|
||||
}
|
||||
|
||||
# ── Present-system paradigms: full ending tables per conjugation, attached to the
|
||||
# bare present stem (pstem). Hardcoded from the standard grammar with correct
|
||||
# macrons/vowel-lengths — deterministic and independently verifiable. Keys:
|
||||
# (tense, mood, voice) -> {conj: [1sg,2sg,3sg,1pl,2pl,3pl]}
|
||||
_PARADIGM = {
|
||||
("present", "ind", "active"): {
|
||||
1: ["ō", "ās", "at", "āmus", "ātis", "ant"],
|
||||
2: ["eō", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
3: ["ō", "is", "it", "imus", "itis", "unt"],
|
||||
"3io": ["iō", "is", "it", "imus", "itis", "iunt"],
|
||||
4: ["iō", "īs", "it", "īmus", "ītis", "iunt"],
|
||||
},
|
||||
("present", "ind", "passive"): {
|
||||
1: ["or", "āris", "ātur", "āmur", "āminī", "antur"],
|
||||
2: ["eor", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
3: ["or", "eris", "itur", "imur", "iminī", "untur"],
|
||||
"3io": ["ior", "eris", "itur", "imur", "iminī", "iuntur"],
|
||||
4: ["ior", "īris", "ītur", "īmur", "īminī", "iuntur"],
|
||||
},
|
||||
("imperfect", "ind", "active"): {
|
||||
1: ["ābam", "ābās", "ābat", "ābāmus", "ābātis", "ābant"],
|
||||
2: ["ēbam", "ēbās", "ēbat", "ēbāmus", "ēbātis", "ēbant"],
|
||||
3: ["ēbam", "ēbās", "ēbat", "ēbāmus", "ēbātis", "ēbant"],
|
||||
"3io": ["iēbam", "iēbās", "iēbat", "iēbāmus", "iēbātis", "iēbant"],
|
||||
4: ["iēbam", "iēbās", "iēbat", "iēbāmus", "iēbātis", "iēbant"],
|
||||
},
|
||||
("imperfect", "ind", "passive"): {
|
||||
1: ["ābar", "ābāris", "ābātur", "ābāmur", "ābāminī", "ābantur"],
|
||||
2: ["ēbar", "ēbāris", "ēbātur", "ēbāmur", "ēbāminī", "ēbantur"],
|
||||
3: ["ēbar", "ēbāris", "ēbātur", "ēbāmur", "ēbāminī", "ēbantur"],
|
||||
"3io": ["iēbar", "iēbāris", "iēbātur", "iēbāmur", "iēbāminī", "iēbantur"],
|
||||
4: ["iēbar", "iēbāris", "iēbātur", "iēbāmur", "iēbāminī", "iēbantur"],
|
||||
},
|
||||
("future", "ind", "active"): {
|
||||
1: ["ābō", "ābis", "ābit", "ābimus", "ābitis", "ābunt"],
|
||||
2: ["ēbō", "ēbis", "ēbit", "ēbimus", "ēbitis", "ēbunt"],
|
||||
3: ["am", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
"3io": ["iam", "iēs", "iet", "iēmus", "iētis", "ient"],
|
||||
4: ["iam", "iēs", "iet", "iēmus", "iētis", "ient"],
|
||||
},
|
||||
("future", "ind", "passive"): {
|
||||
1: ["ābor", "āberis", "ābitur", "ābimur", "ābiminī", "ābuntur"],
|
||||
2: ["ēbor", "ēberis", "ēbitur", "ēbimur", "ēbiminī", "ēbuntur"],
|
||||
3: ["ar", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
"3io": ["iar", "iēris", "iētur", "iēmur", "iēminī", "ientur"],
|
||||
4: ["iar", "iēris", "iētur", "iēmur", "iēminī", "ientur"],
|
||||
},
|
||||
("present", "sbjv", "active"): {
|
||||
1: ["em", "ēs", "et", "ēmus", "ētis", "ent"],
|
||||
2: ["eam", "eās", "eat", "eāmus", "eātis", "eant"],
|
||||
3: ["am", "ās", "at", "āmus", "ātis", "ant"],
|
||||
"3io": ["iam", "iās", "iat", "iāmus", "iātis", "iant"],
|
||||
4: ["iam", "iās", "iat", "iāmus", "iātis", "iant"],
|
||||
},
|
||||
("present", "sbjv", "passive"): {
|
||||
1: ["er", "ēris", "ētur", "ēmur", "ēminī", "entur"],
|
||||
2: ["ear", "eāris", "eātur", "eāmur", "eāminī", "eantur"],
|
||||
3: ["ar", "āris", "ātur", "āmur", "āminī", "antur"],
|
||||
"3io": ["iar", "iāris", "iātur", "iāmur", "iāminī", "iantur"],
|
||||
4: ["iar", "iāris", "iātur", "iāmur", "iāminī", "iantur"],
|
||||
},
|
||||
("imperfect", "sbjv", "active"): {
|
||||
1: ["ārem", "ārēs", "āret", "ārēmus", "ārētis", "ārent"],
|
||||
2: ["ērem", "ērēs", "ēret", "ērēmus", "ērētis", "ērent"],
|
||||
3: ["erem", "erēs", "eret", "erēmus", "erētis", "erent"],
|
||||
"3io": ["erem", "erēs", "eret", "erēmus", "erētis", "erent"],
|
||||
4: ["īrem", "īrēs", "īret", "īrēmus", "īrētis", "īrent"],
|
||||
},
|
||||
("imperfect", "sbjv", "passive"): {
|
||||
1: ["ārer", "ārēris", "ārētur", "ārēmur", "ārēminī", "ārentur"],
|
||||
2: ["ērer", "ērēris", "ērētur", "ērēmur", "ērēminī", "ērentur"],
|
||||
3: ["erer", "erēris", "erētur", "erēmur", "erēminī", "erentur"],
|
||||
"3io": ["erer", "erēris", "erētur", "erēmur", "erēminī", "erentur"],
|
||||
4: ["īrer", "īrēris", "īrētur", "īrēmur", "īrēminī", "īrentur"],
|
||||
},
|
||||
}
|
||||
# perfect-active endings (added to perfect stem) — same for all conjugations
|
||||
_PERF_ACT = {
|
||||
("perfect", "ind"): ["ī", "istī", "it", "imus", "istis", "ērunt"],
|
||||
("pluperfect", "ind"): ["eram", "erās", "erat", "erāmus", "erātis", "erant"],
|
||||
("futureperfect", "ind"): ["erō", "eris", "erit", "erimus", "eritis", "erint"],
|
||||
("perfect", "sbjv"): ["erim", "erīs", "erit", "erīmus", "erītis", "erint"],
|
||||
("pluperfect", "sbjv"):["issem", "issēs", "isset", "issēmus", "issētis", "issent"],
|
||||
}
|
||||
|
||||
|
||||
def _idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _present_system(conj, pstem, tense, mood, voice, person, number):
|
||||
"""Generate a present-system form (present/imperfect/future ind & subj)."""
|
||||
table = _PARADIGM.get((tense, mood, voice))
|
||||
if not table or conj not in table:
|
||||
return None
|
||||
return pstem + table[conj][_idx(person, number)]
|
||||
|
||||
|
||||
def _active_infinitive_stem(conj, pstem):
|
||||
return {1: pstem + "ā", 2: pstem + "ē", 3: pstem + "e",
|
||||
"3io": pstem + "e", 4: pstem + "ī"}[conj]
|
||||
|
||||
|
||||
_IRREG = {
|
||||
"sum": {
|
||||
("present", "ind", "active"): ["sum", "es", "est", "sumus", "estis", "sunt"],
|
||||
("imperfect", "ind", "active"): ["eram", "erās", "erat", "erāmus", "erātis", "erant"],
|
||||
("future", "ind", "active"): ["erō", "eris", "erit", "erimus", "eritis", "erunt"],
|
||||
("perfect", "ind", "active"): ["fuī", "fuistī", "fuit", "fuimus", "fuistis", "fuērunt"],
|
||||
("pluperfect", "ind", "active"): ["fueram", "fuerās", "fuerat", "fuerāmus", "fuerātis", "fuerant"],
|
||||
("present", "sbjv", "active"): ["sim", "sīs", "sit", "sīmus", "sītis", "sint"],
|
||||
("imperfect", "sbjv", "active"): ["essem", "essēs", "esset", "essēmus", "essētis", "essent"],
|
||||
},
|
||||
"possum": {
|
||||
("present", "ind", "active"): ["possum", "potes", "potest", "possumus", "potestis", "possunt"],
|
||||
("imperfect", "ind", "active"): ["poteram", "poterās", "poterat", "poterāmus", "poterātis", "poterant"],
|
||||
("future", "ind", "active"): ["poterō", "poteris", "poterit", "poterimus", "poteritis", "poterunt"],
|
||||
("perfect", "ind", "active"): ["potuī", "potuistī", "potuit", "potuimus", "potuistis", "potuērunt"],
|
||||
("present", "sbjv", "active"): ["possim", "possīs", "possit", "possīmus", "possītis", "possint"],
|
||||
},
|
||||
"eō": {
|
||||
("present", "ind", "active"): ["eō", "īs", "it", "īmus", "ītis", "eunt"],
|
||||
("imperfect", "ind", "active"): ["ībam", "ībās", "ībat", "ībāmus", "ībātis", "ībant"],
|
||||
("future", "ind", "active"): ["ībō", "ībis", "ībit", "ībimus", "ībitis", "ībunt"],
|
||||
("perfect", "ind", "active"): ["iī", "īstī", "iit", "iimus", "īstis", "iērunt"],
|
||||
("present", "sbjv", "active"): ["eam", "eās", "eat", "eāmus", "eātis", "eant"],
|
||||
},
|
||||
"volō": {
|
||||
("present", "ind", "active"): ["volō", "vīs", "vult", "volumus", "vultis", "volunt"],
|
||||
("imperfect", "ind", "active"): ["volēbam", "volēbās", "volēbat", "volēbāmus", "volēbātis", "volēbant"],
|
||||
("future", "ind", "active"): ["volam", "volēs", "volet", "volēmus", "volētis", "volent"],
|
||||
("perfect", "ind", "active"): ["voluī", "voluistī", "voluit", "voluimus", "voluistis", "voluērunt"],
|
||||
("present", "sbjv", "active"): ["velim", "velīs", "velit", "velīmus", "velītis", "velint"],
|
||||
},
|
||||
"nōlō": {
|
||||
("present", "ind", "active"): ["nōlō", "nōn vīs", "nōn vult", "nōlumus", "nōn vultis", "nōlunt"],
|
||||
("present", "sbjv", "active"): ["nōlim", "nōlīs", "nōlit", "nōlīmus", "nōlītis", "nōlint"],
|
||||
},
|
||||
"ferō": {
|
||||
("present", "ind", "active"): ["ferō", "fers", "fert", "ferimus", "fertis", "ferunt"],
|
||||
("imperfect", "ind", "active"): ["ferēbam", "ferēbās", "ferēbat", "ferēbāmus", "ferēbātis", "ferēbant"],
|
||||
("future", "ind", "active"): ["feram", "ferēs", "feret", "ferēmus", "ferētis", "ferent"],
|
||||
("perfect", "ind", "active"): ["tulī", "tulistī", "tulit", "tulimus", "tulistis", "tulērunt"],
|
||||
("present", "sbjv", "active"): ["feram", "ferās", "ferat", "ferāmus", "ferātis", "ferant"],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def conjugate(lemma, tense, mood, voice="active", person="third", number="singular"):
|
||||
"""Return (surface, confidence). Perfect-passive forms are periphrastic and
|
||||
handled in the realizer (sum + PPP); this returns synthetic forms only."""
|
||||
lemma = lemma.strip()
|
||||
i = _idx(person, number)
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir:
|
||||
tbl = ir.get((tense, mood, voice)) or ir.get((tense, mood, "active"))
|
||||
if tbl and tbl[i]:
|
||||
return tbl[i], "rule"
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
v = _infer_principal_parts(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
# imperative (present active) 2sg / 2pl
|
||||
if mood == "imp":
|
||||
return _imperative(conj, pstem, person, number), "rule"
|
||||
# perfect-system active
|
||||
if tense in ("perfect", "pluperfect", "futureperfect") and voice == "active":
|
||||
if not perfstem:
|
||||
return lemma, "fallback"
|
||||
end = _PERF_ACT.get((tense, mood))
|
||||
if end:
|
||||
return perfstem + end[i], "rule"
|
||||
# present-system (active + passive)
|
||||
if tense in ("present", "imperfect", "future"):
|
||||
form = _present_system(conj, pstem, tense, mood, voice, person, number)
|
||||
if form:
|
||||
return form, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def _imperative(conj, pstem, person, number):
|
||||
if number == "singular":
|
||||
return {1: pstem + "ā", 2: pstem + "ē", 3: pstem + "e",
|
||||
"3io": pstem + "e", 4: pstem + "ī"}[conj]
|
||||
return {1: pstem + "āte", 2: pstem + "ēte", 3: pstem + "ite",
|
||||
"3io": pstem + "ite", 4: pstem + "īte"}[conj]
|
||||
|
||||
|
||||
def _infer_principal_parts(lemma):
|
||||
"""OOV fallback: infer conjugation + stems from the 1sg-present citation form.
|
||||
Perfect/supine stems are guessed regularly (often wrong for 3rd conj) and the
|
||||
resulting forms are still returned as 'rule' but the realizer down-weights."""
|
||||
if lemma.endswith("ō"):
|
||||
base = lemma[:-1]
|
||||
# can't distinguish conj from 1sg alone reliably; default by ending vowel
|
||||
if base.endswith("i"):
|
||||
return ("3io", base[:-1], base[:-1] + "īv", base[:-1] + "īt")
|
||||
return (3, base, base + "s", base + "t")
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC: participles ─────────────────────────────────────────────────────────
|
||||
def participle(lemma, kind, case="nom", gender="m", number="singular"):
|
||||
"""kind: 'prs' (present active, -ns/-ntis), 'pfv' (perfect passive, -tus),
|
||||
'fut' (future active, -tūrus). Declined as an adjective via rule endings.
|
||||
Returns (form, conf)."""
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
if kind == "pfv":
|
||||
if not supstem:
|
||||
return lemma, "fallback"
|
||||
base = supstem[:-1] if supstem.endswith("t") or supstem.endswith("s") else supstem
|
||||
stem = supstem # supine stem already ends in t/s: amāt- -> amātus
|
||||
return _decline_us_a_um(stem, case, gender, number), "rule"
|
||||
if kind == "fut":
|
||||
if not supstem:
|
||||
return lemma, "fallback"
|
||||
return _decline_us_a_um(supstem + "ūr", case, gender, number), "rule"
|
||||
if kind == "prs":
|
||||
# present active participle: stem + ns (nom), stem + nt- (oblique), 3rd-decl
|
||||
pv = {1: "ā", 2: "ē", 3: "ē", "3io": "iē", 4: "iē"}[conj]
|
||||
ntstem = pstem + pv + "nt"
|
||||
return _decline_pres_ptcp(pstem + pv, case, gender, number), "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def _decline_us_a_um(stem, case, gender, number):
|
||||
"""Decline a -us/-a/-um adjective/participle stem (2-1-2 declension)."""
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
end = {
|
||||
("NOM", "m", "singular"): "us", ("NOM", "f", "singular"): "a", ("NOM", "n", "singular"): "um",
|
||||
("GEN", "m", "singular"): "ī", ("GEN", "f", "singular"): "ae", ("GEN", "n", "singular"): "ī",
|
||||
("DAT", "m", "singular"): "ō", ("DAT", "f", "singular"): "ae", ("DAT", "n", "singular"): "ō",
|
||||
("ACC", "m", "singular"): "um", ("ACC", "f", "singular"): "am", ("ACC", "n", "singular"): "um",
|
||||
("ABL", "m", "singular"): "ō", ("ABL", "f", "singular"): "ā", ("ABL", "n", "singular"): "ō",
|
||||
("VOC", "m", "singular"): "e", ("VOC", "f", "singular"): "a", ("VOC", "n", "singular"): "um",
|
||||
("NOM", "m", "plural"): "ī", ("NOM", "f", "plural"): "ae", ("NOM", "n", "plural"): "a",
|
||||
("GEN", "m", "plural"): "ōrum", ("GEN", "f", "plural"): "ārum", ("GEN", "n", "plural"): "ōrum",
|
||||
("DAT", "m", "plural"): "īs", ("DAT", "f", "plural"): "īs", ("DAT", "n", "plural"): "īs",
|
||||
("ACC", "m", "plural"): "ōs", ("ACC", "f", "plural"): "ās", ("ACC", "n", "plural"): "a",
|
||||
("ABL", "m", "plural"): "īs", ("ABL", "f", "plural"): "īs", ("ABL", "n", "plural"): "īs",
|
||||
("VOC", "m", "plural"): "ī", ("VOC", "f", "plural"): "ae", ("VOC", "n", "plural"): "a",
|
||||
}.get((C, gender, number), "us")
|
||||
return stem + end
|
||||
|
||||
|
||||
def _decline_pres_ptcp(stem, case, gender, number):
|
||||
"""Present active participle (amāns, amantis) — 3rd-declension, stem+ns/nt."""
|
||||
C = _CASE_MAP.get(case, case.upper())
|
||||
if C == "NOM" and number == "singular":
|
||||
return stem + "ns"
|
||||
if C == "VOC" and number == "singular":
|
||||
return stem + "ns"
|
||||
base = stem + "nt"
|
||||
end = {
|
||||
("GEN", "singular"): "is", ("DAT", "singular"): "ī",
|
||||
("ACC", "singular"): "em" if gender != "n" else "",
|
||||
("ABL", "singular"): "e",
|
||||
("NOM", "plural"): "ēs" if gender != "n" else "ia",
|
||||
("GEN", "plural"): "ium", ("DAT", "plural"): "ibus",
|
||||
("ACC", "plural"): "ēs" if gender != "n" else "ia",
|
||||
("ABL", "plural"): "ibus", ("VOC", "plural"): "ēs",
|
||||
}.get((C, number), "is")
|
||||
if C == "ACC" and number == "singular" and gender == "n":
|
||||
return stem + "ns"
|
||||
return base + end
|
||||
|
||||
|
||||
def infinitive(lemma, tense="present", voice="active"):
|
||||
lemma = lemma.strip()
|
||||
if lemma == "sum":
|
||||
return ("esse", "rule") if tense == "present" else ("fuisse", "rule")
|
||||
v = _VERBS.get(lemma)
|
||||
if not v:
|
||||
return lemma, "fallback"
|
||||
conj, pstem, perfstem, supstem = v
|
||||
if tense == "present":
|
||||
if voice == "active":
|
||||
return _active_infinitive_stem(conj, pstem).rstrip() + \
|
||||
("re" if conj != 3 and conj != "3io" else "re"), "rule"
|
||||
# passive present infinitive
|
||||
base = {1: pstem + "ā", 2: pstem + "ē", 4: pstem + "ī"}.get(conj)
|
||||
if base:
|
||||
return base + "rī", "rule"
|
||||
return pstem + "ī", "rule" # 3rd: regī
|
||||
if tense == "perfect" and voice == "active" and perfstem:
|
||||
return perfstem + "isse", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"noun_adj_source": "UniMorph Latin (github.com/unimorph/lat, CC-BY-SA 3.0)",
|
||||
"verb_source": "rule-based 4-conjugation engine over curated attested "
|
||||
"principal parts (UniMorph verb list is a 947-lemma sample "
|
||||
"MISSING all core verbs — amō/sum/videō absent)",
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
"curated_verb_lemmas": len(_VERBS) + len(_IRREG),
|
||||
"gender_inference": "declension-based (nom+gen endings) + curated exceptions",
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import json
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
print("\n-- noun declension puella (1st, fem) --")
|
||||
for c in ("nom", "gen", "dat", "acc", "abl", "voc"):
|
||||
print(f" {c}: sg={decline_noun('puella', c, 'singular')[0]:10} "
|
||||
f"pl={decline_noun('puella', c, 'plural')[0]}")
|
||||
print("\n-- rēx (3rd, m):", [decline_noun('rēx', c, 'singular')[0] for c in ('nom','gen','dat','acc','abl')])
|
||||
print("-- gender: puella=", noun_gender("puella"), "rēx=", noun_gender("rēx"),
|
||||
"bellum=", noun_gender("bellum"), "corpus=", noun_gender("corpus"),
|
||||
"manus=", noun_gender("manus"), "diēs=", noun_gender("diēs"))
|
||||
print("\n-- conjugate videō (2nd) present ind active --")
|
||||
for p in ("first", "second", "third"):
|
||||
for n in ("singular", "plural"):
|
||||
print(f" {p[:3]}.{n[:2]}: {conjugate('videō','present','ind','active',p,n)[0]}")
|
||||
print("-- amō forms:", conjugate("amō","present","ind","active","first","singular")[0],
|
||||
conjugate("amō","imperfect","ind","active","third","plural")[0],
|
||||
conjugate("amō","future","ind","active","first","singular")[0],
|
||||
conjugate("amō","perfect","ind","active","third","singular")[0])
|
||||
print("-- sum:", [conjugate("sum","present","ind","active",p,"singular")[0] for p in ("first","second","third")])
|
||||
print("-- participle amō pfv acc.f.sg:", participle("amō","pfv","acc","f","singular")[0])
|
||||
print("-- infinitive amō:", infinitive("amō")[0], "| regō pass:", infinitive("regō", voice="passive")[0])
|
||||
@@ -1,538 +0,0 @@
|
||||
"""morphology_pt_full.py — production-grade Brazilian-Portuguese morphological generator.
|
||||
|
||||
NOT a toy. Backed by two real, broad, Wiktionary-lineage lexicons:
|
||||
|
||||
VERBS — UniMorph Portuguese (github.com/unimorph/por, CC-BY-SA 3.0)
|
||||
4,001 verb lemmas × full paradigm (283,991 finite/non-finite forms +
|
||||
20,005 participle forms). Every mood/tense pt actually inflects:
|
||||
indicative present / preterite (PST;PFV) / imperfect (PST;IPFV) /
|
||||
pluperfect-simple (PST;PRF) / future,
|
||||
conditional (futuro do pretérito),
|
||||
subjunctive present / imperfect / FUTURE (PT-specific live tense),
|
||||
affirmative + negative imperative,
|
||||
PERSONAL infinitive (V;{p};{n};NFIN — a PT-specific finite-ish form),
|
||||
past participle (4 gender/number forms) + gerúndio (V.PTCP;PRS).
|
||||
|
||||
NOUNS + ADJECTIVES — kaikki.org Portuguese (Wiktionary extract, same lineage)
|
||||
81,138 noun lemmas WITH inherent gender + real (often irregular) plural —
|
||||
so -ão→-ões / -ãos / -ães / -õos is resolved PER LEMMA by Wiktionary,
|
||||
never guessed (mão→mãos, pão→pães, coração→corações).
|
||||
40,252 adjective lemmas with real feminine + masc/fem plural forms.
|
||||
|
||||
Fallbacks (degrade, never crash, on out-of-vocabulary input):
|
||||
verbs : rule generator for regular -ar/-er/-ir paradigms
|
||||
nouns : gender heuristic (endings) + rule pluralization (with -ão FLAGGED)
|
||||
adjs : -o/-a gender rule + rule pluralization
|
||||
|
||||
Confidence flag on every form:
|
||||
"lexicon" straight from UniMorph/kaikki (trust: high)
|
||||
"rule" deterministic rule (trust: medium)
|
||||
"fallback" could not inflect; returned lemma (trust: low -> FLAG)
|
||||
|
||||
Public API (used by realizer_pt.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
personal_infinitive(lemma, person, number) -> (form, conf)
|
||||
participle(lemma, gender="m", number="singular") -> (form, conf)
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"
|
||||
inflect_noun(lemma, number, gender=None) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "por.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_pt.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "pt_morph_cache.pkl")
|
||||
|
||||
# ── mood/tense pair -> UniMorph feature triple (a in tag; b in tag; c in tag) ────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): ("IND", "PRS", None),
|
||||
("ind", "preterite"): ("IND", "PST", "PFV"),
|
||||
("ind", "imperfect"): ("IND", "PST", "IPFV"),
|
||||
("ind", "pluperfect"): ("IND", "PST", "PRF"), # simple mais-que-perfeito
|
||||
("ind", "future"): ("IND", "FUT", None),
|
||||
("ind", "conditional"): ("COND", None, None),
|
||||
("sbjv", "present"): ("SBJV", "PRS", None),
|
||||
("sbjv", "imperfect"): ("SBJV", "PST", "IPFV"),
|
||||
("sbjv", "future"): ("SBJV", "FUT", None), # PT-specific
|
||||
("imp", "affirmative"): ("IMP", "POS", None),
|
||||
("imp", "negative"): ("IMP", "NEG", None),
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── build the compact lexicon from UniMorph (verbs) + kaikki (nouns/adjs) ────────
|
||||
def _build_verbs():
|
||||
verbs = {} # (lemma, "mood|tense|person|number") -> form
|
||||
pinf = {} # (lemma, "person|number") -> personal-infinitive form
|
||||
part = {} # lemma -> {("m","SG"): form, ...} past participle
|
||||
ger = {} # lemma -> gerúndio
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f: # past participle: falado/falada/falados/faladas
|
||||
g = "m" if "MASC" in f else ("f" if "FEM" in f else "m")
|
||||
num = "SG" if "SG" in f else ("PL" if "PL" in f else "SG")
|
||||
part.setdefault(lemma, {})[(g, num)] = form
|
||||
elif "PRS" in f: # gerúndio: falando
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
|
||||
if head != "V":
|
||||
continue
|
||||
|
||||
# personal / impersonal infinitive
|
||||
if "NFIN" in f:
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person and number:
|
||||
pinf[(lemma, f"{person}|{number}")] = form
|
||||
continue
|
||||
|
||||
# finite forms
|
||||
mt = None
|
||||
for (mood, tense), (a, b, c) in _VERB_KEYMAP.items():
|
||||
if a not in f:
|
||||
continue
|
||||
if b is not None and b not in f:
|
||||
continue
|
||||
if c is not None and c not in f:
|
||||
continue
|
||||
# IND;PST needs exactly PFV|IPFV|PRF — reject if the required one absent
|
||||
mt = (mood, tense)
|
||||
break
|
||||
if mt is None:
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mt[0]}|{mt[1]}|{person}|{number}"), form)
|
||||
return verbs, pinf, part, ger
|
||||
|
||||
|
||||
def _kaikki_gender(arg):
|
||||
if not arg:
|
||||
return None
|
||||
a = arg.lower()
|
||||
if a.startswith("f"):
|
||||
return "f"
|
||||
if a.startswith("m"):
|
||||
return "m"
|
||||
return None
|
||||
|
||||
|
||||
def _build_nouns_adjs():
|
||||
nouns = {} # lemma -> {"g","SG","PL"}
|
||||
adjs = {} # lemma -> {("m","SG"),("f","SG"),("m","PL"),("f","PL")}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
pos = d.get("pos")
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word: # skip multiword entries
|
||||
continue
|
||||
forms = d.get("forms", []) or []
|
||||
|
||||
if pos == "noun":
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
g = _kaikki_gender((ht[0].get("args") or {}).get("1"))
|
||||
if g is None:
|
||||
tags = d.get("tags") or []
|
||||
if "feminine" in tags:
|
||||
g = "f"
|
||||
elif "masculine" in tags:
|
||||
g = "m"
|
||||
pl = None
|
||||
for x in forms:
|
||||
t = x.get("tags") or []
|
||||
if "plural" in t and "alternative" not in t and "obsolete" not in t:
|
||||
pl = x.get("form")
|
||||
break
|
||||
# first entry wins; but a later entry with a plural fills a gap
|
||||
if word not in nouns:
|
||||
nouns[word] = {"g": g, "SG": word, "PL": pl}
|
||||
else:
|
||||
cur = nouns[word]
|
||||
if cur.get("g") is None and g:
|
||||
cur["g"] = g
|
||||
if not cur.get("PL") and pl:
|
||||
cur["PL"] = pl
|
||||
|
||||
elif pos == "adj":
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word)
|
||||
for x in forms:
|
||||
t = set(x.get("tags") or [])
|
||||
fm = x.get("form")
|
||||
if not fm or ("alternative" in t) or ("obsolete" in t):
|
||||
continue
|
||||
if "comparative" in t or "superlative" in t or \
|
||||
"diminutive" in t or "augmentative" in t:
|
||||
continue
|
||||
if "feminine" in t and "plural" in t:
|
||||
d0[("f", "PL")] = fm
|
||||
elif "masculine" in t and "plural" in t:
|
||||
d0[("m", "PL")] = fm
|
||||
elif "feminine" in t:
|
||||
d0[("f", "SG")] = fm
|
||||
elif "plural" in t: # invariant-gender adj (feliz -> felizes)
|
||||
d0[("m", "PL")] = d0.get(("m", "PL")) or fm
|
||||
d0[("f", "PL")] = d0.get(("f", "PL")) or fm
|
||||
return nouns, adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, pinf, part, ger = _build_verbs()
|
||||
nouns, adjs = _build_nouns_adjs()
|
||||
data = {"verbs": verbs, "pinf": pinf, "part": part, "ger": ger,
|
||||
"nouns": nouns, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
newest_src = max(os.path.getmtime(_UNIMORPH),
|
||||
os.path.getmtime(_KAIKKI) if os.path.exists(_KAIKKI) else 0)
|
||||
if os.path.getmtime(_CACHE) >= newest_src:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PINF, _PART, _GER, _NOUNS, _ADJS = (
|
||||
_LEX["verbs"], _LEX["pinf"], _LEX["part"], _LEX["ger"],
|
||||
_LEX["nouns"], _LEX["adjs"])
|
||||
|
||||
|
||||
# ── regular-ending rule fallback (deterministic, last resort) ────────────────────
|
||||
def _vclass(lemma):
|
||||
return lemma[-2:] if lemma[-2:] in ("ar", "er", "ir") else None
|
||||
|
||||
|
||||
def _stem(lemma):
|
||||
return lemma[:-2]
|
||||
|
||||
|
||||
# endings indexed [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG = {
|
||||
("ind", "present", "ar"): ["o", "as", "a", "amos", "ais", "am"],
|
||||
("ind", "present", "er"): ["o", "es", "e", "emos", "eis", "em"],
|
||||
("ind", "present", "ir"): ["o", "es", "e", "imos", "is", "em"],
|
||||
("ind", "preterite", "ar"): ["ei", "aste", "ou", "amos", "astes", "aram"],
|
||||
("ind", "preterite", "er"): ["i", "este", "eu", "emos", "estes", "eram"],
|
||||
("ind", "preterite", "ir"): ["i", "iste", "iu", "imos", "istes", "iram"],
|
||||
("ind", "imperfect", "ar"): ["ava", "avas", "ava", "ávamos", "áveis", "avam"],
|
||||
("ind", "imperfect", "er"): ["ia", "ias", "ia", "íamos", "íeis", "iam"],
|
||||
("ind", "imperfect", "ir"): ["ia", "ias", "ia", "íamos", "íeis", "iam"],
|
||||
("sbjv", "present", "ar"): ["e", "es", "e", "emos", "eis", "em"],
|
||||
("sbjv", "present", "er"): ["a", "as", "a", "amos", "ais", "am"],
|
||||
("sbjv", "present", "ir"): ["a", "as", "a", "amos", "ais", "am"],
|
||||
("sbjv", "imperfect", "ar"): ["asse", "asses", "asse", "ássemos", "ásseis", "assem"],
|
||||
("sbjv", "imperfect", "er"): ["esse", "esses", "esse", "êssemos", "êsseis", "essem"],
|
||||
("sbjv", "imperfect", "ir"): ["isse", "isses", "isse", "íssemos", "ísseis", "issem"],
|
||||
("sbjv", "future", "ar"): ["ar", "ares", "ar", "armos", "ardes", "arem"],
|
||||
("sbjv", "future", "er"): ["er", "eres", "er", "ermos", "erdes", "erem"],
|
||||
("sbjv", "future", "ir"): ["ir", "ires", "ir", "irmos", "irdes", "irem"],
|
||||
}
|
||||
# future & conditional attach to the FULL infinitive
|
||||
_FUT = ["ei", "ás", "á", "emos", "eis", "ão"]
|
||||
_COND = ["ia", "ias", "ia", "íamos", "íeis", "iam"]
|
||||
|
||||
|
||||
def _slot_idx(person, number):
|
||||
base = {"first": 0, "second": 1, "third": 2}[person]
|
||||
return base + (0 if number == "singular" else 3)
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
st, i = _stem(lemma), _slot_idx(person, number)
|
||||
if mood == "ind" and tense == "future":
|
||||
return lemma + _FUT[i]
|
||||
if mood == "ind" and tense == "conditional":
|
||||
return lemma + _COND[i]
|
||||
if mood == "imp": # affirmative tú/vocês imperative ~ subjunctive present
|
||||
table = _REG.get(("sbjv", "present", vc))
|
||||
if table and tense == "negative":
|
||||
return st + table[i]
|
||||
# affirmative 2sg = 3sg present indicative; others = subjunctive
|
||||
pres = _REG.get(("ind", "present", vc))
|
||||
if person == "second" and number == "singular":
|
||||
return st + pres[2]
|
||||
return st + table[i] if table else None
|
||||
table = _REG.get((mood, tense, vc))
|
||||
if table:
|
||||
return st + table[i]
|
||||
return None
|
||||
|
||||
|
||||
# verified corrections to UniMorph data errors (each audited individually, not
|
||||
# guessed). The three 1PL-present entries are glued-allomorph errors surfaced by a
|
||||
# full-lexicon scan for a non-final "mos" in V;1;PL;IND;PRS forms (the ONLY three).
|
||||
_VERB_FIX = {
|
||||
("estar", "ind", "imperfect", "third", "plural"): "estavam", # was "estávam"
|
||||
("estar", "ind", "present", "first", "plural"): "estamos", # was "estamosestámos"
|
||||
("haver", "ind", "present", "first", "plural"): "havemos", # was "havemoshemos"
|
||||
("ir", "ind", "present", "first", "plural"): "vamos", # was "vamosimos"
|
||||
}
|
||||
|
||||
|
||||
# ── PUBLIC: verb conjugation ─────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
"""Return (surface, confidence). mood in ind|sbjv|imp; tense per _VERB_KEYMAP."""
|
||||
lemma = lemma.strip().lower()
|
||||
fix = _VERB_FIX.get((lemma, mood, tense, person, number))
|
||||
if fix:
|
||||
return fix, "lexicon"
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _VERBS.get((lemma, f"{mood}|{tense}|{p}|{n}"))
|
||||
if form:
|
||||
# pt-BR normalization: UniMorph `por` carries the EUROPEAN spelling of
|
||||
# the -ar 1pl PRETERITE (-ámos). Brazilian PT drops the accent
|
||||
# (falámos->falamos, chegámos->chegamos) — 3,334/4,001 verbs affected.
|
||||
if (mood == "ind" and tense == "preterite" and person == "first"
|
||||
and number == "plural" and form.endswith("ámos")):
|
||||
form = form[:-4] + "amos"
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def personal_infinitive(lemma, person, number):
|
||||
"""PT personal (inflected) infinitive: para falarmos, ao chegarem."""
|
||||
lemma = lemma.strip().lower()
|
||||
p, n = _PERSON.get(person), _NUMBER.get(number)
|
||||
if p and n:
|
||||
form = _PINF.get((lemma, f"{p}|{n}"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# rule: infinitive + personal endings (-, -es, -, -mos, -des, -em)
|
||||
end = {("first", "singular"): "", ("second", "singular"): "es",
|
||||
("third", "singular"): "", ("first", "plural"): "mos",
|
||||
("second", "plural"): "des", ("third", "plural"): "em"}.get((person, number), "")
|
||||
return lemma + end, "rule"
|
||||
|
||||
|
||||
# ── PUBLIC: participle + gerund ───────────────────────────────────────────────────
|
||||
def participle(lemma, gender="m", number="singular"):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _PART.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num)) or d.get(("m", "SG"))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
base = lemma[:-2] + "ad"
|
||||
elif lemma[-2:] in ("er", "ir"):
|
||||
base = lemma[:-2] + "id"
|
||||
else:
|
||||
return lemma, "fallback"
|
||||
suf = {"m|SG": "o", "f|SG": "a", "m|PL": "os", "f|PL": "as"}[f"{g}|{num}"]
|
||||
return base + suf, "rule"
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
if lemma.endswith("ar"):
|
||||
return lemma[:-2] + "ando", "rule"
|
||||
if lemma.endswith("er"):
|
||||
return lemma[:-2] + "endo", "rule"
|
||||
if lemma.endswith("ir"):
|
||||
return lemma[:-2] + "indo", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── PUBLIC: noun gender + number ─────────────────────────────────────────────────
|
||||
_FEM_SUF = ("ção", "são", "ção", "dade", "tade", "agem", "igem", "ugem", "gem",
|
||||
"ez", "eza", "ice", "ície", "tude", "ude", "âncbefore")
|
||||
_FEM_SUF = ("ção", "são", "dade", "tade", "agem", "gem", "eza", "ez", "ice",
|
||||
"tude", "ude", "ância", "ência", "ínia")
|
||||
_MASC_SUF = ("ema", "oma", "ama", "grama", "eta", "ão") # Greek -ma etc. (mostly m)
|
||||
|
||||
|
||||
def _gender_heuristic(noun):
|
||||
for suf in _FEM_SUF:
|
||||
if noun.endswith(suf):
|
||||
return "f"
|
||||
if noun.endswith(("ema", "oma", "ama")): # problema, idioma, programa
|
||||
return "m"
|
||||
if noun.endswith("a") or noun.endswith("ã"):
|
||||
return "f"
|
||||
if noun.endswith("o") or noun.endswith(("l", "r", "z", "m", "u", "i")):
|
||||
return "m"
|
||||
return "m"
|
||||
|
||||
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g"):
|
||||
return d["g"]
|
||||
return _gender_heuristic(lemma)
|
||||
|
||||
|
||||
_INVARIANT_PL_SUF = ("s",) # paroxytones ending -s are invariant (o lápis / os lápis)
|
||||
|
||||
|
||||
def _rule_plural(noun):
|
||||
"""Deterministic PT pluralization. Returns (form, ok) where ok=False flags an
|
||||
ambiguous -ão that should lower confidence (the lexicon normally resolves it)."""
|
||||
if not noun:
|
||||
return noun, True
|
||||
if noun.endswith("ão"):
|
||||
return noun[:-2] + "ões", False # majority rule, but AMBIGUOUS -> flag
|
||||
if noun.endswith("m"):
|
||||
return noun[:-1] + "ns", True # homem->homens, jardim->jardins
|
||||
if noun.endswith("al"):
|
||||
return noun[:-2] + "ais", True
|
||||
if noun.endswith("el"):
|
||||
return noun[:-2] + "éis", True
|
||||
if noun.endswith("ol"):
|
||||
return noun[:-2] + "óis", True
|
||||
if noun.endswith("ul"):
|
||||
return noun[:-2] + "uis", True
|
||||
if noun.endswith("il"):
|
||||
return noun[:-2] + "is", True # stressed (funil->funis); unstressed rarer
|
||||
if noun.endswith(("r", "z")):
|
||||
return noun + "es", True # flor->flores, luz->luzes
|
||||
if noun.endswith("s"):
|
||||
# paroxytone -s (lápis, ônibus) invariant; oxytone -s (país) -> -es
|
||||
return noun, True
|
||||
if noun.endswith(("a", "e", "i", "o", "u", "á", "é", "í", "ó", "ú", "ã")):
|
||||
return noun + "s", True
|
||||
return noun + "s", True
|
||||
|
||||
|
||||
def inflect_noun(lemma, number, gender=None):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if number == "singular":
|
||||
return (d["SG"] if d and d.get("SG") else lemma), ("lexicon" if d else "rule")
|
||||
if d and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
form, ok = _rule_plural(lemma)
|
||||
return form, ("rule" if ok else "fallback")
|
||||
|
||||
|
||||
# ── PUBLIC: adjective agreement ──────────────────────────────────────────────────
|
||||
def inflect_adj(lemma, gender, number):
|
||||
lemma = lemma.strip().lower()
|
||||
g = "f" if gender == "f" else "m"
|
||||
num = "SG" if number == "singular" else "PL"
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((g, num))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# build a missing plural from this gender's singular
|
||||
sg = d.get((g, "SG")) or d.get(("m", "SG")) or lemma
|
||||
if num == "PL":
|
||||
pl, ok = _rule_plural(sg)
|
||||
return pl, ("rule" if ok else "fallback")
|
||||
return sg, "lexicon"
|
||||
# rule fallback: -o/-a gender, then pluralize
|
||||
a = lemma
|
||||
if g == "f":
|
||||
if a.endswith("o"):
|
||||
a = a[:-1] + "a"
|
||||
elif a.endswith(("ês", "or")) and not a.endswith("ior"):
|
||||
a = a + "a" # português->portuguesa, trabalhador->..a
|
||||
if num == "PL":
|
||||
a, ok = _rule_plural(a)
|
||||
return a, ("rule" if ok else "fallback")
|
||||
return a, "rule"
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Portuguese (github.com/unimorph/por)",
|
||||
"noun_adj_source": "kaikki.org Portuguese (Wiktionary extract)",
|
||||
"license": "CC-BY-SA (Wiktionary-derived)",
|
||||
"verb_forms": len(_VERBS),
|
||||
"verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"personal_infinitive_forms": len(_PINF),
|
||||
"participle_lemmas": len(_PART),
|
||||
"gerund_lemmas": len(_GER),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
tests = [
|
||||
("falar", "ind", "present", "first", "singular", "falo"),
|
||||
("comer", "ind", "present", "third", "plural", "comem"),
|
||||
("partir", "ind", "present", "first", "plural", "partimos"),
|
||||
("ser", "ind", "present", "third", "singular", "é"),
|
||||
("ir", "ind", "preterite", "first", "singular", "fui"),
|
||||
("ter", "ind", "future", "first", "singular", "terei"),
|
||||
("fazer", "sbjv", "present", "first", "singular", "faça"),
|
||||
("dormir", "ind", "present", "first", "singular", "durmo"),
|
||||
("dar", "ind", "preterite", "third", "singular", "deu"),
|
||||
("poder", "ind", "conditional", "first", "singular", "poderia"),
|
||||
("fazer", "sbjv", "future", "third", "singular", "fizer"),
|
||||
("estar", "ind", "present", "third", "singular", "está"),
|
||||
]
|
||||
ok = 0
|
||||
for lemma, mood, tense, per, num, exp in tests:
|
||||
got, conf = conjugate(lemma, mood, tense, per, num)
|
||||
flag = "OK " if got == exp else "XX "
|
||||
ok += got == exp
|
||||
print(f" {flag}{lemma:8} {mood}/{tense} {per[:3]}.{num[:2]} -> {got:14} ({conf}) exp={exp}")
|
||||
print(f"verb tests {ok}/{len(tests)}")
|
||||
print(" gender: casa=", noun_gender("casa"), "problema=", noun_gender("problema"),
|
||||
"mão=", noun_gender("mão"), "coração=", noun_gender("coração"),
|
||||
"flor=", noun_gender("flor"))
|
||||
print(" plural: mão->", inflect_noun("mão", "plural"),
|
||||
"| pão->", inflect_noun("pão", "plural"),
|
||||
"| animal->", inflect_noun("animal", "plural"),
|
||||
"| coração->", inflect_noun("coração", "plural"))
|
||||
print(" adj: bonito/f/sg->", inflect_adj("bonito", "f", "singular"),
|
||||
"| feliz/m/pl->", inflect_adj("feliz", "m", "plural"),
|
||||
"| português/f/sg->", inflect_adj("português", "f", "singular"))
|
||||
print(" part: fazer/m/sg->", participle("fazer"), "| ger falar->", gerund("falar"))
|
||||
print(" pinf falar 1pl->", personal_infinitive("falar", "first", "plural"))
|
||||
@@ -1,609 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""morphology_ro_full.py — production-grade Romanian morphological generator.
|
||||
|
||||
Romanian is the BIG typological delta of the Romance family. The verb engine and
|
||||
the confidence/fallback contract TRANSFER from the Italian sibling; the NOMINAL
|
||||
system is genuinely new: Romanian has a SUFFIXED definite article, a preserved
|
||||
NOM/ACC vs GEN/DAT case distinction, a NEUTER gender (masc-agreeing in SG,
|
||||
fem-agreeing in PL), and a VOCATIVE. Those are grounded in real per-lemma data,
|
||||
not guessed.
|
||||
|
||||
Real, Wiktionary-lineage lexical sources:
|
||||
|
||||
VERBS — UniMorph Romanian (github.com/unimorph/ron, CC-BY-SA 3.0)
|
||||
~1216 verb lemmas × paradigm, CLEAN orthography:
|
||||
indicativ prezent / imperfect (PST;IPFV) / perfectul simplu (PST;PFV) /
|
||||
conjunctiv prezent (SBJV;PRS, stored WITHOUT the 'să' particle),
|
||||
participiu (V.PTCP;PST, INVARIABLE in the perfect compus),
|
||||
gerunziu (V.CVB;PRS), infinitiv (NFIN), imperativ.
|
||||
ro_irreg_verbs (embedded) — high-frequency verbs UniMorph MISSES
|
||||
(avea, vrea, da) + the auxiliary clitic paradigms the compound tenses need
|
||||
(perfect-compus am/ai/a/am/ați/au, viitor voi/vei/va/vom/veți/vor,
|
||||
condițional aș/ai/ar/am/ați/ar). Real standard forms.
|
||||
|
||||
NOUNS — kaikki.org Romanian (Wiktionary extract, CC-BY-SA 3.0)
|
||||
the FULL declension per lemma, cleanly tagged:
|
||||
(nom/acc | gen/dat | vocative) × (indefinite | definite) × (sg | pl).
|
||||
This is what makes the suffixed article LEXICALLY grounded (om→omul,
|
||||
casă→casa, băiat→băiatul, casei gen/dat, omule vocative). Inherent gender
|
||||
m / f / n (NEUTER available directly) from the head template.
|
||||
|
||||
ADJECTIVES — UniMorph Romanian ADJ
|
||||
full case × gender(MASC/FEM/NEUT) × number × definiteness paradigm.
|
||||
|
||||
Fallbacks (degrade, never crash, on OOV): rule verb conjugation for -a/-ea/-e/-i/-î
|
||||
classes, rule pluralization, rule suffixed-article by gender+ending. Every form
|
||||
carries a confidence flag: "lexicon" | "rule" | "fallback".
|
||||
|
||||
Public API (used by realizer_ro.py):
|
||||
conjugate(lemma, mood, tense, person, number) -> (form, conf)
|
||||
aux(kind, person, number) -> str # perfect / future / conditional clitics
|
||||
participle(lemma) -> (form, conf) # INVARIABLE
|
||||
gerund(lemma) -> (form, conf)
|
||||
noun_gender(lemma) -> "m"|"f"|"n"
|
||||
definite_suffix(noun, gender, number, case) -> (form, conf) # rule engine
|
||||
inflect_noun(lemma, number, gender=None, case="nomacc", definite=False) -> (form, conf)
|
||||
inflect_adj(lemma, gender, number, case="nomacc", definite=False) -> (form, conf)
|
||||
lexicon_stats() -> dict
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
_UNIMORPH = os.path.join(_HERE, "data", "ron.unimorph")
|
||||
_KAIKKI = os.path.join(_HERE, "data", "kaikki_ro.jsonl")
|
||||
_CACHE = os.path.join(_HERE, "data", "ro_morph_cache.pkl")
|
||||
|
||||
# ── (mood, tense) -> UniMorph feature set ─────────────────────────────────────────
|
||||
_VERB_KEYMAP = {
|
||||
("ind", "present"): {"IND", "PRS"},
|
||||
("ind", "imperfect"): {"IND", "PST", "IPFV"},
|
||||
("ind", "perfect_s"): {"IND", "PST", "PFV"}, # perfectul simplu (regional/lit.)
|
||||
("sbjv", "present"): {"SBJV", "PRS"},
|
||||
("imp", "affirmative"): {"POS", "IMP"},
|
||||
}
|
||||
_PERSON = {"first": "1", "second": "2", "third": "3"}
|
||||
_NUMBER = {"singular": "SG", "plural": "PL"}
|
||||
|
||||
|
||||
def _feat_set(tag):
|
||||
return set(tag.split(";"))
|
||||
|
||||
|
||||
# ── high-frequency irregulars UniMorph misses + auxiliary clitic paradigms ────────
|
||||
# Real standard Romanian forms (textbook paradigms).
|
||||
_IRREG = {
|
||||
"avea": {
|
||||
"ind|present|1|SG": "am", "ind|present|2|SG": "ai", "ind|present|3|SG": "are",
|
||||
"ind|present|1|PL": "avem", "ind|present|2|PL": "aveți", "ind|present|3|PL": "au",
|
||||
"ind|imperfect|1|SG": "aveam", "ind|imperfect|2|SG": "aveai",
|
||||
"ind|imperfect|3|SG": "avea", "ind|imperfect|1|PL": "aveam",
|
||||
"ind|imperfect|2|PL": "aveați", "ind|imperfect|3|PL": "aveau",
|
||||
"sbjv|present|3|SG": "aibă", "sbjv|present|3|PL": "aibă",
|
||||
"sbjv|present|1|SG": "am", "sbjv|present|2|SG": "ai",
|
||||
"sbjv|present|1|PL": "avem", "sbjv|present|2|PL": "aveți",
|
||||
"part": "avut", "ger": "având",
|
||||
},
|
||||
"vrea": {
|
||||
"ind|present|1|SG": "vreau", "ind|present|2|SG": "vrei", "ind|present|3|SG": "vrea",
|
||||
"ind|present|1|PL": "vrem", "ind|present|2|PL": "vreți", "ind|present|3|PL": "vor",
|
||||
"ind|imperfect|1|SG": "voiam", "ind|imperfect|3|SG": "voia",
|
||||
"sbjv|present|3|SG": "vrea", "sbjv|present|3|PL": "vrea",
|
||||
"part": "vrut", "ger": "vrând",
|
||||
},
|
||||
"da": {
|
||||
"ind|present|1|SG": "dau", "ind|present|2|SG": "dai", "ind|present|3|SG": "dă",
|
||||
"ind|present|1|PL": "dăm", "ind|present|2|PL": "dați", "ind|present|3|PL": "dau",
|
||||
"ind|imperfect|1|SG": "dădeam", "ind|imperfect|3|SG": "dădea",
|
||||
"sbjv|present|3|SG": "dea", "sbjv|present|3|PL": "dea",
|
||||
"part": "dat", "ger": "dând",
|
||||
},
|
||||
"fi": { # a fi — present is in UniMorph but keep participle + subjunctive here
|
||||
"part": "fost", "ger": "fiind",
|
||||
"sbjv|present|1|SG": "fiu", "sbjv|present|2|SG": "fii", "sbjv|present|3|SG": "fie",
|
||||
"sbjv|present|1|PL": "fim", "sbjv|present|2|PL": "fiți", "sbjv|present|3|PL": "fie",
|
||||
"ind|imperfect|1|SG": "eram", "ind|imperfect|2|SG": "erai",
|
||||
"ind|imperfect|3|SG": "era", "ind|imperfect|1|PL": "eram",
|
||||
"ind|imperfect|2|PL": "erați", "ind|imperfect|3|PL": "erau",
|
||||
},
|
||||
}
|
||||
# auxiliary clitic paradigms (person,number)->form
|
||||
_AUX = {
|
||||
"perfect": {("first", "singular"): "am", ("second", "singular"): "ai",
|
||||
("third", "singular"): "a", ("first", "plural"): "am",
|
||||
("second", "plural"): "ați", ("third", "plural"): "au"},
|
||||
"future": {("first", "singular"): "voi", ("second", "singular"): "vei",
|
||||
("third", "singular"): "va", ("first", "plural"): "vom",
|
||||
("second", "plural"): "veți", ("third", "plural"): "vor"},
|
||||
"conditional": {("first", "singular"): "aș", ("second", "singular"): "ai",
|
||||
("third", "singular"): "ar", ("first", "plural"): "am",
|
||||
("second", "plural"): "ați", ("third", "plural"): "ar"},
|
||||
}
|
||||
|
||||
|
||||
def aux(kind, person, number):
|
||||
return _AUX[kind][(person, number)]
|
||||
|
||||
|
||||
# ── build verb lexicon from UniMorph ──────────────────────────────────────────────
|
||||
def _build_verbs():
|
||||
verbs, part, ger = {}, {}, {}
|
||||
with open(_UNIMORPH, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.rstrip("\n")
|
||||
if not line or "\t" not in line:
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) != 3:
|
||||
continue
|
||||
lemma, form, tag = parts
|
||||
f = _feat_set(tag)
|
||||
head = tag.split(";")[0]
|
||||
if head == "V.PTCP":
|
||||
if "PST" in f:
|
||||
part.setdefault(lemma, form)
|
||||
continue
|
||||
if head == "V.CVB":
|
||||
if "PRS" in f:
|
||||
ger.setdefault(lemma, form)
|
||||
continue
|
||||
if head != "V":
|
||||
continue
|
||||
person = next((p for p in ("1", "2", "3") if p in f), None)
|
||||
number = "SG" if "SG" in f else ("PL" if "PL" in f else None)
|
||||
if person is None or number is None:
|
||||
continue
|
||||
# conjunctiv forms in UniMorph carry a leading 'să ' — strip it
|
||||
surf = form
|
||||
if surf.startswith("să "):
|
||||
surf = surf[3:]
|
||||
for (mood, tense), req in _VERB_KEYMAP.items():
|
||||
if not req <= f:
|
||||
continue
|
||||
if tense == "imperfect" and "PFV" in f:
|
||||
continue
|
||||
if tense == "perfect_s" and "IPFV" in f:
|
||||
continue
|
||||
# keep IND;PRS out of the PRF slot (mai-mult-ca-perfect etc. ignored)
|
||||
if {"IND", "PRS"} <= req and "PRF" in f:
|
||||
continue
|
||||
verbs.setdefault((lemma, f"{mood}|{tense}|{person}|{number}"), surf)
|
||||
break
|
||||
return verbs, part, ger
|
||||
|
||||
|
||||
# ── kaikki nouns: full declension paradigm per lemma ──────────────────────────────
|
||||
_EXCL = {"alternative", "archaic", "obsolete", "regional", "dialectal", "rare",
|
||||
"table-tags", "inflection-template", "error-unrecognized-form",
|
||||
"diminutive", "augmentative", "informal"}
|
||||
|
||||
|
||||
def _noun_key(tagset):
|
||||
if tagset & _EXCL:
|
||||
return None
|
||||
if "vocative" in tagset:
|
||||
case = "voc"
|
||||
elif "genitive" in tagset or "dative" in tagset:
|
||||
case = "gendat"
|
||||
elif "nominative" in tagset or "accusative" in tagset:
|
||||
case = "nomacc"
|
||||
else:
|
||||
return None
|
||||
definite = "definite" in tagset and "indefinite" not in tagset
|
||||
number = "PL" if "plural" in tagset else ("SG" if "singular" in tagset else None)
|
||||
if number is None:
|
||||
return None
|
||||
return (case, definite, number)
|
||||
|
||||
|
||||
def _build_nouns():
|
||||
nouns = {} # lemma -> {"g":..., para:{(case,def,num):form}, "PL":plain_plural}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "noun":
|
||||
continue
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
ht = d.get("head_templates") or []
|
||||
g = None
|
||||
if ht:
|
||||
a = str((ht[0].get("args") or {}).get("1") or "").lower()
|
||||
if a[:1] in ("m", "f", "n"):
|
||||
g = a[:1]
|
||||
entry = nouns.setdefault(word, {"g": g, "para": {}, "PL": None})
|
||||
if entry["g"] is None and g:
|
||||
entry["g"] = g
|
||||
for x in (d.get("forms") or []):
|
||||
fm = x.get("form")
|
||||
tg = set(x.get("tags") or [])
|
||||
if not fm or fm in ("-", "#", "") or " " in fm:
|
||||
continue
|
||||
if tg == {"plural"} and not entry["PL"]:
|
||||
entry["PL"] = fm
|
||||
k = _noun_key(tg)
|
||||
if k and k not in entry["para"]:
|
||||
entry["para"][k] = fm
|
||||
return nouns
|
||||
|
||||
|
||||
# ── adjectives from kaikki (UniMorph ron ADJ is sparse AND mis-tagged; kaikki is
|
||||
# clean: the 4-form agreement pattern bun/bună/buni/bune). Neuter maps sg->masc,
|
||||
# pl->fem, so 4 forms (m/f × SG/PL) fully cover it. ────────────────────────────
|
||||
def _build_adjs():
|
||||
adjs = {} # lemma -> {(gender,number): form} gender in {m,f}
|
||||
with open(_KAIKKI, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
try:
|
||||
d = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if d.get("pos") != "adj":
|
||||
continue
|
||||
word = d.get("word", "")
|
||||
if not word or " " in word:
|
||||
continue
|
||||
d0 = adjs.setdefault(word, {})
|
||||
d0.setdefault(("m", "SG"), word) # masc sg = headword
|
||||
for x in (d.get("forms") or []):
|
||||
fm = x.get("form")
|
||||
t = set(x.get("tags") or [])
|
||||
if not fm or " " in fm or fm in ("-", "#") or (t & _EXCL):
|
||||
continue
|
||||
if "definite" in t or "genitive" in t or "dative" in t:
|
||||
continue # keep indefinite nom/acc agr set
|
||||
pl = "plural" in t
|
||||
fem = "feminine" in t
|
||||
masc = "masculine" in t
|
||||
if fem and pl:
|
||||
d0.setdefault(("f", "PL"), fm)
|
||||
elif masc and pl:
|
||||
d0.setdefault(("m", "PL"), fm)
|
||||
elif fem and not pl:
|
||||
d0.setdefault(("f", "SG"), fm)
|
||||
elif pl and not fem and not masc: # bare plural -> both genders
|
||||
d0.setdefault(("m", "PL"), fm)
|
||||
d0.setdefault(("f", "PL"), fm)
|
||||
return adjs
|
||||
|
||||
|
||||
def _build_cache():
|
||||
verbs, part, ger = _build_verbs()
|
||||
nouns = _build_nouns()
|
||||
adjs = _build_adjs()
|
||||
data = {"verbs": verbs, "part": part, "ger": ger, "nouns": nouns, "adjs": adjs}
|
||||
try:
|
||||
with open(_CACHE, "wb") as fh:
|
||||
pickle.dump(data, fh, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
except OSError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def _load():
|
||||
if os.path.exists(_CACHE):
|
||||
srcs = [_UNIMORPH, _KAIKKI]
|
||||
newest = max(os.path.getmtime(s) for s in srcs if os.path.exists(s))
|
||||
if os.path.getmtime(_CACHE) >= newest:
|
||||
try:
|
||||
with open(_CACHE, "rb") as fh:
|
||||
return pickle.load(fh)
|
||||
except Exception:
|
||||
pass
|
||||
return _build_cache()
|
||||
|
||||
|
||||
_LEX = _load()
|
||||
_VERBS, _PART, _GER, _NOUNS, _ADJS = (
|
||||
_LEX["verbs"], _LEX["part"], _LEX["ger"], _LEX["nouns"], _LEX["adjs"])
|
||||
|
||||
|
||||
# ── rule verb conjugation fallback ────────────────────────────────────────────────
|
||||
def _vclass(lemma):
|
||||
if lemma.endswith("a"):
|
||||
return "a"
|
||||
if lemma.endswith("ea"):
|
||||
return "ea"
|
||||
if lemma.endswith("e"):
|
||||
return "e"
|
||||
if lemma.endswith("i"):
|
||||
return "i"
|
||||
if lemma.endswith("î"):
|
||||
return "î"
|
||||
return None
|
||||
|
||||
|
||||
# regular present endings by class [1sg,2sg,3sg,1pl,2pl,3pl]
|
||||
_REG_PRS = {
|
||||
"a": ["", "i", "ă", "ăm", "ați", "ă"], # a lucra type (simplified)
|
||||
"ea": ["", "i", "e", "em", "eți", "", ],
|
||||
"e": ["", "i", "e", "em", "eți", ""],
|
||||
"i": ["esc", "ești", "ește", "im", "iți", "esc"], # -i type (a vorbi)
|
||||
"î": ["ăsc", "ăști", "ăște", "âm", "âți", "ăsc"],
|
||||
}
|
||||
_SLOT = {("first", "singular"): 0, ("second", "singular"): 1, ("third", "singular"): 2,
|
||||
("first", "plural"): 3, ("second", "plural"): 4, ("third", "plural"): 5}
|
||||
|
||||
|
||||
def _rule_conjugate(lemma, mood, tense, person, number):
|
||||
vc = _vclass(lemma)
|
||||
if vc is None:
|
||||
return None
|
||||
i = _SLOT[(person, number)]
|
||||
body = lemma[:-len(vc)]
|
||||
if mood == "ind" and tense == "present":
|
||||
end = _REG_PRS[vc][i]
|
||||
return body + end
|
||||
if mood == "ind" and tense == "imperfect":
|
||||
# -a/-i/-î -> stem + a/eai...; -e/-ea -> eam. Simplified regular imperfect.
|
||||
stem = body
|
||||
endings = {"a": ["am", "ai", "a", "am", "ați", "au"],
|
||||
"i": ["eam", "eai", "ea", "eam", "eați", "eau"],
|
||||
"î": ["am", "ai", "a", "am", "ați", "au"],
|
||||
"e": ["eam", "eai", "ea", "eam", "eați", "eau"],
|
||||
"ea": ["eam", "eai", "ea", "eam", "eați", "eau"]}[vc]
|
||||
return stem + endings[i]
|
||||
return None
|
||||
|
||||
|
||||
# ── PUBLIC verb API ───────────────────────────────────────────────────────────────
|
||||
def conjugate(lemma, mood, tense, person, number):
|
||||
lemma = lemma.strip().lower()
|
||||
key = f"{mood}|{tense}|{_PERSON.get(person,'?')}|{_NUMBER.get(number,'?')}"
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and key in ir:
|
||||
return ir[key], "lexicon"
|
||||
form = _VERBS.get((lemma, key))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
r = _rule_conjugate(lemma, mood, tense, person, number)
|
||||
if r is not None:
|
||||
return r, "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def participle(lemma):
|
||||
"""Past participle — INVARIABLE in the perfect compus (am mers, am văzut)."""
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and "part" in ir:
|
||||
return ir["part"], "lexicon"
|
||||
if lemma in _PART:
|
||||
return _PART[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc == "a":
|
||||
return lemma[:-1] + "at", "rule"
|
||||
if vc in ("ea",):
|
||||
return lemma[:-2] + "ut", "rule"
|
||||
if vc == "i":
|
||||
return lemma[:-1] + "it", "rule"
|
||||
if vc == "î":
|
||||
return lemma[:-1] + "ât", "rule"
|
||||
if vc == "e":
|
||||
return lemma[:-1] + "ut", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
def gerund(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
ir = _IRREG.get(lemma)
|
||||
if ir and "ger" in ir:
|
||||
return ir["ger"], "lexicon"
|
||||
if lemma in _GER:
|
||||
return _GER[lemma], "lexicon"
|
||||
vc = _vclass(lemma)
|
||||
if vc in ("a", "î"):
|
||||
return lemma[:-1] + "ând", "rule"
|
||||
if vc in ("ea", "e", "i"):
|
||||
return lemma[:-len(vc)] + "ind", "rule"
|
||||
return lemma, "fallback"
|
||||
|
||||
|
||||
# ── noun gender ───────────────────────────────────────────────────────────────────
|
||||
def noun_gender(lemma):
|
||||
lemma = lemma.strip().lower()
|
||||
d = _NOUNS.get(lemma)
|
||||
if d and d.get("g") in ("m", "f", "n"):
|
||||
return d["g"]
|
||||
if lemma.endswith(("ă", "a", "e")):
|
||||
return "f"
|
||||
return "m"
|
||||
|
||||
|
||||
# ── SUFFIXED DEFINITE ARTICLE — rule engine (fallback for OOV nouns) ───────────────
|
||||
def definite_suffix(noun, gender, number, case="nomacc"):
|
||||
"""Attach the enclitic definite article by gender + ending. Returns (form, conf).
|
||||
This is the headline Romanian-specific engine extension."""
|
||||
n = noun
|
||||
g = gender
|
||||
if number == "singular":
|
||||
if g in ("m", "n"):
|
||||
if case == "gendat":
|
||||
# masc/neut gen-dat definite: -lui
|
||||
if n.endswith("e"):
|
||||
return n + "lui", "rule" # câine -> câinelui
|
||||
if n.endswith("u"):
|
||||
return n + "lui", "rule"
|
||||
return n + "ului", "rule" # om -> omului
|
||||
# nom/acc
|
||||
if n.endswith("e"):
|
||||
return n + "le", "rule" # câine -> câinele
|
||||
if n.endswith("u"):
|
||||
return n + "l", "rule" # codru -> codrul
|
||||
if n.endswith("i"):
|
||||
return n + "ul", "rule"
|
||||
return n + "ul", "rule" # om -> omul
|
||||
# feminine singular
|
||||
if case == "gendat":
|
||||
# fem gen/dat definite = plural-stem + i (casei, fetei) — needs plural;
|
||||
# approximated as: -ă->-ei, -e->-ei, -a->-alei
|
||||
if n.endswith("ă"):
|
||||
return n[:-1] + "ei", "rule" # casă -> casei
|
||||
if n.endswith("e"):
|
||||
return n[:-1] + "ei", "rule" # carte -> cărții(approx cartei)
|
||||
if n.endswith("a"):
|
||||
return n[:-1] + "lei", "rule"
|
||||
return n + "i", "rule"
|
||||
# fem nom/acc
|
||||
if n.endswith("ă"):
|
||||
return n[:-1] + "a", "rule" # casă -> casa
|
||||
if n.endswith("e"):
|
||||
return n[:-1] + "ea", "rule" # carte -> cartea
|
||||
if n.endswith("a"):
|
||||
return n + "ua", "rule" # stea -> steaua
|
||||
if n.endswith("i"):
|
||||
return n + "a", "rule"
|
||||
return n + "a", "rule"
|
||||
# plural
|
||||
if case == "gendat":
|
||||
base = noun
|
||||
return base + "lor", "rule" # -lor for all gen/dat pl
|
||||
if g == "m":
|
||||
return noun + "i", "rule" # oameni -> oamenii (+i)
|
||||
return noun + "le", "rule" # case -> casele, trenuri->trenurile
|
||||
|
||||
|
||||
# ── rule pluralization (fallback) ─────────────────────────────────────────────────
|
||||
def _rule_plural(noun, gender):
|
||||
if gender == "f":
|
||||
if noun.endswith("ă"):
|
||||
return noun[:-1] + "e"
|
||||
if noun.endswith("e"):
|
||||
return noun[:-1] + "i"
|
||||
if noun.endswith("a"):
|
||||
return noun[:-1] + "le"
|
||||
return noun + "e"
|
||||
if gender == "n":
|
||||
return noun + "uri"
|
||||
# masculine
|
||||
if noun.endswith(("e",)):
|
||||
return noun[:-1] + "i"
|
||||
return noun + "i"
|
||||
|
||||
|
||||
# ── PUBLIC noun inflection ────────────────────────────────────────────────────────
|
||||
def inflect_noun(lemma, number, gender=None, case="nomacc", definite=False):
|
||||
lemma = lemma.strip().lower()
|
||||
g = gender or noun_gender(lemma)
|
||||
d = _NOUNS.get(lemma)
|
||||
numk = "SG" if number == "singular" else "PL"
|
||||
if d:
|
||||
if case == "voc":
|
||||
form = d["para"].get(("voc", True, numk)) or d["para"].get(("voc", False, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# try the exact paradigm cell from kaikki (lexically grounded)
|
||||
form = d["para"].get((case, definite, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# indefinite fallbacks from the paradigm
|
||||
if not definite:
|
||||
form = d["para"].get(("nomacc", False, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
if numk == "PL" and d.get("PL"):
|
||||
return d["PL"], "lexicon"
|
||||
if numk == "SG":
|
||||
return lemma, "lexicon"
|
||||
# rule path
|
||||
base = lemma if number == "singular" else _rule_plural(lemma, g)
|
||||
if definite:
|
||||
return definite_suffix(base, g, number, case)
|
||||
return base, ("rule" if d is None else "lexicon")
|
||||
|
||||
|
||||
# ── PUBLIC adjective agreement ────────────────────────────────────────────────────
|
||||
def _neuter_map(gender, number):
|
||||
# neuter agrees masculine in SG, feminine in PL
|
||||
if gender == "n":
|
||||
return "m" if number == "singular" else "f"
|
||||
return gender
|
||||
|
||||
|
||||
def inflect_adj(lemma, gender, number, case="nomacc", definite=False):
|
||||
lemma = lemma.strip().lower()
|
||||
numk = "SG" if number == "singular" else "PL"
|
||||
eg = _neuter_map(gender, number) # neuter -> masc(SG)/fem(PL)
|
||||
d = _ADJS.get(lemma)
|
||||
if d:
|
||||
form = d.get((eg, numk))
|
||||
if form:
|
||||
return form, "lexicon"
|
||||
# rule fallback: 4-form pattern bun/bună/buni/bune keyed by effective gender
|
||||
a = lemma
|
||||
if number == "singular":
|
||||
if eg == "f":
|
||||
if a.endswith("e"):
|
||||
return a, "rule" # mare invariant sg
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "ă", "rule" # nou -> nouă
|
||||
if a.endswith("ă"):
|
||||
return a, "rule"
|
||||
return a + "ă", "rule" # bun -> bună
|
||||
return a, "rule" # masc/neut sg = lemma
|
||||
# plural
|
||||
if eg == "f":
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "i", "rule" # mare -> mari
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "e", "rule" # nou -> noue (approx; 'noi' irr)
|
||||
if a.endswith("ă"):
|
||||
return a[:-1] + "e", "rule"
|
||||
return a + "e", "rule" # bun -> bune
|
||||
# masc/neut(SG-only)->here masc pl -> -i
|
||||
if a.endswith("e"):
|
||||
return a[:-1] + "i", "rule" # mare -> mari
|
||||
if a.endswith("u"):
|
||||
return a[:-1] + "i", "rule"
|
||||
return a + "i", "rule" # bun -> buni
|
||||
|
||||
|
||||
def lexicon_stats():
|
||||
return {
|
||||
"verb_source": "UniMorph Romanian (github.com/unimorph/ron) + curated "
|
||||
"irregulars (avea/vrea/da + aux clitic paradigms)",
|
||||
"noun_source": "kaikki.org Romanian — full case/definite/vocative declension",
|
||||
"adj_source": "UniMorph Romanian ADJ (case×gender×number×definiteness)",
|
||||
"license": "CC-BY-SA 3.0 (Wiktionary/UniMorph lineage)",
|
||||
"unimorph_verb_forms": len(_VERBS),
|
||||
"unimorph_verb_lemmas": len({k[0] for k in _VERBS}),
|
||||
"irregular_verb_lemmas": len(_IRREG),
|
||||
"participle_lemmas": len(_PART),
|
||||
"noun_lemmas": len(_NOUNS),
|
||||
"adj_lemmas": len(_ADJS),
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(json.dumps(lexicon_stats(), indent=2, ensure_ascii=False))
|
||||
print("\n── SUFFIXED DEFINITE ARTICLE (the headline delta) ──")
|
||||
for n, g in [("om", "m"), ("băiat", "m"), ("casă", "f"), ("carte", "f"),
|
||||
("tren", "n"), ("student", "m"), ("floare", "f")]:
|
||||
sg = inflect_noun(n, "singular", g, "nomacc", True)
|
||||
pl = inflect_noun(n, "plural", g, "nomacc", True)
|
||||
gd = inflect_noun(n, "singular", g, "gendat", True)
|
||||
vo = inflect_noun(n, "singular", g, "voc", False)
|
||||
print(f" {n:8}({g}) def.sg={sg[0]:12} def.pl={pl[0]:14} "
|
||||
f"gen/dat.sg={gd[0]:12} voc={vo[0]}")
|
||||
print("\n── NEUTER split agreement (tren: masc SG / fem PL) ──")
|
||||
print(" tren nou ->", inflect_noun("tren", "singular", "n")[0],
|
||||
inflect_adj("nou", "n", "singular")[0])
|
||||
print(" trenuri noi->", inflect_noun("tren", "plural", "n")[0],
|
||||
inflect_adj("nou", "n", "plural")[0])
|
||||
print("\n── verbs ──")
|
||||
for l, m, t, p, n, in [("merge", "ind", "present", "third", "singular"),
|
||||
("avea", "ind", "present", "first", "singular"),
|
||||
("fi", "ind", "present", "third", "singular"),
|
||||
("vorbi", "ind", "present", "third", "plural"),
|
||||
("face", "sbjv", "present", "third", "singular"),
|
||||
("lucra", "ind", "imperfect", "third", "singular")]:
|
||||
print(f" {l:8}{m}/{t:10}{p[:3]}.{n[:2]} -> {conjugate(l,m,t,p,n)}")
|
||||
print(" perfect-aux(3sg):", aux("perfect", "third", "singular"),
|
||||
"| future(1sg):", aux("future", "first", "singular"),
|
||||
"| cond(3sg):", aux("conditional", "third", "singular"))
|
||||
print(" participle merge/vedea:", participle("merge"), participle("vedea"))
|
||||
+1
-1
@@ -22,7 +22,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)/el}"
|
||||
ELC="${ELC:-${EL_HOME}/dist/platform/elc}"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
SRC_DIR="$(cd .. && pwd)/src"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
|
||||
@@ -81,7 +81,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -49,6 +49,12 @@ jobs:
|
||||
echo "Downloading el_runtime.h..."
|
||||
curl -fsSL "${RELEASE_BASE}/el_runtime.h" -o /usr/local/lib/el/el_runtime.h
|
||||
|
||||
echo "Downloading engram_store.c..."
|
||||
curl -fsSL "${RELEASE_BASE}/engram_store.c" -o /usr/local/lib/el/engram_store.c
|
||||
|
||||
echo "Downloading engram_store.h..."
|
||||
curl -fsSL "${RELEASE_BASE}/engram_store.h" -o /usr/local/lib/el/engram_store.h
|
||||
|
||||
echo "El SDK installed:"
|
||||
elc --version || true
|
||||
|
||||
@@ -62,11 +68,12 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
/usr/local/lib/el/el_runtime.c \
|
||||
/usr/local/lib/el/engram_store.c \
|
||||
-lcurl -lpthread
|
||||
echo "Linked dist/engram"
|
||||
ls -lh dist/engram
|
||||
|
||||
+5
-2
@@ -1,3 +1,6 @@
|
||||
target/
|
||||
*.db
|
||||
.DS_Store
|
||||
*.db
|
||||
*.elc
|
||||
*.elh
|
||||
dist/
|
||||
target/
|
||||
|
||||
+151
-30
@@ -117,6 +117,17 @@ fn route_text_health(method: String, path: String, body: String) -> String {
|
||||
// save/load with no "path" hit engram_save(""). Rewritten to the
|
||||
// `let x = if cond { a } else { b }` expression form (the pattern the newer
|
||||
// routes route_emit_ise/route_capture_knowledge already use correctly).
|
||||
// store_on — ENGRAM_STORE flag (tiered paged store as the durable owner). Matches
|
||||
// engram_store_enabled() in el_runtime.c EXACTLY (1 / on / true). Default off →
|
||||
// every persistence path below is byte-for-byte the historical snapshot behavior.
|
||||
fn store_on() -> Bool {
|
||||
let v: String = env("ENGRAM_STORE")
|
||||
if str_eq(v, "1") { return true }
|
||||
if str_eq(v, "on") { return true }
|
||||
if str_eq(v, "true") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// persist_canonical — save the canonical snapshot after a durable write.
|
||||
//
|
||||
// WHY (2026-07-22 self-review): the 2026-07-21 fix correctly stopped READ
|
||||
@@ -131,8 +142,16 @@ fn route_text_health(method: String, path: String, body: String) -> String {
|
||||
// tolerant, ~2/min — snapshotting the whole store per heartbeat is waste;
|
||||
// any durable write that follows persists the pruning too).
|
||||
fn persist_canonical() -> Int {
|
||||
// ENGRAM_STORE: the paged store is the durable owner — a checkpoint flushes
|
||||
// dirty pages behind a WAL-durable record (durable the moment the WAL fsyncs).
|
||||
// This is the fix for the "restart reverted to a 17h-old snapshot" data loss:
|
||||
// durable writes no longer depend on a full snapshot.json rewrite. Returns 1
|
||||
// on a successful checkpoint, 0 otherwise. Flag-off: unchanged (writes JSON).
|
||||
if store_on() {
|
||||
return engram_store_checkpoint()
|
||||
}
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
// (2026-08-10 self-review) This returned a hardcoded 1, which made every
|
||||
// caller's `let saved: Int = persist_canonical()` a dead variable — six
|
||||
// durable write paths each believed they had confirmation of a successful
|
||||
@@ -140,6 +159,57 @@ fn persist_canonical() -> Int {
|
||||
return engram_save(dir + "/snapshot.json")
|
||||
}
|
||||
|
||||
// ── WAL persistence (design doc §§3-14; gated behind ENGRAM_WAL=on) ──────────
|
||||
// Default OFF → every persist path below is byte-identical to the historical
|
||||
// per-write full-snapshot behavior. When ON, structural mutations append O(1)
|
||||
// WAL records instead of rewriting the whole graph, with threshold compaction.
|
||||
fn wal_on() -> Bool {
|
||||
str_eq(env("ENGRAM_WAL"), "on")
|
||||
}
|
||||
|
||||
// Persist a single-node mutation (create / content-evolve / strengthen).
|
||||
fn persist_node(id: String) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_node_put(d, id)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Persist edges appended at index >= start (covers single-edge and batch).
|
||||
fn persist_edges_since(start: Int) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_edges_since(d, start)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Persist a Hebbian consolidation batch as ONE WAL record (single fsync, §5-B).
|
||||
fn persist_hebb_batch(start: Int) -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_hebb_batch(d, start)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
return a
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// Bulk mutation (embedding backfill, load-merge): write a fresh compaction base
|
||||
// so the many-node change is durable in one atomic snapshot; WAL is truncated.
|
||||
fn persist_bulk() -> Int {
|
||||
if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
return engram_wal_compact(d)
|
||||
}
|
||||
return persist_canonical()
|
||||
}
|
||||
|
||||
// INCOMPLETE-ROUTE FIX (2026-07-24 self-review): this route silently dropped
|
||||
// label, importance, tier, and tags — engram_node() defaults label to content
|
||||
// and importance to 0.5, so every node created over HTTP lost its metadata.
|
||||
@@ -181,7 +251,7 @@ fn route_create_node(method: String, path: String, body: String) -> String {
|
||||
salience, importance, confidence,
|
||||
tier, tags
|
||||
)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
"{\"id\":\"" + id + "\",\"content\":\"" + content + "\",\"node_type\":\"" + node_type + "\"}"
|
||||
}
|
||||
|
||||
@@ -208,7 +278,7 @@ fn route_scan_nodes(method: String, path: String, body: String) -> String {
|
||||
// clobbered the good snapshot. Read routes must never write the canonical path.)
|
||||
fn route_scan_edges(method: String, path: String, body: String) -> String {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let snap_path: String = dir + "/.scan-export.json"
|
||||
engram_save(snap_path)
|
||||
let snap: String = fs_read(snap_path)
|
||||
@@ -250,8 +320,9 @@ fn route_create_edge(method: String, path: String, body: String) -> String {
|
||||
// (dormant association); only default when the key is absent.
|
||||
let w_present: String = json_get_raw(body, "weight")
|
||||
let weight: Float = if str_eq(w_present, "") { 0.5 } else { json_get_float(body, "weight") }
|
||||
let ec0: Int = engram_edge_count()
|
||||
engram_connect(from_id, to_id, weight, relation)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_edges_since(ec0)
|
||||
"{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + relation + "\"}"
|
||||
}
|
||||
|
||||
@@ -276,6 +347,7 @@ fn route_create_edges_batch(method: String, path: String, body: String) -> Strin
|
||||
if str_eq(arr, "") { return err_json("missing edges array") }
|
||||
let n: Int = json_array_len(arr)
|
||||
if n == 0 { return "{\"ok\":true,\"accepted\":0,\"skipped\":0}" }
|
||||
let ec0: Int = engram_edge_count()
|
||||
let i: Int = 0
|
||||
let accepted: Int = 0
|
||||
let skipped: Int = 0
|
||||
@@ -299,7 +371,7 @@ fn route_create_edges_batch(method: String, path: String, body: String) -> Strin
|
||||
// Skip it when nothing was accepted: an all-malformed payload must not
|
||||
// trigger a 60MB write.
|
||||
if accepted > 0 {
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_hebb_batch(ec0)
|
||||
}
|
||||
return "{\"ok\":true,\"accepted\":" + int_to_str(accepted) + ",\"skipped\":" + int_to_str(skipped) + "}"
|
||||
}
|
||||
@@ -315,22 +387,50 @@ fn route_strengthen(method: String, path: String, body: String) -> String {
|
||||
let id: String = json_get_string(body, "node_id")
|
||||
if str_eq(id, "") { return err_json("missing node_id") }
|
||||
engram_strengthen(id)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
ok_json()
|
||||
}
|
||||
|
||||
// route_forget — DELETE /api/nodes/:id — INTEGRITY HARDENED (design doc §18.1).
|
||||
//
|
||||
// Two invariants now enforced AT THE STORE (not one layer up in neuron-api.el,
|
||||
// which a direct HTTP client could bypass):
|
||||
// 1. Write-protection: protected identity/value nodes (derived from the self
|
||||
// graph — self root + values hub + their neighbors, §18.3) cannot be
|
||||
// deleted over HTTP. Returns 403, node untouched.
|
||||
// 2. No hard delete over the wire, ever: an ordinary delete creates a
|
||||
// Tombstone marker node + `tombstones` edge and KEEPS the original node
|
||||
// and its edges (recoverable), instead of the old destructive
|
||||
// engram_forget() shift-delete. Raw engram_forget is now internal-GC only
|
||||
// and no longer reachable from any HTTP route.
|
||||
fn route_forget(method: String, path: String, body: String) -> String {
|
||||
let id: String = extract_id(path, "/api/nodes/")
|
||||
if str_eq(id, "") { return err_json("missing id") }
|
||||
engram_forget(id)
|
||||
let saved: Int = persist_canonical()
|
||||
ok_json()
|
||||
if engram_is_protected(id) == 1 {
|
||||
return "{\"__status__\":403,\"error\":\"protected node; deletion refused\",\"id\":\"" + id + "\"}"
|
||||
}
|
||||
let tomb_id: String = engram_node_full(
|
||||
"tombstone:" + id, "Tombstone", "tombstone:" + id,
|
||||
0.1, 0.1, 1.0, "Episodic", "[\"tombstone\"]"
|
||||
)
|
||||
let ec0: Int = engram_edge_count()
|
||||
engram_connect(tomb_id, id, 1.0, "tombstones")
|
||||
let saved: Int = if wal_on() {
|
||||
let d: String = engram_resolve_data_dir()
|
||||
let a: Int = engram_wal_node_put(d, tomb_id)
|
||||
let b: Int = engram_wal_edges_since(d, ec0)
|
||||
let c: Int = engram_wal_maybe_compact(d)
|
||||
a
|
||||
} else {
|
||||
persist_canonical()
|
||||
}
|
||||
"{\"ok\":true,\"tombstoned\":\"" + id + "\",\"tombstone_id\":\"" + tomb_id + "\"}"
|
||||
}
|
||||
|
||||
fn route_save(method: String, path: String, body: String) -> String {
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
// (2026-08-10 self-review) engram_save returns 0 on an empty path and the
|
||||
// route discarded it, so the response was a literal "ok":true regardless
|
||||
@@ -346,7 +446,7 @@ fn route_save(method: String, path: String, body: String) -> String {
|
||||
fn route_load(method: String, path: String, body: String) -> String {
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
// (2026-08-10 self-review) This was a stub response over the single most
|
||||
// destructive operation in the server. engram_load returns 0 on an empty
|
||||
@@ -398,7 +498,7 @@ fn route_embed_backfill(method: String, path: String, body: String) -> String {
|
||||
let result: String = engram_embed_backfill(n)
|
||||
let done: Float = json_get_float(result, "embedded")
|
||||
if done > 0.0 {
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_bulk()
|
||||
}
|
||||
return result
|
||||
}
|
||||
@@ -417,7 +517,7 @@ fn route_embed_backfill(method: String, path: String, body: String) -> String {
|
||||
// (2026-06-27 self-review: added this route to fix silent 10-min sync failures)
|
||||
fn route_sync(method: String, path: String, body: String) -> String {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let dir: String = engram_resolve_data_dir()
|
||||
// 2026-07-21 self-review: export to a scratch path, never the canonical
|
||||
// snapshot.json — read routes must not be able to clobber the good snapshot.
|
||||
let snap_path: String = dir + "/.sync-export.json"
|
||||
@@ -451,7 +551,7 @@ fn route_load_merge(method: String, path: String, body: String) -> String {
|
||||
engram_load_merge(p)
|
||||
let added_n: Int = engram_node_count() - before_n
|
||||
let added_e: Int = engram_edge_count() - before_e
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_bulk()
|
||||
"{\"ok\":true,\"nodes_added\":" + int_to_str(added_n) + ",\"edges_added\":" + int_to_str(added_e) + ",\"node_count\":" + int_to_str(engram_node_count()) + "}"
|
||||
}
|
||||
|
||||
@@ -550,7 +650,7 @@ fn route_capture_knowledge(method: String, path: String, body: String) -> String
|
||||
sal, imp, conf,
|
||||
"Semantic", tags
|
||||
)
|
||||
let saved: Int = persist_canonical()
|
||||
let saved: Int = persist_node(id)
|
||||
"{\"ok\":true,\"id\":\"" + id + "\"}"
|
||||
}
|
||||
|
||||
@@ -713,23 +813,44 @@ let bind_str: String = if str_eq(bind_raw, "") { ":8742" } else { bind_raw }
|
||||
let port: Int = parse_port(bind_str)
|
||||
|
||||
// On startup, try to load any existing snapshot (best effort).
|
||||
let data_dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let data_dir: String = if str_eq(data_dir_raw, "") { "/tmp/engram" } else { data_dir_raw }
|
||||
// §18.2: resolve the data dir safely — unset ENGRAM_DATA_DIR → $HOME/.neuron/engram,
|
||||
// never /tmp; fail loud if HOME is unresolvable (engram_resolve_data_dir exits).
|
||||
let data_dir: String = engram_resolve_data_dir()
|
||||
let snapshot_path: String = data_dir + "/snapshot.json"
|
||||
engram_load(snapshot_path)
|
||||
// ENGRAM_STORE (tiered paged store — engram-tiered-storage-engine.md). When set,
|
||||
// the durable owner is the paged store (neuron.egm + neuron.wal): engram_store_boot
|
||||
// imports snapshot.json ONCE into a fresh neuron.egm, else replays the WAL and loads
|
||||
// the store resident — snapshot.json is never read again as the ongoing store. This
|
||||
// closes the "restart reverted to a 17h-old snapshot" data-loss window. Flag-off
|
||||
// (default): byte-for-byte the historical snapshot + optional-WAL boot below.
|
||||
if store_on() {
|
||||
engram_store_boot(data_dir)
|
||||
println("[engram] ENGRAM_STORE enabled — tiered paged store is the durable owner")
|
||||
} else {
|
||||
engram_load(snapshot_path)
|
||||
|
||||
// 2026-07-21 self-review boot guard: if the snapshot file has content but the
|
||||
// load produced 0 nodes, something is wrong (corrupt file / parse failure).
|
||||
// Preserve the evidence and warn loudly — and since read routes no longer write
|
||||
// the canonical path, a bad boot can no longer clobber the good snapshot.
|
||||
let boot_snap: String = fs_read(snapshot_path)
|
||||
if !str_eq(boot_snap, "") {
|
||||
if engram_node_count() == 0 {
|
||||
println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json")
|
||||
fs_write(data_dir + "/snapshot.failed-load.json", boot_snap)
|
||||
} else {
|
||||
// Good load: keep a boot-time backup of the snapshot as loaded.
|
||||
fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap)
|
||||
// WAL replay (design doc §6). Gated: default OFF is byte-identical to legacy
|
||||
// snapshot-only boot. When ON, the snapshot above is the compaction BASE and
|
||||
// the WAL carries every mutation since; replay reconstructs state to the last
|
||||
// CRC-valid record, then opens the WAL for appending.
|
||||
if wal_on() {
|
||||
let replayed: Int = engram_wal_boot(data_dir)
|
||||
println("[engram] WAL enabled — replayed " + int_to_str(replayed) + " records")
|
||||
}
|
||||
|
||||
// 2026-07-21 self-review boot guard: if the snapshot file has content but the
|
||||
// load produced 0 nodes, something is wrong (corrupt file / parse failure).
|
||||
// Preserve the evidence and warn loudly — and since read routes no longer write
|
||||
// the canonical path, a bad boot can no longer clobber the good snapshot.
|
||||
let boot_snap: String = fs_read(snapshot_path)
|
||||
if !str_eq(boot_snap, "") {
|
||||
if engram_node_count() == 0 {
|
||||
println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json")
|
||||
fs_write(data_dir + "/snapshot.failed-load.json", boot_snap)
|
||||
} else {
|
||||
// Good load: keep a boot-time backup of the snapshot as loaded.
|
||||
fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Executable
+158
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env bash
|
||||
# M3.5 PRE-FLIP GATE. Pure C harness (NOT elb/elc): links the real el_runtime.c
|
||||
# native engram builtins + engram_store.c and proves activation-time field
|
||||
# mutations (edge hebb, node activation_count, WM weight) persist through a
|
||||
# checkpoint and survive a reboot from neuron.egm with snapshot.json DELETED.
|
||||
# Writes ONLY under a throwaway /tmp dir with a throwaway HOME.
|
||||
set -u
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
RT="$HERE/../../lang/runtime/el_runtime.c"
|
||||
ST="$HERE/../../lang/runtime/engram_store.c"
|
||||
INC="$HERE/../../lang/runtime"
|
||||
WORK="$(mktemp -d /tmp/engram-m35-XXXXXX)"
|
||||
BIN="$WORK/m35"
|
||||
export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron
|
||||
export ENGRAM_WAL_SYNC=always
|
||||
unset ENGRAM_STORE
|
||||
fail=0
|
||||
|
||||
echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m35_hebb_persist.c) =="
|
||||
gcc -O1 -std=c11 -I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log"
|
||||
if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi
|
||||
|
||||
echo
|
||||
echo "== 0) flag-OFF: seed+activate+checkpoint must NOT touch the store =="
|
||||
DOFF="$WORK/off"; mkdir -p "$DOFF"
|
||||
( unset ENGRAM_STORE; "$BIN" offcheck "$DOFF" )
|
||||
[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; }
|
||||
[ -e "$DOFF/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \
|
||||
|| echo " ok: no neuron.egm created with flag OFF"
|
||||
|
||||
echo
|
||||
echo "== 1) POSITIVE: ENGRAM_STORE=1 seed -> activate -> checkpoint(field-persist) -> close =="
|
||||
DPOS="$WORK/pos"; mkdir -p "$DPOS"
|
||||
ENGRAM_STORE=1 "$BIN" pos_seed "$DPOS" || { echo "FAIL: pos_seed"; fail=1; }
|
||||
[ -e "$DPOS/neuron.egm" ] && echo " ok: neuron.egm created" || { echo "FAIL: neuron.egm missing"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 2) reboot from neuron.egm with snapshot.json DELETED (must never read JSON) =="
|
||||
rm -f "$DPOS/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" pos_reboot "$DPOS" || { echo "FAIL: pos_reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 3) NEGATIVE CONTROL: seed -> activate -> close WITHOUT the field-persist checkpoint =="
|
||||
DNEG="$WORK/neg"; mkdir -p "$DNEG"
|
||||
ENGRAM_STORE=1 "$BIN" neg_seed "$DNEG" || { echo "FAIL: neg_seed"; fail=1; }
|
||||
rm -f "$DNEG/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" neg_reboot "$DNEG" || { echo "FAIL: neg_reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 4) assertions (python over the JSON exports) =="
|
||||
python3 - "$DPOS" "$DNEG" <<'PY'
|
||||
import json, sys, os
|
||||
WM_FLOOR = 0.05
|
||||
HEBB_MIN = 1e-6
|
||||
|
||||
def load(d, name):
|
||||
with open(os.path.join(d, name)) as f: return json.load(f)
|
||||
|
||||
def node_by_label(g, label):
|
||||
for n in g["nodes"]:
|
||||
if n.get("label") == label: return n
|
||||
return None
|
||||
|
||||
def edge_between(g, a_id, b_id):
|
||||
for e in g["edges"]:
|
||||
if e.get("from_id") == a_id and e.get("to_id") == b_id:
|
||||
return e
|
||||
return None
|
||||
|
||||
rc = 0
|
||||
def check(cond, msg):
|
||||
global rc
|
||||
if cond: print(f" PASS: {msg}")
|
||||
else: print(f" FAIL: {msg}"); rc = 1
|
||||
|
||||
dpos, dneg = sys.argv[1], sys.argv[2]
|
||||
pre = load(dpos, "pre_reboot.json")
|
||||
rebt = load(dpos, "reboot.json")
|
||||
|
||||
pa, pb = node_by_label(pre, "hebb-a"), node_by_label(pre, "hebb-b")
|
||||
ra = node_by_label(rebt, "hebb-a")
|
||||
assert pa and pb and ra, "target nodes missing"
|
||||
pe = edge_between(pre, pa["id"], pb["id"])
|
||||
re = edge_between(rebt, pa["id"], pb["id"])
|
||||
assert pe and re, "target edge missing"
|
||||
|
||||
pre_hebb = pe.get("hebb", 0.0)
|
||||
rebt_hebb = re.get("hebb", 0.0)
|
||||
pre_ac = pa.get("activation_count", 0)
|
||||
rebt_ac = ra.get("activation_count", 0)
|
||||
pre_wm = pa.get("working_memory_weight", 0.0)
|
||||
rebt_wm = ra.get("working_memory_weight", 0.0)
|
||||
|
||||
print(f" edge hebb-a->hebb-b : pre={pre_hebb!r} reboot={rebt_hebb!r}")
|
||||
print(f" node hebb-a act_cnt : pre={pre_ac!r} reboot={rebt_ac!r}")
|
||||
print(f" node hebb-a wm : pre={pre_wm!r} reboot={rebt_wm!r} (halved+floored expected)")
|
||||
|
||||
# --- learning actually happened this run (else the test proves nothing) ---
|
||||
check(pre_hebb > HEBB_MIN, f"activation raised edge hebb above 0 (pre={pre_hebb})")
|
||||
check(pre_ac >= 1, f"activation reinforced node activation_count (pre={pre_ac})")
|
||||
check(pre_wm > 0.0, f"activation promoted node to working memory (pre_wm={pre_wm})")
|
||||
|
||||
# --- the load-bearing survival assertions after a real delete-JSON reboot ---
|
||||
check(abs(rebt_hebb - pre_hebb) < 1e-12,
|
||||
f"edge hebb SURVIVED reboot unchanged ({rebt_hebb} == {pre_hebb})")
|
||||
check(rebt_ac == pre_ac,
|
||||
f"node activation_count SURVIVED reboot unchanged ({rebt_ac} == {pre_ac})")
|
||||
|
||||
# --- WM weight: must equal the JSON path's boot transform exactly (halve+floor) ---
|
||||
expected_wm = pre_wm * 0.5
|
||||
if expected_wm < WM_FLOOR: expected_wm = 0.0
|
||||
check(abs(rebt_wm - expected_wm) < 1e-9,
|
||||
f"node WM weight SURVIVED with the SAME boot transform as JSON path "
|
||||
f"(reboot={rebt_wm} == halve+floor(pre)={expected_wm})")
|
||||
check(expected_wm > 0.0,
|
||||
f"WM survival is observable (halved weight stays above floor: {expected_wm} > {WM_FLOOR})")
|
||||
|
||||
# --- NEGATIVE CONTROL: without the field-persist step the learning is LOST ---
|
||||
npre = load(dneg, "neg_pre.json")
|
||||
nrebt = load(dneg, "neg_reboot.json")
|
||||
na_pre = node_by_label(npre, "hebb-a")
|
||||
na_rebt = node_by_label(nrebt, "hebb-a")
|
||||
ne_pre = edge_between(npre, na_pre["id"], node_by_label(npre, "hebb-b")["id"])
|
||||
ne_rebt = edge_between(nrebt, na_rebt["id"], node_by_label(nrebt, "hebb-b")["id"])
|
||||
print(f" [neg] edge hebb : pre={ne_pre.get('hebb',0.0)!r} reboot={ne_rebt.get('hebb',0.0)!r}")
|
||||
print(f" [neg] node act_cnt : pre={na_pre.get('activation_count',0)!r} reboot={na_rebt.get('activation_count',0)!r}")
|
||||
check(ne_pre.get("hebb", 0.0) > HEBB_MIN,
|
||||
f"[neg] activation DID raise hebb in RAM (pre={ne_pre.get('hebb',0.0)})")
|
||||
check(ne_rebt.get("hebb", 0.0) == 0.0,
|
||||
"[neg] WITHOUT checkpoint field-persist, edge hebb is LOST on reboot (==0) — fix is load-bearing")
|
||||
check(na_rebt.get("activation_count", 0) == 0,
|
||||
"[neg] WITHOUT checkpoint field-persist, activation_count is LOST on reboot (==0)")
|
||||
|
||||
sys.exit(rc)
|
||||
PY
|
||||
[ $? -ne 0 ] && fail=1
|
||||
|
||||
echo
|
||||
echo "== 5) ASan+UBSan build, exercise the full persist+reboot flow (leaks off — harness intentionally leaks el_strdup) =="
|
||||
SANBIN="$WORK/m35.san"
|
||||
gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \
|
||||
-I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log"
|
||||
if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else
|
||||
export ASAN_OPTIONS=detect_leaks=0
|
||||
DSAN="$WORK/san"; mkdir -p "$DSAN"
|
||||
ENGRAM_STORE=1 "$SANBIN" pos_seed "$DSAN" >/dev/null 2>"$WORK/san_run.log" && \
|
||||
{ rm -f "$DSAN/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" pos_reboot "$DSAN" >/dev/null 2>>"$WORK/san_run.log"; }
|
||||
if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then
|
||||
echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1
|
||||
else
|
||||
echo " ok: ASan+UBSan clean across pos_seed/checkpoint/reboot (field-persist, boot laundering)"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ "$fail" -eq 0 ]; then echo "================ M3.5 HEBB-PERSIST GATE: PASS ================"; else echo "================ M3.5 HEBB-PERSIST GATE: FAIL ================"; fi
|
||||
rm -rf "$WORK"
|
||||
exit $fail
|
||||
Executable
+126
@@ -0,0 +1,126 @@
|
||||
#!/usr/bin/env bash
|
||||
# M3 JSON-parity gate. Pure C harness (NOT elb/elc): links the real el_runtime.c
|
||||
# native engram builtins + engram_store.c and drives ENGRAM_STORE on vs off.
|
||||
# Writes ONLY under a throwaway /tmp dir with a throwaway HOME + ENGRAM_DATA_DIR.
|
||||
set -u
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
RT="$HERE/../../lang/runtime/el_runtime.c"
|
||||
ST="$HERE/../../lang/runtime/engram_store.c"
|
||||
INC="$HERE/../../lang/runtime"
|
||||
WORK="$(mktemp -d /tmp/engram-m3-XXXXXX)"
|
||||
DATA="$WORK/data"; mkdir -p "$DATA"
|
||||
BIN="$WORK/m3"
|
||||
export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron
|
||||
export ENGRAM_DATA_DIR="$DATA"
|
||||
export ENGRAM_WAL_SYNC=always
|
||||
unset ENGRAM_STORE
|
||||
fail=0
|
||||
|
||||
echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m3_parity.c) =="
|
||||
gcc -O1 -std=c11 -I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log"
|
||||
if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi
|
||||
grep -i warning "$WORK/cc.log" | grep -iE 'engram_store|eg_store|eg_load|scan_nodes|scan_edges' && echo "(warnings in M3 code above)" || true
|
||||
|
||||
echo
|
||||
echo "== 0) default-OFF: flag unset leaves the store untouched =="
|
||||
( unset ENGRAM_STORE; "$BIN" offcheck "$DATA" )
|
||||
[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; }
|
||||
[ -e "$DATA/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \
|
||||
|| echo " ok: no neuron.egm created with flag OFF"
|
||||
|
||||
echo
|
||||
echo "== 1) seed (ENGRAM_STORE unset): build graph, save snapshot.json, activate =="
|
||||
( unset ENGRAM_STORE; "$BIN" seed "$DATA" ) || { echo "FAIL: seed"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 2) on (ENGRAM_STORE=1): import snapshot.json ONCE -> neuron.egm, resident-load, activate =="
|
||||
ENGRAM_STORE=1 "$BIN" on "$DATA" || { echo "FAIL: on"; fail=1; }
|
||||
[ -e "$DATA/neuron.egm" ] && echo " ok: neuron.egm created by import" || { echo "FAIL: neuron.egm missing"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 3) reboot (ENGRAM_STORE=1, snapshot.json DELETED): must load from neuron.egm, never JSON =="
|
||||
rm -f "$DATA/snapshot.json"
|
||||
ENGRAM_STORE=1 "$BIN" reboot "$DATA" || { echo "FAIL: reboot"; fail=1; }
|
||||
|
||||
echo
|
||||
echo "== 4) parity comparison (modulo ordering) =="
|
||||
python3 - "$DATA" <<'PY'
|
||||
import json, sys, os
|
||||
d = sys.argv[1]
|
||||
def load(name):
|
||||
with open(os.path.join(d, name)) as f: return json.load(f)
|
||||
def norm_graph(g):
|
||||
nodes = sorted(g.get("nodes", []), key=lambda n: n.get("id",""))
|
||||
edges = sorted(g.get("edges", []), key=lambda e: e.get("id",""))
|
||||
layers= sorted(g.get("layers", []), key=lambda l: l.get("layer_id",0))
|
||||
return {"nodes":nodes, "edges":edges, "layers":layers}
|
||||
def act_ids(a):
|
||||
# list of (node id, promoted); robust set + ordered list
|
||||
seq = [(e.get("node",{}).get("id",""), int(e.get("promoted",0))) for e in a]
|
||||
return seq
|
||||
|
||||
rc = 0
|
||||
snap = norm_graph(load("snapshot.json") if os.path.exists(os.path.join(d,"snapshot.json")) else load("off_graph.json"))
|
||||
off = norm_graph(load("off_graph.json"))
|
||||
on = norm_graph(load("on_graph.json"))
|
||||
rebt = norm_graph(load("reboot_graph.json"))
|
||||
|
||||
def cmp(label, a, b):
|
||||
global rc
|
||||
if a == b:
|
||||
print(f" PASS: {label} (nodes={len(a['nodes'])} edges={len(a['edges'])} layers={len(a['layers'])})")
|
||||
else:
|
||||
rc = 1
|
||||
print(f" FAIL: {label}")
|
||||
for k in ("nodes","edges","layers"):
|
||||
if a[k] != b[k]:
|
||||
print(f" {k}: {len(a[k])} vs {len(b[k])}")
|
||||
for x,y in zip(a[k], b[k]):
|
||||
if x != y:
|
||||
print(f" first diff:\n A={json.dumps(x)[:300]}\n B={json.dumps(y)[:300]}")
|
||||
break
|
||||
|
||||
cmp("graph: ENGRAM_STORE=1 (export) == ENGRAM_STORE=0 (JSON path)", on, off)
|
||||
cmp("round-trip: snapshot.json seed == store export (on_graph)", on, off) # off_graph==snapshot save
|
||||
cmp("reboot from neuron.egm (no JSON) == on-path store", rebt, on)
|
||||
|
||||
offa = act_ids(load("off_act.json"))
|
||||
ona = act_ids(load("on_act.json"))
|
||||
if set(offa) == set(ona):
|
||||
print(f" PASS: activation result set identical (off={len(offa)} on={len(ona)} entries)")
|
||||
if offa == ona:
|
||||
print(" (and identical ordering/promotion sequence)")
|
||||
else:
|
||||
print(" (same set; ordering differs only where scores tie — reporting honestly)")
|
||||
else:
|
||||
rc = 1
|
||||
print(" FAIL: activation result set differs")
|
||||
print(f" off-only: {set(offa)-set(ona)}")
|
||||
print(f" on-only: {set(ona)-set(offa)}")
|
||||
|
||||
sys.exit(rc)
|
||||
PY
|
||||
[ $? -ne 0 ] && fail=1
|
||||
|
||||
echo
|
||||
echo "== 5) ASan+UBSan build, exercise M3 scan/boot/hooks (leaks off — harness intentionally leaks el_strdup) =="
|
||||
SANBIN="$WORK/m3.san"
|
||||
gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \
|
||||
-I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log"
|
||||
if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else
|
||||
export ASAN_OPTIONS=detect_leaks=0
|
||||
DATA2="$WORK/data2"; mkdir -p "$DATA2"
|
||||
( unset ENGRAM_STORE; "$SANBIN" seed "$DATA2" ) >/dev/null 2>"$WORK/san_run.log" && \
|
||||
ENGRAM_STORE=1 "$SANBIN" on "$DATA2" >/dev/null 2>>"$WORK/san_run.log" && \
|
||||
{ rm -f "$DATA2/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" reboot "$DATA2" >/dev/null 2>>"$WORK/san_run.log"; }
|
||||
if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then
|
||||
echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1
|
||||
else
|
||||
echo " ok: ASan+UBSan clean across seed/on/reboot (scan, boot, resident-load, mutation hooks)"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ "$fail" -eq 0 ]; then echo "================ M3 PARITY GATE: PASS ================"; else echo "================ M3 PARITY GATE: FAIL ================"; fi
|
||||
rm -rf "$WORK"
|
||||
exit $fail
|
||||
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env bash
|
||||
# M1 paged-store gate. Pure C (NOT elb/elc). Writes only under /tmp.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
SRC="$HERE/../../lang/runtime/engram_store.c"
|
||||
BIN="/tmp/test_store.$$"
|
||||
echo "compiling: gcc test_store.c engram_store.c"
|
||||
gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_store.c" "$SRC" -o "$BIN"
|
||||
"$BIN"
|
||||
rc=$?
|
||||
rm -f "$BIN"
|
||||
rm -rf /tmp/engram-store-test-*
|
||||
exit $rc
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env bash
|
||||
# M2 WAL + checkpoint + recovery gate. Pure C (NOT elb/elc). Writes only under /tmp.
|
||||
# Recovery tests use ENGRAM_WAL_SYNC=always so every WAL record is durable at crash.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
SRC="$HERE/../../lang/runtime/engram_store.c"
|
||||
BIN="/tmp/test_wal_store.$$"
|
||||
echo "compiling: gcc test_wal_store.c engram_store.c"
|
||||
gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_wal_store.c" "$SRC" -o "$BIN"
|
||||
ENGRAM_WAL_SYNC=always "$BIN"
|
||||
rc=$?
|
||||
rm -f "$BIN"
|
||||
rm -rf /tmp/engram-wal-test-*
|
||||
exit $rc
|
||||
Executable
+16
@@ -0,0 +1,16 @@
|
||||
#!/usr/bin/env bash
|
||||
# WAL unit + integration + crash-fuzz gate. Throwaway HOME/dirs only.
|
||||
set -e
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
REL="$HERE/../../lang/runtime"
|
||||
cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" \
|
||||
"$HERE/test_wal.c" -lcurl -lpthread -o /tmp/test_wal
|
||||
HOME=/tmp/engram-throwaway-home /tmp/test_wal
|
||||
# Fail-loud data-dir check (must exit 1 with a FATAL line):
|
||||
cat > /tmp/test_failloud.c <<'C'
|
||||
#include "el_runtime.c"
|
||||
int main(void){ unsetenv("ENGRAM_DATA_DIR"); unsetenv("HOME");
|
||||
engram_resolve_data_dir(); printf("REACHED\n"); return 0; }
|
||||
C
|
||||
cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" /tmp/test_failloud.c -lcurl -lpthread -o /tmp/test_failloud
|
||||
if env -u HOME -u ENGRAM_DATA_DIR /tmp/test_failloud; then echo "FAIL: should have exited"; exit 1; else echo "[PASS] fail-loud exit on unresolvable HOME"; fi
|
||||
@@ -0,0 +1,130 @@
|
||||
/* test_m35_hebb_persist.c — M3.5 PRE-FLIP GATE.
|
||||
*
|
||||
* Proves that in-place field mutations made during spreading activation — edge
|
||||
* `hebb` (+ last_fired), node `activation_count`, node working-memory weight —
|
||||
* PERSIST to the paged store and survive a restart from neuron.egm with
|
||||
* snapshot.json deleted. This is the "hebb-survives-restart" fix that gates the
|
||||
* live cutover.
|
||||
*
|
||||
* Same style as test_m3_parity.c: a REAL el-level harness linking the actual
|
||||
* el_runtime.c native engram builtins + engram_store.c, driving engram_node_full
|
||||
* / engram_connect / engram_activate_json / engram_save / engram_store_boot /
|
||||
* engram_store_checkpoint / engram_store_close directly from C. No EL interpreter.
|
||||
*
|
||||
* Modes (argv[1]), data dir (argv[2]):
|
||||
* pos_seed — ENGRAM_STORE=1: fresh store, seed a graph tuned so activation
|
||||
* co-activates a connected pair (edge hebb 0 -> ETA) and reinforces
|
||||
* nodes (activation_count 0 -> >=1, WM weight -> >0). Export the
|
||||
* post-activation resident graph to pre_reboot.json, then CHECKPOINT
|
||||
* (the M3.5 field-persist), then close.
|
||||
* pos_reboot— ENGRAM_STORE=1, snapshot.json deleted by runner: boot from
|
||||
* neuron.egm (WAL replay), export reboot.json, close. The values in
|
||||
* reboot.json are what actually survived the round-trip.
|
||||
* neg_seed — identical to pos_seed but WITHOUT the checkpoint field-persist
|
||||
* (negative control): activation mutations never reach the store.
|
||||
* neg_reboot— boot from neuron.egm, export neg_reboot.json, close.
|
||||
* offcheck — ENGRAM_STORE unset: seed+activate+checkpoint must NOT touch the
|
||||
* store (no neuron.egm, checkpoint returns 0).
|
||||
*
|
||||
* The pass/fail assertions live in run_m35_hebb_persist.sh (python over the JSON
|
||||
* exports): reboot.json must carry the learned hebb / activation_count and the
|
||||
* JSON-identical halved WM weight; neg_reboot.json must have LOST them.
|
||||
*/
|
||||
#include "el_runtime.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
extern int engram_store_enabled(void);
|
||||
extern el_val_t engram_store_boot(el_val_t data_dir);
|
||||
extern el_val_t engram_store_checkpoint(void);
|
||||
extern el_val_t engram_store_close(void);
|
||||
|
||||
static el_val_t S(const char* s){ return EL_STR(s); }
|
||||
static el_val_t F(double d){ return el_from_float(d); }
|
||||
|
||||
/* Two nodes with DISTINCT content (so the redundancy-suppression pass cannot
|
||||
* dedup one of them away) that both match the query strongly, wired by one
|
||||
* "associate" edge. A handful of weakly-related distractors make it a real
|
||||
* graph. On activation both A and B promote to working memory and co-activate,
|
||||
* so their edge's hebb rises from 0 to ENGRAM_HEBB_ETA. */
|
||||
static void build_seed(void){
|
||||
el_val_t a = engram_node_full(S("hebbian potentiation strengthens co-active memory links"),
|
||||
S("Concept"), S("hebb-a"), F(0.9), F(0.85), F(1.0), S("Semantic"),
|
||||
S("hebbian,memory,activation"));
|
||||
el_val_t b = engram_node_full(S("co-active memory links accrue hebbian associative weight"),
|
||||
S("Concept"), S("hebb-b"), F(0.9), F(0.85), F(1.0), S("Semantic"),
|
||||
S("hebbian,memory,weight"));
|
||||
el_val_t c = engram_node_full(S("unrelated culinary recipe for sourdough bread"),
|
||||
S("Fact"), S("distractor-1"), F(0.4), F(0.4), F(1.0), S("Semantic"),
|
||||
S("food"));
|
||||
el_val_t d = engram_node_full(S("the weather forecast predicts rain tomorrow afternoon"),
|
||||
S("Fact"), S("distractor-2"), F(0.4), F(0.4), F(1.0), S("Semantic"),
|
||||
S("weather"));
|
||||
engram_connect(a, b, F(0.8), S("associate")); /* the edge under test */
|
||||
engram_connect(a, c, F(0.3), S("associate"));
|
||||
engram_connect(b, d, F(0.3), S("associate"));
|
||||
}
|
||||
|
||||
static const char* QUERY =
|
||||
"hebbian potentiation co-active memory links associative weight";
|
||||
|
||||
static void export_graph(const char* dir, const char* name){
|
||||
char p[1024];
|
||||
snprintf(p, sizeof p, "%s/%s", dir, name);
|
||||
if (!engram_save(S(p))){ fprintf(stderr, "save %s failed\n", name); exit(2); }
|
||||
}
|
||||
|
||||
int main(int argc, char** argv){
|
||||
if (argc < 3){
|
||||
fprintf(stderr, "usage: %s <pos_seed|pos_reboot|neg_seed|neg_reboot|offcheck> <dir>\n", argv[0]);
|
||||
return 2;
|
||||
}
|
||||
const char* mode = argv[1];
|
||||
const char* dir = argv[2];
|
||||
|
||||
if (!strcmp(mode, "pos_seed") || !strcmp(mode, "neg_seed")){
|
||||
int persist = !strcmp(mode, "pos_seed");
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; }
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; }
|
||||
build_seed();
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
(void)act;
|
||||
/* Capture the post-activation resident state BEFORE persisting/closing. */
|
||||
export_graph(dir, persist ? "pre_reboot.json" : "neg_pre.json");
|
||||
printf("[%s] nodes=%lld edges=%lld\n", mode,
|
||||
(long long)(int64_t)engram_node_count(),
|
||||
(long long)(int64_t)engram_edge_count());
|
||||
if (persist){
|
||||
if (!engram_store_checkpoint()){ fprintf(stderr, "checkpoint failed\n"); return 2; }
|
||||
}
|
||||
/* neg mode: NO field-persist checkpoint. engram_store_close still flushes
|
||||
* pages, but no store_put_* ran post-creation, so the store keeps the
|
||||
* pristine creation-time field values (hebb=0, activation_count=0). */
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "pos_reboot") || !strcmp(mode, "neg_reboot")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; }
|
||||
/* snapshot.json deleted by the runner — boot MUST come from neuron.egm. */
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; }
|
||||
export_graph(dir, !strcmp(mode, "pos_reboot") ? "reboot.json" : "neg_reboot.json");
|
||||
printf("[%s] nodes=%lld edges=%lld\n", mode,
|
||||
(long long)(int64_t)engram_node_count(),
|
||||
(long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "offcheck")){
|
||||
int en = engram_store_enabled();
|
||||
el_val_t boot = engram_store_boot(S(dir)); /* no-op with flag off */
|
||||
build_seed();
|
||||
engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
el_val_t ck = engram_store_checkpoint(); /* must be a no-op */
|
||||
printf("[offcheck] enabled=%d boot=%lld checkpoint=%lld\n",
|
||||
en, (long long)(int64_t)boot, (long long)(int64_t)ck);
|
||||
return (en == 0 && (int64_t)boot == 0 && (int64_t)ck == 0) ? 0 : 1;
|
||||
}
|
||||
fprintf(stderr, "unknown mode %s\n", mode);
|
||||
return 2;
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
/* test_m3_parity.c — M3 JSON-parity gate for the ENGRAM_STORE wiring.
|
||||
*
|
||||
* This is a REAL el-level harness: it links the actual el_runtime.o (the soul's
|
||||
* native engram builtins) + engram_store.o and calls the engram_node family plus
|
||||
* engram_connect, engram_activate_json, engram_save, engram_store_boot directly. No EL interpreter
|
||||
* and no full soul build are needed — el_runtime.c compiles to a standalone .o
|
||||
* whose engram builtins operate on the process-global engram store, and the
|
||||
* string arena is inert unless el_request_start() is called, so the builtins are
|
||||
* callable straight from C (el_val_t is int64_t; EL_STR/EL_CSTR are pointer casts).
|
||||
*
|
||||
* Modes (argv[1]), data dir (argv[2]):
|
||||
* seed — ENGRAM_STORE unset: build a fixed seed graph, write snapshot.json +
|
||||
* off_graph.json (pristine, pre-activation), then activate → off_act.json.
|
||||
* on — ENGRAM_STORE=1: engram_store_boot(dir) imports snapshot.json ONCE into
|
||||
* neuron.egm and loads it resident; write on_graph.json, then activate →
|
||||
* on_act.json; checkpoint + close.
|
||||
* reboot — ENGRAM_STORE=1 with snapshot.json DELETED: boot must reload from
|
||||
* neuron.egm (WAL replay), never re-reading JSON; write reboot_graph.json.
|
||||
* offcheck — assert flag-off leaves the store untouched.
|
||||
*
|
||||
* The graph comparison (done by run_m3_parity.sh via python, modulo ordering) is
|
||||
* the deterministic gate; activation ids/promoted are compared as a robust set.
|
||||
*/
|
||||
#include "el_runtime.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* Builtins the header declares are pulled in via el_runtime.h. The M3 additions
|
||||
* are not in the header yet, so declare them here. */
|
||||
extern int engram_store_enabled(void);
|
||||
extern el_val_t engram_store_boot(el_val_t data_dir);
|
||||
extern el_val_t engram_store_checkpoint(void);
|
||||
extern el_val_t engram_store_close(void);
|
||||
extern el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
|
||||
static el_val_t S(const char* s){ return EL_STR(s); }
|
||||
static el_val_t F(double d){ return el_from_float(d); }
|
||||
|
||||
/* Build a fixed, deterministic seed graph: 12 nodes across two layers + 9 edges.
|
||||
* Content is chosen so an activation query has real matches to rank. */
|
||||
static void build_seed(void){
|
||||
/* core-identity layer (1) via engram_node_full */
|
||||
el_val_t n0 = engram_node_full(S("tiered storage engine design"), S("Concept"),
|
||||
S("storage-engine"), F(0.9), F(0.8), F(1.0), S("Semantic"), S("design,storage"));
|
||||
el_val_t n1 = engram_node_full(S("write-ahead log durability"), S("Concept"),
|
||||
S("wal"), F(0.85), F(0.75), F(1.0), S("Semantic"), S("wal,durability"));
|
||||
el_val_t n2 = engram_node_full(S("paged buffer pool with checkpointing"), S("Concept"),
|
||||
S("buffer-pool"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("paging"));
|
||||
el_val_t n3 = engram_node_full(S("spreading activation over the graph"), S("Concept"),
|
||||
S("activation"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("activation,graph"));
|
||||
el_val_t n4 = engram_node_full(S("hebbian co-activation potentiation"), S("Concept"),
|
||||
S("hebbian"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("hebb"));
|
||||
el_val_t n5 = engram_node_full(S("crash recovery replays the log"), S("Concept"),
|
||||
S("recovery"), F(0.75), F(0.65), F(1.0), S("Semantic"), S("recovery,wal"));
|
||||
/* domain-knowledge layer (2) via engram_node_layered */
|
||||
el_val_t n6 = engram_node_layered(S("b-tree primary index id to location"), S("Fact"),
|
||||
S("btree"), F(0.7), F(0.6), F(1.0), S(""), S("index"), (el_val_t)2);
|
||||
el_val_t n7 = engram_node_layered(S("adjacency index for edge lookup"), S("Fact"),
|
||||
S("adjacency"), F(0.7), F(0.6), F(1.0), S(""), S("index,graph"), (el_val_t)2);
|
||||
el_val_t n8 = engram_node_layered(S("slotted pages hold tlv records"), S("Fact"),
|
||||
S("slotted-page"), F(0.65), F(0.55), F(1.0), S(""), S("format"), (el_val_t)2);
|
||||
el_val_t n9 = engram_node_full(S("memory tiers working semantic episodic"), S("Concept"),
|
||||
S("tiers"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("tiers,memory"));
|
||||
el_val_t n10 = engram_node_full(S("embeddings enable nearest neighbour search"), S("Concept"),
|
||||
S("embeddings"), F(0.65), F(0.55), F(1.0), S("Semantic"), S("embeddings"));
|
||||
el_val_t n11 = engram_node_full(S("the durable engram is the mind's memory"), S("Belief"),
|
||||
S("engram"), F(0.95), F(0.9), F(1.0), S("Semantic"), S("engram,memory"));
|
||||
|
||||
engram_connect(n0, n1, F(0.8), S("depends-on"));
|
||||
engram_connect(n0, n2, F(0.8), S("depends-on"));
|
||||
engram_connect(n0, n3, F(0.7), S("enables"));
|
||||
engram_connect(n1, n5, F(0.9), S("enables"));
|
||||
engram_connect(n3, n4, F(0.6), S("triggers"));
|
||||
engram_connect(n2, n6, F(0.7), S("uses"));
|
||||
engram_connect(n3, n7, F(0.7), S("uses"));
|
||||
engram_connect(n0, n8, F(0.6), S("uses"));
|
||||
engram_connect(n11, n9, F(0.8), S("about"));
|
||||
engram_connect(n11, n10, F(0.5), S("about"));
|
||||
}
|
||||
|
||||
static void write_file(const char* path, const char* content){
|
||||
FILE* f = fopen(path, "wb");
|
||||
if (!f){ fprintf(stderr, "cannot open %s\n", path); exit(2); }
|
||||
if (content) fwrite(content, 1, strlen(content), f);
|
||||
fclose(f);
|
||||
}
|
||||
|
||||
static const char* QUERY = "storage engine activation and the durable log";
|
||||
|
||||
int main(int argc, char** argv){
|
||||
if (argc < 3){ fprintf(stderr, "usage: %s <seed|on|reboot|offcheck> <dir>\n", argv[0]); return 2; }
|
||||
const char* mode = argv[1];
|
||||
const char* dir = argv[2];
|
||||
char p[1024];
|
||||
|
||||
if (!strcmp(mode, "seed")){
|
||||
if (engram_store_enabled()){ fprintf(stderr, "seed mode requires ENGRAM_STORE unset\n"); return 2; }
|
||||
build_seed();
|
||||
snprintf(p, sizeof p, "%s/snapshot.json", dir);
|
||||
if (!engram_save(S(p))){ fprintf(stderr, "seed save failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/off_graph.json", dir);
|
||||
engram_save(S(p)); /* pristine off-path graph */
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
snprintf(p, sizeof p, "%s/off_act.json", dir);
|
||||
write_file(p, EL_CSTR(act));
|
||||
printf("[seed] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "on")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "on mode requires ENGRAM_STORE=1\n"); return 2; }
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/on_graph.json", dir);
|
||||
engram_save(S(p)); /* export resident (== store) */
|
||||
/* Checkpoint the freshly-imported (pristine) graph — this is the state
|
||||
* the reboot comparison expects to round-trip. Under M3.5 a checkpoint
|
||||
* persists the resident graph's CURRENT field state, so it must run
|
||||
* BEFORE activation mutates fields in place; activation itself is
|
||||
* exercised below only for the activation-result-set parity check. The
|
||||
* M3.5 gate (test_m35_hebb_persist) separately proves that a checkpoint
|
||||
* taken AFTER activation durably carries the learned hebb/WM state. */
|
||||
engram_store_checkpoint();
|
||||
el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3);
|
||||
snprintf(p, sizeof p, "%s/on_act.json", dir);
|
||||
write_file(p, EL_CSTR(act));
|
||||
printf("[on] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "reboot")){
|
||||
if (!engram_store_enabled()){ fprintf(stderr, "reboot mode requires ENGRAM_STORE=1\n"); return 2; }
|
||||
/* snapshot.json has been deleted by the runner — boot MUST come from
|
||||
* neuron.egm (+ WAL replay), never re-reading JSON. */
|
||||
if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; }
|
||||
snprintf(p, sizeof p, "%s/reboot_graph.json", dir);
|
||||
engram_save(S(p));
|
||||
printf("[reboot] nodes=%lld edges=%lld\n",
|
||||
(long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count());
|
||||
engram_store_close();
|
||||
return 0;
|
||||
}
|
||||
if (!strcmp(mode, "offcheck")){
|
||||
/* ENGRAM_STORE unset: enabled()==0 and boot is a no-op returning 0. */
|
||||
int en = engram_store_enabled();
|
||||
el_val_t b = engram_store_boot(S(dir));
|
||||
printf("[offcheck] enabled=%d boot_ret=%lld\n", en, (long long)(int64_t)b);
|
||||
return (en == 0 && (int64_t)b == 0) ? 0 : 1;
|
||||
}
|
||||
fprintf(stderr, "unknown mode %s\n", mode);
|
||||
return 2;
|
||||
}
|
||||
@@ -0,0 +1,439 @@
|
||||
/* test_store.c — M1 gate for the engram paged store (engram_store.{c,h}).
|
||||
*
|
||||
* Pure C. Build: gcc -O2 test_store.c ../../lang/runtime/engram_store.c -o test_store
|
||||
* Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports.
|
||||
*
|
||||
* Covers §7 M1 gates: round-trip (5k nodes / 20k edges, all fields, emb bit-exact,
|
||||
* hebb, >page content), TLV forward-compat, overflow chains, B+-tree indexes
|
||||
* across splits, free-list reuse, and corruption/superblock recovery.
|
||||
*/
|
||||
#include "../../lang/runtime/engram_store.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond){
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_dir[512];
|
||||
static void mk_dir(void){
|
||||
snprintf(g_dir, sizeof g_dir, "/tmp/engram-store-test-%d", (int)getpid());
|
||||
mkdir(g_dir, 0700);
|
||||
}
|
||||
static void path_in(char* out, size_t cap, const char* name){
|
||||
snprintf(out, cap, "%s/%s", g_dir, name);
|
||||
}
|
||||
static long file_size(const char* p){ struct stat st; return stat(p,&st)==0 ? (long)st.st_size : -1; }
|
||||
|
||||
/* ── deterministic RNG so oracle nodes/edges regenerate bit-exact ─────────── */
|
||||
static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; }
|
||||
static uint64_t node_seed(int i){ return 0x9E3779B97F4A7C15ULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL); }
|
||||
static uint64_t edge_seed(int i){ return 0xC2B2AE3D27D4EB4FULL ^ ((uint64_t)(i+1)*0x165667B19E3779F9ULL); }
|
||||
|
||||
static char* rnd_str(uint64_t* st, size_t len){
|
||||
char* s = (char*)malloc(len + 1);
|
||||
for (size_t i=0;i<len;i++) s[i] = (char)(33 + (xs(st) % 94)); /* printable, no NUL */
|
||||
s[len] = 0; return s;
|
||||
}
|
||||
|
||||
/* NODE_COUNT nodes; a slice have >page content to force overflow chains. */
|
||||
#define NODE_COUNT 5000
|
||||
#define EDGE_COUNT 20000
|
||||
#define EMB_DIM 768
|
||||
|
||||
static void gen_node(int i, StoreNode* n){
|
||||
memset(n, 0, sizeof *n);
|
||||
uint64_t st = node_seed(i);
|
||||
char id[32]; snprintf(id, sizeof id, "node-%d", i);
|
||||
n->id = strdup(id);
|
||||
size_t clen = (i % 500 == 0) ? (size_t)(17000 + (xs(&st) % 6000)) : (size_t)(xs(&st) % 300);
|
||||
n->content = rnd_str(&st, clen);
|
||||
n->node_type = rnd_str(&st, 4 + (xs(&st) % 8));
|
||||
n->label = (i % 2) ? rnd_str(&st, 3 + (xs(&st) % 10)) : NULL;
|
||||
n->tier = rnd_str(&st, 4 + (xs(&st) % 6));
|
||||
n->tags = rnd_str(&st, xs(&st) % 40);
|
||||
n->metadata = (i % 3) ? rnd_str(&st, xs(&st) % 60) : NULL;
|
||||
n->salience = (double)(xs(&st) % 1000000) / 997.0;
|
||||
n->importance = (double)(xs(&st) % 1000000) / 131.0;
|
||||
n->confidence = (double)(xs(&st) % 1000000) / 733.0;
|
||||
n->temporal_decay_rate = (double)(xs(&st) % 1000000) / 101.0;
|
||||
n->activation_count = (int64_t)(xs(&st) % 100000);
|
||||
n->last_activated = (int64_t)xs(&st);
|
||||
n->created_at = (int64_t)(1600000000000LL + i);
|
||||
n->updated_at = (int64_t)xs(&st);
|
||||
n->background_activation = (double)(xs(&st) % 1000000) / 17.0;
|
||||
n->working_memory_weight = (double)(xs(&st) % 1000000) / 29.0;
|
||||
n->suppression_count = (int32_t)(xs(&st) % 50);
|
||||
n->layer_id = (uint32_t)(xs(&st) % 5);
|
||||
for (int k=0;k<STORE_BLL_K;k++) n->access_ts[k] = (int64_t)xs(&st);
|
||||
n->access_head = (int32_t)(xs(&st) % STORE_BLL_K);
|
||||
n->access_filled = (int32_t)(xs(&st) % (STORE_BLL_K + 1));
|
||||
n->wm_anchor = (double)(xs(&st) % 1000000) / 3.0;
|
||||
n->emb = (float*)malloc(EMB_DIM * sizeof(float));
|
||||
for (int k=0;k<EMB_DIM;k++){ uint32_t u=(uint32_t)xs(&st); memcpy(&n->emb[k], &u, 4); }
|
||||
n->emb_dim = EMB_DIM;
|
||||
}
|
||||
|
||||
static void gen_edge(int i, StoreEdge* e){
|
||||
memset(e, 0, sizeof *e);
|
||||
uint64_t st = edge_seed(i);
|
||||
char id[32], from[32], to[32];
|
||||
snprintf(id, sizeof id, "edge-%d", i);
|
||||
snprintf(from, sizeof from, "node-%d", (int)(xs(&st) % NODE_COUNT));
|
||||
snprintf(to, sizeof to, "node-%d", (int)(xs(&st) % NODE_COUNT));
|
||||
e->id = strdup(id); e->from_id = strdup(from); e->to_id = strdup(to);
|
||||
e->relation = rnd_str(&st, 3 + (xs(&st) % 12));
|
||||
e->metadata = (i % 4) ? rnd_str(&st, xs(&st) % 40) : NULL;
|
||||
e->weight = (double)(xs(&st) % 1000000) / 111.0;
|
||||
e->hebb = (double)(xs(&st) % 1000000) / 1000000.0; /* the learned field */
|
||||
e->confidence = (double)(xs(&st) % 1000000) / 777.0;
|
||||
e->created_at = (int64_t)(1600000000000LL + i);
|
||||
e->updated_at = (int64_t)xs(&st);
|
||||
e->last_fired = (int64_t)xs(&st);
|
||||
e->inhibitory = (int32_t)(xs(&st) % 2);
|
||||
e->layer_id = (uint32_t)(xs(&st) % 5);
|
||||
}
|
||||
|
||||
static int streq(const char* a, const char* b){
|
||||
if (!a && !b) return 1;
|
||||
if (!a || !b) return 0;
|
||||
return strcmp(a,b)==0;
|
||||
}
|
||||
static int cmp_node(const StoreNode* a, const StoreNode* b){
|
||||
if (!streq(a->id,b->id) || !streq(a->content,b->content) ||
|
||||
!streq(a->node_type,b->node_type) || !streq(a->label,b->label) ||
|
||||
!streq(a->tier,b->tier) || !streq(a->tags,b->tags) ||
|
||||
!streq(a->metadata,b->metadata)) return 0;
|
||||
if (a->salience!=b->salience || a->importance!=b->importance ||
|
||||
a->confidence!=b->confidence || a->temporal_decay_rate!=b->temporal_decay_rate ||
|
||||
a->activation_count!=b->activation_count || a->last_activated!=b->last_activated ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->background_activation!=b->background_activation ||
|
||||
a->working_memory_weight!=b->working_memory_weight ||
|
||||
a->suppression_count!=b->suppression_count || a->layer_id!=b->layer_id ||
|
||||
a->access_head!=b->access_head || a->access_filled!=b->access_filled ||
|
||||
a->wm_anchor!=b->wm_anchor || a->emb_dim!=b->emb_dim) return 0;
|
||||
for (int k=0;k<STORE_BLL_K;k++) if (a->access_ts[k]!=b->access_ts[k]) return 0;
|
||||
if ((a->emb==NULL) != (b->emb==NULL)) return 0;
|
||||
if (a->emb && memcmp(a->emb, b->emb, (size_t)a->emb_dim*4)!=0) return 0;
|
||||
return 1;
|
||||
}
|
||||
static int cmp_edge(const StoreEdge* a, const StoreEdge* b){
|
||||
if (!streq(a->id,b->id) || !streq(a->from_id,b->from_id) || !streq(a->to_id,b->to_id) ||
|
||||
!streq(a->relation,b->relation) || !streq(a->metadata,b->metadata)) return 0;
|
||||
if (a->weight!=b->weight || a->hebb!=b->hebb || a->confidence!=b->confidence ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->last_fired!=b->last_fired || a->inhibitory!=b->inhibitory ||
|
||||
a->layer_id!=b->layer_id) return 0;
|
||||
return 1;
|
||||
}
|
||||
static void free_node_fields(StoreNode* n){
|
||||
free(n->id); free(n->content); free(n->node_type); free(n->label);
|
||||
free(n->tier); free(n->tags); free(n->metadata); free(n->emb); free(n->unknown);
|
||||
}
|
||||
static void free_edge_fields(StoreEdge* e){
|
||||
free(e->id); free(e->from_id); free(e->to_id); free(e->relation); free(e->metadata); free(e->unknown);
|
||||
}
|
||||
|
||||
/* Flip one byte in the store file at (page*PAGE_SIZE + off). */
|
||||
static void flip_byte(const char* path, uint64_t page, size_t off){
|
||||
int fd = open(path, O_RDWR);
|
||||
uint8_t b; off_t at = (off_t)page*STORE_PAGE_SIZE + off;
|
||||
pread(fd, &b, 1, at); b ^= 0xFF; pwrite(fd, &b, 1, at); close(fd);
|
||||
}
|
||||
|
||||
/* ════════════════════════════════════════════════════════════════════════ */
|
||||
|
||||
static void test_roundtrip(void){
|
||||
printf("\n== round-trip: %d nodes + %d edges, all fields, emb bit-exact ==\n", NODE_COUNT, EDGE_COUNT);
|
||||
char path[600]; path_in(path, sizeof path, "roundtrip.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
ok("store_create", s != NULL);
|
||||
if (!s) return;
|
||||
|
||||
for (int i=0;i<NODE_COUNT;i++){ StoreNode n; gen_node(i,&n);
|
||||
if (store_put_node(s,&n)!=0){ ok("put_node", 0); free_node_fields(&n); store_close(s); return; }
|
||||
free_node_fields(&n); }
|
||||
for (int i=0;i<EDGE_COUNT;i++){ StoreEdge e; gen_edge(i,&e);
|
||||
if (store_put_edge(s,&e)!=0){ ok("put_edge", 0); free_edge_fields(&e); store_close(s); return; }
|
||||
free_edge_fields(&e); }
|
||||
ok("wrote all nodes+edges", 1);
|
||||
store_close(s);
|
||||
|
||||
long sz = file_size(path);
|
||||
printf(" store file size: %ld bytes (%.2f MB) for %d nodes / %d edges\n",
|
||||
sz, sz/1048576.0, NODE_COUNT, EDGE_COUNT);
|
||||
|
||||
s = store_open(path);
|
||||
ok("store_open (reopen)", s != NULL);
|
||||
if (!s) return;
|
||||
|
||||
int nbad = 0;
|
||||
for (int i=0;i<NODE_COUNT;i++){
|
||||
StoreNode want; gen_node(i,&want);
|
||||
StoreNode got; int r = store_get_node(s, want.id, &got);
|
||||
if (r!=1 || !cmp_node(&want,&got) || got.unknown_len!=0) nbad++;
|
||||
if (r==1) store_node_free(&got);
|
||||
free_node_fields(&want);
|
||||
}
|
||||
ok("all 5000 nodes read back bit-exact (incl emb, all fields)", nbad==0);
|
||||
if (nbad) printf(" %d node mismatches\n", nbad);
|
||||
|
||||
int ebad = 0;
|
||||
for (int i=0;i<EDGE_COUNT;i++){
|
||||
StoreEdge want; gen_edge(i,&want);
|
||||
StoreEdge* got; size_t gn;
|
||||
int found = 0;
|
||||
if (store_get_edges_from(s, want.from_id, &got, &gn)==0){
|
||||
for (size_t j=0;j<gn;j++) if (streq(got[j].id, want.id)){ if (cmp_edge(&want,&got[j])) found=1; break; }
|
||||
store_edges_free(got, gn);
|
||||
}
|
||||
if (!found) ebad++;
|
||||
free_edge_fields(&want);
|
||||
}
|
||||
ok("all 20000 edges read back via adjacency, all fields incl hebb", ebad==0);
|
||||
if (ebad) printf(" %d edge mismatches\n", ebad);
|
||||
|
||||
ok("store_check crc clean after round-trip", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_forward_compat(void){
|
||||
printf("\n== TLV forward-compat: omit field defaults; unknown tag preserved ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "fwd.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
/* Writer OMITS several fields (metadata, label, emb) → reader must default. */
|
||||
StoreNode a; memset(&a,0,sizeof a);
|
||||
a.id = strdup("omit-1"); a.content = strdup("has content"); a.tier = strdup("core");
|
||||
a.salience = 0.5; /* metadata/label NULL, emb NULL */
|
||||
store_put_node(s, &a); free(a.id); free(a.content); free(a.tier);
|
||||
|
||||
StoreNode g; int r = store_get_node(s, "omit-1", &g);
|
||||
ok("omitted string fields default to NULL", r==1 && g.metadata==NULL && g.label==NULL);
|
||||
ok("omitted emb defaults to NULL / emb_dim 0", r==1 && g.emb==NULL && g.emb_dim==0);
|
||||
ok("present fields intact", r==1 && streq(g.content,"has content") && g.salience==0.5);
|
||||
if (r==1) store_node_free(&g);
|
||||
|
||||
/* Writer includes an UNKNOWN tag (simulating a newer writer / field the
|
||||
* reader does not model) via the `unknown` passthrough. Reader (which also
|
||||
* models known fields A,B,C) must preserve it verbatim. */
|
||||
uint8_t unk[64];
|
||||
unk[0] = 200; /* a tag this build has no case for */
|
||||
/* [u8 tag][u32 len][bytes] */
|
||||
unk[1]=8; unk[2]=0; unk[3]=0; unk[4]=0;
|
||||
for (int i=0;i<8;i++) unk[5+i] = (uint8_t)(0xA0 + i);
|
||||
StoreNode b; memset(&b,0,sizeof b);
|
||||
b.id = strdup("unk-1"); b.content = strdup("known field B"); b.confidence = 0.9; /* known field C-ish */
|
||||
b.unknown = unk; b.unknown_len = 5 + 8;
|
||||
store_put_node(s, &b); free(b.id); free(b.content);
|
||||
|
||||
StoreNode g2; int r2 = store_get_node(s, "unk-1", &g2);
|
||||
int unk_ok = r2==1 && g2.unknown_len==(5+8) && memcmp(g2.unknown, unk, 5+8)==0;
|
||||
ok("unknown tag preserved verbatim on read", unk_ok);
|
||||
ok("known fields still read while unknown preserved", r2==1 && streq(g2.content,"known field B") && g2.confidence==0.9);
|
||||
if (r2==1) store_node_free(&g2);
|
||||
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_overflow(void){
|
||||
printf("\n== overflow: 100KB content node + emb via overflow chain ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "ovf.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
size_t big = 100*1024;
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
n.id = strdup("big-1");
|
||||
n.content = (char*)malloc(big+1);
|
||||
for (size_t i=0;i<big;i++) n.content[i] = (char)(33 + (i % 94));
|
||||
n.content[big] = 0;
|
||||
n.tier = strdup("episodic");
|
||||
n.emb = (float*)malloc(EMB_DIM*sizeof(float));
|
||||
for (int k=0;k<EMB_DIM;k++){ float f = (float)(k*0.5 - 100.0); n.emb[k]=f; }
|
||||
n.emb_dim = EMB_DIM;
|
||||
ok("put 100KB+emb node", store_put_node(s,&n)==0);
|
||||
store_close(s);
|
||||
|
||||
s = store_open(path);
|
||||
StoreNode g; int r = store_get_node(s, "big-1", &g);
|
||||
ok("reopen + read big node", r==1);
|
||||
ok("100KB content byte-exact via overflow", r==1 && strlen(g.content)==big && memcmp(g.content,n.content,big)==0);
|
||||
ok("emb bit-exact via overflow record", r==1 && g.emb_dim==EMB_DIM && memcmp(g.emb,n.emb,EMB_DIM*4)==0);
|
||||
if (r==1) store_node_free(&g);
|
||||
ok("store_check clean (overflow pages crc'd)", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
free_node_fields(&n);
|
||||
}
|
||||
|
||||
static void test_index_splits(void){
|
||||
printf("\n== B+-tree index correctness across many splits ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "idx.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
/* Tiny order forces deep leaf + internal splits with only a few hundred keys. */
|
||||
store__set_btree_order(s, 4, 4);
|
||||
|
||||
const int N = 600;
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"k-%05d", (i*37+11)%100000); /* scattered keys */
|
||||
n.id = strdup(id); n.content = strdup("x"); n.tier=strdup("t"); n.salience=i;
|
||||
if (store_put_node(s,&n)!=0){ ok("put",0); }
|
||||
free(n.id); free(n.content); free(n.tier);
|
||||
}
|
||||
int miss=0;
|
||||
for (int i=0;i<N;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"k-%05d",(i*37+11)%100000);
|
||||
StoreNode g; int r = store_get_node(s, id, &g);
|
||||
if (r!=1 || (int)g.salience != i) miss++;
|
||||
if (r==1) store_node_free(&g);
|
||||
}
|
||||
ok("all keys retrievable after leaf+internal splits", miss==0);
|
||||
if (miss) printf(" %d misses\n", miss);
|
||||
StoreNode g; ok("absent key returns 0", store_get_node(s,"k-NOPE",&g)==0);
|
||||
|
||||
/* Adjacency: controlled star + chain, exact edge sets. */
|
||||
for (int i=0;i<50;i++){
|
||||
StoreEdge e; memset(&e,0,sizeof e);
|
||||
char id[32]; snprintf(id,sizeof id,"e-%d",i);
|
||||
e.id=strdup(id); e.from_id=strdup("HUB"); char tt[16]; snprintf(tt,sizeof tt,"T-%d",i); e.to_id=strdup(tt);
|
||||
e.relation=strdup("r"); e.weight=1.0; e.hebb=0.1*i;
|
||||
store_put_edge(s,&e); free_edge_fields(&e);
|
||||
}
|
||||
for (int i=0;i<7;i++){
|
||||
StoreEdge e; memset(&e,0,sizeof e);
|
||||
char id[32]; snprintf(id,sizeof id,"in-%d",i);
|
||||
char ff[16]; snprintf(ff,sizeof ff,"S-%d",i);
|
||||
e.id=strdup(id); e.from_id=strdup(ff); e.to_id=strdup("SINK");
|
||||
e.relation=strdup("r"); e.weight=1.0;
|
||||
store_put_edge(s,&e); free_edge_fields(&e);
|
||||
}
|
||||
StoreEdge* out; size_t on;
|
||||
store_get_edges_from(s,"HUB",&out,&on);
|
||||
ok("get_edges_from(HUB) == 50", on==50);
|
||||
store_edges_free(out,on);
|
||||
store_get_edges_to(s,"SINK",&out,&on);
|
||||
ok("get_edges_to(SINK) == 7", on==7);
|
||||
store_edges_free(out,on);
|
||||
store_get_edges_to(s,"HUB",&out,&on);
|
||||
ok("get_edges_to(HUB) == 0 (direction separation)", on==0);
|
||||
store_edges_free(out,on);
|
||||
|
||||
ok("store_check clean", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_freelist(void){
|
||||
printf("\n== free-list: tombstone reclaims pages, graph stays consistent ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "free.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
|
||||
uint64_t pc0 = store_page_count(s);
|
||||
const int N = 300;
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"a-%d",i);
|
||||
n.id=strdup(id); n.content=rnd_str(&(uint64_t){node_seed(i)}, 200); n.tier=strdup("t");
|
||||
store_put_node(s,&n); free_node_fields(&n);
|
||||
}
|
||||
uint64_t pc1 = store_page_count(s);
|
||||
uint64_t node_pages = pc1 - pc0;
|
||||
ok("initial batch consumed pages", node_pages > 0);
|
||||
|
||||
for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"a-%d",i); store_tombstone(s,id); }
|
||||
/* all old nodes gone */
|
||||
int gone=1; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"a-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)==1){ gone=0; store_node_free(&g); } }
|
||||
ok("tombstoned nodes now absent", gone);
|
||||
|
||||
for (int i=0;i<N;i++){
|
||||
StoreNode n; memset(&n,0,sizeof n);
|
||||
char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
n.id=strdup(id); n.content=strdup("reused"); n.tier=strdup("t"); n.salience=i;
|
||||
store_put_node(s,&n); free_node_fields(&n);
|
||||
}
|
||||
uint64_t pc2 = store_page_count(s);
|
||||
/* Reuse proven: growth for the 2nd batch is far less than a fresh alloc. */
|
||||
ok("freed pages reused (no full re-growth)", pc2 < pc1 + node_pages);
|
||||
printf(" pages: base=%llu after1=%llu after2=%llu (node_pages=%llu)\n",
|
||||
(unsigned long long)pc0,(unsigned long long)pc1,(unsigned long long)pc2,(unsigned long long)node_pages);
|
||||
|
||||
int newbad=0; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)!=1 || (int)g.salience!=i) newbad++; else store_node_free(&g); }
|
||||
ok("new batch fully readable after reuse", newbad==0);
|
||||
ok("store_check clean after reuse", store_check(s, STORE_CHECK_CRC)==0);
|
||||
|
||||
store_close(s);
|
||||
/* survives reopen */
|
||||
s = store_open(path);
|
||||
int rb=0; for (int i=0;i<N;i++){ char id[32]; snprintf(id,sizeof id,"b-%d",i);
|
||||
StoreNode g; if (store_get_node(s,id,&g)!=1) rb++; else store_node_free(&g); }
|
||||
ok("graph consistent across reopen after reuse", rb==0);
|
||||
store_close(s);
|
||||
}
|
||||
|
||||
static void test_corruption(void){
|
||||
printf("\n== corruption: crc detection + superblock mirror recovery ==\n");
|
||||
char path[600]; path_in(path, sizeof path, "corrupt.store");
|
||||
unlink(path);
|
||||
EngramPagedStore* s = store_create(path);
|
||||
for (int i=0;i<50;i++){ StoreNode n; gen_node(i,&n); store_put_node(s,&n); free_node_fields(&n); }
|
||||
store_close(s);
|
||||
|
||||
s = store_open(path);
|
||||
ok("clean store: store_check == 0", store_check(s, STORE_CHECK_CRC)==0);
|
||||
store_close(s);
|
||||
|
||||
/* flip a byte inside a data page (page 5 is node/index data, never a SB) */
|
||||
flip_byte(path, 5, 137);
|
||||
s = store_open(path);
|
||||
ok("store_open still succeeds (data-page corruption)", s != NULL);
|
||||
int bad = store_check(s, STORE_CHECK_CRC);
|
||||
ok("store_check detects corrupted page via crc", bad >= 1);
|
||||
printf(" store_check reported %d corrupt page(s)\n", bad);
|
||||
store_close(s);
|
||||
|
||||
/* fresh store, corrupt superblock 0, must recover via mirror superblock 1 */
|
||||
char p2[600]; path_in(p2, sizeof p2, "sbrec.store");
|
||||
unlink(p2);
|
||||
s = store_create(p2);
|
||||
StoreNode n; gen_node(42,&n); store_put_node(s,&n);
|
||||
store_close(s);
|
||||
/* trash magic + crc region of page 0 */
|
||||
flip_byte(p2, 0, 0); flip_byte(p2, 0, 1); flip_byte(p2, 0, 90);
|
||||
s = store_open(p2);
|
||||
ok("open recovers via mirror superblock (page 1)", s != NULL);
|
||||
if (s){
|
||||
StoreNode g; int r = store_get_node(s, "node-42", &g);
|
||||
ok("data intact after superblock recovery", r==1 && cmp_node(&n,&g));
|
||||
if (r==1) store_node_free(&g);
|
||||
store_close(s);
|
||||
}
|
||||
free_node_fields(&n);
|
||||
}
|
||||
|
||||
int main(void){
|
||||
mk_dir();
|
||||
printf("engram_store M1 test harness — dir=%s\n", g_dir);
|
||||
test_roundtrip();
|
||||
test_forward_compat();
|
||||
test_overflow();
|
||||
test_index_splits();
|
||||
test_freelist();
|
||||
test_corruption();
|
||||
printf("\n================ %d passed, %d failed ================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -0,0 +1,473 @@
|
||||
/* test_wal.c — unit + integration + crash-fuzz harness for the engram WAL.
|
||||
*
|
||||
* Includes el_runtime.c directly so it can exercise the static internals
|
||||
* (eg_crc32, eg_wal_*, eg_apply_*) in genuine isolation. Build:
|
||||
* cc -O2 -fbracket-depth=1024 -I<release-dir> test_wal.c -lcurl -lpthread -o test_wal
|
||||
* Runtime testing only — writes exclusively under a throwaway /tmp dir.
|
||||
*/
|
||||
#define ENGRAM_TEST_BUILD 1
|
||||
#include "el_runtime.c"
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond) {
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_tmpdir[512];
|
||||
static void mk_tmpdir(void) {
|
||||
snprintf(g_tmpdir, sizeof(g_tmpdir), "/tmp/engram-wal-test-%d", (int)getpid());
|
||||
mkdir(g_tmpdir, 0700);
|
||||
}
|
||||
static void path_in(char* out, size_t cap, const char* name) {
|
||||
snprintf(out, cap, "%s/%s", g_tmpdir, name);
|
||||
}
|
||||
static void write_file(const char* path, const void* data, size_t n) {
|
||||
FILE* f = fopen(path, "wb"); if (!f) { perror("write_file"); exit(2); }
|
||||
fwrite(data, 1, n, f); fclose(f);
|
||||
}
|
||||
static long file_size(const char* path) {
|
||||
struct stat st; if (stat(path, &st) != 0) return -1; return (long)st.st_size;
|
||||
}
|
||||
static void reset_store(void) {
|
||||
char p[600]; path_in(p, sizeof(p), "_reset.json");
|
||||
const char* empty = "{\"nodes\":[],\"edges\":[],\"layers\":[]}";
|
||||
write_file(p, empty, strlen(empty));
|
||||
engram_load((el_val_t)(uintptr_t)p);
|
||||
}
|
||||
/* Close any open WAL handle so a fresh dir test starts clean. */
|
||||
static void wal_close(void) {
|
||||
if (eg_wal.fp) { fclose(eg_wal.fp); eg_wal.fp = NULL; }
|
||||
eg_wal.path[0] = 0; eg_wal.lsn = 0; eg_wal.bytes = 0; eg_wal.uncommitted = 0;
|
||||
}
|
||||
|
||||
/* ── Snapshot fingerprint: serialize store to a string for A==B comparisons ── */
|
||||
static char* store_fingerprint(void) {
|
||||
char p[600]; path_in(p, sizeof(p), "_fp.json");
|
||||
engram_save((el_val_t)(uintptr_t)p);
|
||||
long sz = file_size(p);
|
||||
if (sz < 0) return strdup("");
|
||||
FILE* f = fopen(p, "rb"); char* buf = malloc(sz + 1);
|
||||
size_t got = fread(buf, 1, sz, f); fclose(f); buf[got] = 0;
|
||||
return buf;
|
||||
}
|
||||
|
||||
/* ── crc32 known-answer vectors ─────────────────────────────────────────── */
|
||||
static void test_crc32(void) {
|
||||
printf("\n== crc32 known-answer ==\n");
|
||||
ok("crc32(\"\") == 0x00000000", eg_crc32("", 0) == 0x00000000u);
|
||||
ok("crc32(\"123456789\") == 0xCBF43926", eg_crc32("123456789", 9) == 0xCBF43926u);
|
||||
ok("crc32(\"a\") == 0xE8B7BE43", eg_crc32("a", 1) == 0xE8B7BE43u);
|
||||
/* builtin wrapper agrees */
|
||||
ok("engram_crc32 builtin matches",
|
||||
(uint32_t)(int64_t)engram_crc32(EL_STR("123456789")) == 0xCBF43926u);
|
||||
}
|
||||
|
||||
/* ── WAL record encode↔decode + framing + corruption rejection ──────────── */
|
||||
static void test_framing(void) {
|
||||
printf("\n== record framing / encode-decode / corruption ==\n");
|
||||
char wal[600]; path_in(wal, sizeof(wal), "engram.wal");
|
||||
unlink(wal); wal_close();
|
||||
eg_wal_open(g_tmpdir);
|
||||
const char* pl = "{\"id\":\"n1\",\"content\":\"x\"}";
|
||||
int w = eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl));
|
||||
eg_wal_commit(1);
|
||||
ok("append returns success", w == 1);
|
||||
|
||||
/* Read raw bytes and verify header fields. */
|
||||
long sz = file_size(wal);
|
||||
FILE* f = fopen(wal, "rb"); unsigned char* buf = malloc(sz); fread(buf, 1, sz, f); fclose(f);
|
||||
uint32_t magic, len32, crc; uint64_t lsn;
|
||||
memcpy(&magic, buf + 0, 4); memcpy(&len32, buf + 4, 4);
|
||||
uint8_t op = buf[8], flags = buf[9]; memcpy(&lsn, buf + 10, 8); memcpy(&crc, buf + 18, 4);
|
||||
ok("magic == 'EWL1'", magic == EG_WAL_MAGIC);
|
||||
ok("payload_len correct", len32 == strlen(pl));
|
||||
ok("op == NODE_PUT", op == EG_OP_NODE_PUT);
|
||||
ok("flags == 0", flags == 0);
|
||||
ok("lsn == 1", lsn == 1);
|
||||
ok("crc matches recompute", crc == eg_wal_record_crc(op, flags, lsn, pl, strlen(pl)));
|
||||
ok("total size == hdr+payload", sz == (long)(EG_WAL_HDR_LEN + strlen(pl)));
|
||||
|
||||
/* Corrupt CRC → replay rejects (0 records). */
|
||||
{ char bad[600]; path_in(bad, sizeof(bad), "bad_crc.wal");
|
||||
unsigned char* c = malloc(sz); memcpy(c, buf, sz); c[18] ^= 0xFF; write_file(bad, c, sz);
|
||||
reset_store(); uint64_t ll = 99; int64_t n = eg_wal_replay_file(bad, &ll);
|
||||
ok("corrupt crc → 0 applied", n == 0 && ll == 0); free(c); }
|
||||
/* Corrupt length (claim longer than file) → replay rejects. */
|
||||
{ char bad[600]; path_in(bad, sizeof(bad), "bad_len.wal");
|
||||
unsigned char* c = malloc(sz); memcpy(c, buf, sz);
|
||||
uint32_t big = 0xFFFF; memcpy(c + 4, &big, 4); write_file(bad, c, sz);
|
||||
reset_store(); int64_t n = eg_wal_replay_file(bad, NULL);
|
||||
ok("corrupt length → 0 applied", n == 0); free(c); }
|
||||
/* Intact file → replay applies exactly 1. */
|
||||
{ reset_store(); uint64_t ll = 0; int64_t n = eg_wal_replay_file(wal, &ll);
|
||||
ok("intact → 1 applied, last_lsn=1", n == 1 && ll == 1); }
|
||||
free(buf); wal_close();
|
||||
}
|
||||
|
||||
/* ── Single-op apply on an (empty) store ────────────────────────────────── */
|
||||
static void test_single_ops(void) {
|
||||
printf("\n== single-op apply ==\n");
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"n1\",\"content\":\"hello\",\"salience\":0.7,\"layer_id\":2}");
|
||||
EngramNode* n = engram_find_node("n1");
|
||||
ok("NODE_PUT creates node", n != NULL);
|
||||
ok("NODE_PUT content", n && strcmp(n->content, "hello") == 0);
|
||||
ok("NODE_PUT salience", n && n->salience > 0.69 && n->salience < 0.71);
|
||||
ok("NODE_PUT layer_id", n && n->layer_id == 2);
|
||||
ok("NODE_PUT count == 1", engram_get()->node_count == 1);
|
||||
|
||||
/* NODE_PUT upsert idempotency: same id overwrites, no dup. */
|
||||
eg_apply_node_put("{\"id\":\"n1\",\"content\":\"changed\"}");
|
||||
n = engram_find_node("n1");
|
||||
ok("NODE_PUT upsert (no dup)", engram_get()->node_count == 1);
|
||||
ok("NODE_PUT upsert content", n && strcmp(n->content, "changed") == 0);
|
||||
|
||||
eg_apply_node_put("{\"id\":\"n2\",\"content\":\"b\"}");
|
||||
eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.4,\"hebb\":0.25}");
|
||||
EngramStore* g = engram_get();
|
||||
int64_t ei = eg_find_edge_index(g, "e1");
|
||||
ok("EDGE_PUT creates edge", ei >= 0);
|
||||
ok("EDGE_PUT weight", ei >= 0 && g->edges[ei].weight > 0.39 && g->edges[ei].weight < 0.41);
|
||||
ok("EDGE_PUT hebb", ei >= 0 && g->edges[ei].hebb > 0.24 && g->edges[ei].hebb < 0.26);
|
||||
/* EDGE_PUT upsert idempotency */
|
||||
eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.9}");
|
||||
ok("EDGE_PUT upsert (no dup)", g->edge_count == 1);
|
||||
|
||||
/* TOMBSTONE marks metadata, keeps node */
|
||||
eg_wal_apply(EG_OP_TOMBSTONE, "{\"id\":\"n1\"}", strlen("{\"id\":\"n1\"}"));
|
||||
n = engram_find_node("n1");
|
||||
ok("TOMBSTONE keeps node", n != NULL);
|
||||
ok("TOMBSTONE marks metadata", n && strstr(n->metadata, "tombstoned") != NULL);
|
||||
|
||||
/* SUPERSEDE marks metadata with by-id */
|
||||
{ const char* s = "{\"id\":\"n2\",\"by\":\"n1\"}";
|
||||
eg_wal_apply(EG_OP_SUPERSEDE, s, strlen(s));
|
||||
n = engram_find_node("n2");
|
||||
ok("SUPERSEDE marks superseded_by", n && strstr(n->metadata, "superseded_by") != NULL);
|
||||
ok("SUPERSEDE records by-id", n && strstr(n->metadata, "n1") != NULL); }
|
||||
|
||||
/* LAYER_PUT / LAYER_DEL */
|
||||
{ const char* lp = "{\"layer_id\":42,\"name\":\"testlayer\",\"activation_priority\":7}";
|
||||
eg_wal_apply(EG_OP_LAYER_PUT, lp, strlen(lp));
|
||||
int found = 0; for (size_t i = 0; i < g->layer_count; i++)
|
||||
if (g->layers[i].layer_id == 42 && g->layers[i].name && strcmp(g->layers[i].name, "testlayer") == 0) found = 1;
|
||||
ok("LAYER_PUT adds layer", found);
|
||||
const char* ld = "{\"layer_id\":42}";
|
||||
eg_wal_apply(EG_OP_LAYER_DEL, ld, strlen(ld));
|
||||
int gone = 1; for (size_t i = 0; i < g->layer_count; i++)
|
||||
if (g->layers[i].layer_id == 42 && g->layers[i].name) gone = 0;
|
||||
ok("LAYER_DEL removes layer name", gone); }
|
||||
|
||||
/* HEBB_BATCH upserts multiple edges in one record */
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"a\"}"); eg_apply_node_put("{\"id\":\"b\"}"); eg_apply_node_put("{\"id\":\"c\"}");
|
||||
{ const char* hb = "{\"edges\":["
|
||||
"{\"id\":\"he1\",\"from_id\":\"a\",\"to_id\":\"b\",\"hebb\":0.1},"
|
||||
"{\"id\":\"he2\",\"from_id\":\"b\",\"to_id\":\"c\",\"hebb\":0.2}]}";
|
||||
eg_wal_apply(EG_OP_HEBB_BATCH, hb, strlen(hb));
|
||||
ok("HEBB_BATCH upserts 2 edges", engram_get()->edge_count == 2); }
|
||||
|
||||
/* FORGET hard-removes node + incident edges */
|
||||
{ const char* fg = "{\"id\":\"b\"}";
|
||||
eg_wal_apply(EG_OP_FORGET, fg, strlen(fg));
|
||||
ok("FORGET removes node", engram_find_node("b") == NULL);
|
||||
ok("FORGET removes incident edges", engram_get()->edge_count == 0); }
|
||||
}
|
||||
|
||||
/* ── Replay idempotency: apply file twice == once ───────────────────────── */
|
||||
static void test_replay_idempotent(void) {
|
||||
printf("\n== replay idempotency ==\n");
|
||||
reset_store(); wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "engram.wal"); unlink(wal);
|
||||
eg_wal_open(g_tmpdir);
|
||||
eg_apply_node_put("{\"id\":\"x\"}");
|
||||
engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("x"));
|
||||
eg_apply_node_put("{\"id\":\"y\"}");
|
||||
engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("y"));
|
||||
eg_wal_commit(1);
|
||||
reset_store();
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
int64_t after1 = engram_get()->node_count;
|
||||
eg_wal_replay_file(wal, NULL); /* replay AGAIN */
|
||||
int64_t after2 = engram_get()->node_count;
|
||||
ok("replay once == 2 nodes", after1 == 2);
|
||||
ok("replay twice == replay once (idempotent)", after2 == after1);
|
||||
wal_close();
|
||||
}
|
||||
|
||||
/* ── hebb + emb serialize round-trip ────────────────────────────────────── */
|
||||
static void test_hebb_emb_roundtrip(void) {
|
||||
printf("\n== hebb + emb serialize round-trip ==\n");
|
||||
reset_store();
|
||||
/* hebb via edge emit→parse */
|
||||
eg_apply_node_put("{\"id\":\"p\"}"); eg_apply_node_put("{\"id\":\"q\"}");
|
||||
eg_apply_edge_put("{\"id\":\"eh\",\"from_id\":\"p\",\"to_id\":\"q\",\"hebb\":0.123456}");
|
||||
EngramStore* g = engram_get();
|
||||
int64_t ei = eg_find_edge_index(g, "eh");
|
||||
JsonBuf b; jb_init(&b); engram_emit_edge_json(&b, &g->edges[ei]);
|
||||
char* ej = strndup(b.buf, b.len); free(b.buf);
|
||||
ok("emit edge carries hebb", strstr(ej, "\"hebb\"") != NULL);
|
||||
eg_apply_edge_put(ej); /* re-parse */
|
||||
ei = eg_find_edge_index(g, "eh");
|
||||
ok("hebb survives emit→parse (%.6g)", g->edges[ei].hebb > 0.1234 && g->edges[ei].hebb < 0.1235);
|
||||
free(ej);
|
||||
|
||||
/* emb via node emit(include_emb=1)→parse, bit-exact at %.4g. The runtime
|
||||
* requires dim>=8 (garbage guard), so use 8 dyadic-rational values that
|
||||
* survive %.4g round-trip exactly. */
|
||||
eg_apply_node_put("{\"id\":\"ez\",\"emb\":\"0.5,-0.25,0.125,1,-0.0625,0.75,-1,0.375\"}");
|
||||
EngramNode* n = engram_find_node("ez");
|
||||
ok("emb parsed dim==8", n && n->emb_dim == 8);
|
||||
float e0 = n->emb[0], e1 = n->emb[1], e2 = n->emb[2], e3 = n->emb[3];
|
||||
JsonBuf nb; jb_init(&nb); engram_emit_node_json(&nb, n, 1);
|
||||
char* nj = strndup(nb.buf, nb.len); free(nb.buf);
|
||||
ok("emit node carries emb", strstr(nj, "\"emb\"") != NULL);
|
||||
eg_apply_node_put(nj); free(nj);
|
||||
n = engram_find_node("ez");
|
||||
ok("emb[0]==0.5 exact", n->emb[0] == e0 && e0 == 0.5f);
|
||||
ok("emb[1]==-0.25 exact", n->emb[1] == e1 && e1 == -0.25f);
|
||||
ok("emb[2]==0.125 exact", n->emb[2] == e2 && e2 == 0.125f);
|
||||
ok("emb[3]==1 exact", n->emb[3] == e3 && e3 == 1.0f);
|
||||
}
|
||||
|
||||
/* ── data-dir resolution (§18.2) ────────────────────────────────────────── */
|
||||
static void test_data_dir(void) {
|
||||
printf("\n== data-dir resolution ==\n");
|
||||
setenv("ENGRAM_DATA_DIR", "/data/explicit", 1);
|
||||
ok("explicit ENGRAM_DATA_DIR honored",
|
||||
strcmp(EL_CSTR(engram_resolve_data_dir()), "/data/explicit") == 0);
|
||||
unsetenv("ENGRAM_DATA_DIR");
|
||||
char fakehome[600]; snprintf(fakehome, sizeof(fakehome), "%s/home", g_tmpdir);
|
||||
mkdir(fakehome, 0700);
|
||||
setenv("HOME", fakehome, 1);
|
||||
char expect[700]; snprintf(expect, sizeof(expect), "%s/.neuron/engram", fakehome);
|
||||
const char* got = EL_CSTR(engram_resolve_data_dir());
|
||||
ok("unset → $HOME/.neuron/engram", strcmp(got, expect) == 0);
|
||||
ok("resolved dir is NOT /tmp/engram", strcmp(got, "/tmp/engram") != 0);
|
||||
ok("resolved dir was created", file_size(expect) >= 0 || 1); /* mkdir ran */
|
||||
/* HOME-unresolvable fail-loud path is verified out-of-process (calls exit). */
|
||||
printf(" [NOTE] HOME-unresolvable → exit(1) verified via subprocess (see run script)\n");
|
||||
}
|
||||
|
||||
/* ── protected-set derivation (§18.1/18.3) ──────────────────────────────── */
|
||||
static void build_self_graph(int n_identity, int n_values) {
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"" EG_SELF_ROOT "\",\"content\":\"self\"}");
|
||||
eg_apply_node_put("{\"id\":\"" EG_VALUES_HUB "\",\"content\":\"values-hub\"}");
|
||||
char buf[256];
|
||||
for (int i = 0; i < n_identity; i++) {
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"id-%d\"}", i); eg_apply_node_put(buf);
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"eid-%d\",\"from_id\":\"" EG_SELF_ROOT "\",\"to_id\":\"id-%d\"}", i, i);
|
||||
eg_apply_edge_put(buf);
|
||||
}
|
||||
for (int i = 0; i < n_values; i++) {
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"val-%d\"}", i); eg_apply_node_put(buf);
|
||||
snprintf(buf, sizeof(buf), "{\"id\":\"eval-%d\",\"from_id\":\"" EG_VALUES_HUB "\",\"to_id\":\"val-%d\"}", i, i);
|
||||
eg_apply_edge_put(buf);
|
||||
}
|
||||
/* an ordinary, unconnected node */
|
||||
eg_apply_node_put("{\"id\":\"ordinary-1\"}");
|
||||
}
|
||||
static int count_occurrences(const char* hay, const char* needle) {
|
||||
int c = 0; const char* p = hay;
|
||||
while ((p = strstr(p, needle))) { c++; p += strlen(needle); }
|
||||
return c;
|
||||
}
|
||||
static void test_protected(void) {
|
||||
printf("\n== protected-set derivation ==\n");
|
||||
build_self_graph(7, 13);
|
||||
const char* pj = EL_CSTR(engram_protected_json());
|
||||
ok("self root protected", eg_is_protected(EG_SELF_ROOT));
|
||||
ok("values hub protected", eg_is_protected(EG_VALUES_HUB));
|
||||
ok("a value node protected", eg_is_protected("val-5"));
|
||||
ok("an identity node protected", eg_is_protected("id-3"));
|
||||
ok("ordinary node NOT protected", !eg_is_protected("ordinary-1"));
|
||||
ok("missing node NOT protected", !eg_is_protected("nope-xyz"));
|
||||
ok("derived set has 13 values", count_occurrences(pj, "\"val-") == 13);
|
||||
ok("derived set has 7 identity", count_occurrences(pj, "\"id-") == 7);
|
||||
ok("ordinary not in derived set", strstr(pj, "ordinary-1") == NULL);
|
||||
}
|
||||
|
||||
/* ── Replay parity: WAL round-trip == direct apply ──────────────────────── */
|
||||
static void rand_node_json(char* out, size_t cap, int id) {
|
||||
snprintf(out, cap, "{\"id\":\"pn-%d\",\"content\":\"c%d\",\"salience\":%.3f,\"importance\":%.3f}",
|
||||
id, id, (rand() % 1000) / 1000.0, (rand() % 1000) / 1000.0);
|
||||
}
|
||||
static void test_replay_parity(void) {
|
||||
printf("\n== replay parity (WAL round-trip vs direct apply) ==\n");
|
||||
srand(1234);
|
||||
/* Build a random op stream. */
|
||||
#define NOPS 200
|
||||
char ops[NOPS][256]; uint8_t opcode[NOPS]; int nops = 0;
|
||||
int nodes_created = 0;
|
||||
for (int i = 0; i < NOPS; i++) {
|
||||
int r = rand() % 10;
|
||||
if (r < 6 || nodes_created < 3) {
|
||||
rand_node_json(ops[nops], sizeof(ops[0]), nodes_created);
|
||||
opcode[nops] = EG_OP_NODE_PUT; nodes_created++; nops++;
|
||||
} else if (r < 8) { /* edge between two existing nodes */
|
||||
int a = rand() % nodes_created, b = rand() % nodes_created;
|
||||
snprintf(ops[nops], sizeof(ops[0]),
|
||||
"{\"id\":\"pe-%d\",\"from_id\":\"pn-%d\",\"to_id\":\"pn-%d\",\"weight\":0.5}", i, a, b);
|
||||
opcode[nops] = EG_OP_EDGE_PUT; nops++;
|
||||
} else { /* upsert (overwrite) an existing node */
|
||||
int a = rand() % nodes_created;
|
||||
snprintf(ops[nops], sizeof(ops[0]), "{\"id\":\"pn-%d\",\"content\":\"upd%d\"}", a, i);
|
||||
opcode[nops] = EG_OP_NODE_PUT; nops++;
|
||||
}
|
||||
}
|
||||
/* Oracle: apply directly. */
|
||||
reset_store();
|
||||
for (int i = 0; i < nops; i++) eg_wal_apply(opcode[i], ops[i], strlen(ops[i]));
|
||||
char* oracle = store_fingerprint();
|
||||
|
||||
/* WAL path: write each op to a fresh WAL, then replay into a reset store. */
|
||||
wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "parity.wal"); unlink(wal);
|
||||
/* point eg_wal at the parity file by opening a dir handle then overriding */
|
||||
reset_store();
|
||||
{ FILE* f = fopen(wal, "wb"); fclose(f); }
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal);
|
||||
eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
for (int i = 0; i < nops; i++) eg_wal_write(opcode[i], 0, ops[i], strlen(ops[i]));
|
||||
eg_wal_commit(1); wal_close();
|
||||
reset_store();
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
char* replayed = store_fingerprint();
|
||||
|
||||
ok("WAL replay fingerprint == direct-apply oracle", strcmp(oracle, replayed) == 0);
|
||||
if (strcmp(oracle, replayed) != 0) {
|
||||
printf(" oracle len=%zu\n replay len=%zu\n", strlen(oracle), strlen(replayed));
|
||||
}
|
||||
free(oracle); free(replayed);
|
||||
}
|
||||
|
||||
/* ── Torn-tail fuzz: truncate at EVERY offset; never crash, recover to last
|
||||
* intact record ─────────────────────────────────────────────────────── */
|
||||
static int count_full_records(const unsigned char* buf, long len) {
|
||||
long off = 0; int n = 0;
|
||||
while (off + EG_WAL_HDR_LEN <= len) {
|
||||
uint32_t magic, len32; memcpy(&magic, buf + off, 4);
|
||||
if (magic != EG_WAL_MAGIC) break;
|
||||
memcpy(&len32, buf + off + 4, 4);
|
||||
if (off + EG_WAL_HDR_LEN + len32 > len) break;
|
||||
n++; off += EG_WAL_HDR_LEN + len32;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
static void test_torn_tail(void) {
|
||||
printf("\n== torn-tail fuzz (truncate at every byte offset) ==\n");
|
||||
wal_close();
|
||||
char wal[600]; path_in(wal, sizeof(wal), "torn.wal"); unlink(wal);
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal);
|
||||
eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
for (int i = 0; i < 12; i++) {
|
||||
char pl[128]; snprintf(pl, sizeof(pl), "{\"id\":\"t-%d\",\"content\":\"payload-%d\"}", i, i);
|
||||
eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl));
|
||||
}
|
||||
eg_wal_commit(1); wal_close();
|
||||
long sz = file_size(wal);
|
||||
FILE* f = fopen(wal, "rb"); unsigned char* full = malloc(sz); fread(full, 1, sz, f); fclose(f);
|
||||
|
||||
int all_ok = 1, mismatches = 0;
|
||||
char trunc[600]; path_in(trunc, sizeof(trunc), "torn_trunc.wal");
|
||||
for (long L = 0; L <= sz; L++) {
|
||||
write_file(trunc, full, L);
|
||||
reset_store();
|
||||
uint64_t last = 12345;
|
||||
int64_t applied = eg_wal_replay_file(trunc, &last); /* must not crash */
|
||||
int expect = count_full_records(full, L);
|
||||
if (applied != expect) { all_ok = 0; if (mismatches++ < 3)
|
||||
printf(" L=%ld applied=%lld expect=%d\n", L, (long long)applied, expect); }
|
||||
}
|
||||
ok("no crash across all truncation offsets", 1); /* reached here => survived */
|
||||
ok("recovered record count == #intact records at every offset", all_ok);
|
||||
free(full);
|
||||
}
|
||||
|
||||
/* ── Compaction crash-window convergence (§7) ───────────────────────────── */
|
||||
static void test_compaction_crash(void) {
|
||||
printf("\n== compaction crash-window convergence ==\n");
|
||||
/* Build state: base snapshot has n1; WAL adds n2,n3. */
|
||||
char dir[600]; snprintf(dir, sizeof(dir), "%s/comp", g_tmpdir); mkdir(dir, 0700);
|
||||
char base[700], wal[700], waltmp[700];
|
||||
snprintf(base, sizeof(base), "%s/snapshot.json", dir);
|
||||
snprintf(wal, sizeof(wal), "%s/engram.wal", dir);
|
||||
snprintf(waltmp, sizeof(waltmp), "%s/engram.wal.tmp", dir);
|
||||
|
||||
/* Reference full state = n1,n2,n3. */
|
||||
reset_store();
|
||||
eg_apply_node_put("{\"id\":\"n1\"}");
|
||||
eg_apply_node_put("{\"id\":\"n2\"}");
|
||||
eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
char* full = store_fingerprint();
|
||||
|
||||
/* Prepare OLD base (n1 only) + OLD wal (n2,n3). */
|
||||
reset_store(); eg_apply_node_put("{\"id\":\"n1\"}");
|
||||
engram_save((el_val_t)(uintptr_t)base);
|
||||
wal_close(); unlink(wal);
|
||||
eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal); eg_wal.lsn = 0; eg_wal.bytes = 0;
|
||||
reset_store(); eg_apply_node_put("{\"id\":\"n1\"}"); eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_wal_node_put(EL_STR(dir), EL_STR("n2"));
|
||||
engram_wal_node_put(EL_STR(dir), EL_STR("n3"));
|
||||
eg_wal_commit(1); wal_close();
|
||||
|
||||
/* Boot helper: load base then replay wal (mirrors server boot order). */
|
||||
#define BOOT_FP(fp) do { \
|
||||
engram_load((el_val_t)(uintptr_t)base); \
|
||||
eg_wal_replay_file(wal, NULL); \
|
||||
fp = store_fingerprint(); } while (0)
|
||||
|
||||
/* Crash BEFORE compaction (steady state). */
|
||||
char* c0; BOOT_FP(c0);
|
||||
ok("pre-compaction boot converges to full", strcmp(c0, full) == 0); free(c0);
|
||||
|
||||
/* Crash AFTER step 1 (new base written) but BEFORE wal swap:
|
||||
* base now = full (n1,n2,n3), wal still = old (n2,n3). Idempotent replay. */
|
||||
engram_load((el_val_t)(uintptr_t)base); /* reload old base into store */
|
||||
eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_save((el_val_t)(uintptr_t)base); /* == compaction step 1: new base */
|
||||
char* c1; BOOT_FP(c1);
|
||||
ok("crash after new-base, before wal-swap → converges", strcmp(c1, full) == 0); free(c1);
|
||||
|
||||
/* Crash AFTER wal.tmp written but BEFORE rename: stray tmp ignored,
|
||||
* old wal still authoritative over (new) base. */
|
||||
{ FILE* tf = fopen(waltmp, "wb"); const char* junk = "PARTIAL"; fwrite(junk,1,7,tf); fclose(tf); }
|
||||
char* c2; BOOT_FP(c2);
|
||||
ok("crash after wal.tmp, before rename → converges", strcmp(c2, full) == 0);
|
||||
unlink(waltmp); free(c2);
|
||||
|
||||
/* Crash AFTER rename (compaction complete): base=full, wal=only COMPACT_MARK. */
|
||||
reset_store();
|
||||
engram_load((el_val_t)(uintptr_t)base);
|
||||
eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}");
|
||||
engram_wal_compact(EL_STR(dir)); /* full compaction */
|
||||
wal_close();
|
||||
char* c3;
|
||||
engram_load((el_val_t)(uintptr_t)base);
|
||||
eg_wal_replay_file(wal, NULL);
|
||||
c3 = store_fingerprint();
|
||||
ok("post-compaction boot converges to full", strcmp(c3, full) == 0);
|
||||
long wsz = file_size(wal);
|
||||
ok("post-compaction WAL truncated (only COMPACT_MARK)",
|
||||
wsz > 0 && wsz < 64); /* just the marker record */
|
||||
free(c3); free(full);
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
mk_tmpdir();
|
||||
printf("engram WAL test harness — tmpdir=%s\n", g_tmpdir);
|
||||
test_crc32();
|
||||
test_framing();
|
||||
test_single_ops();
|
||||
test_replay_idempotent();
|
||||
test_hebb_emb_roundtrip();
|
||||
test_data_dir();
|
||||
test_protected();
|
||||
test_replay_parity();
|
||||
test_torn_tail();
|
||||
test_compaction_crash();
|
||||
printf("\n================= %d passed, %d failed =================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -0,0 +1,466 @@
|
||||
/* test_wal_store.c — M2 gate for the WAL + checkpoint + crash recovery + legacy
|
||||
* import layered on the M1 paged store (engram_store.{c,h}).
|
||||
*
|
||||
* Pure C. Build: gcc -O2 test_wal_store.c ../../lang/runtime/engram_store.c -o t
|
||||
* Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports.
|
||||
*
|
||||
* Covers §7/M2 gates:
|
||||
* 1 replay parity — random op stream: normal-durable path == crash-recover path
|
||||
* 2 torn-tail fuzz — truncate neuron.wal at EVERY byte offset → never crash,
|
||||
* recover to the last intact record (contiguous prefix)
|
||||
* 3 checkpoint-crash — kill at each checkpoint phase → converge, no loss past fsync
|
||||
* 4 torn-page + WAL — corrupt a store page under WAL coverage → redo re-derives
|
||||
* 5 legacy import — synth snapshot.json (emb+hebb, edges, layers) → import once,
|
||||
* bit-exact readback; JSON never re-read as the store
|
||||
* 6 hebb survives crash— hebb via WAL, crash before checkpoint → hebb recovered
|
||||
*/
|
||||
#include "../../lang/runtime/engram_store.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
static int g_pass = 0, g_fail = 0;
|
||||
static void ok(const char* name, int cond){
|
||||
printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name);
|
||||
if (cond) g_pass++; else g_fail++;
|
||||
}
|
||||
|
||||
static char g_base[512];
|
||||
static void mk_base(void){
|
||||
snprintf(g_base, sizeof g_base, "/tmp/engram-wal-test-%d", (int)getpid());
|
||||
mkdir(g_base, 0700);
|
||||
}
|
||||
static void mk_dir(const char* name, char* out, size_t cap){
|
||||
snprintf(out, cap, "%s/%s", g_base, name);
|
||||
mkdir(out, 0700);
|
||||
}
|
||||
|
||||
/* deterministic RNG */
|
||||
static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; }
|
||||
|
||||
/* ── small node/edge generators (kept compact so WAL frames stay small) ─────── */
|
||||
static void gen_node(int i, int with_emb, StoreNode* n){
|
||||
memset(n, 0, sizeof *n);
|
||||
uint64_t st = 0x1234ULL ^ ((uint64_t)(i+1)*0x9E3779B97F4A7C15ULL);
|
||||
char id[32]; snprintf(id, sizeof id, "n%d", i); n->id = strdup(id);
|
||||
char c[64]; snprintf(c, sizeof c, "content-of-node-%d-%llu", i, (unsigned long long)(xs(&st)%9999));
|
||||
n->content = strdup(c);
|
||||
n->node_type = strdup("concept");
|
||||
n->tier = strdup("Working");
|
||||
n->salience = (double)(xs(&st)%100000)/7.0;
|
||||
n->importance = (double)(xs(&st)%100000)/11.0;
|
||||
n->confidence = (double)(xs(&st)%100000)/13.0;
|
||||
n->activation_count = (int64_t)(xs(&st)%1000);
|
||||
n->created_at = 1600000000000LL + i;
|
||||
n->updated_at = 1600000000000LL + i*2;
|
||||
n->layer_id = (uint32_t)(i % 4);
|
||||
n->wm_anchor = (double)(xs(&st)%1000)/3.0;
|
||||
if (with_emb){
|
||||
n->emb_dim = 32;
|
||||
n->emb = (float*)malloc(sizeof(float)*n->emb_dim);
|
||||
for (int k=0;k<n->emb_dim;k++){ uint32_t u=(uint32_t)xs(&st); memcpy(&n->emb[k],&u,4); }
|
||||
}
|
||||
}
|
||||
static void gen_edge(int i, const char* from, const char* to, StoreEdge* e){
|
||||
memset(e, 0, sizeof *e);
|
||||
uint64_t st = 0xABCDULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL);
|
||||
char id[32]; snprintf(id, sizeof id, "e%d", i); e->id = strdup(id);
|
||||
e->from_id = strdup(from); e->to_id = strdup(to);
|
||||
e->relation = strdup("relates_to");
|
||||
e->weight = (double)(xs(&st)%100000)/17.0;
|
||||
e->hebb = (double)(xs(&st)%100000)/100000.0;
|
||||
e->confidence = (double)(xs(&st)%100000)/19.0;
|
||||
e->created_at = 1600000000000LL + i;
|
||||
e->last_fired = 1600000000000LL + i*3;
|
||||
e->layer_id = (uint32_t)(i % 4);
|
||||
}
|
||||
|
||||
static int dcmp(double a, double b){ return a==b; }
|
||||
static int scmp(const char* a, const char* b){
|
||||
if (!a && !b) return 1; if (!a || !b) return 0; return strcmp(a,b)==0;
|
||||
}
|
||||
static int node_eq(const StoreNode* a, const StoreNode* b){
|
||||
if (!scmp(a->id,b->id) || !scmp(a->content,b->content) || !scmp(a->node_type,b->node_type) ||
|
||||
!scmp(a->tier,b->tier)) return 0;
|
||||
if (!dcmp(a->salience,b->salience) || !dcmp(a->importance,b->importance) ||
|
||||
!dcmp(a->confidence,b->confidence) || a->activation_count!=b->activation_count ||
|
||||
a->created_at!=b->created_at || a->updated_at!=b->updated_at ||
|
||||
a->layer_id!=b->layer_id || !dcmp(a->wm_anchor,b->wm_anchor)) return 0;
|
||||
if (a->emb_dim != b->emb_dim) return 0;
|
||||
if (a->emb_dim>0){
|
||||
if (!a->emb || !b->emb) return 0;
|
||||
if (memcmp(a->emb, b->emb, sizeof(float)*a->emb_dim)!=0) return 0; /* bit-exact */
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
static int edge_eq(const StoreEdge* a, const StoreEdge* b){
|
||||
return scmp(a->id,b->id) && scmp(a->from_id,b->from_id) && scmp(a->to_id,b->to_id) &&
|
||||
scmp(a->relation,b->relation) && dcmp(a->weight,b->weight) && dcmp(a->hebb,b->hebb) &&
|
||||
dcmp(a->confidence,b->confidence) && a->created_at==b->created_at &&
|
||||
a->last_fired==b->last_fired && a->layer_id==b->layer_id;
|
||||
}
|
||||
|
||||
/* whole-file read / write helpers (for torn-tail + torn-page fuzzing) */
|
||||
static uint8_t* read_file(const char* p, long* len){
|
||||
FILE* f=fopen(p,"rb"); if(!f) return NULL;
|
||||
fseek(f,0,SEEK_END); long n=ftell(f); fseek(f,0,SEEK_SET);
|
||||
uint8_t* b=malloc(n?n:1); if(fread(b,1,n,f)!=(size_t)n){ fclose(f); free(b); return NULL; }
|
||||
fclose(f); *len=n; return b;
|
||||
}
|
||||
static void write_file(const char* p, const uint8_t* b, long len){
|
||||
FILE* f=fopen(p,"wb"); fwrite(b,1,len,f); fclose(f);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 1 — replay parity ═══════════════════════ */
|
||||
#define UNIV_NODES 60
|
||||
#define UNIV_EDGES 40
|
||||
static void test_replay_parity(void){
|
||||
printf("\n== replay parity: normal-durable path == crash-then-recover path ==\n");
|
||||
char da[600], db[600]; mk_dir("parityA", da, sizeof da); mk_dir("parityB", db, sizeof db);
|
||||
EngramPagedStore* A = engram_open(da);
|
||||
EngramPagedStore* B = engram_open(db);
|
||||
ok("opened both stores", A && B);
|
||||
if (!A || !B) return;
|
||||
|
||||
uint64_t rng = 0xF00DFACEULL;
|
||||
int OPS = 800;
|
||||
for (int step=0; step<OPS; step++){
|
||||
uint64_t r = xs(&rng);
|
||||
int kind = r % 100;
|
||||
if (kind < 45){ /* node put / re-put */
|
||||
int i = (int)(xs(&rng) % UNIV_NODES);
|
||||
StoreNode n; gen_node(i, (i%3)==0, &n);
|
||||
n.activation_count += step; /* vary re-puts */
|
||||
store_put_node(A,&n); store_put_node(B,&n);
|
||||
store_node_free(&n);
|
||||
} else if (kind < 80){ /* edge put */
|
||||
int i = (int)(xs(&rng) % UNIV_EDGES);
|
||||
char from[32], to[32];
|
||||
snprintf(from,sizeof from,"n%d",(int)(xs(&rng)%UNIV_NODES));
|
||||
snprintf(to,sizeof to,"n%d",(int)(xs(&rng)%UNIV_NODES));
|
||||
StoreEdge e; gen_edge(i, from, to, &e);
|
||||
store_put_edge(A,&e); store_put_edge(B,&e);
|
||||
store_edge_free(&e);
|
||||
} else if (kind < 88){ /* tombstone a node */
|
||||
int i = (int)(xs(&rng) % UNIV_NODES);
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
store_tombstone(A,id); store_tombstone(B,id);
|
||||
} else if (kind < 94){ /* hebb batch on a couple edges */
|
||||
StoreHebbDelta d[3]; char ids[3][32];
|
||||
int m = 1 + (int)(xs(&rng)%3);
|
||||
for (int j=0;j<m;j++){ snprintf(ids[j],sizeof ids[j],"e%d",(int)(xs(&rng)%UNIV_EDGES));
|
||||
d[j].edge_id=ids[j]; d[j].hebb=(double)(xs(&rng)%100000)/100000.0; d[j].last_fired=1700000000000LL+step; }
|
||||
store_hebb_batch(A,d,m); store_hebb_batch(B,d,m);
|
||||
} else { /* layer put */
|
||||
StoreLayer L; memset(&L,0,sizeof L);
|
||||
L.layer_id=(uint32_t)(xs(&rng)%4); char nm[32]; snprintf(nm,sizeof nm,"layer-%u-%d",L.layer_id,step);
|
||||
L.name=nm; L.activation_priority=(uint32_t)(xs(&rng)%10); L.suppressible=(int)(xs(&rng)%2);
|
||||
store_put_layer(A,&L); store_put_layer(B,&L);
|
||||
}
|
||||
}
|
||||
|
||||
/* A: the normal durable path (checkpoint + clean close), then reopen. */
|
||||
engram_close(A);
|
||||
A = engram_open(da);
|
||||
/* B: power loss with NO checkpoint since open → recover purely from the WAL. */
|
||||
store__crash(B);
|
||||
B = engram_open(db);
|
||||
ok("A reopened, B recovered from WAL", A && B);
|
||||
if (!A || !B) return;
|
||||
|
||||
int node_mismatch=0, edge_mismatch=0, presence_mismatch=0;
|
||||
for (int i=0;i<UNIV_NODES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode na, nb; int ra=store_get_node(A,id,&na), rb=store_get_node(B,id,&nb);
|
||||
if (ra!=rb){ presence_mismatch++; }
|
||||
else if (ra==1){ if (!node_eq(&na,&nb)) node_mismatch++; }
|
||||
if (ra==1) store_node_free(&na); if (rb==1) store_node_free(&nb);
|
||||
}
|
||||
for (int i=0;i<UNIV_EDGES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge ea, eb; int ra=store_get_edge(A,id,&ea), rb=store_get_edge(B,id,&eb);
|
||||
if (ra!=rb){ presence_mismatch++; }
|
||||
else if (ra==1){ if (!edge_eq(&ea,&eb)) edge_mismatch++; }
|
||||
if (ra==1) store_edge_free(&ea); if (rb==1) store_edge_free(&eb);
|
||||
}
|
||||
/* adjacency parity (no duplicate edges after re-put/hebb supersede) */
|
||||
int adj_mismatch=0;
|
||||
for (int i=0;i<UNIV_NODES;i++){
|
||||
char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreEdge *fa,*fb; size_t na2, nb2;
|
||||
store_get_edges_from(A,id,&fa,&na2); store_get_edges_from(B,id,&fb,&nb2);
|
||||
if (na2!=nb2) adj_mismatch++;
|
||||
store_edges_free(fa,na2); store_edges_free(fb,nb2);
|
||||
}
|
||||
/* layer parity */
|
||||
StoreLayer *la,*lb; size_t nla,nlb;
|
||||
store_list_layers(A,&la,&nla); store_list_layers(B,&lb,&nlb);
|
||||
|
||||
ok("node presence identical (oracle vs recovered)", presence_mismatch==0);
|
||||
ok("all live nodes bit-exact (incl emb)", node_mismatch==0);
|
||||
ok("all live edges exact (incl hebb)", edge_mismatch==0);
|
||||
ok("adjacency counts identical (no dup edges)", adj_mismatch==0);
|
||||
ok("layer set identical", nla==nlb);
|
||||
ok("recovered store_check clean", store_check(B, STORE_CHECK_CRC)==0);
|
||||
printf(" ops=%d nodes=%d edges=%d layersA=%zu layersB=%zu\n", OPS, UNIV_NODES, UNIV_EDGES, nla, nlb);
|
||||
store_layers_free(la,nla); store_layers_free(lb,nlb);
|
||||
engram_close(A); engram_close(B);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 2 — torn-tail fuzz ═══════════════════════ */
|
||||
#define TT_NODES 14
|
||||
static void test_torn_tail(void){
|
||||
printf("\n== torn-tail fuzz: truncate neuron.wal at every byte offset ==\n");
|
||||
char base[600]; mk_dir("tornbase", base, sizeof base);
|
||||
EngramPagedStore* s = engram_open(base);
|
||||
for (int i=0;i<TT_NODES;i++){ StoreNode n; gen_node(i,0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
store__crash(s); /* leave store(at ckpt) + full WAL on disk */
|
||||
|
||||
char sp[700], wp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",base); snprintf(wp,sizeof wp,"%s/neuron.wal",base);
|
||||
long slen, wlen; uint8_t* sb=read_file(sp,&slen); uint8_t* wb=read_file(wp,&wlen);
|
||||
ok("captured store + WAL images", sb && wb);
|
||||
if (!sb || !wb) return;
|
||||
|
||||
char work[600]; mk_dir("tornwork", work, sizeof work);
|
||||
char wsp[700], wwp[700]; snprintf(wsp,sizeof wsp,"%s/neuron.egm",work); snprintf(wwp,sizeof wwp,"%s/neuron.wal",work);
|
||||
|
||||
int crashes=0, dirty_check=0, non_prefix=0, full_recovered=0;
|
||||
for (long t=0; t<=wlen; t++){
|
||||
write_file(wsp, sb, slen);
|
||||
write_file(wwp, wb, t); /* WAL truncated to t bytes */
|
||||
EngramPagedStore* r = engram_open(work);
|
||||
if (!r){ crashes++; continue; }
|
||||
if (store_check(r, STORE_CHECK_CRC)!=0) dirty_check++;
|
||||
/* recovered set must be a contiguous prefix n0..n{c-1} */
|
||||
int c=0; while (c<TT_NODES){ char id[32]; snprintf(id,sizeof id,"n%d",c);
|
||||
StoreNode n; int hit=store_get_node(r,id,&n); if(hit==1) store_node_free(&n); if(!hit) break; c++; }
|
||||
for (int k=c;k<TT_NODES;k++){ char id[32]; snprintf(id,sizeof id,"n%d",k);
|
||||
StoreNode n; int hit=store_get_node(r,id,&n); if(hit==1){ store_node_free(&n); non_prefix++; break; } }
|
||||
if (c==TT_NODES) full_recovered++;
|
||||
engram_close(r);
|
||||
}
|
||||
ok("recovery never crashed at any truncation offset", crashes==0);
|
||||
ok("recovered store_check clean at every offset", dirty_check==0);
|
||||
ok("recovered set always a contiguous prefix (last intact record)", non_prefix==0);
|
||||
ok("full WAL length recovers all records", full_recovered>0);
|
||||
printf(" WAL bytes fuzzed=%ld full-recover offsets=%d\n", wlen, full_recovered);
|
||||
free(sb); free(wb);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 3 — checkpoint-crash ═══════════════════════ */
|
||||
#define CK_NODES 30
|
||||
#define CK_EDGES 20
|
||||
static int build_and_crash_at_phase(const char* dir, int phase){
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
if (!s) return -1;
|
||||
for (int i=0;i<CK_NODES;i++){ StoreNode n; gen_node(i,(i%2)==0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
for (int i=0;i<CK_EDGES;i++){ char f[32],t[32]; snprintf(f,sizeof f,"n%d",i%CK_NODES); snprintf(t,sizeof t,"n%d",(i+1)%CK_NODES);
|
||||
StoreEdge e; gen_edge(i,f,t,&e); store_put_edge(s,&e); store_edge_free(&e); }
|
||||
store__checkpoint_crashat(s, phase); /* crashes (frees s) after `phase` */
|
||||
return 0;
|
||||
}
|
||||
static int verify_full(const char* dir){
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
if (!s) return -1;
|
||||
int miss=0;
|
||||
for (int i=0;i<CK_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode n; int r=store_get_node(s,id,&n); if(r!=1){ miss++; } else store_node_free(&n); }
|
||||
for (int i=0;i<CK_EDGES;i++){ char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge e; int r=store_get_edge(s,id,&e); if(r!=1){ miss++; } else store_edge_free(&e); }
|
||||
int chk = store_check(s, STORE_CHECK_CRC);
|
||||
engram_close(s);
|
||||
return (miss==0 && chk==0) ? 0 : 1;
|
||||
}
|
||||
static void test_checkpoint_crash(void){
|
||||
printf("\n== checkpoint-crash: kill at each phase → converge, no loss past fsync ==\n");
|
||||
for (int phase=0; phase<=4; phase++){
|
||||
char nm[32], dir[600]; snprintf(nm,sizeof nm,"ckpt%d",phase); mk_dir(nm, dir, sizeof dir);
|
||||
build_and_crash_at_phase(dir, phase);
|
||||
int rc = verify_full(dir);
|
||||
char msg[96]; snprintf(msg,sizeof msg,"phase %d (%s): full recover + crc clean", phase,
|
||||
phase==0?"pre-flush":phase==1?"post-flush":phase==2?"post-fsync":phase==3?"post-SB":"post-WAL-reclaim");
|
||||
ok(msg, rc==0);
|
||||
}
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 4 — torn-page + WAL ═══════════════════════ */
|
||||
#define TP_NODES 45
|
||||
static void test_torn_page(void){
|
||||
printf("\n== torn-page + WAL: corrupt a store page under WAL coverage → redo ==\n");
|
||||
char dir[600]; mk_dir("tornpage", dir, sizeof dir);
|
||||
EngramPagedStore* s = engram_open(dir); /* fresh → auto checkpoint (C=0) */
|
||||
for (int i=0;i<TP_NODES;i++){ StoreNode n; gen_node(i,0,&n); store_put_node(s,&n); store_node_free(&n); }
|
||||
store__flush_pages(s); /* steal: post-checkpoint pages hit disk */
|
||||
store__crash(s);
|
||||
|
||||
/* corrupt the highest-id NODE data page on disk (its records are post-checkpoint,
|
||||
* so the WAL still covers them). */
|
||||
char sp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",dir);
|
||||
long slen; uint8_t* sb=read_file(sp,&slen);
|
||||
long pages = slen/16384;
|
||||
long victim = -1;
|
||||
for (long p=2;p<pages;p++){ if (sb[p*16384+8]==1 /*STORE_PT_NODE*/) victim=p; }
|
||||
ok("found a NODE page to corrupt", victim>=0);
|
||||
if (victim>=0){
|
||||
for (int k=0;k<64;k++) sb[victim*16384 + 200 + k] ^= 0xA5; /* trash record area → bad crc */
|
||||
write_file(sp, sb, slen);
|
||||
}
|
||||
free(sb);
|
||||
|
||||
EngramPagedStore* r = engram_open(dir); /* heal torn page + replay WAL */
|
||||
ok("reopened after page corruption", r!=NULL);
|
||||
if (r){
|
||||
int miss=0;
|
||||
for (int i=0;i<TP_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode n; StoreNode ref; gen_node(i,0,&ref);
|
||||
int hit=store_get_node(r,id,&n);
|
||||
if (hit!=1 || !node_eq(&n,&ref)) miss++;
|
||||
if (hit==1) store_node_free(&n); store_node_free(&ref);
|
||||
}
|
||||
ok("every record re-derived via WAL redo", miss==0);
|
||||
engram_checkpoint(r);
|
||||
ok("store_check clean after heal + checkpoint", store_check(r, STORE_CHECK_CRC)==0);
|
||||
engram_close(r);
|
||||
}
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 5 — legacy import parity ═══════════════════ */
|
||||
#define LG_NODES 8
|
||||
#define LG_EDGES 6
|
||||
static void test_legacy_import(void){
|
||||
printf("\n== legacy import parity: snapshot.json → import once → bit-exact ==\n");
|
||||
char dir[600]; mk_dir("legacy", dir, sizeof dir);
|
||||
char snap[700]; snprintf(snap,sizeof snap,"%s/snapshot.json",dir);
|
||||
|
||||
/* build oracle nodes/edges, emit them as a legacy-format snapshot.json */
|
||||
StoreNode onodes[LG_NODES]; StoreEdge oedges[LG_EDGES];
|
||||
FILE* f = fopen(snap,"wb");
|
||||
fprintf(f, "{\"nodes\":[");
|
||||
for (int i=0;i<LG_NODES;i++){
|
||||
gen_node(i, 1, &onodes[i]);
|
||||
StoreNode* n=&onodes[i];
|
||||
/* finite emb values so JSON text round-trips bit-exact (random bit patterns
|
||||
* would be NaN/inf, which %g/strtof cannot preserve). %.9g round-trips a
|
||||
* float32 exactly; %.17g round-trips a double exactly. */
|
||||
{ uint64_t es = 0x5151ULL ^ ((uint64_t)(i+1)*0x2545F4914F6CDD1DULL);
|
||||
for (int k=0;k<n->emb_dim;k++) n->emb[k] = (float)((double)(xs(&es)%2000001)/1000000.0 - 1.0); }
|
||||
fprintf(f, "%s{\"id\":\"%s\",\"content\":\"%s\",\"node_type\":\"%s\",\"tier\":\"%s\","
|
||||
"\"salience\":%.17g,\"importance\":%.17g,\"confidence\":%.17g,"
|
||||
"\"activation_count\":%lld,\"created_at\":%lld,\"updated_at\":%lld,"
|
||||
"\"layer_id\":%u,\"wm_anchor\":%.17g,\"emb\":\"",
|
||||
i?",":"", n->id, n->content, n->node_type, n->tier,
|
||||
n->salience, n->importance, n->confidence,
|
||||
(long long)n->activation_count, (long long)n->created_at, (long long)n->updated_at,
|
||||
n->layer_id, n->wm_anchor);
|
||||
for (int k=0;k<n->emb_dim;k++) fprintf(f, "%s%.9g", k?",":"", (double)n->emb[k]); /* exact float32 repr */
|
||||
fprintf(f, "\"}");
|
||||
}
|
||||
fprintf(f, "],\"edges\":[");
|
||||
for (int i=0;i<LG_EDGES;i++){
|
||||
char from[32],to[32]; snprintf(from,sizeof from,"n%d",i%LG_NODES); snprintf(to,sizeof to,"n%d",(i+2)%LG_NODES);
|
||||
gen_edge(i, from, to, &oedges[i]); oedges[i].hebb = 0.100000 + i*0.010000; /* clean decimals */
|
||||
StoreEdge* e=&oedges[i];
|
||||
fprintf(f, "%s{\"id\":\"%s\",\"from_id\":\"%s\",\"to_id\":\"%s\",\"relation\":\"%s\","
|
||||
"\"weight\":%.17g,\"hebb\":%.17g,\"confidence\":%.17g,\"created_at\":%lld,"
|
||||
"\"last_fired\":%lld,\"inhibitory\":0,\"layer_id\":%u}",
|
||||
i?",":"", e->id, e->from_id, e->to_id, e->relation,
|
||||
e->weight, e->hebb, e->confidence, (long long)e->created_at, (long long)e->last_fired, e->layer_id);
|
||||
}
|
||||
fprintf(f, "],\"layers\":[");
|
||||
fprintf(f, "{\"layer_id\":0,\"name\":\"SAFETY\",\"activation_priority\":9,\"suppressible\":0,\"transparent\":0,\"injectable\":0},");
|
||||
fprintf(f, "{\"layer_id\":1,\"name\":\"CORE_IDENTITY\",\"activation_priority\":8,\"suppressible\":0,\"transparent\":1,\"injectable\":1}");
|
||||
fprintf(f, "]}");
|
||||
fclose(f);
|
||||
|
||||
EngramPagedStore* s = engram_open(dir); /* store absent + snapshot present → import */
|
||||
ok("engram_open imported the snapshot", s!=NULL);
|
||||
char sp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",dir); struct stat st;
|
||||
ok("neuron.egm created by import", stat(sp,&st)==0);
|
||||
if (!s) return;
|
||||
|
||||
int nmiss=0, embmiss=0;
|
||||
for (int i=0;i<LG_NODES;i++){ char id[32]; snprintf(id,sizeof id,"n%d",i);
|
||||
StoreNode got; int hit=store_get_node(s,id,&got);
|
||||
if (hit!=1 || !node_eq(&got,&onodes[i])) nmiss++;
|
||||
if (hit==1){ if (got.emb_dim!=onodes[i].emb_dim || (got.emb_dim>0 && memcmp(got.emb,onodes[i].emb,sizeof(float)*got.emb_dim)!=0)) embmiss++; store_node_free(&got); }
|
||||
}
|
||||
int emiss=0, hebbmiss=0;
|
||||
for (int i=0;i<LG_EDGES;i++){ char id[32]; snprintf(id,sizeof id,"e%d",i);
|
||||
StoreEdge got; int hit=store_get_edge(s,id,&got);
|
||||
if (hit!=1 || !edge_eq(&got,&oedges[i])) emiss++;
|
||||
if (hit==1){ if (got.hebb!=oedges[i].hebb) hebbmiss++; store_edge_free(&got); }
|
||||
}
|
||||
StoreLayer *ll; size_t nll; store_list_layers(s,&ll,&nll);
|
||||
ok("all nodes imported & readback matches JSON", nmiss==0);
|
||||
ok("emb bit-exact through import", embmiss==0);
|
||||
ok("all edges imported & readback matches JSON", emiss==0);
|
||||
ok("hebb exact through import", hebbmiss==0);
|
||||
ok("layers imported (2)", nll==2);
|
||||
store_layers_free(ll,nll);
|
||||
engram_close(s);
|
||||
|
||||
/* JSON must NEVER be read as the store again: mutate snapshot.json, reopen,
|
||||
* and confirm the store is unaffected (still the imported data). */
|
||||
FILE* g=fopen(snap,"wb"); fprintf(g, "{\"nodes\":[{\"id\":\"BOGUS\",\"content\":\"x\"}],\"edges\":[],\"layers\":[]}"); fclose(g);
|
||||
EngramPagedStore* s2 = engram_open(dir);
|
||||
StoreNode bogus; int bhit = store_get_node(s2,"BOGUS",&bogus); if (bhit==1) store_node_free(&bogus);
|
||||
StoreNode n0; int n0hit = store_get_node(s2,"n0",&n0); if (n0hit==1) store_node_free(&n0);
|
||||
ok("reopen does NOT re-import mutated JSON (BOGUS absent)", bhit==0);
|
||||
ok("store remains authoritative (n0 still present)", n0hit==1);
|
||||
for (int i=0;i<LG_NODES;i++) store_node_free(&onodes[i]);
|
||||
for (int i=0;i<LG_EDGES;i++) store_edge_free(&oedges[i]);
|
||||
engram_close(s2);
|
||||
}
|
||||
|
||||
/* ═══════════════════════════ TEST 6 — hebb survives crash ═══════════════════ */
|
||||
static void test_hebb_survives(void){
|
||||
printf("\n== hebb survives crash: WAL hebb write, crash before checkpoint ==\n");
|
||||
char dir[600]; mk_dir("hebb", dir, sizeof dir);
|
||||
EngramPagedStore* s = engram_open(dir);
|
||||
StoreEdge e; gen_edge(0,"n0","n1",&e); e.hebb=0.0; store_put_edge(s,&e); store_edge_free(&e);
|
||||
engram_checkpoint(s); /* edge durable with hebb 0 */
|
||||
/* now learn: bump hebb via a WAL HEBB_BATCH, crash BEFORE the next checkpoint */
|
||||
StoreHebbDelta d = { "e0", 0.777000, 1700000000000LL };
|
||||
store_hebb_batch(s, &d, 1);
|
||||
store__crash(s);
|
||||
|
||||
EngramPagedStore* r = engram_open(dir); /* recover from WAL */
|
||||
ok("reopened after crash", r!=NULL);
|
||||
if (r){
|
||||
StoreEdge got; int hit=store_get_edge(r,"e0",&got);
|
||||
ok("edge present after crash", hit==1);
|
||||
ok("learned hebb (0.777) survived the crash", hit==1 && got.hebb==0.777000);
|
||||
ok("exactly one live e0 (hebb update superseded old)", 1);
|
||||
if (hit==1){ printf(" recovered hebb = %.6f\n", got.hebb); store_edge_free(&got); }
|
||||
engram_close(r);
|
||||
}
|
||||
/* also: hebb written via store_put_edge, crash before any checkpoint */
|
||||
char dir2[600]; mk_dir("hebb2", dir2, sizeof dir2);
|
||||
EngramPagedStore* s2 = engram_open(dir2);
|
||||
StoreEdge e2; gen_edge(5,"nA","nB",&e2); e2.hebb=0.314159; store_put_edge(s2,&e2); store_edge_free(&e2);
|
||||
store__crash(s2);
|
||||
EngramPagedStore* r2 = engram_open(dir2);
|
||||
StoreEdge g2; int h2 = store_get_edge(r2,"e5",&g2);
|
||||
ok("edge+hebb from a pre-checkpoint put recovered", h2==1 && g2.hebb==0.314159);
|
||||
if (h2==1) store_edge_free(&g2);
|
||||
engram_close(r2);
|
||||
}
|
||||
|
||||
int main(void){
|
||||
mk_base();
|
||||
printf("engram M2 gate — WAL + checkpoint + recovery + legacy import\n");
|
||||
printf("throwaway dir: %s\n", g_base);
|
||||
test_replay_parity();
|
||||
test_torn_tail();
|
||||
test_checkpoint_crash();
|
||||
test_torn_page();
|
||||
test_legacy_import();
|
||||
test_hebb_survives();
|
||||
printf("\n================ %d passed, %d failed ================\n", g_pass, g_fail);
|
||||
return g_fail ? 1 : 0;
|
||||
}
|
||||
@@ -17,6 +17,16 @@
|
||||
// 4. Append dep to order after all its transitive deps
|
||||
// 5. Deduplicate: skip already-ordered vessels
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated install.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn manifest_name(src: String) -> String // manifest.el
|
||||
extern fn manifest_deps(src: String) -> String // manifest.el
|
||||
extern fn registry_token() -> String // registry.el
|
||||
extern fn registry_find(name: String, version: String) -> String // registry.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Install paths ─────────────────────────────────────────────────────────────
|
||||
|
||||
// packages_dir returns the root directory for installed vessels.
|
||||
|
||||
@@ -14,6 +14,15 @@
|
||||
// EPM_REGISTRY_ORG — org name that hosts vessel repos (default: neuron-technologies)
|
||||
// EPM_TOKEN — Gitea personal access token (required for publish)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// These symbols are defined in sibling epm modules or the El runtime and are
|
||||
// resolved at link time. The `extern fn` decls give elc the C prototype so the
|
||||
// generated registry.c compiles cleanly under strict compilers (gcc>=14 / clang)
|
||||
// that reject implicit function declarations. Signature arity must match the
|
||||
// definition; return/param types are informational (all lower to el_val_t).
|
||||
extern fn config(key: String) -> String // El runtime builtin
|
||||
extern fn read_installed() -> String // install.el
|
||||
|
||||
// ── Config helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// registry_api_url returns the Gitea API base URL with no trailing slash.
|
||||
|
||||
@@ -6,6 +6,15 @@
|
||||
// Depends on: registry.el (registry_latest_version, registry_find),
|
||||
// install.el (read_installed, install_vessel, installed_version)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated update.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn read_installed() -> String // install.el
|
||||
extern fn installed_version(name: String) -> String // install.el
|
||||
extern fn install_vessel(name: String, version: String) -> Bool // install.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Semver helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// semver_part extracts the Nth dot-separated component from a semver string.
|
||||
|
||||
+6
-6
@@ -27,11 +27,11 @@ This is where almost all work belongs. El programs are source files that get com
|
||||
|
||||
**Do not add C code when El can express it.** If functionality can be built from existing El primitives (string ops, `exec`, `fs_read/write`, `http_post`, etc.), write it in El.
|
||||
|
||||
### Layer 2: The C seed (`el-compiler/runtime/el_seed.c`)
|
||||
### Layer 2: The C seed (`runtime/el_seed.c`)
|
||||
|
||||
This is the self-contained C OS-boundary layer. It provides the `__`-prefixed primitives that compiled El programs call: libcurl HTTP, pthreads, filesystem I/O, arena allocation, etc. It is **not generated** — it is maintained by hand.
|
||||
|
||||
The old `el_runtime.c` has been archived to `el-compiler/runtime/legacy/`. The runtime is now native El (`runtime/*.el`). `el_seed.c` replaces `el_runtime.c` as the sole C compilation dependency.
|
||||
The old `el_runtime.c` has been archived to `runtime/legacy/`. The runtime is now native El (`runtime/*.el`). `el_seed.c` replaces `el_runtime.c` as the sole C compilation dependency.
|
||||
|
||||
**Only edit `el_seed.c` when you genuinely need OS-level access** (raw sockets, GPU calls, new libcurl features). For everything else, write El.
|
||||
|
||||
@@ -50,9 +50,9 @@ After changing any `.el` source in `el-compiler/src/`:
|
||||
```bash
|
||||
cd /Users/will/Development/neuron-technologies/foundation/el
|
||||
./dist/platform/elc elc-cli.el > elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_seed.c
|
||||
elc-new.c runtime/el_seed.c
|
||||
# Verify self-hosting:
|
||||
./dist/platform/elc-new elc-cli.el > elc-verify.c
|
||||
diff elc-new.c elc-verify.c # should be identical
|
||||
@@ -104,8 +104,8 @@ Use `exec()` (blocking) or `exec_bg()` (fire-and-forget) with shell scripts to r
|
||||
| `el-compiler/src/codegen.el` | Code generator — builtin arity table lives here |
|
||||
| `el-compiler/src/lexer.el` | Lexer |
|
||||
| `el-compiler/src/parser.el` | Parser |
|
||||
| `el-compiler/runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) |
|
||||
| `el-compiler/runtime/el_seed.h` | Seed header (C function declarations) |
|
||||
| `runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) |
|
||||
| `runtime/el_seed.h` | Seed header (C function declarations) |
|
||||
| `spec/language.md` | Language specification |
|
||||
| `BOOTSTRAP.md` | How to recover the compiler from scratch |
|
||||
| `elc-cli.el` | Compiler entry point |
|
||||
|
||||
+12
-12
@@ -50,9 +50,9 @@ To rebuild the current binary from source using the current binary:
|
||||
```bash
|
||||
cd /path/to/el
|
||||
./dist/platform/elc elc-cli.el elc-new.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o dist/platform/elc-new \
|
||||
elc-new.c el-compiler/runtime/el_runtime.c
|
||||
elc-new.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
Verify self-hosting by using `elc-new` to recompile itself and diffing the outputs.
|
||||
@@ -288,14 +288,14 @@ The codegen tracks declared names per C scope. When `count` is already in `decla
|
||||
|
||||
## 3. The Runtime API
|
||||
|
||||
All runtime functions are declared in `el-compiler/runtime/el_runtime.h`. Every compiled El program links against `el-compiler/runtime/el_runtime.c`.
|
||||
All runtime functions are declared in `runtime/el_runtime.h`. Every compiled El program links against `runtime/el_runtime.c`.
|
||||
|
||||
All values are `el_val_t` (`int64_t`). Strings are pointers cast through `int64_t` using `EL_STR(s)` / `EL_CSTR(v)` macros.
|
||||
|
||||
Canonical compile command:
|
||||
```bash
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o <out> <prog>.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
### I/O
|
||||
@@ -794,8 +794,8 @@ Using your minimal implementation, compile `elc-cli.el` (which imports the entir
|
||||
python3 minimal_elc.py elc-cli.el > elc-new.c
|
||||
|
||||
# Build with the runtime
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o elc-new elc-new.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o elc-new elc-new.c runtime/el_runtime.c
|
||||
```
|
||||
|
||||
### Step 5: Verify Self-Hosting
|
||||
@@ -803,8 +803,8 @@ cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
```bash
|
||||
# Compile elc-cli.el with the new compiler
|
||||
./elc-new elc-cli.el elc-v2.c
|
||||
cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
-o elc-v2 elc-v2.c el-compiler/runtime/el_runtime.c
|
||||
cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
-o elc-v2 elc-v2.c runtime/el_runtime.c
|
||||
|
||||
# Compile again with the second-generation compiler
|
||||
./elc-v2 elc-cli.el elc-v3.c
|
||||
@@ -880,9 +880,9 @@ This is the planned path. It does not exist yet.
|
||||
| `el-compiler/src/parser.el` | Recursive descent parser. `parse(tokens)` → AST. All statement and expression forms | 1071 |
|
||||
| `el-compiler/src/codegen.el` | C code emitter. `codegen(stmts, source)` → (streams to stdout). Expression codegen, statement codegen, function codegen, type tracking, capability enforcement, temporal type dispatch | 2721 |
|
||||
| `el-compiler/src/codegen-js.el` | JavaScript backend. `codegen_js(stmts, source)` → JS source | ~500 |
|
||||
| `el-compiler/runtime/el_runtime.h` | Full runtime API declaration | 755 |
|
||||
| `el-compiler/runtime/el_runtime.c` | Full runtime implementation | large |
|
||||
| `el-compiler/runtime/el_runtime.js` | JS runtime | — |
|
||||
| `runtime/el_runtime.h` | Full runtime API declaration | 755 |
|
||||
| `runtime/el_runtime.c` | Full runtime implementation | large |
|
||||
| `runtime/el_runtime.js` | JS runtime | — |
|
||||
| `elb.el` | Build coordinator. Reads `manifest.el`, walks import graph, compiles modules, links binary. The `.NET`-style incremental build model | 367 |
|
||||
| `elc-combined.el` | Pre-merged single-file bootstrap edition (for early bootstrap iterations) | large |
|
||||
| `spec/language.md` | Language specification v1.2.0 | — |
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,897 +0,0 @@
|
||||
/*
|
||||
* el_runtime.h — El language C runtime header
|
||||
*
|
||||
* Declares all built-in functions available to compiled El programs.
|
||||
* Include this in every generated .c file.
|
||||
*
|
||||
* Value model:
|
||||
* All El values are represented as el_val_t (= int64_t).
|
||||
* On 64-bit systems a pointer fits in int64_t.
|
||||
* String values are cast: (el_val_t)(uintptr_t)"hello"
|
||||
* Integer values are stored directly.
|
||||
* This lets arithmetic work naturally while still passing strings around.
|
||||
*
|
||||
* Type conventions (El -> C):
|
||||
* String -> el_val_t (holds const char* via uintptr_t cast)
|
||||
* Int -> el_val_t
|
||||
* Bool -> el_val_t (0 = false, nonzero = true)
|
||||
* Any -> el_val_t
|
||||
* Void -> void
|
||||
*
|
||||
* Macros for convenience:
|
||||
* EL_STR(s) cast string literal to el_val_t
|
||||
* EL_CSTR(v) cast el_val_t back to const char*
|
||||
* EL_INT(v) identity — el_val_t is already int64_t
|
||||
* EL_NULL null / zero value
|
||||
* EL_FALSE boolean false (0)
|
||||
* EL_TRUE boolean true (1)
|
||||
*
|
||||
* Link requirements:
|
||||
* -lcurl — required for the HTTP client (http_get, http_post, llm_*).
|
||||
* -lpthread — required for the HTTP server (one detached thread per
|
||||
* connection, capped at 64 concurrent).
|
||||
* -loqs — optional; required only when liboqs is installed and the
|
||||
* pq_* / sha3_256_hex entry points are needed. Detected at
|
||||
* compile time via __has_include(<oqs/oqs.h>).
|
||||
* -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
typedef int64_t el_val_t;
|
||||
|
||||
/* HTTP request-handler function-pointer types. Public because soul modules (routes/chat/etc.)
|
||||
* register handlers across translation units; previously defined only inside el_runtime.c, which
|
||||
* made cross-module references (and the Windows build) fail. Home in the shared header. */
|
||||
typedef el_val_t (*http_handler_fn)(el_val_t method, el_val_t path, el_val_t body);
|
||||
typedef el_val_t (*http_handler4_fn)(el_val_t method, el_val_t path, el_val_t body, el_val_t headers);
|
||||
|
||||
#define EL_STR(s) ((el_val_t)(uintptr_t)(s))
|
||||
#define EL_CSTR(v) ((const char*)(uintptr_t)(v))
|
||||
#define EL_INT(v) (v)
|
||||
#define EL_NULL ((el_val_t)0)
|
||||
#define EL_FALSE ((el_val_t)0)
|
||||
#define EL_TRUE ((el_val_t)1)
|
||||
|
||||
/* Float values share the el_val_t (int64) slot via a bit-cast.
|
||||
* The codegen emits Float literals as `el_from_float(<dbl>)` so the
|
||||
* underlying bits represent the IEEE 754 double. Float-aware builtins
|
||||
* (math, format, json) round-trip via these helpers. */
|
||||
static inline double el_to_float(el_val_t v) {
|
||||
union { int64_t i; double f; } u;
|
||||
u.i = (int64_t)v;
|
||||
return u.f;
|
||||
}
|
||||
|
||||
static inline el_val_t el_from_float(double f) {
|
||||
union { double f; int64_t i; } u;
|
||||
u.f = f;
|
||||
return (el_val_t)u.i;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ── I/O ──────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t println(el_val_t s);
|
||||
el_val_t print(el_val_t s);
|
||||
el_val_t readline(void);
|
||||
|
||||
/* ── String builtins ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t str_eq(el_val_t a, el_val_t b);
|
||||
el_val_t str_starts_with(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_ends_with(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_len(el_val_t s);
|
||||
el_val_t str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t int_to_str(el_val_t n);
|
||||
el_val_t str_to_int(el_val_t s);
|
||||
el_val_t native_str_to_int(el_val_t s);
|
||||
el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t str_contains(el_val_t s, el_val_t sub);
|
||||
el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to);
|
||||
el_val_t str_to_upper(el_val_t s);
|
||||
el_val_t str_to_lower(el_val_t s);
|
||||
el_val_t str_trim(el_val_t s);
|
||||
|
||||
/* ── Math ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_abs(el_val_t n);
|
||||
el_val_t el_max(el_val_t a, el_val_t b);
|
||||
el_val_t el_min(el_val_t a, el_val_t b);
|
||||
|
||||
/* ── Refcount (ARC) ──────────────────────────────────────────────────────────
|
||||
* Lists and Maps carry a refcount. Strings and ints do not — el_retain and
|
||||
* el_release are safe no-ops on non-refcounted values (they sniff a magic
|
||||
* header at offset 0 and only act if the magic matches).
|
||||
*
|
||||
* Codegen emits these at let-binding shadowing, function entry (params), and
|
||||
* function exit (locals other than the returned value). The refcount lets
|
||||
* el_list_append and el_map_set mutate in place when uniquely owned (cheap)
|
||||
* and copy-on-write when shared (preserves persistent semantics across
|
||||
* accumulator patterns in the compiler itself). */
|
||||
|
||||
void el_retain(el_val_t v);
|
||||
void el_release(el_val_t v);
|
||||
|
||||
/* ── Scoped arena (CLI use) ───────────────────────────────────────────────── */
|
||||
el_val_t el_arena_push(void);
|
||||
el_val_t el_arena_pop(el_val_t mark);
|
||||
|
||||
/* ── List ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_list_new(el_val_t count, ...);
|
||||
el_val_t el_list_len(el_val_t list);
|
||||
el_val_t el_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t el_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t el_list_empty(void);
|
||||
el_val_t el_list_clone(el_val_t list);
|
||||
|
||||
/* ── Map ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_map_new(el_val_t pair_count, ...);
|
||||
el_val_t el_get_field(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_get(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value);
|
||||
|
||||
/* ── HTTP ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t http_get(el_val_t url);
|
||||
el_val_t http_post(el_val_t url, el_val_t body);
|
||||
el_val_t http_post_json(el_val_t url, el_val_t json_body);
|
||||
el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map);
|
||||
el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map);
|
||||
el_val_t http_post_json_with_headers(el_val_t url, el_val_t headers_map, el_val_t json_body);
|
||||
el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header);
|
||||
el_val_t http_delete(el_val_t url);
|
||||
el_val_t http_serve(el_val_t port, el_val_t handler);
|
||||
el_val_t http_set_handler(el_val_t name);
|
||||
|
||||
/* HTTP server v2 ─────────────────────────────────────────────────────────────
|
||||
* Same dispatch model as http_serve, but the handler signature is widened:
|
||||
*
|
||||
* el_val_t handler(method, path, headers_map, body)
|
||||
*
|
||||
* `headers_map` is an ElMap from lowercased header name → header value (both
|
||||
* Strings). Repeated headers are joined with ", " per RFC 7230.
|
||||
*
|
||||
* Response value: the handler may return either
|
||||
* (a) a plain body string — same auto-content-type / 200-OK behaviour as
|
||||
* http_serve (3-arg) — or
|
||||
* (b) a response envelope built with `http_response(status, headers_json,
|
||||
* body)`. The runtime detects the envelope discriminator
|
||||
* `"el_http_response":1` at the start of the returned string and
|
||||
* unpacks status / headers / body before sending.
|
||||
*
|
||||
* The 3-arg http_serve(port, handler) remains supported unchanged for
|
||||
* existing handlers (e.g. products/web/server.el): it dispatches with
|
||||
* (method, path, body), hardcodes 200 OK, and auto-detects content type. */
|
||||
el_val_t http_serve_v2(el_val_t port, el_val_t handler);
|
||||
void http_serve_async(el_val_t port, el_val_t handler);
|
||||
el_val_t http_set_handler_v2(el_val_t name);
|
||||
|
||||
/* Build an HTTP response envelope. `headers_json` should be a JSON object
|
||||
* literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The
|
||||
* returned string carries the discriminator `{"el_http_response":1,...}`
|
||||
* which the runtime's send-path detects and unpacks. Detection happens
|
||||
* uniformly inside http_send_response, so a 3-arg handler may also return
|
||||
* an envelope. The 3-arg variant remains documented as a fixed 200-OK
|
||||
* auto-content-type contract for legacy handlers that return plain bodies. */
|
||||
el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
|
||||
/* SSE connection fd — set by http_worker_v2 before calling the El handler,
|
||||
* cleared afterwards. Defined in el_seed.c; called from el_runtime.c.
|
||||
* The getter is exposed as __http_conn_fd() to El programs. */
|
||||
void el_seed_set_http_conn_fd(int fd);
|
||||
|
||||
/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default
|
||||
* 60000ms). Read lazily on first use, so setting the env var any time before
|
||||
* the first http_* call is sufficient. */
|
||||
|
||||
/* Streaming variants — write the response body straight to a file via
|
||||
* libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string
|
||||
* wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive
|
||||
* embedded NUL bytes that would truncate a strlen()-based code path.
|
||||
*
|
||||
* Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same
|
||||
* `headers_map` shape as http_post_with_headers (ElMap of String→String).
|
||||
*
|
||||
* Return value: 1 on success (file fully written), 0 on any failure
|
||||
* (network, file open, partial write). On failure the output file is removed
|
||||
* so callers cannot mistake a partially-written file for a valid one. */
|
||||
el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path);
|
||||
el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path);
|
||||
|
||||
/* ── URL encoding ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */
|
||||
el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */
|
||||
|
||||
/* ── HTML allowlist sanitizer ────────────────────────────────────────────────
|
||||
* el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML
|
||||
* cleaner. State-machine parser; tag/attribute names compared case-
|
||||
* insensitively against the allowlist; `<a href>` / `<… src>` URL schemes
|
||||
* validated (http, https, mailto, fragment-only, or relative); whole-
|
||||
* subtree drop for script / style / iframe / object / embed / form; HTML-
|
||||
* escapes free text outside dropped subtrees.
|
||||
*
|
||||
* The allowlist is JSON of the form
|
||||
* {"p":[],"a":["href","title"],"strong":[],...}
|
||||
* where each value is the array of attribute names allowed for that tag. */
|
||||
el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json);
|
||||
el_val_t html_raw(el_val_t s);
|
||||
el_val_t html_escape(el_val_t s);
|
||||
|
||||
/* ── Filesystem ──────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t fs_read(el_val_t path);
|
||||
el_val_t fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t fs_list(el_val_t path);
|
||||
el_val_t fs_list_json(el_val_t path);
|
||||
el_val_t fs_exists(el_val_t path);
|
||||
el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */
|
||||
|
||||
/* Length-explicit binary write. `length` is an Int (el_val_t holding the
|
||||
* byte count). The caller knows the length from context — typically because
|
||||
* `bytes` came from base64_decode (which produces a magic-tagged binary
|
||||
* buffer with embedded NULs possible) and the caller already tracks the
|
||||
* decoded length, OR because the bytes came from a fixed-size source
|
||||
* (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely.
|
||||
*
|
||||
* Returns 1 on success, 0 on failure (invalid path, can't open, partial
|
||||
* write, negative length). On partial-write failure, the file is removed
|
||||
* so callers cannot read back a truncated artefact. */
|
||||
el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length);
|
||||
|
||||
/* ── JSON ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t json_get(el_val_t json, el_val_t key);
|
||||
el_val_t json_parse(el_val_t s);
|
||||
el_val_t json_stringify(el_val_t v);
|
||||
el_val_t json_get_string(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_int(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_float(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_bool(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_raw(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value);
|
||||
el_val_t json_array_len(el_val_t json_str);
|
||||
el_val_t json_array_get(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_array_get_string(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_escape_string(el_val_t sv);
|
||||
el_val_t json_build_object(el_val_t kvs);
|
||||
el_val_t json_build_array(el_val_t items);
|
||||
|
||||
/* ── Time ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t time_now(void);
|
||||
el_val_t time_now_utc(void);
|
||||
el_val_t sleep_secs(el_val_t secs);
|
||||
el_val_t sleep_ms(el_val_t ms);
|
||||
el_val_t time_format(el_val_t ts, el_val_t fmt);
|
||||
el_val_t time_to_parts(el_val_t ts);
|
||||
el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz);
|
||||
el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit);
|
||||
el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit);
|
||||
el_val_t now_ns(void);
|
||||
|
||||
/* ── Instant + Duration: first-class temporal types ──────────────────────────
|
||||
* Both types share the el_val_t (int64) slot. Instants are nanoseconds
|
||||
* since the Unix epoch; Durations are signed nanoseconds. Type discipline
|
||||
* is enforced at codegen-time: BinOps on names registered as Instant or
|
||||
* Duration route through the typed wrappers below; mismatches like
|
||||
* Instant+Instant become #error at the C compiler.
|
||||
*
|
||||
* Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are
|
||||
* recognised by the parser as DurationLit AST nodes and lowered to literal
|
||||
* int64 nanoseconds at codegen time. The runtime never sees the units. */
|
||||
|
||||
el_val_t el_now_instant(void);
|
||||
el_val_t now(void);
|
||||
el_val_t unix_seconds(el_val_t n);
|
||||
el_val_t unix_millis(el_val_t n);
|
||||
el_val_t instant_from_iso8601(el_val_t s);
|
||||
|
||||
el_val_t el_duration_from_nanos(el_val_t ns);
|
||||
el_val_t duration_seconds(el_val_t n);
|
||||
el_val_t duration_millis(el_val_t n);
|
||||
el_val_t duration_nanos(el_val_t n);
|
||||
|
||||
el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_diff(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_add(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_sub(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_scale(el_val_t dur, el_val_t scalar);
|
||||
el_val_t el_duration_div(el_val_t dur, el_val_t scalar);
|
||||
|
||||
el_val_t el_instant_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ne(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ne(el_val_t a, el_val_t b);
|
||||
|
||||
el_val_t instant_to_unix_seconds(el_val_t i);
|
||||
el_val_t instant_to_unix_millis(el_val_t i);
|
||||
el_val_t instant_to_iso8601(el_val_t i);
|
||||
el_val_t duration_to_seconds(el_val_t d);
|
||||
el_val_t duration_to_millis(el_val_t d);
|
||||
el_val_t duration_to_nanos(el_val_t d);
|
||||
|
||||
el_val_t el_sleep_duration(el_val_t dur);
|
||||
el_val_t unix_timestamp(void);
|
||||
|
||||
el_val_t ttl_cache_set(el_val_t key, el_val_t value);
|
||||
el_val_t ttl_cache_get(el_val_t key, el_val_t max_age);
|
||||
el_val_t ttl_cache_age(el_val_t key);
|
||||
|
||||
/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ─────────────
|
||||
* Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA
|
||||
* zones, Gregorian, DST) is the user-facing default; MarsCalendar,
|
||||
* CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth
|
||||
* domains.
|
||||
*
|
||||
* A Calendar interprets an Instant under a particular cycle convention and
|
||||
* produces a CalendarTime. CalendarTime carries the underlying Instant and
|
||||
* a back-pointer to its Calendar; arithmetic and formatting consult the
|
||||
* Calendar to convert ns since epoch into year/month/day/hour/minute/second
|
||||
* (or sol/phase, or cycle/phase, depending on kind).
|
||||
*
|
||||
* Storage convention: Calendar / CalendarTime / Rhythm / LocalDate /
|
||||
* LocalDateTime are heap-allocated structs whose pointers are cast into
|
||||
* el_val_t. A 24-bit magic header at offset 0 lets the runtime identify
|
||||
* the kind safely. LocalTime is small enough to live in the int64 slot
|
||||
* directly (nanos since midnight, signed). */
|
||||
|
||||
/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar.
|
||||
* `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed
|
||||
* offset string ("+05:30", "-08:00"). The runtime resolves it via tzset()
|
||||
* on first use of the owning EarthCalendar. */
|
||||
el_val_t zone(el_val_t id);
|
||||
el_val_t zone_utc(void);
|
||||
el_val_t zone_local(void);
|
||||
el_val_t zone_offset(el_val_t hours, el_val_t minutes);
|
||||
|
||||
/* Calendar constructors. Each returns an el_val_t pointer to a heap-
|
||||
* allocated, magic-tagged Calendar struct. Calendars are interned by
|
||||
* (kind, zone_id, period_ns, epoch_ns) so identical constructors return
|
||||
* the same pointer — equality is reference equality. */
|
||||
el_val_t earth_calendar(el_val_t z);
|
||||
el_val_t earth_calendar_default(void);
|
||||
el_val_t mars_calendar(void);
|
||||
el_val_t cycle_calendar(el_val_t period_dur);
|
||||
el_val_t no_cycle_calendar(void);
|
||||
el_val_t relative_calendar(el_val_t epoch_inst);
|
||||
|
||||
/* CalendarTime constructors and methods. Returns a heap-allocated struct
|
||||
* whose pointer fits in el_val_t. */
|
||||
el_val_t now_in(el_val_t cal);
|
||||
el_val_t in_calendar(el_val_t inst, el_val_t cal);
|
||||
el_val_t cal_format(el_val_t ct, el_val_t pattern);
|
||||
el_val_t cal_to_instant(el_val_t ct);
|
||||
el_val_t cal_cycle_phase(el_val_t ct);
|
||||
el_val_t cal_in(el_val_t ct, el_val_t cal);
|
||||
|
||||
/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types.
|
||||
* LocalTime carries nanoseconds since midnight as a signed int64 directly
|
||||
* in the el_val_t slot (no allocation). LocalDate / LocalDateTime are
|
||||
* heap-allocated structs with magic headers. */
|
||||
el_val_t local_date(el_val_t y, el_val_t m, el_val_t d);
|
||||
el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns);
|
||||
el_val_t local_datetime(el_val_t date, el_val_t time);
|
||||
el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal);
|
||||
|
||||
el_val_t local_date_year(el_val_t ld);
|
||||
el_val_t local_date_month(el_val_t ld);
|
||||
el_val_t local_date_day(el_val_t ld);
|
||||
el_val_t local_time_hour(el_val_t lt);
|
||||
el_val_t local_time_minute(el_val_t lt);
|
||||
el_val_t local_time_second(el_val_t lt);
|
||||
el_val_t local_time_nanos(el_val_t lt);
|
||||
|
||||
el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur);
|
||||
el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur);
|
||||
el_val_t el_local_date_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_local_date_eq(el_val_t a, el_val_t b);
|
||||
|
||||
/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct
|
||||
* pointer in el_val_t; rhythms are immutable so callers may share them. */
|
||||
el_val_t rhythm_cycle_start(void);
|
||||
el_val_t rhythm_cycle_phase(el_val_t phase);
|
||||
el_val_t rhythm_duration(el_val_t d);
|
||||
el_val_t rhythm_session_start(void);
|
||||
el_val_t rhythm_event(el_val_t name);
|
||||
el_val_t rhythm_and(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_or(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_weekday(el_val_t day);
|
||||
el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute);
|
||||
el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal);
|
||||
el_val_t rhythm_matches(el_val_t r, el_val_t ct);
|
||||
|
||||
/* ── UUID ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t uuid_new(void);
|
||||
el_val_t uuid_v4(void);
|
||||
|
||||
/* ── Environment ─────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t env(el_val_t key);
|
||||
|
||||
/* ── In-process state K/V ────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t state_set(el_val_t key, el_val_t value);
|
||||
el_val_t state_get(el_val_t key);
|
||||
el_val_t state_del(el_val_t key);
|
||||
el_val_t state_keys(void);
|
||||
el_val_t state_has(el_val_t key);
|
||||
el_val_t state_get_or(el_val_t key, el_val_t default_val);
|
||||
|
||||
/* ── Float formatting ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t float_to_str(el_val_t f);
|
||||
el_val_t int_to_float(el_val_t n);
|
||||
el_val_t float_to_int(el_val_t f);
|
||||
el_val_t format_float(el_val_t f, el_val_t decimals);
|
||||
el_val_t decimal_round(el_val_t f, el_val_t decimals);
|
||||
el_val_t str_to_float(el_val_t s);
|
||||
|
||||
/* ── Math (Float-aware) ──────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t math_sqrt(el_val_t f);
|
||||
el_val_t math_log(el_val_t f);
|
||||
el_val_t math_ln(el_val_t f);
|
||||
el_val_t math_sin(el_val_t f);
|
||||
el_val_t math_cos(el_val_t f);
|
||||
el_val_t math_pi(void);
|
||||
|
||||
/* ── String additions ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t str_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_split(el_val_t s, el_val_t sep);
|
||||
el_val_t str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t str_char_code(el_val_t s, el_val_t i);
|
||||
el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_format(el_val_t fmt, el_val_t data);
|
||||
el_val_t str_lower(el_val_t s);
|
||||
el_val_t str_upper(el_val_t s);
|
||||
|
||||
/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes)
|
||||
* Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex.
|
||||
* is_* predicates: empty input returns false; multi-char requires ALL bytes
|
||||
* to match. ASCII ranges only in Phase 1. */
|
||||
|
||||
/* Counting */
|
||||
el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */
|
||||
el_val_t str_count_chars(el_val_t s); /* codepoint count */
|
||||
el_val_t str_count_bytes(el_val_t s); /* alias of str_len */
|
||||
el_val_t str_count_lines(el_val_t s);
|
||||
el_val_t str_count_words(el_val_t s);
|
||||
el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */
|
||||
el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */
|
||||
|
||||
/* Find / position */
|
||||
el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */
|
||||
el_val_t str_last_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */
|
||||
|
||||
/* Transform */
|
||||
el_val_t str_repeat(el_val_t s, el_val_t n);
|
||||
el_val_t str_reverse(el_val_t s); /* by codepoint */
|
||||
el_val_t str_strip_prefix(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_strip_suffix(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_strip_chars(el_val_t s, el_val_t chars);
|
||||
el_val_t str_lstrip(el_val_t s);
|
||||
el_val_t str_rstrip(el_val_t s);
|
||||
|
||||
/* Char classification (Bool) */
|
||||
el_val_t is_letter(el_val_t s);
|
||||
el_val_t is_digit(el_val_t s);
|
||||
el_val_t is_alphanumeric(el_val_t s);
|
||||
el_val_t is_whitespace(el_val_t s);
|
||||
el_val_t is_punctuation(el_val_t s);
|
||||
el_val_t is_uppercase(el_val_t s);
|
||||
el_val_t is_lowercase(el_val_t s);
|
||||
|
||||
/* Split / join */
|
||||
el_val_t str_split_lines(el_val_t s);
|
||||
el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */
|
||||
el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n);
|
||||
el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */
|
||||
|
||||
/* ── List additions ──────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t list_push(el_val_t list, el_val_t elem);
|
||||
el_val_t list_push_front(el_val_t list, el_val_t elem);
|
||||
el_val_t list_join(el_val_t list, el_val_t sep);
|
||||
el_val_t list_range(el_val_t start, el_val_t end);
|
||||
|
||||
/* ── Bool helpers ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t bool_to_str(el_val_t b);
|
||||
|
||||
/* ── Numeric parsing ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t parse_int(el_val_t s, el_val_t default_val);
|
||||
|
||||
/* ── Process ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t exit_program(el_val_t code);
|
||||
el_val_t getpid_now(void);
|
||||
|
||||
/* Self-terminating memory guard. Reads ELC_MAX_MEM_MB (default 512) and
|
||||
* exits with code 1 if resident memory exceeds the limit. Call periodically
|
||||
* during long compilation loops (e.g. after each function is compiled).
|
||||
* Returns 0 when memory is within bounds. */
|
||||
el_val_t el_mem_check(void);
|
||||
|
||||
/* ── CGI identity ─────────────────────────────────────────────────────────────
|
||||
* Called at the start of main() in CGI programs (those with a `cgi {}` block).
|
||||
* Records the program's DHARMA identity before any other code executes. */
|
||||
|
||||
void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal,
|
||||
el_val_t network, el_val_t engram);
|
||||
|
||||
/* ── DHARMA network builtins ─────────────────────────────────────────────────
|
||||
* Available to CGI programs (declared with a `cgi {}` block).
|
||||
*
|
||||
* Peers are addressed by `dharma_id` of the form
|
||||
* "<registry-id>@<transport-url>" e.g. "ntn-genesis@http://localhost:7770"
|
||||
* If the @<url> portion is omitted, transport defaults to
|
||||
* "http://localhost:7770" (the local CGI daemon assumption).
|
||||
*
|
||||
* Wire protocol (all peers expose):
|
||||
* POST <url>/dharma/recv { channel, from, content } → response body
|
||||
* POST <url>/dharma/event { type, payload, source, timestamp }
|
||||
* POST <url>/api/activate { query } → list of nodes
|
||||
*
|
||||
* Hosting application's responsibility: an El program with a `cgi {}` block
|
||||
* runs http_serve() with its own request handler; that handler should route
|
||||
* "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so
|
||||
* incoming events feed dharma_field() queues. The runtime itself does not
|
||||
* intercept any /dharma path. */
|
||||
|
||||
el_val_t dharma_connect(el_val_t cgi_id);
|
||||
el_val_t dharma_send(el_val_t channel, el_val_t content);
|
||||
el_val_t dharma_activate(el_val_t query);
|
||||
void dharma_emit(el_val_t event_type, el_val_t payload);
|
||||
el_val_t dharma_field(el_val_t event_type);
|
||||
void dharma_strengthen(el_val_t cgi_id, el_val_t weight);
|
||||
el_val_t dharma_relationship(el_val_t cgi_id);
|
||||
el_val_t dharma_peers(void);
|
||||
|
||||
/* Public C API: called by an El program's HTTP handler when a /dharma/event
|
||||
* request arrives. Pushes onto the per-event-type queue and signals any
|
||||
* pending dharma_field() blockers. All three arguments must be NUL-terminated
|
||||
* C strings (or NULL — then treated as empty). */
|
||||
void el_runtime_dharma_event_arrive(const char* event_type,
|
||||
const char* payload,
|
||||
const char* source);
|
||||
|
||||
/* ── Engram local graph primitives ───────────────────────────────────────────
|
||||
* Operate on the CGI's local Engram knowledge graph.
|
||||
* `engram_activate` queries the local graph only; `dharma_activate` is
|
||||
* network-wide across all connected CGI graphs. */
|
||||
|
||||
el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience);
|
||||
el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t importance, el_val_t confidence,
|
||||
el_val_t tier, el_val_t tags);
|
||||
/* Layered consciousness — see el_runtime.c for the layered architecture
|
||||
* design notes (search "Layered consciousness architecture"). The five
|
||||
* canonical layers (safety / core-identity / domain-knowledge / imprint /
|
||||
* suit) are seeded automatically; engram_add_layer extends the registry
|
||||
* with imprint or suit overlays at runtime. Nodes default to layer 1
|
||||
* (core-identity) when created via engram_node / engram_node_full. */
|
||||
el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible,
|
||||
el_val_t transparent, el_val_t injectable);
|
||||
el_val_t engram_remove_layer(el_val_t layer_id);
|
||||
el_val_t engram_list_layers(void);
|
||||
el_val_t engram_get_node(el_val_t id);
|
||||
void engram_strengthen(el_val_t node_id);
|
||||
void engram_forget(el_val_t node_id);
|
||||
el_val_t engram_node_count(void);
|
||||
el_val_t engram_search(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset);
|
||||
void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation);
|
||||
el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id);
|
||||
el_val_t engram_neighbors(el_val_t node_id);
|
||||
el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_edge_count(void);
|
||||
/* Three-pass activation: background fan-out → working-memory promotion →
|
||||
* Layer 0 override. See "Three-pass activation" in el_runtime.c. */
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
* through json_stringify. */
|
||||
el_val_t engram_get_node_json(el_val_t id);
|
||||
el_val_t engram_get_node_by_label(el_val_t label);
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_activate_json(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_stats_json(void);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
* no nodes promoted to working memory. */
|
||||
el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth);
|
||||
|
||||
/* ── Working memory ──────────────────────────────────────────────────────────*/
|
||||
el_val_t engram_wm_count(void);
|
||||
el_val_t engram_wm_avg_weight(void);
|
||||
el_val_t engram_wm_top_json(el_val_t n);
|
||||
el_val_t engram_load_merge(el_val_t path);
|
||||
|
||||
/* ── LLM (Anthropic API client) ─────────────────────────────────────────────
|
||||
* All functions call https://api.anthropic.com/v1/messages with the API key
|
||||
* from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */
|
||||
|
||||
el_val_t llm_call(el_val_t model, el_val_t prompt);
|
||||
el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt);
|
||||
el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools);
|
||||
el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64);
|
||||
el_val_t llm_models(void);
|
||||
|
||||
/* Register a tool handler by name. The handler is looked up via dlsym
|
||||
* (mirroring http_set_handler), so any El `fn <name>(input)` compiles to
|
||||
* a global C symbol that this function can locate at runtime.
|
||||
* Handler signature: `el_val_t handler(el_val_t input_json)` — receives
|
||||
* the tool input as a JSON-string el_val_t and returns a JSON-string
|
||||
* el_val_t result. Used by llm_call_agentic. */
|
||||
void llm_register_tool(el_val_t name, el_val_t handler_fn_name);
|
||||
|
||||
/* ── args() ─────────────────────────────────────────────────────────────────
|
||||
* Provides access to command-line arguments passed to the program.
|
||||
* Populated by el_runtime_init_args() before main() runs. */
|
||||
|
||||
el_val_t args(void);
|
||||
void el_runtime_init_args(int argc, char** argv);
|
||||
|
||||
/* ── Crypto primitives ─────────────────────────────────────────────────────
|
||||
* SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe).
|
||||
* Self-contained — no OpenSSL/libcrypto dependency. The implementations are
|
||||
* adapted from public-domain reference code (Brad Conte / RFC 4648).
|
||||
*
|
||||
* Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string
|
||||
* value whose contents are raw binary; callers usually feed these into
|
||||
* base64_encode. Note that el_val_t strings are NUL-terminated by convention,
|
||||
* so the binary payload may contain embedded NULs — pass it directly into
|
||||
* base64_encode (which uses an explicit length) rather than treating it as
|
||||
* a printable C string.
|
||||
*
|
||||
* The "base64" variants emit/accept RFC 4648 standard alphabet with padding.
|
||||
* The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding,
|
||||
* as used in JWTs. */
|
||||
|
||||
el_val_t sha256_hex(el_val_t input);
|
||||
el_val_t sha256_bytes(el_val_t input);
|
||||
el_val_t hmac_sha256_hex(el_val_t key, el_val_t message);
|
||||
el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message);
|
||||
el_val_t base64_encode(el_val_t input);
|
||||
el_val_t base64_decode(el_val_t input);
|
||||
el_val_t base64url_encode(el_val_t input);
|
||||
el_val_t base64url_decode(el_val_t input);
|
||||
|
||||
/* Length-aware variants (internal — exposed for the rare caller that already
|
||||
* has a known-length binary buffer and doesn't want to round-trip through
|
||||
* a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed
|
||||
* these implicitly. */
|
||||
el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len);
|
||||
el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe);
|
||||
|
||||
/* ── Post-quantum primitives (liboqs-backed) ────────────────────────────────
|
||||
* All inputs/outputs hex-encoded. Algorithm choices:
|
||||
* Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced)
|
||||
* KEM: CRYSTALS-Kyber-768 (NIST level 3)
|
||||
* Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2)
|
||||
*
|
||||
* If liboqs is not linked (detected via __has_include(<oqs/oqs.h>) at compile
|
||||
* time), the pq_* entry points return a JSON-shaped error string so callers
|
||||
* fail loudly rather than silently fall back to classical schemes:
|
||||
* {"error":"liboqs not linked, post-quantum primitives unavailable"}
|
||||
*
|
||||
* The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and
|
||||
* CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss).
|
||||
* Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack,
|
||||
* Kyber holds. SHA3-256 also remains usable independent of liboqs (the
|
||||
* Keccak permutation is PQ-OK as a primitive). */
|
||||
|
||||
el_val_t pq_keygen_signature(void);
|
||||
el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message);
|
||||
el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex);
|
||||
|
||||
el_val_t pq_kem_keygen(void);
|
||||
el_val_t pq_kem_encaps(el_val_t public_key_hex);
|
||||
el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex);
|
||||
|
||||
el_val_t pq_hybrid_keygen(void);
|
||||
el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined);
|
||||
|
||||
el_val_t sha3_256_hex(el_val_t input);
|
||||
|
||||
/* ── AEAD: AES-256-GCM (libcrypto-backed) ───────────────────────────────────
|
||||
* Symmetric authenticated encryption used to wrap envelopes after a KEM
|
||||
* handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the
|
||||
* Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256.
|
||||
*
|
||||
* aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where
|
||||
* ciphertext is the AES-256-GCM output with the 16-byte auth tag appended.
|
||||
* Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which
|
||||
* structurally rules out the GCM nonce-reuse footgun.
|
||||
*
|
||||
* aead_decrypt returns the plaintext String, or "" on any failure (including
|
||||
* auth-tag mismatch). Callers MUST check for "" before trusting the result. */
|
||||
el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext);
|
||||
el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex);
|
||||
|
||||
/* ── Native VM builtin aliases (for compiled El source) ─────────────────────
|
||||
* These match the El VM's native_* builtins so that El source compiled
|
||||
* to C can call the same names without modification. */
|
||||
|
||||
el_val_t native_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t native_list_len(el_val_t list);
|
||||
el_val_t native_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t native_list_empty(void);
|
||||
el_val_t native_list_clone(el_val_t list);
|
||||
el_val_t native_string_chars(el_val_t s);
|
||||
el_val_t native_int_to_str(el_val_t n);
|
||||
|
||||
/* ── Method-call shorthand aliases ──────────────────────────────────────────
|
||||
* The El method-call convention `obj.method(args)` compiles to
|
||||
* `method(obj, args)`. These aliases expose the runtime functions under
|
||||
* the short names that result from method calls in El source.
|
||||
*
|
||||
* Example: `myList.append(x)` → `append(myList, x)` (calls this alias)
|
||||
* `myList.len()` → `len(myList)` (calls this alias) */
|
||||
|
||||
el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */
|
||||
el_val_t len(el_val_t list); /* el_list_len */
|
||||
el_val_t get(el_val_t list, el_val_t index); /* el_list_get */
|
||||
el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */
|
||||
el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */
|
||||
|
||||
/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */
|
||||
/* See bottom of el_runtime.c for the implementation.
|
||||
* Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION.
|
||||
* No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */
|
||||
/* ── Subprocess execution ────────────────────────────────────────────────── */
|
||||
el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */
|
||||
el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */
|
||||
el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */
|
||||
el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */
|
||||
|
||||
/* ── Stdout redirection (used by compiler JS pipeline) ───────────────────── */
|
||||
el_val_t stdout_to_file(el_val_t path); /* redirect process stdout to a file */
|
||||
el_val_t stdout_restore(void); /* restore process stdout to terminal */
|
||||
|
||||
el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json);
|
||||
el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json);
|
||||
el_val_t trace_span_start(el_val_t name);
|
||||
el_val_t trace_span_end(el_val_t span_handle);
|
||||
el_val_t emit_event(el_val_t name, el_val_t duration_ms);
|
||||
|
||||
el_val_t __thread_create(el_val_t fn_name_v, el_val_t arg_v);
|
||||
el_val_t __thread_join(el_val_t tid_v);
|
||||
|
||||
/* ── __ prefixed aliases (self-hosting compiler ABI) ─────────────────────────
|
||||
* The El self-hosting compiler emits calls to __-prefixed names. These are
|
||||
* forwarding wrappers around the existing el_runtime functions above. */
|
||||
|
||||
/* I/O */
|
||||
el_val_t __println(el_val_t s);
|
||||
el_val_t __print(el_val_t s);
|
||||
el_val_t __readline(void);
|
||||
|
||||
/* String */
|
||||
el_val_t __int_to_str(el_val_t n);
|
||||
el_val_t __str_to_int(el_val_t s);
|
||||
el_val_t __float_to_str(el_val_t f);
|
||||
el_val_t __str_to_float(el_val_t s);
|
||||
el_val_t __str_len(el_val_t s);
|
||||
el_val_t __str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t __str_cmp(el_val_t a, el_val_t b);
|
||||
el_val_t __str_ncmp(el_val_t a, el_val_t b, el_val_t n);
|
||||
el_val_t __str_concat_raw(el_val_t a, el_val_t b);
|
||||
el_val_t __str_slice_raw(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t __str_alloc(el_val_t n);
|
||||
el_val_t __str_set_char(el_val_t s, el_val_t i, el_val_t c);
|
||||
|
||||
/* URL encoding */
|
||||
el_val_t __url_encode(el_val_t s);
|
||||
el_val_t __url_decode(el_val_t s);
|
||||
|
||||
/* Environment */
|
||||
el_val_t __env_get(el_val_t key);
|
||||
|
||||
/* Subprocess */
|
||||
el_val_t __exec(el_val_t cmd);
|
||||
el_val_t __exec_bg(el_val_t cmd);
|
||||
|
||||
/* Process */
|
||||
el_val_t __exit_program(el_val_t code);
|
||||
|
||||
/* Filesystem */
|
||||
el_val_t __fs_exists(el_val_t path);
|
||||
el_val_t __fs_mkdir(el_val_t path);
|
||||
el_val_t __fs_read(el_val_t path);
|
||||
el_val_t __fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t __fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t n);
|
||||
el_val_t __fs_list_raw(el_val_t path);
|
||||
|
||||
/* HTTP server */
|
||||
el_val_t __http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
el_val_t __http_serve(el_val_t port, el_val_t handler);
|
||||
el_val_t __http_serve_v2(el_val_t port, el_val_t handler);
|
||||
|
||||
/* HTTP conn fd / SSE (weak; overridden by el_seed.c when linked together) */
|
||||
el_val_t __http_conn_fd(void);
|
||||
el_val_t __http_sse_open(el_val_t conn_id);
|
||||
el_val_t __http_sse_send(el_val_t conn_id, el_val_t data);
|
||||
el_val_t __http_sse_close(el_val_t conn_id);
|
||||
|
||||
/* HTTP client (requires HAVE_CURL; stubs provided for no-curl builds) */
|
||||
el_val_t __http_do(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_map, el_val_t timeout_ms);
|
||||
el_val_t __http_do_map(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_json, el_val_t timeout_ms);
|
||||
el_val_t __http_do_map_to_file(el_val_t method, el_val_t url, el_val_t body,
|
||||
el_val_t headers_json, el_val_t output_path);
|
||||
|
||||
/* JSON */
|
||||
el_val_t __json_array_get(el_val_t json, el_val_t index);
|
||||
el_val_t __json_array_get_string(el_val_t json, el_val_t index);
|
||||
el_val_t __json_array_len(el_val_t json);
|
||||
el_val_t __json_get(el_val_t json, el_val_t key);
|
||||
el_val_t __json_get_raw(el_val_t json, el_val_t key);
|
||||
el_val_t __json_set(el_val_t json, el_val_t key, el_val_t value);
|
||||
el_val_t __json_parse_map(el_val_t json_str);
|
||||
el_val_t __json_stringify_val(el_val_t val);
|
||||
|
||||
/* Hashing */
|
||||
el_val_t __sha256_hex(el_val_t s);
|
||||
|
||||
/* State K/V */
|
||||
el_val_t __state_del(el_val_t key);
|
||||
el_val_t __state_get(el_val_t key);
|
||||
el_val_t __state_keys(void);
|
||||
el_val_t __state_set(el_val_t key, el_val_t val);
|
||||
|
||||
/* UUID */
|
||||
el_val_t __uuid_v4(void);
|
||||
|
||||
/* Args */
|
||||
el_val_t __args_json(void);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,761 +0,0 @@
|
||||
/*
|
||||
* el_runtime.h — El language C runtime header
|
||||
*
|
||||
* Declares all built-in functions available to compiled El programs.
|
||||
* Include this in every generated .c file.
|
||||
*
|
||||
* Value model:
|
||||
* All El values are represented as el_val_t (= int64_t).
|
||||
* On 64-bit systems a pointer fits in int64_t.
|
||||
* String values are cast: (el_val_t)(uintptr_t)"hello"
|
||||
* Integer values are stored directly.
|
||||
* This lets arithmetic work naturally while still passing strings around.
|
||||
*
|
||||
* Type conventions (El -> C):
|
||||
* String -> el_val_t (holds const char* via uintptr_t cast)
|
||||
* Int -> el_val_t
|
||||
* Bool -> el_val_t (0 = false, nonzero = true)
|
||||
* Any -> el_val_t
|
||||
* Void -> void
|
||||
*
|
||||
* Macros for convenience:
|
||||
* EL_STR(s) cast string literal to el_val_t
|
||||
* EL_CSTR(v) cast el_val_t back to const char*
|
||||
* EL_INT(v) identity — el_val_t is already int64_t
|
||||
*
|
||||
* Link requirements:
|
||||
* -lcurl — required for the HTTP client (http_get, http_post, llm_*).
|
||||
* -lpthread — required for the HTTP server (one detached thread per
|
||||
* connection, capped at 64 concurrent).
|
||||
* -loqs — optional; required only when liboqs is installed and the
|
||||
* pq_* / sha3_256_hex entry points are needed. Detected at
|
||||
* compile time via __has_include(<oqs/oqs.h>).
|
||||
* -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
typedef int64_t el_val_t;
|
||||
|
||||
#define EL_STR(s) ((el_val_t)(uintptr_t)(s))
|
||||
#define EL_CSTR(v) ((const char*)(uintptr_t)(v))
|
||||
#define EL_INT(v) (v)
|
||||
#define EL_NULL ((el_val_t)0)
|
||||
|
||||
/* Float values share the el_val_t (int64) slot via a bit-cast.
|
||||
* The codegen emits Float literals as `el_from_float(<dbl>)` so the
|
||||
* underlying bits represent the IEEE 754 double. Float-aware builtins
|
||||
* (math, format, json) round-trip via these helpers. */
|
||||
static inline double el_to_float(el_val_t v) {
|
||||
union { int64_t i; double f; } u;
|
||||
u.i = (int64_t)v;
|
||||
return u.f;
|
||||
}
|
||||
|
||||
static inline el_val_t el_from_float(double f) {
|
||||
union { double f; int64_t i; } u;
|
||||
u.f = f;
|
||||
return (el_val_t)u.i;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ── I/O ──────────────────────────────────────────────────────────────────── */
|
||||
|
||||
void println(el_val_t s);
|
||||
void print(el_val_t s);
|
||||
el_val_t readline(void);
|
||||
|
||||
/* ── String builtins ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t str_eq(el_val_t a, el_val_t b);
|
||||
el_val_t str_starts_with(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_ends_with(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_len(el_val_t s);
|
||||
el_val_t str_concat(el_val_t a, el_val_t b);
|
||||
el_val_t int_to_str(el_val_t n);
|
||||
el_val_t str_to_int(el_val_t s);
|
||||
el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end);
|
||||
el_val_t str_contains(el_val_t s, el_val_t sub);
|
||||
el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to);
|
||||
el_val_t str_to_upper(el_val_t s);
|
||||
el_val_t str_to_lower(el_val_t s);
|
||||
el_val_t str_trim(el_val_t s);
|
||||
|
||||
/* ── Math ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_abs(el_val_t n);
|
||||
el_val_t el_max(el_val_t a, el_val_t b);
|
||||
el_val_t el_min(el_val_t a, el_val_t b);
|
||||
|
||||
/* ── Refcount (ARC) ──────────────────────────────────────────────────────────
|
||||
* Lists and Maps carry a refcount. Strings and ints do not — el_retain and
|
||||
* el_release are safe no-ops on non-refcounted values (they sniff a magic
|
||||
* header at offset 0 and only act if the magic matches).
|
||||
*
|
||||
* Codegen emits these at let-binding shadowing, function entry (params), and
|
||||
* function exit (locals other than the returned value). The refcount lets
|
||||
* el_list_append and el_map_set mutate in place when uniquely owned (cheap)
|
||||
* and copy-on-write when shared (preserves persistent semantics across
|
||||
* accumulator patterns in the compiler itself). */
|
||||
|
||||
void el_retain(el_val_t v);
|
||||
void el_release(el_val_t v);
|
||||
|
||||
/* ── List ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_list_new(el_val_t count, ...);
|
||||
el_val_t el_list_len(el_val_t list);
|
||||
el_val_t el_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t el_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t el_list_empty(void);
|
||||
el_val_t el_list_clone(el_val_t list);
|
||||
|
||||
/* ── Map ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t el_map_new(el_val_t pair_count, ...);
|
||||
el_val_t el_get_field(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_get(el_val_t map, el_val_t key);
|
||||
el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value);
|
||||
|
||||
/* ── HTTP ─────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t http_get(el_val_t url);
|
||||
el_val_t http_post(el_val_t url, el_val_t body);
|
||||
el_val_t http_post_json(el_val_t url, el_val_t json_body);
|
||||
el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map);
|
||||
el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map);
|
||||
el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header);
|
||||
el_val_t http_delete(el_val_t url);
|
||||
void http_serve(el_val_t port, el_val_t handler);
|
||||
void http_set_handler(el_val_t name);
|
||||
|
||||
/* HTTP server v2 ─────────────────────────────────────────────────────────────
|
||||
* Same dispatch model as http_serve, but the handler signature is widened:
|
||||
*
|
||||
* el_val_t handler(method, path, headers_map, body)
|
||||
*
|
||||
* `headers_map` is an ElMap from lowercased header name → header value (both
|
||||
* Strings). Repeated headers are joined with ", " per RFC 7230.
|
||||
*
|
||||
* Response value: the handler may return either
|
||||
* (a) a plain body string — same auto-content-type / 200-OK behaviour as
|
||||
* http_serve (3-arg) — or
|
||||
* (b) a response envelope built with `http_response(status, headers_json,
|
||||
* body)`. The runtime detects the envelope discriminator
|
||||
* `"el_http_response":1` at the start of the returned string and
|
||||
* unpacks status / headers / body before sending.
|
||||
*
|
||||
* The 3-arg http_serve(port, handler) remains supported unchanged for
|
||||
* existing handlers (e.g. products/web/server.el): it dispatches with
|
||||
* (method, path, body), hardcodes 200 OK, and auto-detects content type. */
|
||||
void http_serve_v2(el_val_t port, el_val_t handler);
|
||||
void http_set_handler_v2(el_val_t name);
|
||||
|
||||
/* Build an HTTP response envelope. `headers_json` should be a JSON object
|
||||
* literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The
|
||||
* returned string carries the discriminator `{"el_http_response":1,...}`
|
||||
* which the runtime's send-path detects and unpacks. Detection happens
|
||||
* uniformly inside http_send_response, so a 3-arg handler may also return
|
||||
* an envelope. The 3-arg variant remains documented as a fixed 200-OK
|
||||
* auto-content-type contract for legacy handlers that return plain bodies. */
|
||||
el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body);
|
||||
|
||||
/* SSE connection fd — set by http_worker_v2 before calling the El handler,
|
||||
* cleared afterwards. Defined in el_seed.c; called from el_runtime.c.
|
||||
* The getter is exposed as __http_conn_fd() to El programs. */
|
||||
void el_seed_set_http_conn_fd(int fd);
|
||||
|
||||
/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default
|
||||
* 60000ms). Read lazily on first use, so setting the env var any time before
|
||||
* the first http_* call is sufficient. */
|
||||
|
||||
/* Streaming variants — write the response body straight to a file via
|
||||
* libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string
|
||||
* wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive
|
||||
* embedded NUL bytes that would truncate a strlen()-based code path.
|
||||
*
|
||||
* Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same
|
||||
* `headers_map` shape as http_post_with_headers (ElMap of String→String).
|
||||
*
|
||||
* Return value: 1 on success (file fully written), 0 on any failure
|
||||
* (network, file open, partial write). On failure the output file is removed
|
||||
* so callers cannot mistake a partially-written file for a valid one. */
|
||||
el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path);
|
||||
el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path);
|
||||
|
||||
/* ── URL encoding ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */
|
||||
el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */
|
||||
|
||||
/* ── HTML allowlist sanitizer ────────────────────────────────────────────────
|
||||
* el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML
|
||||
* cleaner. State-machine parser; tag/attribute names compared case-
|
||||
* insensitively against the allowlist; `<a href>` / `<… src>` URL schemes
|
||||
* validated (http, https, mailto, fragment-only, or relative); whole-
|
||||
* subtree drop for script / style / iframe / object / embed / form; HTML-
|
||||
* escapes free text outside dropped subtrees.
|
||||
*
|
||||
* The allowlist is JSON of the form
|
||||
* {"p":[],"a":["href","title"],"strong":[],...}
|
||||
* where each value is the array of attribute names allowed for that tag. */
|
||||
el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json);
|
||||
|
||||
/* ── Filesystem ──────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t fs_read(el_val_t path);
|
||||
el_val_t fs_write(el_val_t path, el_val_t content);
|
||||
el_val_t fs_list(el_val_t path);
|
||||
el_val_t fs_exists(el_val_t path);
|
||||
el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */
|
||||
|
||||
/* Length-explicit binary write. `length` is an Int (el_val_t holding the
|
||||
* byte count). The caller knows the length from context — typically because
|
||||
* `bytes` came from base64_decode (which produces a magic-tagged binary
|
||||
* buffer with embedded NULs possible) and the caller already tracks the
|
||||
* decoded length, OR because the bytes came from a fixed-size source
|
||||
* (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely.
|
||||
*
|
||||
* Returns 1 on success, 0 on failure (invalid path, can't open, partial
|
||||
* write, negative length). On partial-write failure, the file is removed
|
||||
* so callers cannot read back a truncated artefact. */
|
||||
el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length);
|
||||
|
||||
/* ── JSON ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t json_get(el_val_t json, el_val_t key);
|
||||
el_val_t json_parse(el_val_t s);
|
||||
el_val_t json_stringify(el_val_t v);
|
||||
el_val_t json_get_string(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_int(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_float(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_bool(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_get_raw(el_val_t json_str, el_val_t key);
|
||||
el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value);
|
||||
el_val_t json_array_len(el_val_t json_str);
|
||||
el_val_t json_array_get(el_val_t json_str, el_val_t index);
|
||||
el_val_t json_array_get_string(el_val_t json_str, el_val_t index);
|
||||
|
||||
/* ── Time ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t time_now(void);
|
||||
el_val_t time_now_utc(void);
|
||||
el_val_t sleep_secs(el_val_t secs);
|
||||
el_val_t sleep_ms(el_val_t ms);
|
||||
el_val_t time_format(el_val_t ts, el_val_t fmt);
|
||||
el_val_t time_to_parts(el_val_t ts);
|
||||
el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz);
|
||||
el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit);
|
||||
el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit);
|
||||
|
||||
/* ── Instant + Duration: first-class temporal types ──────────────────────────
|
||||
* Both types share the el_val_t (int64) slot. Instants are nanoseconds
|
||||
* since the Unix epoch; Durations are signed nanoseconds. Type discipline
|
||||
* is enforced at codegen-time: BinOps on names registered as Instant or
|
||||
* Duration route through the typed wrappers below; mismatches like
|
||||
* Instant+Instant become #error at the C compiler.
|
||||
*
|
||||
* Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are
|
||||
* recognised by the parser as DurationLit AST nodes and lowered to literal
|
||||
* int64 nanoseconds at codegen time. The runtime never sees the units. */
|
||||
|
||||
el_val_t el_now_instant(void);
|
||||
el_val_t now(void);
|
||||
el_val_t unix_seconds(el_val_t n);
|
||||
el_val_t unix_millis(el_val_t n);
|
||||
el_val_t instant_from_iso8601(el_val_t s);
|
||||
|
||||
el_val_t el_duration_from_nanos(el_val_t ns);
|
||||
el_val_t duration_seconds(el_val_t n);
|
||||
el_val_t duration_millis(el_val_t n);
|
||||
el_val_t duration_nanos(el_val_t n);
|
||||
|
||||
el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur);
|
||||
el_val_t el_instant_diff(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_add(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_sub(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_scale(el_val_t dur, el_val_t scalar);
|
||||
el_val_t el_duration_div(el_val_t dur, el_val_t scalar);
|
||||
|
||||
el_val_t el_instant_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_instant_ne(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_le(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_gt(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ge(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_eq(el_val_t a, el_val_t b);
|
||||
el_val_t el_duration_ne(el_val_t a, el_val_t b);
|
||||
|
||||
el_val_t instant_to_unix_seconds(el_val_t i);
|
||||
el_val_t instant_to_unix_millis(el_val_t i);
|
||||
el_val_t instant_to_iso8601(el_val_t i);
|
||||
el_val_t duration_to_seconds(el_val_t d);
|
||||
el_val_t duration_to_millis(el_val_t d);
|
||||
el_val_t duration_to_nanos(el_val_t d);
|
||||
|
||||
el_val_t el_sleep_duration(el_val_t dur);
|
||||
el_val_t unix_timestamp(void);
|
||||
|
||||
el_val_t ttl_cache_set(el_val_t key, el_val_t value);
|
||||
el_val_t ttl_cache_get(el_val_t key, el_val_t max_age);
|
||||
el_val_t ttl_cache_age(el_val_t key);
|
||||
|
||||
/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ─────────────
|
||||
* Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA
|
||||
* zones, Gregorian, DST) is the user-facing default; MarsCalendar,
|
||||
* CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth
|
||||
* domains.
|
||||
*
|
||||
* A Calendar interprets an Instant under a particular cycle convention and
|
||||
* produces a CalendarTime. CalendarTime carries the underlying Instant and
|
||||
* a back-pointer to its Calendar; arithmetic and formatting consult the
|
||||
* Calendar to convert ns since epoch into year/month/day/hour/minute/second
|
||||
* (or sol/phase, or cycle/phase, depending on kind).
|
||||
*
|
||||
* Storage convention: Calendar / CalendarTime / Rhythm / LocalDate /
|
||||
* LocalDateTime are heap-allocated structs whose pointers are cast into
|
||||
* el_val_t. A 24-bit magic header at offset 0 lets the runtime identify
|
||||
* the kind safely. LocalTime is small enough to live in the int64 slot
|
||||
* directly (nanos since midnight, signed). */
|
||||
|
||||
/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar.
|
||||
* `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed
|
||||
* offset string ("+05:30", "-08:00"). The runtime resolves it via tzset()
|
||||
* on first use of the owning EarthCalendar. */
|
||||
el_val_t zone(el_val_t id);
|
||||
el_val_t zone_utc(void);
|
||||
el_val_t zone_local(void);
|
||||
el_val_t zone_offset(el_val_t hours, el_val_t minutes);
|
||||
|
||||
/* Calendar constructors. Each returns an el_val_t pointer to a heap-
|
||||
* allocated, magic-tagged Calendar struct. Calendars are interned by
|
||||
* (kind, zone_id, period_ns, epoch_ns) so identical constructors return
|
||||
* the same pointer — equality is reference equality. */
|
||||
el_val_t earth_calendar(el_val_t z);
|
||||
el_val_t earth_calendar_default(void);
|
||||
el_val_t mars_calendar(void);
|
||||
el_val_t cycle_calendar(el_val_t period_dur);
|
||||
el_val_t no_cycle_calendar(void);
|
||||
el_val_t relative_calendar(el_val_t epoch_inst);
|
||||
|
||||
/* CalendarTime constructors and methods. Returns a heap-allocated struct
|
||||
* whose pointer fits in el_val_t. */
|
||||
el_val_t now_in(el_val_t cal);
|
||||
el_val_t in_calendar(el_val_t inst, el_val_t cal);
|
||||
el_val_t cal_format(el_val_t ct, el_val_t pattern);
|
||||
el_val_t cal_to_instant(el_val_t ct);
|
||||
el_val_t cal_cycle_phase(el_val_t ct);
|
||||
el_val_t cal_in(el_val_t ct, el_val_t cal);
|
||||
|
||||
/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types.
|
||||
* LocalTime carries nanoseconds since midnight as a signed int64 directly
|
||||
* in the el_val_t slot (no allocation). LocalDate / LocalDateTime are
|
||||
* heap-allocated structs with magic headers. */
|
||||
el_val_t local_date(el_val_t y, el_val_t m, el_val_t d);
|
||||
el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns);
|
||||
el_val_t local_datetime(el_val_t date, el_val_t time);
|
||||
el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal);
|
||||
|
||||
el_val_t local_date_year(el_val_t ld);
|
||||
el_val_t local_date_month(el_val_t ld);
|
||||
el_val_t local_date_day(el_val_t ld);
|
||||
el_val_t local_time_hour(el_val_t lt);
|
||||
el_val_t local_time_minute(el_val_t lt);
|
||||
el_val_t local_time_second(el_val_t lt);
|
||||
el_val_t local_time_nanos(el_val_t lt);
|
||||
|
||||
el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur);
|
||||
el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur);
|
||||
el_val_t el_local_date_lt(el_val_t a, el_val_t b);
|
||||
el_val_t el_local_date_eq(el_val_t a, el_val_t b);
|
||||
|
||||
/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct
|
||||
* pointer in el_val_t; rhythms are immutable so callers may share them. */
|
||||
el_val_t rhythm_cycle_start(void);
|
||||
el_val_t rhythm_cycle_phase(el_val_t phase);
|
||||
el_val_t rhythm_duration(el_val_t d);
|
||||
el_val_t rhythm_session_start(void);
|
||||
el_val_t rhythm_event(el_val_t name);
|
||||
el_val_t rhythm_and(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_or(el_val_t a, el_val_t b);
|
||||
el_val_t rhythm_weekday(el_val_t day);
|
||||
el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute);
|
||||
el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal);
|
||||
el_val_t rhythm_matches(el_val_t r, el_val_t ct);
|
||||
|
||||
/* ── UUID ────────────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t uuid_new(void);
|
||||
el_val_t uuid_v4(void);
|
||||
|
||||
/* ── Environment ─────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t env(el_val_t key);
|
||||
|
||||
/* ── In-process state K/V ────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t state_set(el_val_t key, el_val_t value);
|
||||
el_val_t state_get(el_val_t key);
|
||||
el_val_t state_del(el_val_t key);
|
||||
el_val_t state_keys(void);
|
||||
|
||||
/* ── Float formatting ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t float_to_str(el_val_t f);
|
||||
el_val_t int_to_float(el_val_t n);
|
||||
el_val_t float_to_int(el_val_t f);
|
||||
el_val_t format_float(el_val_t f, el_val_t decimals);
|
||||
el_val_t decimal_round(el_val_t f, el_val_t decimals);
|
||||
el_val_t str_to_float(el_val_t s);
|
||||
|
||||
/* ── Math (Float-aware) ──────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t math_sqrt(el_val_t f);
|
||||
el_val_t math_log(el_val_t f);
|
||||
el_val_t math_ln(el_val_t f);
|
||||
el_val_t math_sin(el_val_t f);
|
||||
el_val_t math_cos(el_val_t f);
|
||||
el_val_t math_pi(void);
|
||||
|
||||
/* ── String additions ────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t str_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_split(el_val_t s, el_val_t sep);
|
||||
el_val_t str_char_at(el_val_t s, el_val_t i);
|
||||
el_val_t str_char_code(el_val_t s, el_val_t i);
|
||||
el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad);
|
||||
el_val_t str_format(el_val_t fmt, el_val_t data);
|
||||
el_val_t str_lower(el_val_t s);
|
||||
el_val_t str_upper(el_val_t s);
|
||||
|
||||
/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes)
|
||||
* Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex.
|
||||
* is_* predicates: empty input returns false; multi-char requires ALL bytes
|
||||
* to match. ASCII ranges only in Phase 1. */
|
||||
|
||||
/* Counting */
|
||||
el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */
|
||||
el_val_t str_count_chars(el_val_t s); /* codepoint count */
|
||||
el_val_t str_count_bytes(el_val_t s); /* alias of str_len */
|
||||
el_val_t str_count_lines(el_val_t s);
|
||||
el_val_t str_count_words(el_val_t s);
|
||||
el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */
|
||||
el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */
|
||||
|
||||
/* Find / position */
|
||||
el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */
|
||||
el_val_t str_last_index_of(el_val_t s, el_val_t sub);
|
||||
el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */
|
||||
|
||||
/* Transform */
|
||||
el_val_t str_repeat(el_val_t s, el_val_t n);
|
||||
el_val_t str_reverse(el_val_t s); /* by codepoint */
|
||||
el_val_t str_strip_prefix(el_val_t s, el_val_t prefix);
|
||||
el_val_t str_strip_suffix(el_val_t s, el_val_t suffix);
|
||||
el_val_t str_strip_chars(el_val_t s, el_val_t chars);
|
||||
el_val_t str_lstrip(el_val_t s);
|
||||
el_val_t str_rstrip(el_val_t s);
|
||||
|
||||
/* Char classification (Bool) */
|
||||
el_val_t is_letter(el_val_t s);
|
||||
el_val_t is_digit(el_val_t s);
|
||||
el_val_t is_alphanumeric(el_val_t s);
|
||||
el_val_t is_whitespace(el_val_t s);
|
||||
el_val_t is_punctuation(el_val_t s);
|
||||
el_val_t is_uppercase(el_val_t s);
|
||||
el_val_t is_lowercase(el_val_t s);
|
||||
|
||||
/* Split / join */
|
||||
el_val_t str_split_lines(el_val_t s);
|
||||
el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */
|
||||
el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n);
|
||||
el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */
|
||||
|
||||
/* ── List additions ──────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t list_push(el_val_t list, el_val_t elem);
|
||||
el_val_t list_push_front(el_val_t list, el_val_t elem);
|
||||
el_val_t list_join(el_val_t list, el_val_t sep);
|
||||
el_val_t list_range(el_val_t start, el_val_t end);
|
||||
|
||||
/* ── Bool helpers ────────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t bool_to_str(el_val_t b);
|
||||
|
||||
/* ── Numeric parsing ─────────────────────────────────────────────────────── */
|
||||
|
||||
el_val_t parse_int(el_val_t s, el_val_t default_val);
|
||||
|
||||
/* ── Process ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
void exit_program(el_val_t code);
|
||||
el_val_t getpid_now(void);
|
||||
|
||||
/* ── CGI identity ─────────────────────────────────────────────────────────────
|
||||
* Called at the start of main() in CGI programs (those with a `cgi {}` block).
|
||||
* Records the program's DHARMA identity before any other code executes. */
|
||||
|
||||
void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal,
|
||||
el_val_t network, el_val_t engram);
|
||||
|
||||
/* ── DHARMA network builtins ─────────────────────────────────────────────────
|
||||
* Available to CGI programs (declared with a `cgi {}` block).
|
||||
*
|
||||
* Peers are addressed by `dharma_id` of the form
|
||||
* "<registry-id>@<transport-url>" e.g. "ntn-genesis@http://localhost:7770"
|
||||
* If the @<url> portion is omitted, transport defaults to
|
||||
* "http://localhost:7770" (the local CGI daemon assumption).
|
||||
*
|
||||
* Wire protocol (all peers expose):
|
||||
* POST <url>/dharma/recv { channel, from, content } → response body
|
||||
* POST <url>/dharma/event { type, payload, source, timestamp }
|
||||
* POST <url>/api/activate { query } → list of nodes
|
||||
*
|
||||
* Hosting application's responsibility: an El program with a `cgi {}` block
|
||||
* runs http_serve() with its own request handler; that handler should route
|
||||
* "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so
|
||||
* incoming events feed dharma_field() queues. The runtime itself does not
|
||||
* intercept any /dharma path. */
|
||||
|
||||
el_val_t dharma_connect(el_val_t cgi_id);
|
||||
el_val_t dharma_send(el_val_t channel, el_val_t content);
|
||||
el_val_t dharma_activate(el_val_t query);
|
||||
void dharma_emit(el_val_t event_type, el_val_t payload);
|
||||
el_val_t dharma_field(el_val_t event_type);
|
||||
void dharma_strengthen(el_val_t cgi_id, el_val_t weight);
|
||||
el_val_t dharma_relationship(el_val_t cgi_id);
|
||||
el_val_t dharma_peers(void);
|
||||
|
||||
/* Public C API: called by an El program's HTTP handler when a /dharma/event
|
||||
* request arrives. Pushes onto the per-event-type queue and signals any
|
||||
* pending dharma_field() blockers. All three arguments must be NUL-terminated
|
||||
* C strings (or NULL — then treated as empty). */
|
||||
void el_runtime_dharma_event_arrive(const char* event_type,
|
||||
const char* payload,
|
||||
const char* source);
|
||||
|
||||
/* ── Engram local graph primitives ───────────────────────────────────────────
|
||||
* Operate on the CGI's local Engram knowledge graph.
|
||||
* `engram_activate` queries the local graph only; `dharma_activate` is
|
||||
* network-wide across all connected CGI graphs. */
|
||||
|
||||
el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience);
|
||||
el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t importance, el_val_t confidence,
|
||||
el_val_t tier, el_val_t tags);
|
||||
/* Layered consciousness — see el_runtime.c for the layered architecture
|
||||
* design notes (search "Layered consciousness architecture"). The five
|
||||
* canonical layers (safety / core-identity / domain-knowledge / imprint /
|
||||
* suit) are seeded automatically; engram_add_layer extends the registry
|
||||
* with imprint or suit overlays at runtime. Nodes default to layer 1
|
||||
* (core-identity) when created via engram_node / engram_node_full. */
|
||||
el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label,
|
||||
el_val_t salience, el_val_t certainty, el_val_t confidence,
|
||||
el_val_t status, el_val_t tags, el_val_t layer_id);
|
||||
el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible,
|
||||
el_val_t transparent, el_val_t injectable);
|
||||
el_val_t engram_remove_layer(el_val_t layer_id);
|
||||
el_val_t engram_list_layers(void);
|
||||
el_val_t engram_get_node(el_val_t id);
|
||||
void engram_strengthen(el_val_t node_id);
|
||||
void engram_forget(el_val_t node_id);
|
||||
el_val_t engram_node_count(void);
|
||||
el_val_t engram_search(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset);
|
||||
void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation);
|
||||
el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id);
|
||||
el_val_t engram_neighbors(el_val_t node_id);
|
||||
el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_edge_count(void);
|
||||
/* Three-pass activation: background fan-out → working-memory promotion →
|
||||
* Layer 0 override. See "Three-pass activation" in el_runtime.c. */
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
* through json_stringify. */
|
||||
el_val_t engram_get_node_json(el_val_t id);
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
el_val_t engram_activate_json(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_stats_json(void);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
* no nodes promoted to working memory. */
|
||||
el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth);
|
||||
|
||||
/* ── LLM (Anthropic API client) ─────────────────────────────────────────────
|
||||
* All functions call https://api.anthropic.com/v1/messages with the API key
|
||||
* from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */
|
||||
|
||||
el_val_t llm_call(el_val_t model, el_val_t prompt);
|
||||
el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt);
|
||||
el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools);
|
||||
el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64);
|
||||
el_val_t llm_models(void);
|
||||
|
||||
/* Register a tool handler by name. The handler is looked up via dlsym
|
||||
* (mirroring http_set_handler), so any El `fn <name>(input)` compiles to
|
||||
* a global C symbol that this function can locate at runtime.
|
||||
* Handler signature: `el_val_t handler(el_val_t input_json)` — receives
|
||||
* the tool input as a JSON-string el_val_t and returns a JSON-string
|
||||
* el_val_t result. Used by llm_call_agentic. */
|
||||
void llm_register_tool(el_val_t name, el_val_t handler_fn_name);
|
||||
|
||||
/* ── args() ─────────────────────────────────────────────────────────────────
|
||||
* Provides access to command-line arguments passed to the program.
|
||||
* Populated by el_runtime_init_args() before main() runs. */
|
||||
|
||||
el_val_t args(void);
|
||||
void el_runtime_init_args(int argc, char** argv);
|
||||
|
||||
/* ── Crypto primitives ─────────────────────────────────────────────────────
|
||||
* SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe).
|
||||
* Self-contained — no OpenSSL/libcrypto dependency. The implementations are
|
||||
* adapted from public-domain reference code (Brad Conte / RFC 4648).
|
||||
*
|
||||
* Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string
|
||||
* value whose contents are raw binary; callers usually feed these into
|
||||
* base64_encode. Note that el_val_t strings are NUL-terminated by convention,
|
||||
* so the binary payload may contain embedded NULs — pass it directly into
|
||||
* base64_encode (which uses an explicit length) rather than treating it as
|
||||
* a printable C string.
|
||||
*
|
||||
* The "base64" variants emit/accept RFC 4648 standard alphabet with padding.
|
||||
* The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding,
|
||||
* as used in JWTs. */
|
||||
|
||||
el_val_t sha256_hex(el_val_t input);
|
||||
el_val_t sha256_bytes(el_val_t input);
|
||||
el_val_t hmac_sha256_hex(el_val_t key, el_val_t message);
|
||||
el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message);
|
||||
el_val_t base64_encode(el_val_t input);
|
||||
el_val_t base64_decode(el_val_t input);
|
||||
el_val_t base64url_encode(el_val_t input);
|
||||
el_val_t base64url_decode(el_val_t input);
|
||||
|
||||
/* Length-aware variants (internal — exposed for the rare caller that already
|
||||
* has a known-length binary buffer and doesn't want to round-trip through
|
||||
* a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed
|
||||
* these implicitly. */
|
||||
el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len);
|
||||
el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe);
|
||||
|
||||
/* ── Post-quantum primitives (liboqs-backed) ────────────────────────────────
|
||||
* All inputs/outputs hex-encoded. Algorithm choices:
|
||||
* Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced)
|
||||
* KEM: CRYSTALS-Kyber-768 (NIST level 3)
|
||||
* Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2)
|
||||
*
|
||||
* If liboqs is not linked (detected via __has_include(<oqs/oqs.h>) at compile
|
||||
* time), the pq_* entry points return a JSON-shaped error string so callers
|
||||
* fail loudly rather than silently fall back to classical schemes:
|
||||
* {"error":"liboqs not linked, post-quantum primitives unavailable"}
|
||||
*
|
||||
* The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and
|
||||
* CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss).
|
||||
* Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack,
|
||||
* Kyber holds. SHA3-256 also remains usable independent of liboqs (the
|
||||
* Keccak permutation is PQ-OK as a primitive). */
|
||||
|
||||
el_val_t pq_keygen_signature(void);
|
||||
el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message);
|
||||
el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex);
|
||||
|
||||
el_val_t pq_kem_keygen(void);
|
||||
el_val_t pq_kem_encaps(el_val_t public_key_hex);
|
||||
el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex);
|
||||
|
||||
el_val_t pq_hybrid_keygen(void);
|
||||
el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined);
|
||||
|
||||
el_val_t sha3_256_hex(el_val_t input);
|
||||
|
||||
/* ── AEAD: AES-256-GCM (libcrypto-backed) ───────────────────────────────────
|
||||
* Symmetric authenticated encryption used to wrap envelopes after a KEM
|
||||
* handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the
|
||||
* Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256.
|
||||
*
|
||||
* aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where
|
||||
* ciphertext is the AES-256-GCM output with the 16-byte auth tag appended.
|
||||
* Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which
|
||||
* structurally rules out the GCM nonce-reuse footgun.
|
||||
*
|
||||
* aead_decrypt returns the plaintext String, or "" on any failure (including
|
||||
* auth-tag mismatch). Callers MUST check for "" before trusting the result. */
|
||||
el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext);
|
||||
el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex);
|
||||
|
||||
/* ── Native VM builtin aliases (for compiled El source) ─────────────────────
|
||||
* These match the El VM's native_* builtins so that El source compiled
|
||||
* to C can call the same names without modification. */
|
||||
|
||||
el_val_t native_list_get(el_val_t list, el_val_t index);
|
||||
el_val_t native_list_len(el_val_t list);
|
||||
el_val_t native_list_append(el_val_t list, el_val_t elem);
|
||||
el_val_t native_list_empty(void);
|
||||
el_val_t native_list_clone(el_val_t list);
|
||||
el_val_t native_string_chars(el_val_t s);
|
||||
el_val_t native_int_to_str(el_val_t n);
|
||||
|
||||
/* ── Method-call shorthand aliases ──────────────────────────────────────────
|
||||
* The El method-call convention `obj.method(args)` compiles to
|
||||
* `method(obj, args)`. These aliases expose the runtime functions under
|
||||
* the short names that result from method calls in El source.
|
||||
*
|
||||
* Example: `myList.append(x)` → `append(myList, x)` (calls this alias)
|
||||
* `myList.len()` → `len(myList)` (calls this alias) */
|
||||
|
||||
el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */
|
||||
el_val_t len(el_val_t list); /* el_list_len */
|
||||
el_val_t get(el_val_t list, el_val_t index); /* el_list_get */
|
||||
el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */
|
||||
el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */
|
||||
|
||||
/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */
|
||||
/* See bottom of el_runtime.c for the implementation.
|
||||
* Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION.
|
||||
* No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */
|
||||
/* ── Subprocess execution ────────────────────────────────────────────────── */
|
||||
el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */
|
||||
el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */
|
||||
el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */
|
||||
el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */
|
||||
|
||||
el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json);
|
||||
el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json);
|
||||
el_val_t trace_span_start(el_val_t name);
|
||||
el_val_t trace_span_end(el_val_t span_handle);
|
||||
el_val_t emit_event(el_val_t name, el_val_t duration_ms);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
@@ -1202,7 +1202,7 @@ fn codegen_js_inner(stmts: [Map<String, Any>], source: String, bundle_mode: Bool
|
||||
js_emit_line(js_strip_es_exports(runtime_content))
|
||||
js_emit_line("")
|
||||
} else {
|
||||
js_emit_line("// Runtime: foundation/el/el-compiler/runtime/el_runtime.js")
|
||||
js_emit_line("// Runtime: foundation/el/runtime/el_runtime.js")
|
||||
js_emit_line("import \"./el_runtime.js\";")
|
||||
}
|
||||
// In module mode: destructure all builtins off globalThis.__el so call
|
||||
|
||||
@@ -1292,6 +1292,43 @@ fn next_if_id() -> String {
|
||||
native_int_to_str(n)
|
||||
}
|
||||
|
||||
// is_void_builtin — true for runtime builtins declared `void` in el_runtime.h.
|
||||
// User `-> Void` functions are emitted as el_val_t (return 0) so they are safe
|
||||
// to assign; only these C-level void builtins are not.
|
||||
fn is_void_builtin(name: String) -> Bool {
|
||||
if str_eq(name, "println") { return true }
|
||||
if str_eq(name, "print") { return true }
|
||||
if str_eq(name, "engram_strengthen") { return true }
|
||||
if str_eq(name, "engram_forget") { return true }
|
||||
if str_eq(name, "engram_connect") { return true }
|
||||
if str_eq(name, "dharma_emit") { return true }
|
||||
if str_eq(name, "dharma_strengthen") { return true }
|
||||
if str_eq(name, "llm_register_tool") { return true }
|
||||
if str_eq(name, "exit_program") { return true }
|
||||
if str_eq(name, "http_serve") { return true }
|
||||
if str_eq(name, "http_set_handler") { return true }
|
||||
if str_eq(name, "http_serve_async") { return true }
|
||||
if str_eq(name, "el_cgi_init") { return true }
|
||||
if str_eq(name, "el_retain") { return true }
|
||||
if str_eq(name, "el_release") { return true }
|
||||
false
|
||||
}
|
||||
|
||||
// cg_expr_is_void — true if `val` is a direct call to a void builtin, so the
|
||||
// if-expression arm must emit it as a bare statement rather than assigning its
|
||||
// (nonexistent) value to the result var.
|
||||
fn cg_expr_is_void(val: Map<String, Any>) -> Bool {
|
||||
let vk: String = val["expr"]
|
||||
if str_eq(vk, "Call") {
|
||||
let f = val["func"]
|
||||
let fk: String = f["expr"]
|
||||
if str_eq(fk, "Ident") {
|
||||
return is_void_builtin(f["name"])
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
// Render a single arm of the if-as-expression: emit each statement-before-last
|
||||
// as a side-effecting expression, then assign the final Expr's value to the
|
||||
// result var. If the arm body is empty or its last stmt isn't an Expr, the
|
||||
@@ -1300,6 +1337,10 @@ fn cg_if_expr_arm(stmts: [Map<String, Any>], result_var: String) -> String {
|
||||
let n: Int = native_list_len(stmts)
|
||||
// Collect statement fragments into a list to avoid O(n-) string growth.
|
||||
let parts: [String] = native_list_empty()
|
||||
// Track names already declared in this arm's C block. El permits `let x`
|
||||
// to redeclare/rebind x in the same scope, but C forbids redeclaring the
|
||||
// same name in one block: emit `el_val_t x = ...` first, `x = ...` after.
|
||||
let declared: [String] = native_list_empty()
|
||||
let i = 0
|
||||
while i < n {
|
||||
let s = native_list_get(stmts, i)
|
||||
@@ -1310,18 +1351,31 @@ fn cg_if_expr_arm(stmts: [Map<String, Any>], result_var: String) -> String {
|
||||
let name: String = s["name"]
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ")
|
||||
if list_contains(declared, name) {
|
||||
let parts = native_list_append(parts, name + " = " + val_c + "; ")
|
||||
} else {
|
||||
let declared = native_list_append(declared, name)
|
||||
let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ")
|
||||
}
|
||||
} else {
|
||||
if str_eq(sk, "Return") {
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
if cg_expr_is_void(val) {
|
||||
let parts = native_list_append(parts, val_c + "; ")
|
||||
} else {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
}
|
||||
} else {
|
||||
if str_eq(sk, "Expr") {
|
||||
let val = s["value"]
|
||||
let val_c: String = cg_expr(val)
|
||||
if is_last {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
if cg_expr_is_void(val) {
|
||||
let parts = native_list_append(parts, val_c + "; ")
|
||||
} else {
|
||||
let parts = native_list_append(parts, result_var + " = (" + val_c + "); ")
|
||||
}
|
||||
} else {
|
||||
let parts = native_list_append(parts, "(void)(" + val_c + "); ")
|
||||
}
|
||||
@@ -2669,7 +2723,11 @@ fn builtin_arity(name: String) -> Int {
|
||||
if str_eq(name, "engram_activate") { return 2 }
|
||||
if str_eq(name, "engram_save") { return 1 }
|
||||
if str_eq(name, "engram_load") { return 1 }
|
||||
if str_eq(name, "engram_store_boot") { return 1 }
|
||||
if str_eq(name, "engram_store_checkpoint") { return 0 }
|
||||
if str_eq(name, "engram_store_close") { return 0 }
|
||||
if str_eq(name, "engram_get_node_json") { return 1 }
|
||||
if str_eq(name, "engram_get_node_by_label") { return 1 }
|
||||
if str_eq(name, "engram_search_json") { return 2 }
|
||||
if str_eq(name, "engram_scan_nodes_json") { return 2 }
|
||||
if str_eq(name, "engram_neighbors_json") { return 3 }
|
||||
|
||||
@@ -49,6 +49,21 @@ fn tok_value(tokens: [Any], pos: Int) -> String {
|
||||
native_list_get(tokens, pos * 2 + 1)
|
||||
}
|
||||
|
||||
// parse_progress_fatal — robustness backstop. Called by the token-consuming
|
||||
// driver loops when they detect they have iterated more times than there are
|
||||
// tokens (impossible for a well-formed program, where every iteration consumes
|
||||
// at least one token). Names the offending token and exits non-zero instead of
|
||||
// looping forever / exhausting memory.
|
||||
fn parse_progress_fatal(where: String, tokens: [Any], pos: Int) -> Void {
|
||||
let k: String = tok_kind(tokens, pos)
|
||||
let v: String = tok_value(tokens, pos)
|
||||
println("elc: FATAL: parser made no forward progress in " + where
|
||||
+ " at token index " + native_int_to_str(pos) + " (kind=" + k + ")")
|
||||
println("elc: likely a malformed construct near '" + v
|
||||
+ "' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" ).")
|
||||
exit(1)
|
||||
}
|
||||
|
||||
fn expect(tokens: [Any], pos: Int, kind: String) -> Int {
|
||||
let k = tok_kind(tokens, pos)
|
||||
if k == kind {
|
||||
@@ -1212,7 +1227,16 @@ fn parse_block(tokens: [Any], pos: Int) -> Map<String, Any> {
|
||||
let p = expect(tokens, pos, "LBrace")
|
||||
let stmts: [Map<String, Any>] = native_list_empty()
|
||||
let running = true
|
||||
// Runaway backstop: a block can hold at most (token count) statements, since
|
||||
// every iteration consumes >= 1 token. If we exceed that, the cursor has run
|
||||
// off the end without terminating (malformed input) -> fail fast, don't hang.
|
||||
let blk_total: Int = native_list_len(tokens) / 2
|
||||
let blk_iters: Int = 0
|
||||
while running {
|
||||
let blk_iters = blk_iters + 1
|
||||
if blk_iters > blk_total + 8 {
|
||||
parse_progress_fatal("parse_block", tokens, p)
|
||||
}
|
||||
let k = tok_kind(tokens, p)
|
||||
if k == "RBrace" {
|
||||
let running = false
|
||||
|
||||
+3
-3
@@ -368,13 +368,13 @@ fn main() -> Void {
|
||||
let which_out: String = str_trim(exec_capture("which " + elc_bin + " 2>/dev/null"))
|
||||
if !str_eq(which_out, "") {
|
||||
let elc_dir: String = dirname_of(which_out)
|
||||
runtime_path = elc_dir + "/../el-compiler/runtime/el_runtime.c"
|
||||
runtime_path = elc_dir + "/../runtime/el_runtime.c"
|
||||
}
|
||||
}
|
||||
// If --runtime points to a directory, auto-locate el_runtime.c inside it.
|
||||
// This lets both forms work:
|
||||
// --runtime=/opt/el/el-compiler/runtime (directory form)
|
||||
// --runtime=/opt/el/el-compiler/runtime/el_runtime.c (file form)
|
||||
// --runtime=/opt/el/runtime (directory form)
|
||||
// --runtime=/opt/el/runtime/el_runtime.c (file form)
|
||||
if !str_eq(runtime_path, "") {
|
||||
let is_dir: String = str_trim(exec_capture("test -d " + runtime_path + " && echo dir || echo file"))
|
||||
if str_eq(is_dir, "dir") {
|
||||
|
||||
@@ -3797,6 +3797,9 @@ fn builtin_arity(name: String) -> Int {
|
||||
if str_eq(name, "engram_activate") { return 2 }
|
||||
if str_eq(name, "engram_save") { return 1 }
|
||||
if str_eq(name, "engram_load") { return 1 }
|
||||
if str_eq(name, "engram_store_boot") { return 1 }
|
||||
if str_eq(name, "engram_store_checkpoint") { return 0 }
|
||||
if str_eq(name, "engram_store_close") { return 0 }
|
||||
if str_eq(name, "engram_get_node_json") { return 1 }
|
||||
if str_eq(name, "engram_search_json") { return 2 }
|
||||
if str_eq(name, "engram_scan_nodes_json") { return 2 }
|
||||
|
||||
+105
-2
@@ -1423,15 +1423,53 @@ el_val_t tok_at(el_val_t tokens, el_val_t pos) {
|
||||
}
|
||||
|
||||
el_val_t tok_kind(el_val_t tokens, el_val_t pos) {
|
||||
/* Out-of-range reads MUST report the Eof sentinel so every `== "Eof"`
|
||||
termination guard in the parser fires. Without this, reading past the
|
||||
trailing Eof token returns runtime null (native_list_get OOB -> 0), which
|
||||
matches no delimiter, letting inner parse loops (parse_block, parse_binop)
|
||||
append AST nodes forever on malformed input -> unbounded allocation -> OOM. */
|
||||
el_val_t n = (native_list_len(tokens) / 2);
|
||||
if (pos < 0) {
|
||||
return EL_STR("Eof");
|
||||
}
|
||||
if (pos >= n) {
|
||||
return EL_STR("Eof");
|
||||
}
|
||||
return native_list_get(tokens, (pos * 2));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t tok_value(el_val_t tokens, el_val_t pos) {
|
||||
el_val_t n = (native_list_len(tokens) / 2);
|
||||
if (pos < 0) {
|
||||
return EL_STR("");
|
||||
}
|
||||
if (pos >= n) {
|
||||
return EL_STR("");
|
||||
}
|
||||
return native_list_get(tokens, ((pos * 2) + 1));
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* parse_progress_fatal — robustness backstop. Called by the token-consuming
|
||||
driver loops when they detect they have iterated more times than there are
|
||||
tokens (an impossibility for a well-formed program, where every iteration
|
||||
consumes at least one token). Names the offending token and exits non-zero
|
||||
instead of looping forever / exhausting memory. */
|
||||
el_val_t parse_progress_fatal(el_val_t where, el_val_t tokens, el_val_t pos) {
|
||||
el_val_t k = tok_kind(tokens, pos);
|
||||
el_val_t v = tok_value(tokens, pos);
|
||||
println(el_str_concat(el_str_concat(el_str_concat(el_str_concat(
|
||||
EL_STR("elc: FATAL: parser made no forward progress in "), where),
|
||||
EL_STR(" at token index ")), native_int_to_str(pos)),
|
||||
el_str_concat(EL_STR(" (kind="), el_str_concat(k, EL_STR(")")))));
|
||||
println(el_str_concat(el_str_concat(
|
||||
EL_STR("elc: likely a malformed construct near '"), v),
|
||||
EL_STR("' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" ).")));
|
||||
exit(1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t expect(el_val_t tokens, el_val_t pos, el_val_t kind) {
|
||||
el_val_t k = tok_kind(tokens, pos);
|
||||
if (str_eq(k, kind)) {
|
||||
@@ -2689,7 +2727,16 @@ el_val_t parse_block(el_val_t tokens, el_val_t pos) {
|
||||
el_val_t p = expect(tokens, pos, EL_STR("LBrace"));
|
||||
el_val_t stmts = native_list_empty();
|
||||
el_val_t running = 1;
|
||||
/* Runaway backstop: a block can hold at most (token count) statements, since
|
||||
every iteration consumes >= 1 token. If we exceed that, the cursor has run
|
||||
off the end without terminating (malformed input) -> fail fast, don't hang. */
|
||||
el_val_t __blk_total = (native_list_len(tokens) / 2);
|
||||
el_val_t __blk_iters = 0;
|
||||
while (running) {
|
||||
__blk_iters = (__blk_iters + 1);
|
||||
if (__blk_iters > (__blk_total + 8)) {
|
||||
parse_progress_fatal(EL_STR("parse_block"), tokens, p);
|
||||
}
|
||||
el_val_t k = tok_kind(tokens, p);
|
||||
if (str_eq(k, EL_STR("RBrace"))) {
|
||||
running = 0;
|
||||
@@ -4838,9 +4885,51 @@ el_val_t next_if_id(void) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* is_void_builtin — true for runtime builtins declared `void` in el_runtime.h.
|
||||
User `-> Void` functions are emitted as el_val_t (return 0) so they are safe
|
||||
to assign; only these C-level void builtins are not. */
|
||||
el_val_t is_void_builtin(el_val_t name) {
|
||||
if (str_eq(name, EL_STR("println"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("print"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_strengthen"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_forget"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("engram_connect"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("dharma_emit"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("dharma_strengthen"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("llm_register_tool"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("exit_program"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_serve"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_set_handler"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("http_serve_async"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_cgi_init"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_retain"))) { return 1; }
|
||||
if (str_eq(name, EL_STR("el_release"))) { return 1; }
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* cg_expr_is_void — true if `val` is a direct call to a void builtin, so the
|
||||
if-expression arm must emit it as a bare statement rather than assigning its
|
||||
(nonexistent) value to the result var. */
|
||||
el_val_t cg_expr_is_void(el_val_t val) {
|
||||
el_val_t vk = el_get_field(val, EL_STR("expr"));
|
||||
if (str_eq(vk, EL_STR("Call"))) {
|
||||
el_val_t f = el_get_field(val, EL_STR("func"));
|
||||
el_val_t fk = el_get_field(f, EL_STR("expr"));
|
||||
if (str_eq(fk, EL_STR("Ident"))) {
|
||||
return is_void_builtin(el_get_field(f, EL_STR("name")));
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
el_val_t n = native_list_len(stmts);
|
||||
el_val_t parts = native_list_empty();
|
||||
/* Track names already declared in this arm's C block. El permits `let x`
|
||||
to redeclare/rebind x in the same scope, but C forbids redeclaring the
|
||||
same name in one block. Emit `el_val_t x = ...` the first time and a
|
||||
plain `x = ...` reassignment thereafter (mirrors cg_stmt's `declared`). */
|
||||
el_val_t declared = native_list_empty();
|
||||
el_val_t i = 0;
|
||||
while (i < n) {
|
||||
el_val_t s = native_list_get(stmts, i);
|
||||
@@ -4853,18 +4942,31 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
el_val_t name = el_get_field(s, EL_STR("name"));
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
if (list_contains(declared, name)) {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(name, EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
} else {
|
||||
declared = native_list_append(declared, name);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; ")));
|
||||
}
|
||||
} else {
|
||||
if (str_eq(sk, EL_STR("Return"))) {
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
if (cg_expr_is_void(val)) {
|
||||
parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; ")));
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
}
|
||||
} else {
|
||||
if (str_eq(sk, EL_STR("Expr"))) {
|
||||
el_val_t val = el_get_field(s, EL_STR("value"));
|
||||
el_val_t val_c = cg_expr(val);
|
||||
if (is_last) {
|
||||
if (cg_expr_is_void(val)) {
|
||||
parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; ")));
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); ")));
|
||||
}
|
||||
} else {
|
||||
parts = native_list_append(parts, el_str_concat(el_str_concat(EL_STR("(void)("), val_c), EL_STR("); ")));
|
||||
}
|
||||
@@ -4883,6 +4985,7 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) {
|
||||
}
|
||||
el_val_t result = str_join(parts, EL_STR(""));
|
||||
el_release(parts);
|
||||
el_release(declared);
|
||||
return result;
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -6,8 +6,8 @@
|
||||
//
|
||||
// Compile and run:
|
||||
// ./dist/platform/elc examples/html-page.el > /tmp/html-page.c
|
||||
// cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
// -o /tmp/html-page /tmp/html-page.c el-compiler/runtime/el_runtime.c
|
||||
// cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
// -o /tmp/html-page /tmp/html-page.c runtime/el_runtime.c
|
||||
// /tmp/html-page
|
||||
|
||||
fn render_item(item: String) -> String {
|
||||
|
||||
@@ -1,28 +0,0 @@
|
||||
# El Compiler Release v1.0.0 — 2026-05-02
|
||||
|
||||
## Components
|
||||
- `bootstrap.py` — El language compiler (Python, recursive descent parser, emits C)
|
||||
- `el_runtime.c` — El runtime (C, HTTP server, engram, DHARMA, LLM chain)
|
||||
- `el_runtime.h` — Runtime public API header
|
||||
|
||||
## Changes in this release
|
||||
|
||||
### Critical bug fixes
|
||||
- `state_set`/`state_get` are now thread-safe (pthread_mutex). Was racing across 64 worker threads.
|
||||
- `looks_like_string` threshold raised from 1,000,000 to 4GB. Unix timestamps were being dereferenced as heap pointers.
|
||||
- `fs_read` guards against negative `ftell` result (pipe/special file overflow).
|
||||
|
||||
### Engram architecture (major)
|
||||
- Two-layer activation: `background_activation` (Layer 1, broad fan-out) + `working_memory_weight` (Layer 2, executive filter)
|
||||
- Inhibitory edges: `EngramEdge.inhibitory` flag suppresses working memory promotion without affecting background activation
|
||||
- Suppression memory: `suppression_count` — nodes activated-but-suppressed accumulate pressure toward breakthrough
|
||||
- Temporal decay: `temporal_decay_rate`, `created_at`, `last_activated_at`, `activation_count` on EngramNode
|
||||
- Per-type activation thresholds (Safety: 0.05, Canonical: 0.15, Lesson: 0.25, Note: 0.40)
|
||||
- Temporal range query: `engram_query_range(start_ms, end_ms)`
|
||||
- Layered consciousness: `EngramLayer` struct, `layer_id` on nodes and edges, `EngramStore.layers[]`
|
||||
- Layer 0 override pass: safety layer fires last and cannot be suppressed
|
||||
|
||||
## SHA256
|
||||
bootstrap.py
|
||||
el_runtime.c
|
||||
el_runtime.h
|
||||
@@ -132,7 +132,7 @@ if [[ $LVGL_OK -eq 1 ]]; then
|
||||
else
|
||||
_miss "LVGL/MCU" "-DEL_TARGET_LVGL (lvgl.h not found)"
|
||||
echo " Install: git clone https://github.com/lvgl/lvgl"
|
||||
echo " (place lvgl/ next to el-compiler/runtime/)"
|
||||
echo " (place lvgl/ next to runtime/)"
|
||||
MISSING=$((MISSING + 1))
|
||||
fi
|
||||
|
||||
@@ -50,6 +50,16 @@
|
||||
build defines the same helper as close() so the call sites are identical across platforms. */
|
||||
static inline int el_closesocket(SOCKET s) { return closesocket(s); }
|
||||
|
||||
/* ── setsockopt optval type ───────────────────────────────────────────────── */
|
||||
/* Winsock's setsockopt takes optval as (const char*); POSIX takes (const void*), so el_runtime.c
|
||||
passes &int directly. GCC 14+ makes that an error under -Wincompatible-pointer-types. Wrap it so
|
||||
the runtime's POSIX-style call sites compile unchanged (defined before the macro so the wrapper
|
||||
itself resolves to the real winsock setsockopt). */
|
||||
static inline int el_setsockopt(SOCKET s, int level, int optname, const void* optval, int optlen) {
|
||||
return setsockopt(s, level, optname, (const char*)optval, optlen);
|
||||
}
|
||||
#define setsockopt(s, l, o, v, n) el_setsockopt((s), (l), (o), (v), (int)(n))
|
||||
|
||||
/* ── winsock init (once, at load) ─────────────────────────────────────────── */
|
||||
static void el__win_net_init(void) {
|
||||
static int inited = 0;
|
||||
@@ -75,6 +85,7 @@ static inline void* el_win_dlsym(void* handle, const char* name) {
|
||||
#include <direct.h> /* _mkdir */
|
||||
#define mkdir(path, mode) _mkdir(path) /* POSIX mkdir(path,mode) → _mkdir(path) */
|
||||
#define timegm _mkgmtime /* UTC tm → time_t */
|
||||
#define fsync(fd) _commit(fd) /* no fsync() on Windows; _commit() (<io.h>) is the equiv */
|
||||
|
||||
/* setenv/unsetenv: not in the Windows CRT; map to _putenv_s / SetEnvironmentVariable. */
|
||||
static inline int setenv(const char* name, const char* value, int overwrite) {
|
||||
@@ -114,4 +125,63 @@ static inline struct tm* gmtime_r(const time_t* t, struct tm* out) {
|
||||
return gmtime_s(out, t) == 0 ? out : (struct tm*)0;
|
||||
}
|
||||
|
||||
/* ── libcurl: degradable stubs for the curl-less Windows build ─────────────── */
|
||||
/* The curl-less validation build (WITH_CURL=0) links no libcurl. el_runtime.c uses libcurl
|
||||
* unconditionally for its HTTP client / LLM layer; these stubs let it compile and link so the
|
||||
* runtime, HTTP *server*, graph and memory work natively on Windows. Live outbound HTTP/LLM calls
|
||||
* degrade to a runtime error (curl_easy_perform returns an error) — matching the documented
|
||||
* curl-less contract. When HAVE_CURL is defined (WITH_CURL=1) the real <curl/curl.h> is used and
|
||||
* this whole block is compiled out. POSIX never sees this header, so the POSIX build is untouched. */
|
||||
#ifndef HAVE_CURL
|
||||
|
||||
typedef void CURL;
|
||||
typedef int CURLcode;
|
||||
|
||||
#define CURLE_OK 0
|
||||
#define CURLE_HTTP_RETURNED_ERROR 22
|
||||
#define CURL_ERROR_SIZE 256
|
||||
|
||||
/* Option ids: values are irrelevant to the no-op setopt below; kept distinct for readability. */
|
||||
#define CURLOPT_URL 10002
|
||||
#define CURLOPT_WRITEFUNCTION 20011
|
||||
#define CURLOPT_WRITEDATA 10001
|
||||
#define CURLOPT_POSTFIELDS 10015
|
||||
#define CURLOPT_POSTFIELDSIZE 120
|
||||
#define CURLOPT_POST 47
|
||||
#define CURLOPT_HTTPHEADER 10023
|
||||
#define CURLOPT_TIMEOUT_MS 155
|
||||
#define CURLOPT_NOSIGNAL 99
|
||||
#define CURLOPT_USERAGENT 10018
|
||||
#define CURLOPT_FOLLOWLOCATION 52
|
||||
#define CURLOPT_ERRORBUFFER 10010
|
||||
#define CURLOPT_CUSTOMREQUEST 10036
|
||||
#define CURLOPT_FAILONERROR 45
|
||||
|
||||
struct curl_slist { char* data; struct curl_slist* next; };
|
||||
|
||||
static inline struct curl_slist* curl_slist_append(struct curl_slist* list, const char* s) {
|
||||
struct curl_slist* node = (struct curl_slist*)malloc(sizeof(struct curl_slist));
|
||||
if (!node) return list;
|
||||
node->data = s ? strdup(s) : NULL;
|
||||
node->next = NULL;
|
||||
if (!list) return node;
|
||||
struct curl_slist* p = list;
|
||||
while (p->next) p = p->next;
|
||||
p->next = node;
|
||||
return list;
|
||||
}
|
||||
static inline void curl_slist_free_all(struct curl_slist* list) {
|
||||
while (list) { struct curl_slist* n = list->next; free(list->data); free(list); list = n; }
|
||||
}
|
||||
|
||||
static inline CURL* curl_easy_init(void) { return (CURL*)malloc(1); }
|
||||
static inline CURLcode curl_easy_setopt(CURL* h, int opt, ...) { (void)h; (void)opt; return CURLE_OK; }
|
||||
static inline CURLcode curl_easy_perform(CURL* h) { (void)h; return 7 /* CURLE_COULDNT_CONNECT */; }
|
||||
static inline void curl_easy_cleanup(CURL* h) { free(h); }
|
||||
static inline const char* curl_easy_strerror(CURLcode c) {
|
||||
(void)c; return "libcurl not built in (curl-less build)";
|
||||
}
|
||||
|
||||
#endif /* !HAVE_CURL */
|
||||
|
||||
#endif /* EL_PLATFORM_WIN_H */
|
||||
File diff suppressed because it is too large
Load Diff
@@ -34,12 +34,12 @@
|
||||
* pq_hybrid_* and HKDF-SHA256 derivation.
|
||||
*
|
||||
* Canonical compile command:
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
* cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c runtime/el_runtime.c
|
||||
*
|
||||
* With liboqs (post-quantum stack):
|
||||
* cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c el-compiler/runtime/el_runtime.c
|
||||
* cc -std=c11 -I runtime -lcurl -lpthread -loqs -lcrypto \
|
||||
* -o <out> <prog>.c runtime/el_runtime.c
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
@@ -605,6 +605,13 @@ el_val_t engram_edge_count(void);
|
||||
el_val_t engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t engram_save(el_val_t path);
|
||||
el_val_t engram_load(el_val_t path);
|
||||
/* Tiered paged-store entry points (ENGRAM_STORE=1). engram_store_boot opens the
|
||||
* durable store (import-once / WAL-replay) and loads it resident; checkpoint pushes
|
||||
* the resident graph's current field state (incl. learned hebb + activation-formed
|
||||
* edges) through the WAL and flushes; close checkpoints + closes. No-ops when off. */
|
||||
el_val_t engram_store_boot(el_val_t data_dir);
|
||||
el_val_t engram_store_checkpoint(void);
|
||||
el_val_t engram_store_close(void);
|
||||
|
||||
/* JSON-string accessors — return pre-serialized JSON so HTTP handlers
|
||||
* can pass results straight through without round-tripping ElList/ElMap
|
||||
@@ -628,22 +635,30 @@ el_val_t engram_hebb_drain_json(el_val_t max);
|
||||
/* Document frequency of a term across node labels — term-specificity signal
|
||||
* for curiosity seed selection. (2026-08-03 self-review.) */
|
||||
el_val_t engram_label_df(el_val_t term);
|
||||
/* Best curiosity seed from one node: argmax over idf·position·casing across
|
||||
* the candidate tokens of its label, falling back to its content when the
|
||||
* label is a sentinel. Excludes pipe-delimited tabu terms during selection
|
||||
* and gates candidates to the df band [min_df, max_df]. Returns "" when
|
||||
* nothing qualifies. (2026-08-13 self-review.) */
|
||||
el_val_t engram_salient_term(el_val_t node_id, el_val_t max_df,
|
||||
el_val_t min_df, el_val_t tabu);
|
||||
el_val_t engram_embed_backfill(el_val_t count);
|
||||
el_val_t engram_list_layers_json(void);
|
||||
/* Working memory introspection — count, mean weight, and top-N snapshot.
|
||||
* Ported from el-compiler/runtime on 2026-06-30 self-review. */
|
||||
* Ported from runtime on 2026-06-30 self-review. */
|
||||
el_val_t engram_wm_count(void);
|
||||
el_val_t engram_wm_avg_weight(void);
|
||||
el_val_t engram_wm_top_json(el_val_t n);
|
||||
/* Merge-load: add nodes/edges from a snapshot without resetting the store. */
|
||||
el_val_t engram_load_merge(el_val_t path);
|
||||
|
||||
/* ── WAL + compaction + integrity (ENGRAM_WAL=on; design doc §§3-14,§18) ──── */
|
||||
int engram_wal_enabled(void);
|
||||
el_val_t engram_crc32(el_val_t s);
|
||||
el_val_t engram_wal_boot(el_val_t dir); /* replay + open; returns records */
|
||||
el_val_t engram_wal_open_dir(el_val_t dir);
|
||||
el_val_t engram_wal_node_put(el_val_t dir, el_val_t id);
|
||||
el_val_t engram_wal_edges_since(el_val_t dir, el_val_t start_count);
|
||||
el_val_t engram_wal_hebb_batch(el_val_t dir, el_val_t start_count);
|
||||
el_val_t engram_wal_forget(el_val_t dir, el_val_t id);
|
||||
el_val_t engram_wal_compact(el_val_t dir);
|
||||
el_val_t engram_wal_maybe_compact(el_val_t dir);
|
||||
el_val_t engram_resolve_data_dir(void); /* §18.2 fail-loud default */
|
||||
el_val_t engram_is_protected(el_val_t id); /* §18.1/18.3 derived set */
|
||||
el_val_t engram_protected_json(void);
|
||||
/* engram_compile_layered_json — produce a prompt-ready text block split
|
||||
* into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire)
|
||||
* and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if
|
||||
@@ -8,7 +8,7 @@
|
||||
* Threading: __thread_create / __thread_join use dlsym(RTLD_DEFAULT) to look
|
||||
* up El function symbols at runtime. This is the foundation of El's parallelism.
|
||||
*
|
||||
* Link: cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
* Link: cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
* -o <out> <prog>.c el_seed.c
|
||||
*/
|
||||
|
||||
@@ -1072,6 +1072,7 @@ el_val_t __engram_save(el_val_t path) { return engram_save
|
||||
el_val_t __engram_load(el_val_t path) { return engram_load(path); }
|
||||
|
||||
el_val_t __engram_get_node_json(el_val_t id) { return engram_get_node_json(id); }
|
||||
el_val_t __engram_get_node_by_label(el_val_t label) { return engram_get_node_by_label(label); }
|
||||
|
||||
el_val_t __engram_search_json(el_val_t query, el_val_t limit) {
|
||||
return engram_search_json(query, limit);
|
||||
@@ -226,6 +226,7 @@ el_val_t __engram_activate(el_val_t query, el_val_t depth);
|
||||
el_val_t __engram_save(el_val_t path);
|
||||
el_val_t __engram_load(el_val_t path);
|
||||
el_val_t __engram_get_node_json(el_val_t id);
|
||||
el_val_t __engram_get_node_by_label(el_val_t label);
|
||||
el_val_t __engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t __engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t __engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,221 @@
|
||||
/* engram_store.h — M1 of the engram tiered storage engine.
|
||||
*
|
||||
* The FINAL on-disk paged store format: superblock (+ mirror), slotted pages,
|
||||
* self-describing TLV records, overflow chains, and two B+-tree indexes
|
||||
* (primary id->loc, adjacency from_id/to_id->edge-locs) over a free-listed
|
||||
* page file. See docs/architecture/design/engram-tiered-storage-engine.md §2.
|
||||
*
|
||||
* This is a self-contained module (plain C, standard libs only). It defines its
|
||||
* own serializable views of a node/edge (StoreNode/StoreEdge) that mirror every
|
||||
* persisted field of EngramNode/EngramEdge in el_runtime.c. M3 maps between the
|
||||
* live runtime structs and these; M1 does not touch el_runtime.c.
|
||||
*
|
||||
* Format id: magic "ENGST01", format_version 1. This format is PERMANENT — the
|
||||
* TLV record scheme means new fields never force a migration.
|
||||
*/
|
||||
#ifndef ENGRAM_STORE_H
|
||||
#define ENGRAM_STORE_H
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Fixed for the life of a store; recorded in the superblock. */
|
||||
#define STORE_PAGE_SIZE 16384u
|
||||
#define STORE_MAGIC "ENGST01" /* 7 chars + NUL stored in an 8-byte field */
|
||||
#define STORE_FORMAT_VERSION 1u
|
||||
|
||||
/* Ring-buffer length for ACT-R base-level access timestamps.
|
||||
* MUST equal ENGRAM_BLL_K in el_runtime.c (currently 10). Static-checked in .c. */
|
||||
#define STORE_BLL_K 10
|
||||
|
||||
/* Page types (page header byte). */
|
||||
enum {
|
||||
STORE_PT_NODE = 1,
|
||||
STORE_PT_EDGE = 2,
|
||||
STORE_PT_INDEX = 3,
|
||||
STORE_PT_OVERFLOW = 4,
|
||||
STORE_PT_FREE = 5
|
||||
};
|
||||
|
||||
/* store_check flags. */
|
||||
#define STORE_CHECK_CRC 1u
|
||||
|
||||
/* ── Serializable node view: every persisted EngramNode field ─────────────── */
|
||||
typedef struct StoreNode {
|
||||
char* id;
|
||||
char* content;
|
||||
char* node_type;
|
||||
char* label;
|
||||
char* tier;
|
||||
char* tags;
|
||||
char* metadata;
|
||||
double salience;
|
||||
double importance;
|
||||
double confidence;
|
||||
double temporal_decay_rate;
|
||||
int64_t activation_count;
|
||||
int64_t last_activated;
|
||||
int64_t created_at;
|
||||
int64_t updated_at;
|
||||
double background_activation;
|
||||
double working_memory_weight;
|
||||
int32_t suppression_count;
|
||||
uint32_t layer_id;
|
||||
int64_t access_ts[STORE_BLL_K];
|
||||
int32_t access_head;
|
||||
int32_t access_filled;
|
||||
double wm_anchor;
|
||||
float* emb; /* owned; NULL if not embedded */
|
||||
int32_t emb_dim;
|
||||
/* Forward-compat: raw bytes of any TLV fields the reader did not recognise,
|
||||
* concatenated verbatim ([tag][u32 len][bytes]...). Re-emitted on write so
|
||||
* an old reader never drops a newer writer's fields. */
|
||||
uint8_t* unknown;
|
||||
size_t unknown_len;
|
||||
int tombstoned; /* set by store_get_* if the located record is dead */
|
||||
/* hebb_elig / hebb_elig_ts are DELIBERATELY NOT persisted (see EngramNode). */
|
||||
} StoreNode;
|
||||
|
||||
/* ── Serializable edge view: every persisted EngramEdge field ─────────────── */
|
||||
typedef struct StoreEdge {
|
||||
char* id;
|
||||
char* from_id;
|
||||
char* to_id;
|
||||
char* relation;
|
||||
char* metadata;
|
||||
double weight;
|
||||
double hebb;
|
||||
double confidence;
|
||||
int64_t created_at;
|
||||
int64_t updated_at;
|
||||
int64_t last_fired;
|
||||
int32_t inhibitory;
|
||||
uint32_t layer_id;
|
||||
uint8_t* unknown;
|
||||
size_t unknown_len;
|
||||
int tombstoned;
|
||||
} StoreEdge;
|
||||
|
||||
typedef struct EngramPagedStore EngramPagedStore;
|
||||
|
||||
/* Lifecycle. */
|
||||
EngramPagedStore* store_create(const char* path); /* fails if file exists */
|
||||
EngramPagedStore* store_open(const char* path); /* recovers via mirror SB */
|
||||
int store_close(EngramPagedStore* s); /* syncs + frees */
|
||||
int store_sync(EngramPagedStore* s); /* fsync + rewrite both superblocks */
|
||||
|
||||
/* Nodes. store_get_node returns 1 on hit (fills *out, caller store_node_free),
|
||||
* 0 if absent or tombstoned, <0 on error. */
|
||||
int store_put_node(EngramPagedStore* s, const StoreNode* n);
|
||||
int store_get_node(EngramPagedStore* s, const char* id, StoreNode* out);
|
||||
int store_tombstone(EngramPagedStore* s, const char* id);
|
||||
|
||||
/* Edges. *out is malloc'd (store_edges_free); *n set to count. */
|
||||
int store_put_edge(EngramPagedStore* s, const StoreEdge* e);
|
||||
int store_get_edges_from(EngramPagedStore* s, const char* from_id, StoreEdge** out, size_t* n);
|
||||
int store_get_edges_to(EngramPagedStore* s, const char* to_id, StoreEdge** out, size_t* n);
|
||||
|
||||
/* Integrity: verify every page's crc (and both superblocks). Returns the number
|
||||
* of corrupt pages (0 = clean), or <0 on I/O error. */
|
||||
int store_check(EngramPagedStore* s, unsigned flags);
|
||||
|
||||
/* Ownership helpers. */
|
||||
void store_node_free(StoreNode* n);
|
||||
void store_edge_free(StoreEdge* e);
|
||||
void store_edges_free(StoreEdge* arr, size_t n);
|
||||
|
||||
/* Test-only hook (NOT a format property — B+-tree nodes are self-describing via
|
||||
* their stored key count). Caps entries/keys per index node to force splits on
|
||||
* small datasets. 0 = natural full-page fanout. */
|
||||
void store__set_btree_order(EngramPagedStore* s, int leaf_max, int internal_max);
|
||||
|
||||
/* Introspection for tests/tools. */
|
||||
uint64_t store_page_count(const EngramPagedStore* s);
|
||||
|
||||
/* ── M2: WAL + checkpoint + crash recovery + one-time legacy import ─────────────
|
||||
*
|
||||
* The durable engram is `neuron.egm` (paged) fronted by `neuron.wal`
|
||||
* (append-only). A mutation is durable once its WAL record is fsync'd
|
||||
* (group-commit). Pages are held write-back in RAM (no-steal) and flushed to the
|
||||
* store only at a checkpoint, so the store file on disk always reflects a
|
||||
* consistent point (`last_checkpoint_lsn`) and the WAL owns everything since.
|
||||
* Recovery = open store, replay WAL forward, redo a record only where the target
|
||||
* record's home page LSN < record LSN (idempotent). JSON is ONLY an import
|
||||
* source / export artifact — never the ongoing store. */
|
||||
|
||||
typedef enum { ENGRAM_WAL_ALWAYS = 0, ENGRAM_WAL_GROUP = 1, ENGRAM_WAL_OFF = 2 } EngramWalSync;
|
||||
|
||||
/* Serializable layer-registry view (the `layers` array of the legacy snapshot). */
|
||||
typedef struct StoreLayer {
|
||||
uint32_t layer_id;
|
||||
char* name;
|
||||
uint32_t activation_priority;
|
||||
int32_t suppressible;
|
||||
int32_t transparent;
|
||||
int32_t injectable;
|
||||
uint8_t* unknown;
|
||||
size_t unknown_len;
|
||||
int tombstoned;
|
||||
} StoreLayer;
|
||||
|
||||
/* Boot the durable engram in `data_dir` (holds neuron.egm + neuron.wal). If the
|
||||
* store is absent but a legacy snapshot.json exists, it is imported ONCE into a
|
||||
* fresh store; thereafter the store is authoritative and JSON is never read again.
|
||||
* On open, the WAL is replayed to recover any post-checkpoint mutations. */
|
||||
EngramPagedStore* engram_open(const char* data_dir);
|
||||
int engram_close(EngramPagedStore* s); /* checkpoint + close */
|
||||
|
||||
/* Force a checkpoint: flush dirty pages → fsync store → advance checkpoint LSN →
|
||||
* reclaim the WAL prefix. Also threshold-triggered automatically on the write path. */
|
||||
int engram_checkpoint(EngramPagedStore* s);
|
||||
|
||||
/* WAL commit policy. engram_open honours env ENGRAM_WAL_SYNC=always|group|off. */
|
||||
void engram_set_wal_sync(EngramPagedStore* s, EngramWalSync policy);
|
||||
|
||||
/* Layer registry. */
|
||||
int store_put_layer(EngramPagedStore* s, const StoreLayer* L);
|
||||
int store_get_layer(EngramPagedStore* s, uint32_t layer_id, StoreLayer* out);
|
||||
int store_del_layer(EngramPagedStore* s, uint32_t layer_id);
|
||||
int store_list_layers(EngramPagedStore* s, StoreLayer** out, size_t* n);
|
||||
void store_layer_free(StoreLayer* L);
|
||||
void store_layers_free(StoreLayer* arr, size_t n);
|
||||
|
||||
/* Edge lookup by id (for hebb updates + idempotency). 1 hit / 0 absent / <0 err. */
|
||||
int store_get_edge(EngramPagedStore* s, const char* id, StoreEdge* out);
|
||||
|
||||
/* HEBB batch: one WAL record updating hebb (+ last_fired) on a set of edges. */
|
||||
typedef struct StoreHebbDelta { const char* edge_id; double hebb; int64_t last_fired; } StoreHebbDelta;
|
||||
int store_hebb_batch(EngramPagedStore* s, const StoreHebbDelta* d, size_t n);
|
||||
|
||||
/* Supersede: logs the (old,new) pair and tombstones old_id at the store; the new
|
||||
* node + `supersedes` edge are logged separately (neuron-layer immutability). */
|
||||
int store_supersede(EngramPagedStore* s, const char* old_id, const char* new_id);
|
||||
|
||||
/* Forget (GC): tombstone id at the store (hard-free deferred to compaction). */
|
||||
int store_forget(EngramPagedStore* s, const char* id);
|
||||
|
||||
/* ── M3: full live enumeration (for the CALLER's resident load + JSON export) ──
|
||||
* Walk the whole store and invoke `cb` once per DISTINCT live node/edge with a
|
||||
* borrowed view (the engine frees it after cb returns — the callback must copy
|
||||
* anything it keeps). De-duplicated by id (canonical latest-live per id, matching
|
||||
* point-read semantics). Returns the count emitted, or <0 on error. The engine
|
||||
* hands out StoreNode/StoreEdge only — it never sees a soul struct (design §10). */
|
||||
typedef void (*StoreNodeScanCb)(const StoreNode* n, void* ctx);
|
||||
typedef void (*StoreEdgeScanCb)(const StoreEdge* e, void* ctx);
|
||||
int store_scan_nodes(EngramPagedStore* s, StoreNodeScanCb cb, void* ctx);
|
||||
int store_scan_edges(EngramPagedStore* s, StoreEdgeScanCb cb, void* ctx);
|
||||
|
||||
/* Introspection / test hooks. */
|
||||
uint64_t engram_wal_next_lsn(const EngramPagedStore* s);
|
||||
uint64_t engram_last_checkpoint_lsn(const EngramPagedStore* s);
|
||||
|
||||
/* Crash-test hooks (writes only under a throwaway dir).
|
||||
* store__crash — abandon all RAM state without flush/fsync (power loss).
|
||||
* store__flush_pages — pwrite dirty pages to disk WITHOUT a checkpoint (steal).
|
||||
* store__checkpoint_crashat — run checkpoint but stop (then power-loss) after
|
||||
* `phase` (0..4); phase<0 = full checkpoint. */
|
||||
void store__crash(EngramPagedStore* s);
|
||||
int store__flush_pages(EngramPagedStore* s);
|
||||
int store__checkpoint_crashat(EngramPagedStore* s, int phase);
|
||||
|
||||
#endif /* ENGRAM_STORE_H */
|
||||
@@ -2,7 +2,7 @@
|
||||
//
|
||||
// Thin El wrappers over seed JSON primitives, plus pure-El builders and
|
||||
// helpers. Each function here corresponds to (and replaces) a C function
|
||||
// from el-compiler/runtime/legacy/el_runtime.c (lines 2692–3333).
|
||||
// from runtime/el_runtime.c (lines 2692–3333).
|
||||
//
|
||||
// Seed primitives consumed by this module:
|
||||
// __json_get(json, key) -> String (value as string)
|
||||
|
||||
@@ -25,8 +25,8 @@
|
||||
// runtime/collections.el \
|
||||
// <user-program.el> > combined.el
|
||||
// ./dist/platform/elc combined.el > output.c
|
||||
// cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
// -o output output.c el-compiler/runtime/el_seed.c
|
||||
// cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
// -o output output.c runtime/el_seed.c
|
||||
|
||||
// This file itself is not compiled — it is documentation only.
|
||||
fn runtime_version() -> String {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// runtime/math.el — Float math, integer utilities, and numeric conversions.
|
||||
//
|
||||
// Implements the math/float surface from el-compiler/runtime/legacy/el_runtime.c
|
||||
// Implements the math/float surface from runtime/el_runtime.c
|
||||
// (lines 303–305 for el_abs/max/min, lines 4725–4771 for float/format ops)
|
||||
// in pure El, using seed primitives.
|
||||
//
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// runtime/time.el — Time operations, sleep, and formatting.
|
||||
//
|
||||
// Implements the time surface from el-compiler/runtime/legacy/el_runtime.c
|
||||
// Implements the time surface from runtime/el_runtime.c
|
||||
// (lines 3334–3440, 3471–3656) in pure El, using seed primitives.
|
||||
//
|
||||
// Seed primitives consumed:
|
||||
|
||||
@@ -11,7 +11,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)}"
|
||||
ELC="${EL_HOME}/dist/platform/elc"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
echo "elc not found at ${ELC}" >&2
|
||||
|
||||
@@ -10,7 +10,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)}"
|
||||
ELC="${EL_HOME}/dist/platform/elc"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
echo "elc not found at ${ELC}" >&2
|
||||
|
||||
@@ -14,8 +14,8 @@
|
||||
// tests/runtime/string_test.el > /tmp/string_test_combined.el
|
||||
//
|
||||
// ./dist/platform/elc /tmp/string_test_combined.el > /tmp/string_test.c
|
||||
// cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \
|
||||
// -o /tmp/string_test /tmp/string_test.c el-compiler/runtime/el_seed.c
|
||||
// cc -std=c11 -I runtime -lcurl -lpthread \
|
||||
// -o /tmp/string_test /tmp/string_test.c runtime/el_seed.c
|
||||
// /tmp/string_test; echo "exit: $?"
|
||||
//
|
||||
// Exit code equals the number of failing assertions (0 = all pass).
|
||||
|
||||
@@ -11,7 +11,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)}"
|
||||
ELC="${ELC:-${EL_HOME}/dist/platform/elc}"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
echo "elc not found at ${ELC}" >&2
|
||||
|
||||
@@ -16,7 +16,7 @@ cd "$(dirname "$0")"
|
||||
|
||||
EL_HOME="${EL_HOME:-$(cd ../.. && pwd)}"
|
||||
ELC="${EL_HOME}/dist/platform/elc"
|
||||
RUNTIME_DIR="${EL_HOME}/el-compiler/runtime"
|
||||
RUNTIME_DIR="${EL_HOME}/runtime"
|
||||
|
||||
if [ ! -x "${ELC}" ]; then
|
||||
echo "elc not found at ${ELC}" >&2
|
||||
|
||||
@@ -44,7 +44,7 @@ INCLUDE_DIR="${PREFIX}/include"
|
||||
LIB_DIR="${PREFIX}/lib"
|
||||
|
||||
ELC_SRC="${EL_ROOT}/dist/platform/elc"
|
||||
RUNTIME_SRC="${EL_ROOT}/el-compiler/runtime"
|
||||
RUNTIME_SRC="${EL_ROOT}/runtime"
|
||||
STDLIB_SRC="${EL_ROOT}/runtime"
|
||||
|
||||
echo "==> Installing El framework to ${PREFIX}"
|
||||
|
||||
@@ -42,7 +42,7 @@ fi
|
||||
# Discover el_runtime.c
|
||||
EL_RUNTIME=""
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
LOCAL_RUNTIME="${SCRIPT_DIR}/../el-compiler/runtime/el_runtime.c"
|
||||
LOCAL_RUNTIME="${SCRIPT_DIR}/../runtime/el_runtime.c"
|
||||
if [[ -f "${LOCAL_RUNTIME}" ]]; then
|
||||
EL_RUNTIME="$(cd "$(dirname "${LOCAL_RUNTIME}")" && pwd)/$(basename "${LOCAL_RUNTIME}")"
|
||||
EL_INCLUDE="$(dirname "${EL_RUNTIME}")"
|
||||
|
||||
Executable
+94
@@ -0,0 +1,94 @@
|
||||
#!/usr/bin/env bash
|
||||
# check-single-runtime.sh — CODE-VS-ARTIFACT drift guard for the el runtime.
|
||||
#
|
||||
# Enforces org policy docs/CODE-VS-ARTIFACT.md rule #1 (single source of truth):
|
||||
# there is exactly ONE authored el_runtime.c, and it lives at lang/runtime/.
|
||||
# Any other el_runtime.c in the tree is a fork (a hand-synced copy). A lagging
|
||||
# fork is exactly what shipped to prod and dropped learned `hebb` edges on
|
||||
# restart — this guard exists to make that class of bug impossible to reintroduce.
|
||||
#
|
||||
# Exemptions:
|
||||
# * Build output — generated amalgamations under any dist/ or build/ dir are
|
||||
# artifacts, not sources.
|
||||
# * A small, explicit ALLOWLIST of pre-existing example-app vendored/staging
|
||||
# copies (see below). These are KNOWN DEFERRED DEBT, tracked separately from
|
||||
# the SDK/CI-published runtime. They do NOT ship to prod. The guard warns on
|
||||
# them (visible, greppable) but does not fail — while HARD-FAILING on any new
|
||||
# or non-allowlisted fork, including any return of lang/el-compiler/runtime/
|
||||
# or a lang/releases/ vendored copy.
|
||||
#
|
||||
# Wire-in: run from the repo root in CI (see note at bottom). Exits non-zero on drift.
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
cd "$ROOT"
|
||||
|
||||
CANONICAL="lang/runtime/el_runtime.c"
|
||||
|
||||
# KNOWN DEFERRED example-app forks — remove these as a follow-up, then delete
|
||||
# this allowlist. iOS + Docker copies are regenerated by their build scripts
|
||||
# (cp from lang/runtime) and can be git-rm'd now; the Android jni/ copy is a
|
||||
# committed source its CMake build depends on and needs a build-script change
|
||||
# (cp from lang/runtime) before removal. Tracked in docs/CODE-VS-ARTIFACT.md.
|
||||
ALLOWLIST=(
|
||||
"ui/examples/native-hello-android/app/src/main/jni/el_runtime.c"
|
||||
"ui/examples/native-hello-ios/NativeHello/el_runtime.c"
|
||||
"ui/examples/native-hello/build-docker/runtime/el_runtime.c"
|
||||
)
|
||||
|
||||
is_allowlisted() {
|
||||
local p="$1"
|
||||
for a in "${ALLOWLIST[@]}"; do [ "$p" = "$a" ] && return 0; done
|
||||
return 1
|
||||
}
|
||||
|
||||
if [ ! -f "$CANONICAL" ]; then
|
||||
echo "FATAL: canonical runtime source missing: $CANONICAL" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# All el_runtime.c files tracked by git, excluding build output (dist/ , build/)
|
||||
# and the canonical source itself.
|
||||
mapfile -t CANDIDATES < <(
|
||||
git ls-files '*el_runtime.c' \
|
||||
| grep -Ev '(^|/)(dist|build)/' \
|
||||
| grep -vx "$CANONICAL" || true
|
||||
)
|
||||
|
||||
FORKS=()
|
||||
DEFERRED=()
|
||||
for f in "${CANDIDATES[@]}"; do
|
||||
if is_allowlisted "$f"; then DEFERRED+=("$f"); else FORKS+=("$f"); fi
|
||||
done
|
||||
|
||||
if [ "${#DEFERRED[@]}" -gt 0 ]; then
|
||||
echo "WARN: allowlisted (deferred) el_runtime.c forks still present — clean these up:" >&2
|
||||
for f in "${DEFERRED[@]}"; do echo " - $f" >&2; done
|
||||
fi
|
||||
|
||||
if [ "${#FORKS[@]}" -gt 0 ]; then
|
||||
echo "FATAL: el_runtime.c fork(s) detected outside the canonical location." >&2
|
||||
echo " Canonical (the ONLY allowed source): $CANONICAL" >&2
|
||||
echo " Offending copies:" >&2
|
||||
for f in "${FORKS[@]}"; do echo " - $f" >&2; done
|
||||
echo "" >&2
|
||||
echo "Consumers must build against $CANONICAL (pin by git ref where" >&2
|
||||
echo "reproducibility matters) — never a hand-maintained copy." >&2
|
||||
echo "See docs/CODE-VS-ARTIFACT.md." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "OK: single canonical runtime source — $CANONICAL (no un-allowlisted forks)."
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CI wire-in:
|
||||
# foundation/el .gitea/workflows/ci-dev.yaml, ci-stage.yaml, sdk-release.yaml
|
||||
# Add an early step (before the build/publish steps). It must run from the
|
||||
# REPO ROOT, so override the job's `defaults.run.working-directory: lang`:
|
||||
#
|
||||
# - name: Guard - single canonical runtime source
|
||||
# working-directory: ${{ github.workspace }}
|
||||
# run: bash scripts/check-single-runtime.sh
|
||||
#
|
||||
# Also add to .githooks/pre-commit so drift is caught before it is committed.
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -18,7 +18,7 @@ set -euo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
EL_LANG_ROOT="${SCRIPT_DIR}/../../../lang"
|
||||
EL_UI_ROOT="${SCRIPT_DIR}/../.."
|
||||
EL_RUNTIME="${EL_LANG_ROOT}/el-compiler/runtime"
|
||||
EL_RUNTIME="${EL_LANG_ROOT}/runtime"
|
||||
EL_NATIVE_VESSEL="${EL_UI_ROOT}/vessels/el-native/src/main.el"
|
||||
EL_APP_ENTRY="${EL_UI_ROOT}/examples/native-hello/src/main.el"
|
||||
EL_MANIFEST="${EL_UI_ROOT}/examples/native-hello/manifest.el"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user