diff --git a/.gitea/workflows/ci-dev.yaml b/.gitea/workflows/ci-dev.yaml index 092a80e..d4194b8 100644 --- a/.gitea/workflows/ci-dev.yaml +++ b/.gitea/workflows/ci-dev.yaml @@ -39,9 +39,9 @@ jobs: run: | dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elc-gen2.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/platform/elc chmod +x dist/platform/elc @@ -54,9 +54,9 @@ jobs: mkdir -p dist/bin dist/platform/elc elb.el > dist/elb.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elb.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/bin/elb chmod +x dist/bin/elb @@ -91,7 +91,7 @@ jobs: - name: Precompile el_runtime.o run: | set -euo pipefail - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" gcc -O2 -c -I "$RUNTIME" "$RUNTIME/el_runtime.c" \ -o /tmp/el_runtime.o echo "el_runtime.o compiled" @@ -100,7 +100,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core @@ -110,7 +110,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text @@ -120,7 +120,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string @@ -130,7 +130,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math @@ -140,7 +140,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state @@ -150,7 +150,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time @@ -160,7 +160,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json @@ -170,7 +170,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env @@ -180,7 +180,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c /tmp/el_runtime.o \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs @@ -191,7 +191,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/epm @@ -202,7 +202,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/el-install @@ -251,7 +251,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-c \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.c + --source=runtime/el_runtime.c gcloud artifacts generic upload \ --repository=foundation-dev \ @@ -259,7 +259,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-h \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.h + --source=runtime/el_runtime.h gcloud artifacts generic upload \ --repository=foundation-dev \ @@ -267,7 +267,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-js \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.js + --source=runtime/el_runtime.js echo "Published El SDK version=${VERSION} to foundation-dev" # Keep key alive for the ci-base rebuild step below @@ -306,9 +306,9 @@ jobs: FROM ${BASE} COPY dist/platform/elc /opt/el/dist/platform/elc COPY dist/bin/elb /opt/el/dist/bin/elb - COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c - COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h - COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js + COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c + COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h + COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb EOF diff --git a/.gitea/workflows/ci-stage.yaml b/.gitea/workflows/ci-stage.yaml index 281e9bb..3280050 100644 --- a/.gitea/workflows/ci-stage.yaml +++ b/.gitea/workflows/ci-stage.yaml @@ -46,9 +46,9 @@ jobs: run: | dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elc-gen2.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/platform/elc chmod +x dist/platform/elc @@ -84,7 +84,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core @@ -94,7 +94,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text @@ -104,7 +104,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string @@ -114,7 +114,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math @@ -124,7 +124,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state @@ -134,7 +134,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time @@ -144,7 +144,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json @@ -154,7 +154,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env @@ -164,7 +164,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs @@ -176,9 +176,9 @@ jobs: mkdir -p dist/bin dist/platform/elc elb.el > dist/elb.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elb.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/bin/elb chmod +x dist/bin/elb @@ -189,7 +189,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/epm @@ -200,7 +200,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/el-install @@ -244,7 +244,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-c \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.c + --source=runtime/el_runtime.c gcloud artifacts generic upload \ --repository=foundation-stage \ @@ -252,7 +252,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-h \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.h + --source=runtime/el_runtime.h echo "Published El SDK version=${VERSION} to foundation-stage" # Keep key alive for the ci-base rebuild step below @@ -290,9 +290,9 @@ jobs: FROM ${BASE} COPY dist/platform/elc /opt/el/dist/platform/elc COPY dist/bin/elb /opt/el/dist/bin/elb - COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c - COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h - COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js + COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c + COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h + COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb EOF diff --git a/.gitea/workflows/sdk-release.yaml b/.gitea/workflows/sdk-release.yaml index 66d9c3a..d8eb0c7 100644 --- a/.gitea/workflows/sdk-release.yaml +++ b/.gitea/workflows/sdk-release.yaml @@ -47,9 +47,9 @@ jobs: mkdir -p dist/platform dist/platform/elc-linux-amd64 elc-cli.el > dist/elc-gen2.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elc-gen2.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/platform/elc chmod +x dist/platform/elc @@ -62,9 +62,9 @@ jobs: mkdir -p dist/bin dist/platform/elc elb.el > dist/elb.c gcc -O2 \ - -I el-compiler/runtime \ + -I runtime \ dist/elb.c \ - el-compiler/runtime/el_runtime.c \ + runtime/el_runtime.c \ -lcurl -lssl -lcrypto -lpthread -lm \ -o dist/bin/elb chmod +x dist/bin/elb @@ -75,7 +75,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd ../epm && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/epm @@ -86,7 +86,7 @@ jobs: run: | ABS_ELB="$(pwd)/dist/bin/elb" ABS_ELC="$(pwd)/dist/platform/elc" - ABS_RUNTIME="$(pwd)/el-compiler/runtime" + ABS_RUNTIME="$(pwd)/runtime" ABS_OUT="$(pwd)/dist/bin" (cd tools/install && "$ABS_ELB" --clean --elc="$ABS_ELC" --runtime="$ABS_RUNTIME" --out="$ABS_OUT") chmod +x dist/bin/el-install @@ -121,7 +121,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core @@ -131,7 +131,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text @@ -141,7 +141,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string @@ -151,7 +151,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math @@ -161,7 +161,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state @@ -171,7 +171,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time @@ -181,7 +181,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json @@ -191,7 +191,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env @@ -201,7 +201,7 @@ jobs: run: | set -euo pipefail ELC="$(pwd)/dist/platform/elc" - RUNTIME="$(pwd)/el-compiler/runtime" + RUNTIME="$(pwd)/runtime" "$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \ -lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs @@ -216,8 +216,10 @@ jobs: cp lang/dist/platform/elc dist/sdk/bin/elc cp lang/dist/bin/elb dist/sdk/bin/elb cp lang/dist/bin/epm dist/sdk/bin/epm - cp lang/el-compiler/runtime/el_runtime.c dist/sdk/runtime/ - cp lang/el-compiler/runtime/el_runtime.h dist/sdk/runtime/ + cp lang/runtime/el_runtime.c dist/sdk/runtime/ + cp lang/runtime/el_runtime.h dist/sdk/runtime/ + cp lang/runtime/engram_store.c dist/sdk/runtime/ + cp lang/runtime/engram_store.h dist/sdk/runtime/ cp lang/runtime/*.el dist/sdk/runtime/ tar -czf dist/el-sdk-latest.tar.gz -C dist/sdk . echo "SDK tarball bundled: dist/el-sdk-latest.tar.gz" @@ -274,8 +276,10 @@ jobs: # Per-file assets (downstream CI needs these individually) upload_asset lang/dist/platform/elc elc - upload_asset lang/el-compiler/runtime/el_runtime.c el_runtime.c - upload_asset lang/el-compiler/runtime/el_runtime.h el_runtime.h + upload_asset lang/runtime/el_runtime.c el_runtime.c + upload_asset lang/runtime/el_runtime.h el_runtime.h + upload_asset lang/runtime/engram_store.c engram_store.c + upload_asset lang/runtime/engram_store.h engram_store.h # SDK bundle and installer binary upload_asset dist/el-sdk-latest.tar.gz el-sdk-latest.tar.gz @@ -328,7 +332,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-c \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.c + --source=runtime/el_runtime.c gcloud artifacts generic upload \ --repository=foundation-prod \ @@ -336,7 +340,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-h \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.h + --source=runtime/el_runtime.h gcloud artifacts generic upload \ --repository=foundation-prod \ @@ -344,7 +348,7 @@ jobs: --project=neuron-785695 \ --package=el-runtime-js \ --version="${VERSION}" \ - --source=el-compiler/runtime/el_runtime.js + --source=runtime/el_runtime.js echo "Published El SDK version=${VERSION} to foundation-prod" # Keep key alive for the ci-base rebuild step below @@ -382,9 +386,9 @@ jobs: FROM ${BASE} COPY dist/platform/elc /opt/el/dist/platform/elc COPY dist/bin/elb /opt/el/dist/bin/elb - COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c - COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h - COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js + COPY runtime/el_runtime.c /opt/el/runtime/el_runtime.c + COPY runtime/el_runtime.h /opt/el/runtime/el_runtime.h + COPY runtime/el_runtime.js /opt/el/runtime/el_runtime.js RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb EOF diff --git a/.githooks/pre-commit b/.githooks/pre-commit index bbea4e6..bc78c05 100755 --- a/.githooks/pre-commit +++ b/.githooks/pre-commit @@ -6,13 +6,13 @@ set -euo pipefail ROOT="$(git rev-parse --show-toplevel)" LANG_DIR="$ROOT/lang" -RUNTIME="$LANG_DIR/el-compiler/runtime" +RUNTIME="$LANG_DIR/runtime" ELC="$LANG_DIR/dist/platform/elc" # If elc isn't built yet, skip with a warning rather than blocking if [ ! -x "$ELC" ]; then echo "⚠ elc not found at lang/dist/platform/elc — skipping pre-commit tests" - echo " Build it first: cd lang && gcc -O2 -I el-compiler/runtime dist/elc-bootstrap.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I el-compiler/runtime /tmp/elc.c el-compiler/runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc" + echo " Build it first: cd lang && gcc -O2 -I runtime dist/elc-bootstrap.c runtime/el_runtime.c -lcurl -lpthread -o dist/elc-gen2 && ./dist/elc-gen2 el-compiler/src/compiler.el > /tmp/elc.c && gcc -O2 -I runtime /tmp/elc.c runtime/el_runtime.c -lcurl -lpthread -o dist/platform/elc" exit 0 fi diff --git a/elp/tests/run.sh b/elp/tests/run.sh index ed423b6..9653953 100755 --- a/elp/tests/run.sh +++ b/elp/tests/run.sh @@ -22,7 +22,7 @@ cd "$(dirname "$0")" EL_HOME="${EL_HOME:-$(cd ../.. && pwd)/el}" ELC="${ELC:-${EL_HOME}/dist/platform/elc}" -RUNTIME_DIR="${EL_HOME}/el-compiler/runtime" +RUNTIME_DIR="${EL_HOME}/runtime" SRC_DIR="$(cd .. && pwd)/src" if [ ! -x "${ELC}" ]; then diff --git a/engram/.gitea/workflows/engram-release.yaml b/engram/.gitea/workflows/engram-release.yaml index aa02e9f..fcd241f 100644 --- a/engram/.gitea/workflows/engram-release.yaml +++ b/engram/.gitea/workflows/engram-release.yaml @@ -49,6 +49,12 @@ jobs: echo "Downloading el_runtime.h..." curl -fsSL "${RELEASE_BASE}/el_runtime.h" -o /usr/local/lib/el/el_runtime.h + echo "Downloading engram_store.c..." + curl -fsSL "${RELEASE_BASE}/engram_store.c" -o /usr/local/lib/el/engram_store.c + + echo "Downloading engram_store.h..." + curl -fsSL "${RELEASE_BASE}/engram_store.h" -o /usr/local/lib/el/engram_store.h + echo "El SDK installed:" elc --version || true @@ -67,6 +73,7 @@ jobs: -o dist/engram \ dist/engram.c \ /usr/local/lib/el/el_runtime.c \ + /usr/local/lib/el/engram_store.c \ -lcurl -lpthread echo "Linked dist/engram" ls -lh dist/engram diff --git a/engram/.gitignore b/engram/.gitignore index e88e923..8ddf8ec 100644 --- a/engram/.gitignore +++ b/engram/.gitignore @@ -1,3 +1,6 @@ -target/ -*.db .DS_Store +*.db +*.elc +*.elh +dist/ +target/ diff --git a/engram/dist/engram b/engram/dist/engram index 988e3cd..cfde628 100755 Binary files a/engram/dist/engram and b/engram/dist/engram differ diff --git a/engram/dist/engram.c b/engram/dist/engram.c index f43fca5..eebdc34 100644 --- a/engram/dist/engram.c +++ b/engram/dist/engram.c @@ -10,6 +10,8 @@ el_val_t query_param(el_val_t path, el_val_t key); el_val_t query_int(el_val_t path, el_val_t key, el_val_t default_val); el_val_t extract_id(el_val_t path, el_val_t prefix); el_val_t route_stats(el_val_t method, el_val_t path, el_val_t body); +el_val_t route_act_stats(el_val_t method, el_val_t path, el_val_t body); +el_val_t route_text_health(el_val_t method, el_val_t path, el_val_t body); el_val_t persist_canonical(void); el_val_t route_create_node(el_val_t method, el_val_t path, el_val_t body); el_val_t route_get_node(el_val_t method, el_val_t path, el_val_t body); @@ -18,16 +20,19 @@ el_val_t route_scan_edges(el_val_t method, el_val_t path, el_val_t body); el_val_t route_search(el_val_t method, el_val_t path, el_val_t body); el_val_t route_activate(el_val_t method, el_val_t path, el_val_t body); el_val_t route_create_edge(el_val_t method, el_val_t path, el_val_t body); +el_val_t route_create_edges_batch(el_val_t method, el_val_t path, el_val_t body); el_val_t route_neighbors(el_val_t method, el_val_t path, el_val_t body); el_val_t route_strengthen(el_val_t method, el_val_t path, el_val_t body); el_val_t route_forget(el_val_t method, el_val_t path, el_val_t body); el_val_t route_save(el_val_t method, el_val_t path, el_val_t body); el_val_t route_load(el_val_t method, el_val_t path, el_val_t body); el_val_t route_health(el_val_t method, el_val_t path, el_val_t body); +el_val_t route_embed_backfill(el_val_t method, el_val_t path, el_val_t body); el_val_t route_sync(el_val_t method, el_val_t path, el_val_t body); el_val_t route_load_merge(el_val_t method, el_val_t path, el_val_t body); el_val_t route_emit_ise(el_val_t method, el_val_t path, el_val_t body); el_val_t route_capture_knowledge(el_val_t method, el_val_t path, el_val_t body); +el_val_t route_similarity(el_val_t method, el_val_t path, el_val_t body); el_val_t check_auth_ok(el_val_t method, el_val_t body); el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body); @@ -116,11 +121,20 @@ el_val_t route_stats(el_val_t method, el_val_t path, el_val_t body) { return 0; } +el_val_t route_act_stats(el_val_t method, el_val_t path, el_val_t body) { + return engram_act_stats_json(); + return 0; +} + +el_val_t route_text_health(el_val_t method, el_val_t path, el_val_t body) { + return engram_text_health_json(); + return 0; +} + el_val_t persist_canonical(void) { el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); el_val_t dir = ({ el_val_t _if_result_1 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_1 = (EL_STR("/tmp/engram")); } else { _if_result_1 = (dir_raw); } _if_result_1; }); - engram_save(el_str_concat(dir, EL_STR("/snapshot.json"))); - return 1; + return engram_save(el_str_concat(dir, EL_STR("/snapshot.json"))); return 0; } @@ -128,9 +142,18 @@ el_val_t route_create_node(el_val_t method, el_val_t path, el_val_t body) { el_val_t content = json_get_string(body, EL_STR("content")); el_val_t nt_raw = json_get_string(body, EL_STR("node_type")); el_val_t node_type = ({ el_val_t _if_result_2 = 0; if (str_eq(nt_raw, EL_STR(""))) { _if_result_2 = (EL_STR("Memory")); } else { _if_result_2 = (nt_raw); } _if_result_2; }); - el_val_t sal_raw = json_get_float(body, EL_STR("salience")); - el_val_t salience = ({ el_val_t _if_result_3 = 0; if ((sal_raw == el_from_float(0.0))) { _if_result_3 = (el_from_float(0.5)); } else { _if_result_3 = (sal_raw); } _if_result_3; }); - el_val_t id = engram_node(content, node_type, salience); + el_val_t sal_present = json_get_raw(body, EL_STR("salience")); + el_val_t salience = ({ el_val_t _if_result_3 = 0; if (str_eq(sal_present, EL_STR(""))) { _if_result_3 = (el_from_float(0.5)); } else { _if_result_3 = (json_get_float(body, EL_STR("salience"))); } _if_result_3; }); + el_val_t label_raw = json_get_string(body, EL_STR("label")); + el_val_t label = ({ el_val_t _if_result_4 = 0; if (str_eq(label_raw, EL_STR(""))) { _if_result_4 = (content); } else { _if_result_4 = (label_raw); } _if_result_4; }); + el_val_t imp_present = json_get_raw(body, EL_STR("importance")); + el_val_t importance = ({ el_val_t _if_result_5 = 0; if (str_eq(imp_present, EL_STR(""))) { _if_result_5 = (el_from_float(0.5)); } else { _if_result_5 = (json_get_float(body, EL_STR("importance"))); } _if_result_5; }); + el_val_t conf_present = json_get_raw(body, EL_STR("confidence")); + el_val_t confidence = ({ el_val_t _if_result_6 = 0; if (str_eq(conf_present, EL_STR(""))) { _if_result_6 = (el_from_float(1.0)); } else { _if_result_6 = (json_get_float(body, EL_STR("confidence"))); } _if_result_6; }); + el_val_t tier_raw = json_get_string(body, EL_STR("tier")); + el_val_t tier = ({ el_val_t _if_result_7 = 0; if (str_eq(tier_raw, EL_STR(""))) { _if_result_7 = (EL_STR("Working")); } else { _if_result_7 = (tier_raw); } _if_result_7; }); + el_val_t tags = json_get_string(body, EL_STR("tags")); + el_val_t id = engram_node_full(content, node_type, label, salience, importance, confidence, tier, tags); el_val_t saved = persist_canonical(); return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"id\":\""), id), EL_STR("\",\"content\":\"")), content), EL_STR("\",\"node_type\":\"")), node_type), EL_STR("\"}")); return 0; @@ -158,7 +181,7 @@ el_val_t route_scan_nodes(el_val_t method, el_val_t path, el_val_t body) { el_val_t route_scan_edges(el_val_t method, el_val_t path, el_val_t body) { el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); - el_val_t dir = ({ el_val_t _if_result_4 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_4 = (EL_STR("/tmp/engram")); } else { _if_result_4 = (dir_raw); } _if_result_4; }); + el_val_t dir = ({ el_val_t _if_result_8 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_8 = (EL_STR("/tmp/engram")); } else { _if_result_8 = (dir_raw); } _if_result_8; }); el_val_t snap_path = el_str_concat(dir, EL_STR("/.scan-export.json")); engram_save(snap_path); el_val_t snap = fs_read(snap_path); @@ -174,22 +197,22 @@ el_val_t route_scan_edges(el_val_t method, el_val_t path, el_val_t body) { } el_val_t route_search(el_val_t method, el_val_t path, el_val_t body) { - el_val_t q = ({ el_val_t _if_result_5 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_5 = (query_param(path, EL_STR("q"))); } else { _if_result_5 = (json_get_string(body, EL_STR("query"))); } _if_result_5; }); + el_val_t q = ({ el_val_t _if_result_9 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_9 = (query_param(path, EL_STR("q"))); } else { _if_result_9 = (json_get_string(body, EL_STR("query"))); } _if_result_9; }); el_val_t lim_url = query_int(path, EL_STR("limit"), 0); el_val_t lim_body = json_get_int(body, EL_STR("limit")); - el_val_t lim_either = ({ el_val_t _if_result_6 = 0; if ((lim_url > 0)) { _if_result_6 = (lim_url); } else { _if_result_6 = (lim_body); } _if_result_6; }); - el_val_t limit = ({ el_val_t _if_result_7 = 0; if ((lim_either > 0)) { _if_result_7 = (lim_either); } else { _if_result_7 = (20); } _if_result_7; }); + el_val_t lim_either = ({ el_val_t _if_result_10 = 0; if ((lim_url > 0)) { _if_result_10 = (lim_url); } else { _if_result_10 = (lim_body); } _if_result_10; }); + el_val_t limit = ({ el_val_t _if_result_11 = 0; if ((lim_either > 0)) { _if_result_11 = (lim_either); } else { _if_result_11 = (20); } _if_result_11; }); return engram_search_json(q, limit); return 0; } el_val_t route_activate(el_val_t method, el_val_t path, el_val_t body) { - el_val_t q = ({ el_val_t _if_result_8 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_8 = (query_param(path, EL_STR("q"))); } else { _if_result_8 = (json_get_string(body, EL_STR("query"))); } _if_result_8; }); + el_val_t q = ({ el_val_t _if_result_12 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_12 = (query_param(path, EL_STR("q"))); } else { _if_result_12 = (json_get_string(body, EL_STR("query"))); } _if_result_12; }); if (str_eq(q, EL_STR(""))) { return err_json(EL_STR("missing query")); } - el_val_t d_raw = ({ el_val_t _if_result_9 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_9 = (query_int(path, EL_STR("depth"), 3)); } else { _if_result_9 = (json_get_int(body, EL_STR("depth"))); } _if_result_9; }); - el_val_t depth = ({ el_val_t _if_result_10 = 0; if ((d_raw > 0)) { _if_result_10 = (d_raw); } else { _if_result_10 = (3); } _if_result_10; }); + el_val_t d_raw = ({ el_val_t _if_result_13 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_13 = (query_int(path, EL_STR("depth"), 3)); } else { _if_result_13 = (json_get_int(body, EL_STR("depth"))); } _if_result_13; }); + el_val_t depth = ({ el_val_t _if_result_14 = 0; if ((d_raw > 0)) { _if_result_14 = (d_raw); } else { _if_result_14 = (3); } _if_result_14; }); return el_str_concat(el_str_concat(EL_STR("{\"results\":"), engram_activate_json(q, depth)), EL_STR("}")); return 0; } @@ -198,15 +221,50 @@ el_val_t route_create_edge(el_val_t method, el_val_t path, el_val_t body) { el_val_t from_id = json_get_string(body, EL_STR("from_id")); el_val_t to_id = json_get_string(body, EL_STR("to_id")); el_val_t rel_raw = json_get_string(body, EL_STR("relation")); - el_val_t relation = ({ el_val_t _if_result_11 = 0; if (str_eq(rel_raw, EL_STR(""))) { _if_result_11 = (EL_STR("associates")); } else { _if_result_11 = (rel_raw); } _if_result_11; }); - el_val_t w_raw = json_get_float(body, EL_STR("weight")); - el_val_t weight = ({ el_val_t _if_result_12 = 0; if ((w_raw == el_from_float(0.0))) { _if_result_12 = (el_from_float(0.5)); } else { _if_result_12 = (w_raw); } _if_result_12; }); + el_val_t relation = ({ el_val_t _if_result_15 = 0; if (str_eq(rel_raw, EL_STR(""))) { _if_result_15 = (EL_STR("associates")); } else { _if_result_15 = (rel_raw); } _if_result_15; }); + el_val_t w_present = json_get_raw(body, EL_STR("weight")); + el_val_t weight = ({ el_val_t _if_result_16 = 0; if (str_eq(w_present, EL_STR(""))) { _if_result_16 = (el_from_float(0.5)); } else { _if_result_16 = (json_get_float(body, EL_STR("weight"))); } _if_result_16; }); engram_connect(from_id, to_id, weight, relation); el_val_t saved = persist_canonical(); return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"from_id\":\""), from_id), EL_STR("\",\"to_id\":\"")), to_id), EL_STR("\",\"relation\":\"")), relation), EL_STR("\"}")); return 0; } +el_val_t route_create_edges_batch(el_val_t method, el_val_t path, el_val_t body) { + el_val_t arr = json_get_raw(body, EL_STR("edges")); + if (str_eq(arr, EL_STR(""))) { + return err_json(EL_STR("missing edges array")); + } + el_val_t n = json_array_len(arr); + if (n == 0) { + return EL_STR("{\"ok\":true,\"accepted\":0,\"skipped\":0}"); + } + el_val_t i = 0; + el_val_t accepted = 0; + el_val_t skipped = 0; + while (i < n) { + el_val_t item = json_array_get(arr, i); + el_val_t from_id = json_get_string(item, EL_STR("from_id")); + el_val_t to_id = json_get_string(item, EL_STR("to_id")); + if (str_eq(from_id, EL_STR("")) || str_eq(to_id, EL_STR(""))) { + skipped = (skipped + 1); + } else { + el_val_t rel_raw = json_get_string(item, EL_STR("relation")); + el_val_t relation = ({ el_val_t _if_result_17 = 0; if (str_eq(rel_raw, EL_STR(""))) { _if_result_17 = (EL_STR("associates")); } else { _if_result_17 = (rel_raw); } _if_result_17; }); + el_val_t w_present = json_get_raw(item, EL_STR("weight")); + el_val_t weight = ({ el_val_t _if_result_18 = 0; if (str_eq(w_present, EL_STR(""))) { _if_result_18 = (el_from_float(0.5)); } else { _if_result_18 = (json_get_float(item, EL_STR("weight"))); } _if_result_18; }); + engram_connect(from_id, to_id, weight, relation); + accepted = (accepted + 1); + } + i = (i + 1); + } + if (accepted > 0) { + el_val_t saved = persist_canonical(); + } + return el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"accepted\":"), int_to_str(accepted)), EL_STR(",\"skipped\":")), int_to_str(skipped)), EL_STR("}")); + return 0; +} + el_val_t route_neighbors(el_val_t method, el_val_t path, el_val_t body) { el_val_t id = extract_id(path, EL_STR("/api/neighbors/")); if (str_eq(id, EL_STR(""))) { @@ -242,36 +300,51 @@ el_val_t route_forget(el_val_t method, el_val_t path, el_val_t body) { el_val_t route_save(el_val_t method, el_val_t path, el_val_t body) { el_val_t p_raw = json_get_string(body, EL_STR("path")); el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); - el_val_t dir = ({ el_val_t _if_result_13 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_13 = (EL_STR("/tmp/engram")); } else { _if_result_13 = (dir_raw); } _if_result_13; }); - el_val_t p = ({ el_val_t _if_result_14 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_14 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_14 = (p_raw); } _if_result_14; }); - engram_save(p); - return el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"path\":\""), p), EL_STR("\"}")); + el_val_t dir = ({ el_val_t _if_result_19 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_19 = (EL_STR("/tmp/engram")); } else { _if_result_19 = (dir_raw); } _if_result_19; }); + el_val_t p = ({ el_val_t _if_result_20 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_20 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_20 = (p_raw); } _if_result_20; }); + el_val_t sv = engram_save(p); + el_val_t sv_ok = ({ el_val_t _if_result_21 = 0; if ((sv == 0)) { _if_result_21 = (EL_STR("false")); } else { _if_result_21 = (EL_STR("true")); } _if_result_21; }); + return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":"), sv_ok), EL_STR(",\"path\":\"")), p), EL_STR("\",\"node_count\":")), int_to_str(engram_node_count())), EL_STR(",\"edge_count\":")), int_to_str(engram_edge_count())), EL_STR("}")); return 0; } el_val_t route_load(el_val_t method, el_val_t path, el_val_t body) { el_val_t p_raw = json_get_string(body, EL_STR("path")); el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); - el_val_t dir = ({ el_val_t _if_result_15 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_15 = (EL_STR("/tmp/engram")); } else { _if_result_15 = (dir_raw); } _if_result_15; }); - el_val_t p = ({ el_val_t _if_result_16 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_16 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_16 = (p_raw); } _if_result_16; }); - engram_load(p); - return ok_json(); + el_val_t dir = ({ el_val_t _if_result_22 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_22 = (EL_STR("/tmp/engram")); } else { _if_result_22 = (dir_raw); } _if_result_22; }); + el_val_t p = ({ el_val_t _if_result_23 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_23 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_23 = (p_raw); } _if_result_23; }); + el_val_t ld = engram_load(p); + el_val_t ld_ok = ({ el_val_t _if_result_24 = 0; if ((ld == 0)) { _if_result_24 = (EL_STR("false")); } else { _if_result_24 = (EL_STR("true")); } _if_result_24; }); + el_val_t nc_after = engram_node_count(); + el_val_t hollow = ({ el_val_t _if_result_25 = 0; if ((nc_after == 0)) { _if_result_25 = (EL_STR("true")); } else { _if_result_25 = (EL_STR("false")); } _if_result_25; }); + return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":"), ld_ok), EL_STR(",\"path\":\"")), p), EL_STR("\",\"node_count\":")), int_to_str(nc_after)), EL_STR(",\"edge_count\":")), int_to_str(engram_edge_count())), EL_STR(",\"hollow\":")), hollow), EL_STR("}")); return 0; } el_val_t route_health(el_val_t method, el_val_t path, el_val_t body) { - return EL_STR("{\"status\":\"ok\",\"engine\":\"engram-runtime-native\"}"); + return el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"status\":\"ok\",\"engine\":\"engram-runtime-native\",\"node_count\":"), int_to_str(engram_node_count())), EL_STR(",\"edge_count\":")), int_to_str(engram_edge_count())), EL_STR("}")); + return 0; +} + +el_val_t route_embed_backfill(el_val_t method, el_val_t path, el_val_t body) { + el_val_t n = query_int(path, EL_STR("n"), 32); + el_val_t result = engram_embed_backfill(n); + el_val_t done = json_get_float(result, EL_STR("embedded")); + if (done > el_from_float(0.0)) { + el_val_t saved = persist_canonical(); + } + return result; return 0; } el_val_t route_sync(el_val_t method, el_val_t path, el_val_t body) { el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); - el_val_t dir = ({ el_val_t _if_result_17 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_17 = (EL_STR("/tmp/engram")); } else { _if_result_17 = (dir_raw); } _if_result_17; }); + el_val_t dir = ({ el_val_t _if_result_26 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_26 = (EL_STR("/tmp/engram")); } else { _if_result_26 = (dir_raw); } _if_result_26; }); el_val_t snap_path = el_str_concat(dir, EL_STR("/.sync-export.json")); engram_save(snap_path); el_val_t snap = fs_read(snap_path); if (str_eq(snap, EL_STR(""))) { - return EL_STR("{\"nodes\":[],\"edges\":[]}"); + return err_json(EL_STR("sync export failed: snapshot unreadable")); } return snap; return 0; @@ -305,7 +378,7 @@ el_val_t route_emit_ise(el_val_t method, el_val_t path, el_val_t body) { el_val_t conf = el_from_float(0.8); el_val_t id = engram_node_full(content, EL_STR("InternalStateEvent"), EL_STR("state-event"), sal, imp, conf, EL_STR("Episodic"), EL_STR("[\"internal-state\",\"InternalStateEvent\"]")); el_val_t ret_raw = env(EL_STR("ENGRAM_ISE_RETENTION_MS")); - el_val_t ret_ms = ({ el_val_t _if_result_18 = 0; if (str_eq(ret_raw, EL_STR(""))) { _if_result_18 = (172800000); } else { _if_result_18 = (str_to_int(ret_raw)); } _if_result_18; }); + el_val_t ret_ms = ({ el_val_t _if_result_27 = 0; if (str_eq(ret_raw, EL_STR(""))) { _if_result_27 = (172800000); } else { _if_result_27 = (str_to_int(ret_raw)); } _if_result_27; }); el_val_t pruned = engram_prune_telemetry(ret_ms); return el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"id\":\""), id), EL_STR("\",\"pruned\":")), int_to_str(pruned)), EL_STR("}")); return 0; @@ -317,21 +390,21 @@ el_val_t route_capture_knowledge(el_val_t method, el_val_t path, el_val_t body) return err_json(EL_STR("missing content")); } el_val_t title = json_get_string(body, EL_STR("title")); - el_val_t label = ({ el_val_t _if_result_19 = 0; if (str_eq(title, EL_STR(""))) { _if_result_19 = (str_slice(content, 0, 60)); } else { _if_result_19 = (title); } _if_result_19; }); + el_val_t label = ({ el_val_t _if_result_28 = 0; if (str_eq(title, EL_STR(""))) { _if_result_28 = (str_slice(content, 0, 60)); } else { _if_result_28 = (title); } _if_result_28; }); el_val_t category_raw = json_get_string(body, EL_STR("category")); - el_val_t category = ({ el_val_t _if_result_20 = 0; if (str_eq(category_raw, EL_STR(""))) { _if_result_20 = (EL_STR("other")); } else { _if_result_20 = (category_raw); } _if_result_20; }); + el_val_t category = ({ el_val_t _if_result_29 = 0; if (str_eq(category_raw, EL_STR(""))) { _if_result_29 = (EL_STR("other")); } else { _if_result_29 = (category_raw); } _if_result_29; }); el_val_t ktier_raw = json_get_string(body, EL_STR("tier")); - el_val_t ktier = ({ el_val_t _if_result_21 = 0; if (str_eq(ktier_raw, EL_STR(""))) { _if_result_21 = (EL_STR("note")); } else { _if_result_21 = (ktier_raw); } _if_result_21; }); + el_val_t ktier = ({ el_val_t _if_result_30 = 0; if (str_eq(ktier_raw, EL_STR(""))) { _if_result_30 = (EL_STR("note")); } else { _if_result_30 = (ktier_raw); } _if_result_30; }); el_val_t project = json_get_string(body, EL_STR("project")); el_val_t tags_raw = json_get_raw(body, EL_STR("tags")); - el_val_t tags_base = ({ el_val_t _if_result_22 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_22 = (EL_STR("[]")); } else { _if_result_22 = (tags_raw); } _if_result_22; }); + el_val_t tags_base = ({ el_val_t _if_result_31 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_31 = (EL_STR("[]")); } else { _if_result_31 = (tags_raw); } _if_result_31; }); el_val_t base_len = str_len(tags_base); el_val_t head = str_slice(tags_base, 0, (base_len - 1)); - el_val_t sep = ({ el_val_t _if_result_23 = 0; if (str_eq(head, EL_STR("["))) { _if_result_23 = (EL_STR("")); } else { _if_result_23 = (EL_STR(",")); } _if_result_23; }); + el_val_t sep = ({ el_val_t _if_result_32 = 0; if (str_eq(head, EL_STR("["))) { _if_result_32 = (EL_STR("")); } else { _if_result_32 = (EL_STR(",")); } _if_result_32; }); el_val_t safe_cat = str_replace(category, EL_STR("\""), EL_STR("'")); el_val_t safe_tier = str_replace(ktier, EL_STR("\""), EL_STR("'")); el_val_t safe_proj = str_replace(project, EL_STR("\""), EL_STR("'")); - el_val_t proj_tag = ({ el_val_t _if_result_24 = 0; if (str_eq(safe_proj, EL_STR(""))) { _if_result_24 = (EL_STR("")); } else { _if_result_24 = (el_str_concat(el_str_concat(EL_STR(",\"project:"), safe_proj), EL_STR("\""))); } _if_result_24; }); + el_val_t proj_tag = ({ el_val_t _if_result_33 = 0; if (str_eq(safe_proj, EL_STR(""))) { _if_result_33 = (EL_STR("")); } else { _if_result_33 = (el_str_concat(el_str_concat(EL_STR(",\"project:"), safe_proj), EL_STR("\""))); } _if_result_33; }); el_val_t tags = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(head, sep), EL_STR("\"category:")), safe_cat), EL_STR("\",\"tier:")), safe_tier), EL_STR("\"")), proj_tag), EL_STR("]")); el_val_t sal = el_from_float(0.5); el_val_t imp = el_from_float(0.5); @@ -342,6 +415,20 @@ el_val_t route_capture_knowledge(el_val_t method, el_val_t path, el_val_t body) return 0; } +el_val_t route_similarity(el_val_t method, el_val_t path, el_val_t body) { + el_val_t a = query_param(path, EL_STR("a")); + el_val_t b = query_param(path, EL_STR("b")); + if (str_eq(a, EL_STR(""))) { + return err_json(EL_STR("missing a")); + } + if (str_eq(b, EL_STR(""))) { + return err_json(EL_STR("missing b")); + } + el_val_t sim = engram_cosine_sim(a, b); + return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"a\":\""), a), EL_STR("\",\"b\":\"")), b), EL_STR("\",\"cosine\":")), float_to_str(sim)), EL_STR("}")); + return 0; +} + el_val_t check_auth_ok(el_val_t method, el_val_t body) { el_val_t key = env(EL_STR("ENGRAM_API_KEY")); if (str_eq(key, EL_STR(""))) { @@ -377,6 +464,12 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) { if (str_eq(method, EL_STR("GET")) && (str_eq(clean, EL_STR("/api/stats")) || str_eq(clean, EL_STR("/stats")))) { return route_stats(method, path, body); } + if (str_eq(method, EL_STR("GET")) && (str_eq(clean, EL_STR("/api/act-stats")) || str_eq(clean, EL_STR("/act-stats")))) { + return route_act_stats(method, path, body); + } + if (str_eq(method, EL_STR("GET")) && (str_eq(clean, EL_STR("/api/text-health")) || str_eq(clean, EL_STR("/text-health")))) { + return route_text_health(method, path, body); + } if (str_eq(method, EL_STR("POST")) && (str_eq(clean, EL_STR("/api/nodes")) || str_eq(clean, EL_STR("/nodes")))) { return route_create_node(method, path, body); } @@ -395,6 +488,9 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) { if (str_eq(method, EL_STR("POST")) && (str_eq(clean, EL_STR("/api/edges")) || str_eq(clean, EL_STR("/edges")))) { return route_create_edge(method, path, body); } + if (str_eq(method, EL_STR("POST")) && (str_eq(clean, EL_STR("/api/edges/batch")) || str_eq(clean, EL_STR("/edges/batch")))) { + return route_create_edges_batch(method, path, body); + } if (str_eq(method, EL_STR("GET")) && str_starts_with(clean, EL_STR("/api/neighbors/"))) { return route_neighbors(method, path, body); } @@ -425,6 +521,12 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) { if (str_eq(method, EL_STR("GET")) && str_eq(clean, EL_STR("/api/sync"))) { return route_sync(method, path, body); } + if (str_eq(clean, EL_STR("/api/embed-backfill"))) { + return route_embed_backfill(method, path, body); + } + if (str_eq(method, EL_STR("GET")) && str_starts_with(clean, EL_STR("/api/similarity"))) { + return route_similarity(method, path, body); + } return el_str_concat(el_str_concat(EL_STR("{\"error\":\"not found\",\"path\":\""), clean), EL_STR("\"}")); return 0; } @@ -432,10 +534,10 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) { int main(int _argc, char** _argv) { el_runtime_init_args(_argc, _argv); bind_raw = env(EL_STR("ENGRAM_BIND")); - bind_str = ({ el_val_t _if_result_25 = 0; if (str_eq(bind_raw, EL_STR(""))) { _if_result_25 = (EL_STR(":8742")); } else { _if_result_25 = (bind_raw); } _if_result_25; }); + bind_str = ({ el_val_t _if_result_34 = 0; if (str_eq(bind_raw, EL_STR(""))) { _if_result_34 = (EL_STR(":8742")); } else { _if_result_34 = (bind_raw); } _if_result_34; }); port = parse_port(bind_str); data_dir_raw = env(EL_STR("ENGRAM_DATA_DIR")); - data_dir = ({ el_val_t _if_result_26 = 0; if (str_eq(data_dir_raw, EL_STR(""))) { _if_result_26 = (EL_STR("/tmp/engram")); } else { _if_result_26 = (data_dir_raw); } _if_result_26; }); + data_dir = ({ el_val_t _if_result_35 = 0; if (str_eq(data_dir_raw, EL_STR(""))) { _if_result_35 = (EL_STR("/tmp/engram")); } else { _if_result_35 = (data_dir_raw); } _if_result_35; }); snapshot_path = el_str_concat(data_dir, EL_STR("/snapshot.json")); engram_load(snapshot_path); boot_snap = fs_read(snapshot_path); diff --git a/engram/src/server.el b/engram/src/server.el index 3372ff6..06d4d40 100644 --- a/engram/src/server.el +++ b/engram/src/server.el @@ -76,6 +76,38 @@ fn route_stats(method: String, path: String, body: String) -> String { engram_stats_json() } +// route_act_stats — GET /api/act-stats +// (2026-08-04 self-review) engram_act_stats_json() has existed since the +// 2026-07-27 review but was reachable ONLY through the soul daemon's heartbeat +// binding. Every activation-layer gauge — WM evictions, breakthroughs, embedder +// breaker state, context drift, and now the Hebbian counters — was therefore +// invisible unless the soul happened to be running and its ISEs were read back +// out of the store. Diagnosing the activation layer required a working soul, +// which is exactly backwards: the lower layer should be observable on its own. +// This review needed it to verify link formation and could not get at it. One +// line of plumbing, and the whole activation layer becomes directly diagnosable. +fn route_act_stats(method: String, path: String, body: String) -> String { + engram_act_stats_json() +} + +// route_text_health — GET /api/text-health +// (2026-08-08 self-review) The daily census half of the text-integrity gauge. +// Today's review found that the JSON parser had been replacing every \uXXXX +// escape with a literal '?' for at least two months: 3,119 of 4,081 +// non-telemetry nodes (76%) were damaged, including the self traversal root +// and every values node, and NOTHING detected it — because every gauge in the +// system measured whether the machinery was running, and none measured whether +// the text it carried was intact. No snapshot on disk predates the damage, so +// it cannot be undone; it can only be made impossible to repeat quietly. +// +// The parser is fixed. This route is the standing check: `damaged` should now +// hold flat at its historical floor and never climb. `write_damaged` (also on +// the heartbeat as txt_damaged) is the live regression signal — non-zero means +// a write path is mangling text right now. +fn route_text_health(method: String, path: String, body: String) -> String { + engram_text_health_json() +} + // (2026-07-18 self-review) Scoping sweep: `let` inside an if-block creates an // inner scope only — it does NOT mutate the outer binding (documented with // evidence in awareness.el, 2026-05-25). Every default/reassignment below used @@ -85,6 +117,17 @@ fn route_stats(method: String, path: String, body: String) -> String { // save/load with no "path" hit engram_save(""). Rewritten to the // `let x = if cond { a } else { b }` expression form (the pattern the newer // routes route_emit_ise/route_capture_knowledge already use correctly). +// store_on — ENGRAM_STORE flag (tiered paged store as the durable owner). Matches +// engram_store_enabled() in el_runtime.c EXACTLY (1 / on / true). Default off → +// every persistence path below is byte-for-byte the historical snapshot behavior. +fn store_on() -> Bool { + let v: String = env("ENGRAM_STORE") + if str_eq(v, "1") { return true } + if str_eq(v, "on") { return true } + if str_eq(v, "true") { return true } + return false +} + // persist_canonical — save the canonical snapshot after a durable write. // // WHY (2026-07-22 self-review): the 2026-07-21 fix correctly stopped READ @@ -99,20 +142,116 @@ fn route_stats(method: String, path: String, body: String) -> String { // tolerant, ~2/min — snapshotting the whole store per heartbeat is waste; // any durable write that follows persists the pruning too). fn persist_canonical() -> Int { + // ENGRAM_STORE: the paged store is the durable owner — a checkpoint flushes + // dirty pages behind a WAL-durable record (durable the moment the WAL fsyncs). + // This is the fix for the "restart reverted to a 17h-old snapshot" data loss: + // durable writes no longer depend on a full snapshot.json rewrite. Returns 1 + // on a successful checkpoint, 0 otherwise. Flag-off: unchanged (writes JSON). + if store_on() { + return engram_store_checkpoint() + } let dir_raw: String = env("ENGRAM_DATA_DIR") - let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw } - engram_save(dir + "/snapshot.json") - return 1 + let dir: String = engram_resolve_data_dir() + // (2026-08-10 self-review) This returned a hardcoded 1, which made every + // caller's `let saved: Int = persist_canonical()` a dead variable — six + // durable write paths each believed they had confirmation of a successful + // canonical persist and none of them had any. Propagate the real result. + return engram_save(dir + "/snapshot.json") } +// ── WAL persistence (design doc §§3-14; gated behind ENGRAM_WAL=on) ────────── +// Default OFF → every persist path below is byte-identical to the historical +// per-write full-snapshot behavior. When ON, structural mutations append O(1) +// WAL records instead of rewriting the whole graph, with threshold compaction. +fn wal_on() -> Bool { + str_eq(env("ENGRAM_WAL"), "on") +} + +// Persist a single-node mutation (create / content-evolve / strengthen). +fn persist_node(id: String) -> Int { + if wal_on() { + let d: String = engram_resolve_data_dir() + let a: Int = engram_wal_node_put(d, id) + let c: Int = engram_wal_maybe_compact(d) + return a + } + return persist_canonical() +} + +// Persist edges appended at index >= start (covers single-edge and batch). +fn persist_edges_since(start: Int) -> Int { + if wal_on() { + let d: String = engram_resolve_data_dir() + let a: Int = engram_wal_edges_since(d, start) + let c: Int = engram_wal_maybe_compact(d) + return a + } + return persist_canonical() +} + +// Persist a Hebbian consolidation batch as ONE WAL record (single fsync, §5-B). +fn persist_hebb_batch(start: Int) -> Int { + if wal_on() { + let d: String = engram_resolve_data_dir() + let a: Int = engram_wal_hebb_batch(d, start) + let c: Int = engram_wal_maybe_compact(d) + return a + } + return persist_canonical() +} + +// Bulk mutation (embedding backfill, load-merge): write a fresh compaction base +// so the many-node change is durable in one atomic snapshot; WAL is truncated. +fn persist_bulk() -> Int { + if wal_on() { + let d: String = engram_resolve_data_dir() + return engram_wal_compact(d) + } + return persist_canonical() +} + +// INCOMPLETE-ROUTE FIX (2026-07-24 self-review): this route silently dropped +// label, importance, tier, and tags — engram_node() defaults label to content +// and importance to 0.5, so every node created over HTTP lost its metadata. +// Observed live: the soul's boot-counter write-back landed with +// label="soul:boot_count:99" (content), importance 0.5, no tags. Honor the +// full field set via engram_node_full when any of them is supplied. +// PRESENCE-AWARE DEFAULTS (2026-08-01 self-review): the old pattern +// `if x == 0.0 { default }` made a legitimate 0.0 unrepresentable — a caller +// setting salience/importance/weight to zero silently got 0.5. json_get_raw +// returns "" when the key is ABSENT and the raw token when present, so +// absence and zero are now distinguishable. Also: confidence was hardcoded +// to 1.0 regardless of input — every HTTP-created node claimed full +// epistemic confidence. Now honored from the payload (default 1.0). fn route_create_node(method: String, path: String, body: String) -> String { let content: String = json_get_string(body, "content") let nt_raw: String = json_get_string(body, "node_type") let node_type: String = if str_eq(nt_raw, "") { "Memory" } else { nt_raw } - let sal_raw: Float = json_get_float(body, "salience") - let salience: Float = if sal_raw == 0.0 { 0.5 } else { sal_raw } - let id: String = engram_node(content, node_type, salience) - let saved: Int = persist_canonical() + let sal_present: String = json_get_raw(body, "salience") + let salience: Float = if str_eq(sal_present, "") { 0.5 } else { json_get_float(body, "salience") } + let label_raw: String = json_get_string(body, "label") + let label: String = if str_eq(label_raw, "") { content } else { label_raw } + let imp_present: String = json_get_raw(body, "importance") + let importance: Float = if str_eq(imp_present, "") { 0.5 } else { json_get_float(body, "importance") } + let conf_present: String = json_get_raw(body, "confidence") + let confidence: Float = if str_eq(conf_present, "") { 1.0 } else { json_get_float(body, "confidence") } + let tier_raw: String = json_get_string(body, "tier") + let tier: String = if str_eq(tier_raw, "") { "Working" } else { tier_raw } + let tags: String = json_get_string(body, "tags") + // NO el_from_float WRAPPER (2026-08-01 self-review): salience/importance/ + // confidence are already Float (el_val_t) values — json_get_float and + // Float literals both encode. Wrapping them in el_from_float AGAIN + // reinterpreted the boxed bits as a raw double, producing garbage that + // failed engram_decode_score's range check and clamped every HTTP-created + // node to defaults (salience 0.9 in → 0.5 stored; confidence 0.6 in → 1.0 + // stored — verified live). route_emit_ise always passed Floats bare and + // its 0.3/0.3/0.8 stored correctly; this call now does the same. + let id: String = engram_node_full( + content, node_type, label, + salience, importance, confidence, + tier, tags + ) + let saved: Int = persist_node(id) "{\"id\":\"" + id + "\",\"content\":\"" + content + "\",\"node_type\":\"" + node_type + "\"}" } @@ -139,7 +278,7 @@ fn route_scan_nodes(method: String, path: String, body: String) -> String { // clobbered the good snapshot. Read routes must never write the canonical path.) fn route_scan_edges(method: String, path: String, body: String) -> String { let dir_raw: String = env("ENGRAM_DATA_DIR") - let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw } + let dir: String = engram_resolve_data_dir() let snap_path: String = dir + "/.scan-export.json" engram_save(snap_path) let snap: String = fs_read(snap_path) @@ -177,13 +316,66 @@ fn route_create_edge(method: String, path: String, body: String) -> String { let to_id: String = json_get_string(body, "to_id") let rel_raw: String = json_get_string(body, "relation") let relation: String = if str_eq(rel_raw, "") { "associates" } else { rel_raw } - let w_raw: Float = json_get_float(body, "weight") - let weight: Float = if w_raw == 0.0 { 0.5 } else { w_raw } + // Presence-aware (2026-08-01): weight 0.0 is a legitimate edge weight + // (dormant association); only default when the key is absent. + let w_present: String = json_get_raw(body, "weight") + let weight: Float = if str_eq(w_present, "") { 0.5 } else { json_get_float(body, "weight") } + let ec0: Int = engram_edge_count() engram_connect(from_id, to_id, weight, relation) - let saved: Int = persist_canonical() + let saved: Int = persist_edges_since(ec0) "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + relation + "\"}" } +// route_create_edges_batch — POST /api/edges/batch {"edges":[{from_id,to_id,relation,weight}, ...]} +// +// WHY THIS EXISTS (2026-08-07 self-review). persist_canonical() writes the +// FULL canonical snapshot — 60MB at current graph size — and route_create_edge +// calls it once per edge. That is correct for the interactive one-edge case and +// ruinous for any bulk write: the soul's Hebbian consolidation path delivers +// ~14 associations per 8-minute heartbeat, which through the single-edge route +// would be ~840MB of disk writes per beat, ~150GB/day, to persist 14 edges. +// +// The fix is not to weaken durability — it is to make the unit of durability +// the BATCH. Connect every edge, then snapshot exactly once. Same guarantee +// (nothing acknowledged is lost to a restart), 1/N the writes. Empty or +// malformed entries are skipped rather than aborting the batch: a consolidation +// payload is best-effort by design, and one bad id should not cost the other 13. +// +// Returns the accepted count so the caller can tell delivery from silence. +fn route_create_edges_batch(method: String, path: String, body: String) -> String { + let arr: String = json_get_raw(body, "edges") + if str_eq(arr, "") { return err_json("missing edges array") } + let n: Int = json_array_len(arr) + if n == 0 { return "{\"ok\":true,\"accepted\":0,\"skipped\":0}" } + let ec0: Int = engram_edge_count() + let i: Int = 0 + let accepted: Int = 0 + let skipped: Int = 0 + while i < n { + let item: String = json_array_get(arr, i) + let from_id: String = json_get_string(item, "from_id") + let to_id: String = json_get_string(item, "to_id") + if str_eq(from_id, "") || str_eq(to_id, "") { + let skipped = skipped + 1 + } else { + let rel_raw: String = json_get_string(item, "relation") + let relation: String = if str_eq(rel_raw, "") { "associates" } else { rel_raw } + let w_present: String = json_get_raw(item, "weight") + let weight: Float = if str_eq(w_present, "") { 0.5 } else { json_get_float(item, "weight") } + engram_connect(from_id, to_id, weight, relation) + let accepted = accepted + 1 + } + let i = i + 1 + } + // ONE snapshot for the whole batch — the entire point of this route. + // Skip it when nothing was accepted: an all-malformed payload must not + // trigger a 60MB write. + if accepted > 0 { + let saved: Int = persist_hebb_batch(ec0) + } + return "{\"ok\":true,\"accepted\":" + int_to_str(accepted) + ",\"skipped\":" + int_to_str(skipped) + "}" +} + fn route_neighbors(method: String, path: String, body: String) -> String { let id: String = extract_id(path, "/api/neighbors/") if str_eq(id, "") { return err_json("missing id") } @@ -195,38 +387,120 @@ fn route_strengthen(method: String, path: String, body: String) -> String { let id: String = json_get_string(body, "node_id") if str_eq(id, "") { return err_json("missing node_id") } engram_strengthen(id) - let saved: Int = persist_canonical() + let saved: Int = persist_node(id) ok_json() } +// route_forget — DELETE /api/nodes/:id — INTEGRITY HARDENED (design doc §18.1). +// +// Two invariants now enforced AT THE STORE (not one layer up in neuron-api.el, +// which a direct HTTP client could bypass): +// 1. Write-protection: protected identity/value nodes (derived from the self +// graph — self root + values hub + their neighbors, §18.3) cannot be +// deleted over HTTP. Returns 403, node untouched. +// 2. No hard delete over the wire, ever: an ordinary delete creates a +// Tombstone marker node + `tombstones` edge and KEEPS the original node +// and its edges (recoverable), instead of the old destructive +// engram_forget() shift-delete. Raw engram_forget is now internal-GC only +// and no longer reachable from any HTTP route. fn route_forget(method: String, path: String, body: String) -> String { let id: String = extract_id(path, "/api/nodes/") if str_eq(id, "") { return err_json("missing id") } - engram_forget(id) - let saved: Int = persist_canonical() - ok_json() + if engram_is_protected(id) == 1 { + return "{\"__status__\":403,\"error\":\"protected node; deletion refused\",\"id\":\"" + id + "\"}" + } + let tomb_id: String = engram_node_full( + "tombstone:" + id, "Tombstone", "tombstone:" + id, + 0.1, 0.1, 1.0, "Episodic", "[\"tombstone\"]" + ) + let ec0: Int = engram_edge_count() + engram_connect(tomb_id, id, 1.0, "tombstones") + let saved: Int = if wal_on() { + let d: String = engram_resolve_data_dir() + let a: Int = engram_wal_node_put(d, tomb_id) + let b: Int = engram_wal_edges_since(d, ec0) + let c: Int = engram_wal_maybe_compact(d) + a + } else { + persist_canonical() + } + "{\"ok\":true,\"tombstoned\":\"" + id + "\",\"tombstone_id\":\"" + tomb_id + "\"}" } fn route_save(method: String, path: String, body: String) -> String { let p_raw: String = json_get_string(body, "path") let dir_raw: String = env("ENGRAM_DATA_DIR") - let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw } + let dir: String = engram_resolve_data_dir() let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw } - engram_save(p) - "{\"ok\":true,\"path\":\"" + p + "\"}" + // (2026-08-10 self-review) engram_save returns 0 on an empty path and the + // route discarded it, so the response was a literal "ok":true regardless + // of whether anything was written. Report the actual result AND the counts + // that were supposed to have been written — the same move that made + // route_health honest on 2026-08-01. A caller can now tell "saved 13k + // nodes" from "saved nothing and said ok". + let sv: Int = engram_save(p) + let sv_ok: String = if sv == 0 { "false" } else { "true" } + "{\"ok\":" + sv_ok + ",\"path\":\"" + p + "\",\"node_count\":" + int_to_str(engram_node_count()) + ",\"edge_count\":" + int_to_str(engram_edge_count()) + "}" } fn route_load(method: String, path: String, body: String) -> String { let p_raw: String = json_get_string(body, "path") let dir_raw: String = env("ENGRAM_DATA_DIR") - let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw } + let dir: String = engram_resolve_data_dir() let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw } - engram_load(p) - ok_json() + // (2026-08-10 self-review) This was a stub response over the single most + // destructive operation in the server. engram_load returns 0 on an empty + // path, an unopenable file, a zero-length file, or malloc failure — and + // this route answered ok_json() in every one of those cases. + // + // Precise failure shape (el_runtime.c:9890): the fopen guard runs BEFORE + // the store reset, so a MISSING path is genuinely safe — it returns 0 with + // the graph intact. The dangerous case is a readable-but-malformed file: + // the reset loop frees every node and edge FIRST, then parses, so a + // truncated or non-snapshot JSON leaves a hollow store — and the caller + // was told "ok":true. With 37 GB of stale dated snapshots sitting in the + // data dir as tempting restore targets, "restore reported success and + // silently emptied the graph" is a live risk, not a hypothetical one. + // + // Fix: surface the return value AND the resulting counts. node_count=0 + // after a load is the unambiguous hollow-store signal (same convention + // route_health adopted 2026-08-01). Callers can now verify a restore + // instead of trusting it. + let ld: Int = engram_load(p) + let ld_ok: String = if ld == 0 { "false" } else { "true" } + let nc_after: Int = engram_node_count() + let hollow: String = if nc_after == 0 { "true" } else { "false" } + "{\"ok\":" + ld_ok + ",\"path\":\"" + p + "\",\"node_count\":" + int_to_str(nc_after) + ",\"edge_count\":" + int_to_str(engram_edge_count()) + ",\"hollow\":" + hollow + "}" } +// (2026-08-01 self-review) Health previously returned a hardcoded literal — +// it reported "ok" even when the snapshot failed to load and the store was +// empty. Now reports live counts so a monitor can distinguish "up and +// loaded" from "up and hollow" (node_count=0 after boot = failed load). fn route_health(method: String, path: String, body: String) -> String { - "{\"status\":\"ok\",\"engine\":\"engram-runtime-native\"}" + "{\"status\":\"ok\",\"engine\":\"engram-runtime-native\",\"node_count\":" + int_to_str(engram_node_count()) + ",\"edge_count\":" + int_to_str(engram_edge_count()) + "}" +} + +// route_embed_backfill — GET/POST /api/embed-backfill?n=48 +// +// (2026-07-25 self-review) The lazy embedding backfill runs only inside +// engram_activate, and nothing in production calls /api/activate on this +// store — the soul's curiosity loop activates its own in-process graph. +// After a restart from a snapshot without vectors, embedded_count stalled +// at 93/12175 and would never recover. This route lets the soul's +// heartbeat pump the backfill explicitly (48/min clears a 12k backlog in +// ~4h). Persists the canonical snapshot whenever new vectors were +// generated — the 2026-07-25 regression happened precisely because 3747 +// in-RAM embeddings were never snapshotted before a restart. Self- +// limiting: once coverage is full, embedded=0 and no save occurs. +fn route_embed_backfill(method: String, path: String, body: String) -> String { + let n: Int = query_int(path, "n", 32) + let result: String = engram_embed_backfill(n) + let done: Float = json_get_float(result, "embedded") + if done > 0.0 { + let saved: Int = persist_bulk() + } + return result } // route_sync — return a snapshot of non-ISE/non-Working nodes for the soul daemon @@ -243,13 +517,22 @@ fn route_health(method: String, path: String, body: String) -> String { // (2026-06-27 self-review: added this route to fix silent 10-min sync failures) fn route_sync(method: String, path: String, body: String) -> String { let dir_raw: String = env("ENGRAM_DATA_DIR") - let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw } + let dir: String = engram_resolve_data_dir() // 2026-07-21 self-review: export to a scratch path, never the canonical // snapshot.json — read routes must not be able to clobber the good snapshot. let snap_path: String = dir + "/.sync-export.json" engram_save(snap_path) let snap: String = fs_read(snap_path) - if str_eq(snap, "") { return "{\"nodes\":[],\"edges\":[]}" } + // 2026-08-02 self-review: this used to return {"nodes":[],"edges":[]} when + // the export/read failed. The soul's sync_ok test (awareness.el) only + // checks for "" and "{}", so that placeholder PASSED as a healthy sync: + // soul.last_sync_ok_ts got stamped, sync_age_ms stayed green, the + // sync_empty warn ISE never fired, and engram_sync reported added:0 + // forever. A totally broken sync was indistinguishable from a quiet + // healthy one — the exact failure class this route was added to fix in + // the first place (see 2026-06-27 note above). Return a real error so the + // failure is loud on both sides. + if str_eq(snap, "") { return err_json("sync export failed: snapshot unreadable") } return snap } @@ -268,7 +551,7 @@ fn route_load_merge(method: String, path: String, body: String) -> String { engram_load_merge(p) let added_n: Int = engram_node_count() - before_n let added_e: Int = engram_edge_count() - before_e - let saved: Int = persist_canonical() + let saved: Int = persist_bulk() "{\"ok\":true,\"nodes_added\":" + int_to_str(added_n) + ",\"edges_added\":" + int_to_str(added_e) + ",\"node_count\":" + int_to_str(engram_node_count()) + "}" } @@ -367,10 +650,29 @@ fn route_capture_knowledge(method: String, path: String, body: String) -> String sal, imp, conf, "Semantic", tags ) - let saved: Int = persist_canonical() + let saved: Int = persist_node(id) "{\"ok\":true,\"id\":\"" + id + "\"}" } +// route_similarity — GET /api/similarity?a=&b= +// +// (2026-08-01 self-review) engram_cosine_sim was added 2026-07-24 +// (bl-b2d1c944) with the stated purpose of exposing semantic distance to +// "EL code and the introspection API" — but it had ZERO callers anywhere: +// no route, no soul-daemon use. The activation path uses embeddings +// internally (semantic seeding, Pass-2 additive term), but there was no way +// to probe pairwise node similarity from outside. This closes that: cosine +// in [-1,1], or -2 when either node is missing or not yet embedded (so +// "not comparable" is distinguishable from "genuinely orthogonal" 0.0). +fn route_similarity(method: String, path: String, body: String) -> String { + let a: String = query_param(path, "a") + let b: String = query_param(path, "b") + if str_eq(a, "") { return err_json("missing a") } + if str_eq(b, "") { return err_json("missing b") } + let sim: Float = engram_cosine_sim(a, b) + "{\"a\":\"" + a + "\",\"b\":\"" + b + "\",\"cosine\":" + float_to_str(sim) + "}" +} + // ── Auth ────────────────────────────────────────────────────────────────────── fn check_auth_ok(method: String, body: String) -> Bool { @@ -417,6 +719,12 @@ fn handle_request(method: String, path: String, body: String) -> String { if str_eq(method, "GET") && (str_eq(clean, "/api/stats") || str_eq(clean, "/stats")) { return route_stats(method, path, body) } + if str_eq(method, "GET") && (str_eq(clean, "/api/act-stats") || str_eq(clean, "/act-stats")) { + return route_act_stats(method, path, body) + } + if str_eq(method, "GET") && (str_eq(clean, "/api/text-health") || str_eq(clean, "/text-health")) { + return route_text_health(method, path, body) + } // Nodes if str_eq(method, "POST") && (str_eq(clean, "/api/nodes") || str_eq(clean, "/nodes")) { @@ -439,6 +747,13 @@ fn handle_request(method: String, path: String, body: String) -> String { if str_eq(method, "POST") && (str_eq(clean, "/api/edges") || str_eq(clean, "/edges")) { return route_create_edge(method, path, body) } + // Batch edge write — one snapshot for the whole payload. Must be tested + // BEFORE nothing else claims it; the exact-match on "/api/edges" above + // does not catch "/api/edges/batch", so order is not load-bearing here, + // but keeping the two adjacent keeps them from drifting apart. + if str_eq(method, "POST") && (str_eq(clean, "/api/edges/batch") || str_eq(clean, "/edges/batch")) { + return route_create_edges_batch(method, path, body) + } if str_eq(method, "GET") && str_starts_with(clean, "/api/neighbors/") { return route_neighbors(method, path, body) } @@ -478,6 +793,16 @@ fn handle_request(method: String, path: String, body: String) -> String { return route_sync(method, path, body) } + // Embedding backfill — pumped by the soul heartbeat (2026-07-25) + if str_eq(clean, "/api/embed-backfill") { + return route_embed_backfill(method, path, body) + } + + // Semantic similarity probe (2026-08-01) + if str_eq(method, "GET") && str_starts_with(clean, "/api/similarity") { + return route_similarity(method, path, body) + } + "{\"error\":\"not found\",\"path\":\"" + clean + "\"}" } @@ -488,23 +813,44 @@ let bind_str: String = if str_eq(bind_raw, "") { ":8742" } else { bind_raw } let port: Int = parse_port(bind_str) // On startup, try to load any existing snapshot (best effort). -let data_dir_raw: String = env("ENGRAM_DATA_DIR") -let data_dir: String = if str_eq(data_dir_raw, "") { "/tmp/engram" } else { data_dir_raw } +// §18.2: resolve the data dir safely — unset ENGRAM_DATA_DIR → $HOME/.neuron/engram, +// never /tmp; fail loud if HOME is unresolvable (engram_resolve_data_dir exits). +let data_dir: String = engram_resolve_data_dir() let snapshot_path: String = data_dir + "/snapshot.json" -engram_load(snapshot_path) +// ENGRAM_STORE (tiered paged store — engram-tiered-storage-engine.md). When set, +// the durable owner is the paged store (neuron.egm + neuron.wal): engram_store_boot +// imports snapshot.json ONCE into a fresh neuron.egm, else replays the WAL and loads +// the store resident — snapshot.json is never read again as the ongoing store. This +// closes the "restart reverted to a 17h-old snapshot" data-loss window. Flag-off +// (default): byte-for-byte the historical snapshot + optional-WAL boot below. +if store_on() { + engram_store_boot(data_dir) + println("[engram] ENGRAM_STORE enabled — tiered paged store is the durable owner") +} else { + engram_load(snapshot_path) -// 2026-07-21 self-review boot guard: if the snapshot file has content but the -// load produced 0 nodes, something is wrong (corrupt file / parse failure). -// Preserve the evidence and warn loudly — and since read routes no longer write -// the canonical path, a bad boot can no longer clobber the good snapshot. -let boot_snap: String = fs_read(snapshot_path) -if !str_eq(boot_snap, "") { - if engram_node_count() == 0 { - println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json") - fs_write(data_dir + "/snapshot.failed-load.json", boot_snap) - } else { - // Good load: keep a boot-time backup of the snapshot as loaded. - fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap) + // WAL replay (design doc §6). Gated: default OFF is byte-identical to legacy + // snapshot-only boot. When ON, the snapshot above is the compaction BASE and + // the WAL carries every mutation since; replay reconstructs state to the last + // CRC-valid record, then opens the WAL for appending. + if wal_on() { + let replayed: Int = engram_wal_boot(data_dir) + println("[engram] WAL enabled — replayed " + int_to_str(replayed) + " records") + } + + // 2026-07-21 self-review boot guard: if the snapshot file has content but the + // load produced 0 nodes, something is wrong (corrupt file / parse failure). + // Preserve the evidence and warn loudly — and since read routes no longer write + // the canonical path, a bad boot can no longer clobber the good snapshot. + let boot_snap: String = fs_read(snapshot_path) + if !str_eq(boot_snap, "") { + if engram_node_count() == 0 { + println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json") + fs_write(data_dir + "/snapshot.failed-load.json", boot_snap) + } else { + // Good load: keep a boot-time backup of the snapshot as loaded. + fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap) + } } } diff --git a/engram/test/run_m35_hebb_persist.sh b/engram/test/run_m35_hebb_persist.sh new file mode 100755 index 0000000..1ce7dd3 --- /dev/null +++ b/engram/test/run_m35_hebb_persist.sh @@ -0,0 +1,158 @@ +#!/usr/bin/env bash +# M3.5 PRE-FLIP GATE. Pure C harness (NOT elb/elc): links the real el_runtime.c +# native engram builtins + engram_store.c and proves activation-time field +# mutations (edge hebb, node activation_count, WM weight) persist through a +# checkpoint and survive a reboot from neuron.egm with snapshot.json DELETED. +# Writes ONLY under a throwaway /tmp dir with a throwaway HOME. +set -u +HERE="$(cd "$(dirname "$0")" && pwd)" +RT="$HERE/../../lang/runtime/el_runtime.c" +ST="$HERE/../../lang/runtime/engram_store.c" +INC="$HERE/../../lang/runtime" +WORK="$(mktemp -d /tmp/engram-m35-XXXXXX)" +BIN="$WORK/m35" +export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron +export ENGRAM_WAL_SYNC=always +unset ENGRAM_STORE +fail=0 + +echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m35_hebb_persist.c) ==" +gcc -O1 -std=c11 -I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log" +if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi + +echo +echo "== 0) flag-OFF: seed+activate+checkpoint must NOT touch the store ==" +DOFF="$WORK/off"; mkdir -p "$DOFF" +( unset ENGRAM_STORE; "$BIN" offcheck "$DOFF" ) +[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; } +[ -e "$DOFF/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \ + || echo " ok: no neuron.egm created with flag OFF" + +echo +echo "== 1) POSITIVE: ENGRAM_STORE=1 seed -> activate -> checkpoint(field-persist) -> close ==" +DPOS="$WORK/pos"; mkdir -p "$DPOS" +ENGRAM_STORE=1 "$BIN" pos_seed "$DPOS" || { echo "FAIL: pos_seed"; fail=1; } +[ -e "$DPOS/neuron.egm" ] && echo " ok: neuron.egm created" || { echo "FAIL: neuron.egm missing"; fail=1; } + +echo +echo "== 2) reboot from neuron.egm with snapshot.json DELETED (must never read JSON) ==" +rm -f "$DPOS/snapshot.json" +ENGRAM_STORE=1 "$BIN" pos_reboot "$DPOS" || { echo "FAIL: pos_reboot"; fail=1; } + +echo +echo "== 3) NEGATIVE CONTROL: seed -> activate -> close WITHOUT the field-persist checkpoint ==" +DNEG="$WORK/neg"; mkdir -p "$DNEG" +ENGRAM_STORE=1 "$BIN" neg_seed "$DNEG" || { echo "FAIL: neg_seed"; fail=1; } +rm -f "$DNEG/snapshot.json" +ENGRAM_STORE=1 "$BIN" neg_reboot "$DNEG" || { echo "FAIL: neg_reboot"; fail=1; } + +echo +echo "== 4) assertions (python over the JSON exports) ==" +python3 - "$DPOS" "$DNEG" <<'PY' +import json, sys, os +WM_FLOOR = 0.05 +HEBB_MIN = 1e-6 + +def load(d, name): + with open(os.path.join(d, name)) as f: return json.load(f) + +def node_by_label(g, label): + for n in g["nodes"]: + if n.get("label") == label: return n + return None + +def edge_between(g, a_id, b_id): + for e in g["edges"]: + if e.get("from_id") == a_id and e.get("to_id") == b_id: + return e + return None + +rc = 0 +def check(cond, msg): + global rc + if cond: print(f" PASS: {msg}") + else: print(f" FAIL: {msg}"); rc = 1 + +dpos, dneg = sys.argv[1], sys.argv[2] +pre = load(dpos, "pre_reboot.json") +rebt = load(dpos, "reboot.json") + +pa, pb = node_by_label(pre, "hebb-a"), node_by_label(pre, "hebb-b") +ra = node_by_label(rebt, "hebb-a") +assert pa and pb and ra, "target nodes missing" +pe = edge_between(pre, pa["id"], pb["id"]) +re = edge_between(rebt, pa["id"], pb["id"]) +assert pe and re, "target edge missing" + +pre_hebb = pe.get("hebb", 0.0) +rebt_hebb = re.get("hebb", 0.0) +pre_ac = pa.get("activation_count", 0) +rebt_ac = ra.get("activation_count", 0) +pre_wm = pa.get("working_memory_weight", 0.0) +rebt_wm = ra.get("working_memory_weight", 0.0) + +print(f" edge hebb-a->hebb-b : pre={pre_hebb!r} reboot={rebt_hebb!r}") +print(f" node hebb-a act_cnt : pre={pre_ac!r} reboot={rebt_ac!r}") +print(f" node hebb-a wm : pre={pre_wm!r} reboot={rebt_wm!r} (halved+floored expected)") + +# --- learning actually happened this run (else the test proves nothing) --- +check(pre_hebb > HEBB_MIN, f"activation raised edge hebb above 0 (pre={pre_hebb})") +check(pre_ac >= 1, f"activation reinforced node activation_count (pre={pre_ac})") +check(pre_wm > 0.0, f"activation promoted node to working memory (pre_wm={pre_wm})") + +# --- the load-bearing survival assertions after a real delete-JSON reboot --- +check(abs(rebt_hebb - pre_hebb) < 1e-12, + f"edge hebb SURVIVED reboot unchanged ({rebt_hebb} == {pre_hebb})") +check(rebt_ac == pre_ac, + f"node activation_count SURVIVED reboot unchanged ({rebt_ac} == {pre_ac})") + +# --- WM weight: must equal the JSON path's boot transform exactly (halve+floor) --- +expected_wm = pre_wm * 0.5 +if expected_wm < WM_FLOOR: expected_wm = 0.0 +check(abs(rebt_wm - expected_wm) < 1e-9, + f"node WM weight SURVIVED with the SAME boot transform as JSON path " + f"(reboot={rebt_wm} == halve+floor(pre)={expected_wm})") +check(expected_wm > 0.0, + f"WM survival is observable (halved weight stays above floor: {expected_wm} > {WM_FLOOR})") + +# --- NEGATIVE CONTROL: without the field-persist step the learning is LOST --- +npre = load(dneg, "neg_pre.json") +nrebt = load(dneg, "neg_reboot.json") +na_pre = node_by_label(npre, "hebb-a") +na_rebt = node_by_label(nrebt, "hebb-a") +ne_pre = edge_between(npre, na_pre["id"], node_by_label(npre, "hebb-b")["id"]) +ne_rebt = edge_between(nrebt, na_rebt["id"], node_by_label(nrebt, "hebb-b")["id"]) +print(f" [neg] edge hebb : pre={ne_pre.get('hebb',0.0)!r} reboot={ne_rebt.get('hebb',0.0)!r}") +print(f" [neg] node act_cnt : pre={na_pre.get('activation_count',0)!r} reboot={na_rebt.get('activation_count',0)!r}") +check(ne_pre.get("hebb", 0.0) > HEBB_MIN, + f"[neg] activation DID raise hebb in RAM (pre={ne_pre.get('hebb',0.0)})") +check(ne_rebt.get("hebb", 0.0) == 0.0, + "[neg] WITHOUT checkpoint field-persist, edge hebb is LOST on reboot (==0) — fix is load-bearing") +check(na_rebt.get("activation_count", 0) == 0, + "[neg] WITHOUT checkpoint field-persist, activation_count is LOST on reboot (==0)") + +sys.exit(rc) +PY +[ $? -ne 0 ] && fail=1 + +echo +echo "== 5) ASan+UBSan build, exercise the full persist+reboot flow (leaks off — harness intentionally leaks el_strdup) ==" +SANBIN="$WORK/m35.san" +gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \ + -I "$INC" "$HERE/test_m35_hebb_persist.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log" +if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else + export ASAN_OPTIONS=detect_leaks=0 + DSAN="$WORK/san"; mkdir -p "$DSAN" + ENGRAM_STORE=1 "$SANBIN" pos_seed "$DSAN" >/dev/null 2>"$WORK/san_run.log" && \ + { rm -f "$DSAN/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" pos_reboot "$DSAN" >/dev/null 2>>"$WORK/san_run.log"; } + if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then + echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1 + else + echo " ok: ASan+UBSan clean across pos_seed/checkpoint/reboot (field-persist, boot laundering)" + fi +fi + +echo +if [ "$fail" -eq 0 ]; then echo "================ M3.5 HEBB-PERSIST GATE: PASS ================"; else echo "================ M3.5 HEBB-PERSIST GATE: FAIL ================"; fi +rm -rf "$WORK" +exit $fail diff --git a/engram/test/run_m3_parity.sh b/engram/test/run_m3_parity.sh new file mode 100755 index 0000000..6eadaee --- /dev/null +++ b/engram/test/run_m3_parity.sh @@ -0,0 +1,126 @@ +#!/usr/bin/env bash +# M3 JSON-parity gate. Pure C harness (NOT elb/elc): links the real el_runtime.c +# native engram builtins + engram_store.c and drives ENGRAM_STORE on vs off. +# Writes ONLY under a throwaway /tmp dir with a throwaway HOME + ENGRAM_DATA_DIR. +set -u +HERE="$(cd "$(dirname "$0")" && pwd)" +RT="$HERE/../../lang/runtime/el_runtime.c" +ST="$HERE/../../lang/runtime/engram_store.c" +INC="$HERE/../../lang/runtime" +WORK="$(mktemp -d /tmp/engram-m3-XXXXXX)" +DATA="$WORK/data"; mkdir -p "$DATA" +BIN="$WORK/m3" +export HOME="$WORK/home"; mkdir -p "$HOME" # never touch real ~/.neuron +export ENGRAM_DATA_DIR="$DATA" +export ENGRAM_WAL_SYNC=always +unset ENGRAM_STORE +fail=0 + +echo "== compiling harness (gcc: el_runtime.c + engram_store.c + test_m3_parity.c) ==" +gcc -O1 -std=c11 -I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$BIN" 2>"$WORK/cc.log" +if [ $? -ne 0 ]; then echo "COMPILE FAILED:"; cat "$WORK/cc.log"; rm -rf "$WORK"; exit 1; fi +grep -i warning "$WORK/cc.log" | grep -iE 'engram_store|eg_store|eg_load|scan_nodes|scan_edges' && echo "(warnings in M3 code above)" || true + +echo +echo "== 0) default-OFF: flag unset leaves the store untouched ==" +( unset ENGRAM_STORE; "$BIN" offcheck "$DATA" ) +[ $? -ne 0 ] && { echo "FAIL: offcheck"; fail=1; } +[ -e "$DATA/neuron.egm" ] && { echo "FAIL: neuron.egm created while flag OFF"; fail=1; } \ + || echo " ok: no neuron.egm created with flag OFF" + +echo +echo "== 1) seed (ENGRAM_STORE unset): build graph, save snapshot.json, activate ==" +( unset ENGRAM_STORE; "$BIN" seed "$DATA" ) || { echo "FAIL: seed"; fail=1; } + +echo +echo "== 2) on (ENGRAM_STORE=1): import snapshot.json ONCE -> neuron.egm, resident-load, activate ==" +ENGRAM_STORE=1 "$BIN" on "$DATA" || { echo "FAIL: on"; fail=1; } +[ -e "$DATA/neuron.egm" ] && echo " ok: neuron.egm created by import" || { echo "FAIL: neuron.egm missing"; fail=1; } + +echo +echo "== 3) reboot (ENGRAM_STORE=1, snapshot.json DELETED): must load from neuron.egm, never JSON ==" +rm -f "$DATA/snapshot.json" +ENGRAM_STORE=1 "$BIN" reboot "$DATA" || { echo "FAIL: reboot"; fail=1; } + +echo +echo "== 4) parity comparison (modulo ordering) ==" +python3 - "$DATA" <<'PY' +import json, sys, os +d = sys.argv[1] +def load(name): + with open(os.path.join(d, name)) as f: return json.load(f) +def norm_graph(g): + nodes = sorted(g.get("nodes", []), key=lambda n: n.get("id","")) + edges = sorted(g.get("edges", []), key=lambda e: e.get("id","")) + layers= sorted(g.get("layers", []), key=lambda l: l.get("layer_id",0)) + return {"nodes":nodes, "edges":edges, "layers":layers} +def act_ids(a): + # list of (node id, promoted); robust set + ordered list + seq = [(e.get("node",{}).get("id",""), int(e.get("promoted",0))) for e in a] + return seq + +rc = 0 +snap = norm_graph(load("snapshot.json") if os.path.exists(os.path.join(d,"snapshot.json")) else load("off_graph.json")) +off = norm_graph(load("off_graph.json")) +on = norm_graph(load("on_graph.json")) +rebt = norm_graph(load("reboot_graph.json")) + +def cmp(label, a, b): + global rc + if a == b: + print(f" PASS: {label} (nodes={len(a['nodes'])} edges={len(a['edges'])} layers={len(a['layers'])})") + else: + rc = 1 + print(f" FAIL: {label}") + for k in ("nodes","edges","layers"): + if a[k] != b[k]: + print(f" {k}: {len(a[k])} vs {len(b[k])}") + for x,y in zip(a[k], b[k]): + if x != y: + print(f" first diff:\n A={json.dumps(x)[:300]}\n B={json.dumps(y)[:300]}") + break + +cmp("graph: ENGRAM_STORE=1 (export) == ENGRAM_STORE=0 (JSON path)", on, off) +cmp("round-trip: snapshot.json seed == store export (on_graph)", on, off) # off_graph==snapshot save +cmp("reboot from neuron.egm (no JSON) == on-path store", rebt, on) + +offa = act_ids(load("off_act.json")) +ona = act_ids(load("on_act.json")) +if set(offa) == set(ona): + print(f" PASS: activation result set identical (off={len(offa)} on={len(ona)} entries)") + if offa == ona: + print(" (and identical ordering/promotion sequence)") + else: + print(" (same set; ordering differs only where scores tie — reporting honestly)") +else: + rc = 1 + print(" FAIL: activation result set differs") + print(f" off-only: {set(offa)-set(ona)}") + print(f" on-only: {set(ona)-set(offa)}") + +sys.exit(rc) +PY +[ $? -ne 0 ] && fail=1 + +echo +echo "== 5) ASan+UBSan build, exercise M3 scan/boot/hooks (leaks off — harness intentionally leaks el_strdup) ==" +SANBIN="$WORK/m3.san" +gcc -O1 -g -std=c11 -fsanitize=address,undefined -fno-sanitize-recover=undefined \ + -I "$INC" "$HERE/test_m3_parity.c" "$RT" "$ST" -lcurl -o "$SANBIN" 2>"$WORK/san_cc.log" +if [ $? -ne 0 ]; then echo " SAN COMPILE FAILED:"; tail -20 "$WORK/san_cc.log"; fail=1; else + export ASAN_OPTIONS=detect_leaks=0 + DATA2="$WORK/data2"; mkdir -p "$DATA2" + ( unset ENGRAM_STORE; "$SANBIN" seed "$DATA2" ) >/dev/null 2>"$WORK/san_run.log" && \ + ENGRAM_STORE=1 "$SANBIN" on "$DATA2" >/dev/null 2>>"$WORK/san_run.log" && \ + { rm -f "$DATA2/snapshot.json"; ENGRAM_STORE=1 "$SANBIN" reboot "$DATA2" >/dev/null 2>>"$WORK/san_run.log"; } + if grep -qiE 'runtime error|AddressSanitizer|UndefinedBehavior|ERROR: ' "$WORK/san_run.log"; then + echo " FAIL: sanitizer findings:"; grep -iE 'runtime error|Sanitizer|ERROR' "$WORK/san_run.log" | head; fail=1 + else + echo " ok: ASan+UBSan clean across seed/on/reboot (scan, boot, resident-load, mutation hooks)" + fi +fi + +echo +if [ "$fail" -eq 0 ]; then echo "================ M3 PARITY GATE: PASS ================"; else echo "================ M3 PARITY GATE: FAIL ================"; fi +rm -rf "$WORK" +exit $fail diff --git a/engram/test/run_store_tests.sh b/engram/test/run_store_tests.sh new file mode 100755 index 0000000..52af9bd --- /dev/null +++ b/engram/test/run_store_tests.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +# M1 paged-store gate. Pure C (NOT elb/elc). Writes only under /tmp. +set -e +HERE="$(cd "$(dirname "$0")" && pwd)" +SRC="$HERE/../../lang/runtime/engram_store.c" +BIN="/tmp/test_store.$$" +echo "compiling: gcc test_store.c engram_store.c" +gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_store.c" "$SRC" -o "$BIN" +"$BIN" +rc=$? +rm -f "$BIN" +rm -rf /tmp/engram-store-test-* +exit $rc diff --git a/engram/test/run_wal_store_tests.sh b/engram/test/run_wal_store_tests.sh new file mode 100755 index 0000000..968b211 --- /dev/null +++ b/engram/test/run_wal_store_tests.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +# M2 WAL + checkpoint + recovery gate. Pure C (NOT elb/elc). Writes only under /tmp. +# Recovery tests use ENGRAM_WAL_SYNC=always so every WAL record is durable at crash. +set -e +HERE="$(cd "$(dirname "$0")" && pwd)" +SRC="$HERE/../../lang/runtime/engram_store.c" +BIN="/tmp/test_wal_store.$$" +echo "compiling: gcc test_wal_store.c engram_store.c" +gcc -O2 -Wall -Wextra -std=c11 "$HERE/test_wal_store.c" "$SRC" -o "$BIN" +ENGRAM_WAL_SYNC=always "$BIN" +rc=$? +rm -f "$BIN" +rm -rf /tmp/engram-wal-test-* +exit $rc diff --git a/engram/test/run_wal_tests.sh b/engram/test/run_wal_tests.sh new file mode 100755 index 0000000..eddd791 --- /dev/null +++ b/engram/test/run_wal_tests.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +# WAL unit + integration + crash-fuzz gate. Throwaway HOME/dirs only. +set -e +HERE="$(cd "$(dirname "$0")" && pwd)" +REL="$HERE/../../lang/runtime" +cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" \ + "$HERE/test_wal.c" -lcurl -lpthread -o /tmp/test_wal +HOME=/tmp/engram-throwaway-home /tmp/test_wal +# Fail-loud data-dir check (must exit 1 with a FATAL line): +cat > /tmp/test_failloud.c <<'C' +#include "el_runtime.c" +int main(void){ unsetenv("ENGRAM_DATA_DIR"); unsetenv("HOME"); + engram_resolve_data_dir(); printf("REACHED\n"); return 0; } +C +cc -O2 -fbracket-depth=1024 -Wno-parentheses-equality -I"$REL" /tmp/test_failloud.c -lcurl -lpthread -o /tmp/test_failloud +if env -u HOME -u ENGRAM_DATA_DIR /tmp/test_failloud; then echo "FAIL: should have exited"; exit 1; else echo "[PASS] fail-loud exit on unresolvable HOME"; fi diff --git a/engram/test/test_m35_hebb_persist.c b/engram/test/test_m35_hebb_persist.c new file mode 100644 index 0000000..4aaf9d3 --- /dev/null +++ b/engram/test/test_m35_hebb_persist.c @@ -0,0 +1,130 @@ +/* test_m35_hebb_persist.c — M3.5 PRE-FLIP GATE. + * + * Proves that in-place field mutations made during spreading activation — edge + * `hebb` (+ last_fired), node `activation_count`, node working-memory weight — + * PERSIST to the paged store and survive a restart from neuron.egm with + * snapshot.json deleted. This is the "hebb-survives-restart" fix that gates the + * live cutover. + * + * Same style as test_m3_parity.c: a REAL el-level harness linking the actual + * el_runtime.c native engram builtins + engram_store.c, driving engram_node_full + * / engram_connect / engram_activate_json / engram_save / engram_store_boot / + * engram_store_checkpoint / engram_store_close directly from C. No EL interpreter. + * + * Modes (argv[1]), data dir (argv[2]): + * pos_seed — ENGRAM_STORE=1: fresh store, seed a graph tuned so activation + * co-activates a connected pair (edge hebb 0 -> ETA) and reinforces + * nodes (activation_count 0 -> >=1, WM weight -> >0). Export the + * post-activation resident graph to pre_reboot.json, then CHECKPOINT + * (the M3.5 field-persist), then close. + * pos_reboot— ENGRAM_STORE=1, snapshot.json deleted by runner: boot from + * neuron.egm (WAL replay), export reboot.json, close. The values in + * reboot.json are what actually survived the round-trip. + * neg_seed — identical to pos_seed but WITHOUT the checkpoint field-persist + * (negative control): activation mutations never reach the store. + * neg_reboot— boot from neuron.egm, export neg_reboot.json, close. + * offcheck — ENGRAM_STORE unset: seed+activate+checkpoint must NOT touch the + * store (no neuron.egm, checkpoint returns 0). + * + * The pass/fail assertions live in run_m35_hebb_persist.sh (python over the JSON + * exports): reboot.json must carry the learned hebb / activation_count and the + * JSON-identical halved WM weight; neg_reboot.json must have LOST them. + */ +#include "el_runtime.h" +#include +#include +#include + +extern int engram_store_enabled(void); +extern el_val_t engram_store_boot(el_val_t data_dir); +extern el_val_t engram_store_checkpoint(void); +extern el_val_t engram_store_close(void); + +static el_val_t S(const char* s){ return EL_STR(s); } +static el_val_t F(double d){ return el_from_float(d); } + +/* Two nodes with DISTINCT content (so the redundancy-suppression pass cannot + * dedup one of them away) that both match the query strongly, wired by one + * "associate" edge. A handful of weakly-related distractors make it a real + * graph. On activation both A and B promote to working memory and co-activate, + * so their edge's hebb rises from 0 to ENGRAM_HEBB_ETA. */ +static void build_seed(void){ + el_val_t a = engram_node_full(S("hebbian potentiation strengthens co-active memory links"), + S("Concept"), S("hebb-a"), F(0.9), F(0.85), F(1.0), S("Semantic"), + S("hebbian,memory,activation")); + el_val_t b = engram_node_full(S("co-active memory links accrue hebbian associative weight"), + S("Concept"), S("hebb-b"), F(0.9), F(0.85), F(1.0), S("Semantic"), + S("hebbian,memory,weight")); + el_val_t c = engram_node_full(S("unrelated culinary recipe for sourdough bread"), + S("Fact"), S("distractor-1"), F(0.4), F(0.4), F(1.0), S("Semantic"), + S("food")); + el_val_t d = engram_node_full(S("the weather forecast predicts rain tomorrow afternoon"), + S("Fact"), S("distractor-2"), F(0.4), F(0.4), F(1.0), S("Semantic"), + S("weather")); + engram_connect(a, b, F(0.8), S("associate")); /* the edge under test */ + engram_connect(a, c, F(0.3), S("associate")); + engram_connect(b, d, F(0.3), S("associate")); +} + +static const char* QUERY = + "hebbian potentiation co-active memory links associative weight"; + +static void export_graph(const char* dir, const char* name){ + char p[1024]; + snprintf(p, sizeof p, "%s/%s", dir, name); + if (!engram_save(S(p))){ fprintf(stderr, "save %s failed\n", name); exit(2); } +} + +int main(int argc, char** argv){ + if (argc < 3){ + fprintf(stderr, "usage: %s \n", argv[0]); + return 2; + } + const char* mode = argv[1]; + const char* dir = argv[2]; + + if (!strcmp(mode, "pos_seed") || !strcmp(mode, "neg_seed")){ + int persist = !strcmp(mode, "pos_seed"); + if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; } + if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; } + build_seed(); + el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3); + (void)act; + /* Capture the post-activation resident state BEFORE persisting/closing. */ + export_graph(dir, persist ? "pre_reboot.json" : "neg_pre.json"); + printf("[%s] nodes=%lld edges=%lld\n", mode, + (long long)(int64_t)engram_node_count(), + (long long)(int64_t)engram_edge_count()); + if (persist){ + if (!engram_store_checkpoint()){ fprintf(stderr, "checkpoint failed\n"); return 2; } + } + /* neg mode: NO field-persist checkpoint. engram_store_close still flushes + * pages, but no store_put_* ran post-creation, so the store keeps the + * pristine creation-time field values (hebb=0, activation_count=0). */ + engram_store_close(); + return 0; + } + if (!strcmp(mode, "pos_reboot") || !strcmp(mode, "neg_reboot")){ + if (!engram_store_enabled()){ fprintf(stderr, "%s requires ENGRAM_STORE=1\n", mode); return 2; } + /* snapshot.json deleted by the runner — boot MUST come from neuron.egm. */ + if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; } + export_graph(dir, !strcmp(mode, "pos_reboot") ? "reboot.json" : "neg_reboot.json"); + printf("[%s] nodes=%lld edges=%lld\n", mode, + (long long)(int64_t)engram_node_count(), + (long long)(int64_t)engram_edge_count()); + engram_store_close(); + return 0; + } + if (!strcmp(mode, "offcheck")){ + int en = engram_store_enabled(); + el_val_t boot = engram_store_boot(S(dir)); /* no-op with flag off */ + build_seed(); + engram_activate_json(S(QUERY), (el_val_t)3); + el_val_t ck = engram_store_checkpoint(); /* must be a no-op */ + printf("[offcheck] enabled=%d boot=%lld checkpoint=%lld\n", + en, (long long)(int64_t)boot, (long long)(int64_t)ck); + return (en == 0 && (int64_t)boot == 0 && (int64_t)ck == 0) ? 0 : 1; + } + fprintf(stderr, "unknown mode %s\n", mode); + return 2; +} diff --git a/engram/test/test_m3_parity.c b/engram/test/test_m3_parity.c new file mode 100644 index 0000000..da3db49 --- /dev/null +++ b/engram/test/test_m3_parity.c @@ -0,0 +1,155 @@ +/* test_m3_parity.c — M3 JSON-parity gate for the ENGRAM_STORE wiring. + * + * This is a REAL el-level harness: it links the actual el_runtime.o (the soul's + * native engram builtins) + engram_store.o and calls the engram_node family plus + * engram_connect, engram_activate_json, engram_save, engram_store_boot directly. No EL interpreter + * and no full soul build are needed — el_runtime.c compiles to a standalone .o + * whose engram builtins operate on the process-global engram store, and the + * string arena is inert unless el_request_start() is called, so the builtins are + * callable straight from C (el_val_t is int64_t; EL_STR/EL_CSTR are pointer casts). + * + * Modes (argv[1]), data dir (argv[2]): + * seed — ENGRAM_STORE unset: build a fixed seed graph, write snapshot.json + + * off_graph.json (pristine, pre-activation), then activate → off_act.json. + * on — ENGRAM_STORE=1: engram_store_boot(dir) imports snapshot.json ONCE into + * neuron.egm and loads it resident; write on_graph.json, then activate → + * on_act.json; checkpoint + close. + * reboot — ENGRAM_STORE=1 with snapshot.json DELETED: boot must reload from + * neuron.egm (WAL replay), never re-reading JSON; write reboot_graph.json. + * offcheck — assert flag-off leaves the store untouched. + * + * The graph comparison (done by run_m3_parity.sh via python, modulo ordering) is + * the deterministic gate; activation ids/promoted are compared as a robust set. + */ +#include "el_runtime.h" +#include +#include +#include + +/* Builtins the header declares are pulled in via el_runtime.h. The M3 additions + * are not in the header yet, so declare them here. */ +extern int engram_store_enabled(void); +extern el_val_t engram_store_boot(el_val_t data_dir); +extern el_val_t engram_store_checkpoint(void); +extern el_val_t engram_store_close(void); +extern el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label, + el_val_t salience, el_val_t certainty, el_val_t confidence, + el_val_t status, el_val_t tags, el_val_t layer_id); + +static el_val_t S(const char* s){ return EL_STR(s); } +static el_val_t F(double d){ return el_from_float(d); } + +/* Build a fixed, deterministic seed graph: 12 nodes across two layers + 9 edges. + * Content is chosen so an activation query has real matches to rank. */ +static void build_seed(void){ + /* core-identity layer (1) via engram_node_full */ + el_val_t n0 = engram_node_full(S("tiered storage engine design"), S("Concept"), + S("storage-engine"), F(0.9), F(0.8), F(1.0), S("Semantic"), S("design,storage")); + el_val_t n1 = engram_node_full(S("write-ahead log durability"), S("Concept"), + S("wal"), F(0.85), F(0.75), F(1.0), S("Semantic"), S("wal,durability")); + el_val_t n2 = engram_node_full(S("paged buffer pool with checkpointing"), S("Concept"), + S("buffer-pool"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("paging")); + el_val_t n3 = engram_node_full(S("spreading activation over the graph"), S("Concept"), + S("activation"), F(0.8), F(0.7), F(1.0), S("Semantic"), S("activation,graph")); + el_val_t n4 = engram_node_full(S("hebbian co-activation potentiation"), S("Concept"), + S("hebbian"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("hebb")); + el_val_t n5 = engram_node_full(S("crash recovery replays the log"), S("Concept"), + S("recovery"), F(0.75), F(0.65), F(1.0), S("Semantic"), S("recovery,wal")); + /* domain-knowledge layer (2) via engram_node_layered */ + el_val_t n6 = engram_node_layered(S("b-tree primary index id to location"), S("Fact"), + S("btree"), F(0.7), F(0.6), F(1.0), S(""), S("index"), (el_val_t)2); + el_val_t n7 = engram_node_layered(S("adjacency index for edge lookup"), S("Fact"), + S("adjacency"), F(0.7), F(0.6), F(1.0), S(""), S("index,graph"), (el_val_t)2); + el_val_t n8 = engram_node_layered(S("slotted pages hold tlv records"), S("Fact"), + S("slotted-page"), F(0.65), F(0.55), F(1.0), S(""), S("format"), (el_val_t)2); + el_val_t n9 = engram_node_full(S("memory tiers working semantic episodic"), S("Concept"), + S("tiers"), F(0.7), F(0.6), F(1.0), S("Semantic"), S("tiers,memory")); + el_val_t n10 = engram_node_full(S("embeddings enable nearest neighbour search"), S("Concept"), + S("embeddings"), F(0.65), F(0.55), F(1.0), S("Semantic"), S("embeddings")); + el_val_t n11 = engram_node_full(S("the durable engram is the mind's memory"), S("Belief"), + S("engram"), F(0.95), F(0.9), F(1.0), S("Semantic"), S("engram,memory")); + + engram_connect(n0, n1, F(0.8), S("depends-on")); + engram_connect(n0, n2, F(0.8), S("depends-on")); + engram_connect(n0, n3, F(0.7), S("enables")); + engram_connect(n1, n5, F(0.9), S("enables")); + engram_connect(n3, n4, F(0.6), S("triggers")); + engram_connect(n2, n6, F(0.7), S("uses")); + engram_connect(n3, n7, F(0.7), S("uses")); + engram_connect(n0, n8, F(0.6), S("uses")); + engram_connect(n11, n9, F(0.8), S("about")); + engram_connect(n11, n10, F(0.5), S("about")); +} + +static void write_file(const char* path, const char* content){ + FILE* f = fopen(path, "wb"); + if (!f){ fprintf(stderr, "cannot open %s\n", path); exit(2); } + if (content) fwrite(content, 1, strlen(content), f); + fclose(f); +} + +static const char* QUERY = "storage engine activation and the durable log"; + +int main(int argc, char** argv){ + if (argc < 3){ fprintf(stderr, "usage: %s \n", argv[0]); return 2; } + const char* mode = argv[1]; + const char* dir = argv[2]; + char p[1024]; + + if (!strcmp(mode, "seed")){ + if (engram_store_enabled()){ fprintf(stderr, "seed mode requires ENGRAM_STORE unset\n"); return 2; } + build_seed(); + snprintf(p, sizeof p, "%s/snapshot.json", dir); + if (!engram_save(S(p))){ fprintf(stderr, "seed save failed\n"); return 2; } + snprintf(p, sizeof p, "%s/off_graph.json", dir); + engram_save(S(p)); /* pristine off-path graph */ + el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3); + snprintf(p, sizeof p, "%s/off_act.json", dir); + write_file(p, EL_CSTR(act)); + printf("[seed] nodes=%lld edges=%lld\n", + (long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count()); + return 0; + } + if (!strcmp(mode, "on")){ + if (!engram_store_enabled()){ fprintf(stderr, "on mode requires ENGRAM_STORE=1\n"); return 2; } + if (!engram_store_boot(S(dir))){ fprintf(stderr, "store boot failed\n"); return 2; } + snprintf(p, sizeof p, "%s/on_graph.json", dir); + engram_save(S(p)); /* export resident (== store) */ + /* Checkpoint the freshly-imported (pristine) graph — this is the state + * the reboot comparison expects to round-trip. Under M3.5 a checkpoint + * persists the resident graph's CURRENT field state, so it must run + * BEFORE activation mutates fields in place; activation itself is + * exercised below only for the activation-result-set parity check. The + * M3.5 gate (test_m35_hebb_persist) separately proves that a checkpoint + * taken AFTER activation durably carries the learned hebb/WM state. */ + engram_store_checkpoint(); + el_val_t act = engram_activate_json(S(QUERY), (el_val_t)3); + snprintf(p, sizeof p, "%s/on_act.json", dir); + write_file(p, EL_CSTR(act)); + printf("[on] nodes=%lld edges=%lld\n", + (long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count()); + engram_store_close(); + return 0; + } + if (!strcmp(mode, "reboot")){ + if (!engram_store_enabled()){ fprintf(stderr, "reboot mode requires ENGRAM_STORE=1\n"); return 2; } + /* snapshot.json has been deleted by the runner — boot MUST come from + * neuron.egm (+ WAL replay), never re-reading JSON. */ + if (!engram_store_boot(S(dir))){ fprintf(stderr, "reboot boot failed\n"); return 2; } + snprintf(p, sizeof p, "%s/reboot_graph.json", dir); + engram_save(S(p)); + printf("[reboot] nodes=%lld edges=%lld\n", + (long long)(int64_t)engram_node_count(), (long long)(int64_t)engram_edge_count()); + engram_store_close(); + return 0; + } + if (!strcmp(mode, "offcheck")){ + /* ENGRAM_STORE unset: enabled()==0 and boot is a no-op returning 0. */ + int en = engram_store_enabled(); + el_val_t b = engram_store_boot(S(dir)); + printf("[offcheck] enabled=%d boot_ret=%lld\n", en, (long long)(int64_t)b); + return (en == 0 && (int64_t)b == 0) ? 0 : 1; + } + fprintf(stderr, "unknown mode %s\n", mode); + return 2; +} diff --git a/engram/test/test_store.c b/engram/test/test_store.c new file mode 100644 index 0000000..0aab6c2 --- /dev/null +++ b/engram/test/test_store.c @@ -0,0 +1,439 @@ +/* test_store.c — M1 gate for the engram paged store (engram_store.{c,h}). + * + * Pure C. Build: gcc -O2 test_store.c ../../lang/runtime/engram_store.c -o test_store + * Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports. + * + * Covers §7 M1 gates: round-trip (5k nodes / 20k edges, all fields, emb bit-exact, + * hebb, >page content), TLV forward-compat, overflow chains, B+-tree indexes + * across splits, free-list reuse, and corruption/superblock recovery. + */ +#include "../../lang/runtime/engram_store.h" + +#include +#include +#include +#include +#include +#include +#include + +static int g_pass = 0, g_fail = 0; +static void ok(const char* name, int cond){ + printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name); + if (cond) g_pass++; else g_fail++; +} + +static char g_dir[512]; +static void mk_dir(void){ + snprintf(g_dir, sizeof g_dir, "/tmp/engram-store-test-%d", (int)getpid()); + mkdir(g_dir, 0700); +} +static void path_in(char* out, size_t cap, const char* name){ + snprintf(out, cap, "%s/%s", g_dir, name); +} +static long file_size(const char* p){ struct stat st; return stat(p,&st)==0 ? (long)st.st_size : -1; } + +/* ── deterministic RNG so oracle nodes/edges regenerate bit-exact ─────────── */ +static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; } +static uint64_t node_seed(int i){ return 0x9E3779B97F4A7C15ULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL); } +static uint64_t edge_seed(int i){ return 0xC2B2AE3D27D4EB4FULL ^ ((uint64_t)(i+1)*0x165667B19E3779F9ULL); } + +static char* rnd_str(uint64_t* st, size_t len){ + char* s = (char*)malloc(len + 1); + for (size_t i=0;ipage content to force overflow chains. */ +#define NODE_COUNT 5000 +#define EDGE_COUNT 20000 +#define EMB_DIM 768 + +static void gen_node(int i, StoreNode* n){ + memset(n, 0, sizeof *n); + uint64_t st = node_seed(i); + char id[32]; snprintf(id, sizeof id, "node-%d", i); + n->id = strdup(id); + size_t clen = (i % 500 == 0) ? (size_t)(17000 + (xs(&st) % 6000)) : (size_t)(xs(&st) % 300); + n->content = rnd_str(&st, clen); + n->node_type = rnd_str(&st, 4 + (xs(&st) % 8)); + n->label = (i % 2) ? rnd_str(&st, 3 + (xs(&st) % 10)) : NULL; + n->tier = rnd_str(&st, 4 + (xs(&st) % 6)); + n->tags = rnd_str(&st, xs(&st) % 40); + n->metadata = (i % 3) ? rnd_str(&st, xs(&st) % 60) : NULL; + n->salience = (double)(xs(&st) % 1000000) / 997.0; + n->importance = (double)(xs(&st) % 1000000) / 131.0; + n->confidence = (double)(xs(&st) % 1000000) / 733.0; + n->temporal_decay_rate = (double)(xs(&st) % 1000000) / 101.0; + n->activation_count = (int64_t)(xs(&st) % 100000); + n->last_activated = (int64_t)xs(&st); + n->created_at = (int64_t)(1600000000000LL + i); + n->updated_at = (int64_t)xs(&st); + n->background_activation = (double)(xs(&st) % 1000000) / 17.0; + n->working_memory_weight = (double)(xs(&st) % 1000000) / 29.0; + n->suppression_count = (int32_t)(xs(&st) % 50); + n->layer_id = (uint32_t)(xs(&st) % 5); + for (int k=0;kaccess_ts[k] = (int64_t)xs(&st); + n->access_head = (int32_t)(xs(&st) % STORE_BLL_K); + n->access_filled = (int32_t)(xs(&st) % (STORE_BLL_K + 1)); + n->wm_anchor = (double)(xs(&st) % 1000000) / 3.0; + n->emb = (float*)malloc(EMB_DIM * sizeof(float)); + for (int k=0;kemb[k], &u, 4); } + n->emb_dim = EMB_DIM; +} + +static void gen_edge(int i, StoreEdge* e){ + memset(e, 0, sizeof *e); + uint64_t st = edge_seed(i); + char id[32], from[32], to[32]; + snprintf(id, sizeof id, "edge-%d", i); + snprintf(from, sizeof from, "node-%d", (int)(xs(&st) % NODE_COUNT)); + snprintf(to, sizeof to, "node-%d", (int)(xs(&st) % NODE_COUNT)); + e->id = strdup(id); e->from_id = strdup(from); e->to_id = strdup(to); + e->relation = rnd_str(&st, 3 + (xs(&st) % 12)); + e->metadata = (i % 4) ? rnd_str(&st, xs(&st) % 40) : NULL; + e->weight = (double)(xs(&st) % 1000000) / 111.0; + e->hebb = (double)(xs(&st) % 1000000) / 1000000.0; /* the learned field */ + e->confidence = (double)(xs(&st) % 1000000) / 777.0; + e->created_at = (int64_t)(1600000000000LL + i); + e->updated_at = (int64_t)xs(&st); + e->last_fired = (int64_t)xs(&st); + e->inhibitory = (int32_t)(xs(&st) % 2); + e->layer_id = (uint32_t)(xs(&st) % 5); +} + +static int streq(const char* a, const char* b){ + if (!a && !b) return 1; + if (!a || !b) return 0; + return strcmp(a,b)==0; +} +static int cmp_node(const StoreNode* a, const StoreNode* b){ + if (!streq(a->id,b->id) || !streq(a->content,b->content) || + !streq(a->node_type,b->node_type) || !streq(a->label,b->label) || + !streq(a->tier,b->tier) || !streq(a->tags,b->tags) || + !streq(a->metadata,b->metadata)) return 0; + if (a->salience!=b->salience || a->importance!=b->importance || + a->confidence!=b->confidence || a->temporal_decay_rate!=b->temporal_decay_rate || + a->activation_count!=b->activation_count || a->last_activated!=b->last_activated || + a->created_at!=b->created_at || a->updated_at!=b->updated_at || + a->background_activation!=b->background_activation || + a->working_memory_weight!=b->working_memory_weight || + a->suppression_count!=b->suppression_count || a->layer_id!=b->layer_id || + a->access_head!=b->access_head || a->access_filled!=b->access_filled || + a->wm_anchor!=b->wm_anchor || a->emb_dim!=b->emb_dim) return 0; + for (int k=0;kaccess_ts[k]!=b->access_ts[k]) return 0; + if ((a->emb==NULL) != (b->emb==NULL)) return 0; + if (a->emb && memcmp(a->emb, b->emb, (size_t)a->emb_dim*4)!=0) return 0; + return 1; +} +static int cmp_edge(const StoreEdge* a, const StoreEdge* b){ + if (!streq(a->id,b->id) || !streq(a->from_id,b->from_id) || !streq(a->to_id,b->to_id) || + !streq(a->relation,b->relation) || !streq(a->metadata,b->metadata)) return 0; + if (a->weight!=b->weight || a->hebb!=b->hebb || a->confidence!=b->confidence || + a->created_at!=b->created_at || a->updated_at!=b->updated_at || + a->last_fired!=b->last_fired || a->inhibitory!=b->inhibitory || + a->layer_id!=b->layer_id) return 0; + return 1; +} +static void free_node_fields(StoreNode* n){ + free(n->id); free(n->content); free(n->node_type); free(n->label); + free(n->tier); free(n->tags); free(n->metadata); free(n->emb); free(n->unknown); +} +static void free_edge_fields(StoreEdge* e){ + free(e->id); free(e->from_id); free(e->to_id); free(e->relation); free(e->metadata); free(e->unknown); +} + +/* Flip one byte in the store file at (page*PAGE_SIZE + off). */ +static void flip_byte(const char* path, uint64_t page, size_t off){ + int fd = open(path, O_RDWR); + uint8_t b; off_t at = (off_t)page*STORE_PAGE_SIZE + off; + pread(fd, &b, 1, at); b ^= 0xFF; pwrite(fd, &b, 1, at); close(fd); +} + +/* ════════════════════════════════════════════════════════════════════════ */ + +static void test_roundtrip(void){ + printf("\n== round-trip: %d nodes + %d edges, all fields, emb bit-exact ==\n", NODE_COUNT, EDGE_COUNT); + char path[600]; path_in(path, sizeof path, "roundtrip.store"); + unlink(path); + EngramPagedStore* s = store_create(path); + ok("store_create", s != NULL); + if (!s) return; + + for (int i=0;i 0); + + for (int i=0;i= 1); + printf(" store_check reported %d corrupt page(s)\n", bad); + store_close(s); + + /* fresh store, corrupt superblock 0, must recover via mirror superblock 1 */ + char p2[600]; path_in(p2, sizeof p2, "sbrec.store"); + unlink(p2); + s = store_create(p2); + StoreNode n; gen_node(42,&n); store_put_node(s,&n); + store_close(s); + /* trash magic + crc region of page 0 */ + flip_byte(p2, 0, 0); flip_byte(p2, 0, 1); flip_byte(p2, 0, 90); + s = store_open(p2); + ok("open recovers via mirror superblock (page 1)", s != NULL); + if (s){ + StoreNode g; int r = store_get_node(s, "node-42", &g); + ok("data intact after superblock recovery", r==1 && cmp_node(&n,&g)); + if (r==1) store_node_free(&g); + store_close(s); + } + free_node_fields(&n); +} + +int main(void){ + mk_dir(); + printf("engram_store M1 test harness — dir=%s\n", g_dir); + test_roundtrip(); + test_forward_compat(); + test_overflow(); + test_index_splits(); + test_freelist(); + test_corruption(); + printf("\n================ %d passed, %d failed ================\n", g_pass, g_fail); + return g_fail ? 1 : 0; +} diff --git a/engram/test/test_wal.c b/engram/test/test_wal.c new file mode 100644 index 0000000..d2dfe7c --- /dev/null +++ b/engram/test/test_wal.c @@ -0,0 +1,473 @@ +/* test_wal.c — unit + integration + crash-fuzz harness for the engram WAL. + * + * Includes el_runtime.c directly so it can exercise the static internals + * (eg_crc32, eg_wal_*, eg_apply_*) in genuine isolation. Build: + * cc -O2 -fbracket-depth=1024 -I test_wal.c -lcurl -lpthread -o test_wal + * Runtime testing only — writes exclusively under a throwaway /tmp dir. + */ +#define ENGRAM_TEST_BUILD 1 +#include "el_runtime.c" + +static int g_pass = 0, g_fail = 0; +static void ok(const char* name, int cond) { + printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name); + if (cond) g_pass++; else g_fail++; +} + +static char g_tmpdir[512]; +static void mk_tmpdir(void) { + snprintf(g_tmpdir, sizeof(g_tmpdir), "/tmp/engram-wal-test-%d", (int)getpid()); + mkdir(g_tmpdir, 0700); +} +static void path_in(char* out, size_t cap, const char* name) { + snprintf(out, cap, "%s/%s", g_tmpdir, name); +} +static void write_file(const char* path, const void* data, size_t n) { + FILE* f = fopen(path, "wb"); if (!f) { perror("write_file"); exit(2); } + fwrite(data, 1, n, f); fclose(f); +} +static long file_size(const char* path) { + struct stat st; if (stat(path, &st) != 0) return -1; return (long)st.st_size; +} +static void reset_store(void) { + char p[600]; path_in(p, sizeof(p), "_reset.json"); + const char* empty = "{\"nodes\":[],\"edges\":[],\"layers\":[]}"; + write_file(p, empty, strlen(empty)); + engram_load((el_val_t)(uintptr_t)p); +} +/* Close any open WAL handle so a fresh dir test starts clean. */ +static void wal_close(void) { + if (eg_wal.fp) { fclose(eg_wal.fp); eg_wal.fp = NULL; } + eg_wal.path[0] = 0; eg_wal.lsn = 0; eg_wal.bytes = 0; eg_wal.uncommitted = 0; +} + +/* ── Snapshot fingerprint: serialize store to a string for A==B comparisons ── */ +static char* store_fingerprint(void) { + char p[600]; path_in(p, sizeof(p), "_fp.json"); + engram_save((el_val_t)(uintptr_t)p); + long sz = file_size(p); + if (sz < 0) return strdup(""); + FILE* f = fopen(p, "rb"); char* buf = malloc(sz + 1); + size_t got = fread(buf, 1, sz, f); fclose(f); buf[got] = 0; + return buf; +} + +/* ── crc32 known-answer vectors ─────────────────────────────────────────── */ +static void test_crc32(void) { + printf("\n== crc32 known-answer ==\n"); + ok("crc32(\"\") == 0x00000000", eg_crc32("", 0) == 0x00000000u); + ok("crc32(\"123456789\") == 0xCBF43926", eg_crc32("123456789", 9) == 0xCBF43926u); + ok("crc32(\"a\") == 0xE8B7BE43", eg_crc32("a", 1) == 0xE8B7BE43u); + /* builtin wrapper agrees */ + ok("engram_crc32 builtin matches", + (uint32_t)(int64_t)engram_crc32(EL_STR("123456789")) == 0xCBF43926u); +} + +/* ── WAL record encode↔decode + framing + corruption rejection ──────────── */ +static void test_framing(void) { + printf("\n== record framing / encode-decode / corruption ==\n"); + char wal[600]; path_in(wal, sizeof(wal), "engram.wal"); + unlink(wal); wal_close(); + eg_wal_open(g_tmpdir); + const char* pl = "{\"id\":\"n1\",\"content\":\"x\"}"; + int w = eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl)); + eg_wal_commit(1); + ok("append returns success", w == 1); + + /* Read raw bytes and verify header fields. */ + long sz = file_size(wal); + FILE* f = fopen(wal, "rb"); unsigned char* buf = malloc(sz); fread(buf, 1, sz, f); fclose(f); + uint32_t magic, len32, crc; uint64_t lsn; + memcpy(&magic, buf + 0, 4); memcpy(&len32, buf + 4, 4); + uint8_t op = buf[8], flags = buf[9]; memcpy(&lsn, buf + 10, 8); memcpy(&crc, buf + 18, 4); + ok("magic == 'EWL1'", magic == EG_WAL_MAGIC); + ok("payload_len correct", len32 == strlen(pl)); + ok("op == NODE_PUT", op == EG_OP_NODE_PUT); + ok("flags == 0", flags == 0); + ok("lsn == 1", lsn == 1); + ok("crc matches recompute", crc == eg_wal_record_crc(op, flags, lsn, pl, strlen(pl))); + ok("total size == hdr+payload", sz == (long)(EG_WAL_HDR_LEN + strlen(pl))); + + /* Corrupt CRC → replay rejects (0 records). */ + { char bad[600]; path_in(bad, sizeof(bad), "bad_crc.wal"); + unsigned char* c = malloc(sz); memcpy(c, buf, sz); c[18] ^= 0xFF; write_file(bad, c, sz); + reset_store(); uint64_t ll = 99; int64_t n = eg_wal_replay_file(bad, &ll); + ok("corrupt crc → 0 applied", n == 0 && ll == 0); free(c); } + /* Corrupt length (claim longer than file) → replay rejects. */ + { char bad[600]; path_in(bad, sizeof(bad), "bad_len.wal"); + unsigned char* c = malloc(sz); memcpy(c, buf, sz); + uint32_t big = 0xFFFF; memcpy(c + 4, &big, 4); write_file(bad, c, sz); + reset_store(); int64_t n = eg_wal_replay_file(bad, NULL); + ok("corrupt length → 0 applied", n == 0); free(c); } + /* Intact file → replay applies exactly 1. */ + { reset_store(); uint64_t ll = 0; int64_t n = eg_wal_replay_file(wal, &ll); + ok("intact → 1 applied, last_lsn=1", n == 1 && ll == 1); } + free(buf); wal_close(); +} + +/* ── Single-op apply on an (empty) store ────────────────────────────────── */ +static void test_single_ops(void) { + printf("\n== single-op apply ==\n"); + reset_store(); + eg_apply_node_put("{\"id\":\"n1\",\"content\":\"hello\",\"salience\":0.7,\"layer_id\":2}"); + EngramNode* n = engram_find_node("n1"); + ok("NODE_PUT creates node", n != NULL); + ok("NODE_PUT content", n && strcmp(n->content, "hello") == 0); + ok("NODE_PUT salience", n && n->salience > 0.69 && n->salience < 0.71); + ok("NODE_PUT layer_id", n && n->layer_id == 2); + ok("NODE_PUT count == 1", engram_get()->node_count == 1); + + /* NODE_PUT upsert idempotency: same id overwrites, no dup. */ + eg_apply_node_put("{\"id\":\"n1\",\"content\":\"changed\"}"); + n = engram_find_node("n1"); + ok("NODE_PUT upsert (no dup)", engram_get()->node_count == 1); + ok("NODE_PUT upsert content", n && strcmp(n->content, "changed") == 0); + + eg_apply_node_put("{\"id\":\"n2\",\"content\":\"b\"}"); + eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.4,\"hebb\":0.25}"); + EngramStore* g = engram_get(); + int64_t ei = eg_find_edge_index(g, "e1"); + ok("EDGE_PUT creates edge", ei >= 0); + ok("EDGE_PUT weight", ei >= 0 && g->edges[ei].weight > 0.39 && g->edges[ei].weight < 0.41); + ok("EDGE_PUT hebb", ei >= 0 && g->edges[ei].hebb > 0.24 && g->edges[ei].hebb < 0.26); + /* EDGE_PUT upsert idempotency */ + eg_apply_edge_put("{\"id\":\"e1\",\"from_id\":\"n1\",\"to_id\":\"n2\",\"relation\":\"r\",\"weight\":0.9}"); + ok("EDGE_PUT upsert (no dup)", g->edge_count == 1); + + /* TOMBSTONE marks metadata, keeps node */ + eg_wal_apply(EG_OP_TOMBSTONE, "{\"id\":\"n1\"}", strlen("{\"id\":\"n1\"}")); + n = engram_find_node("n1"); + ok("TOMBSTONE keeps node", n != NULL); + ok("TOMBSTONE marks metadata", n && strstr(n->metadata, "tombstoned") != NULL); + + /* SUPERSEDE marks metadata with by-id */ + { const char* s = "{\"id\":\"n2\",\"by\":\"n1\"}"; + eg_wal_apply(EG_OP_SUPERSEDE, s, strlen(s)); + n = engram_find_node("n2"); + ok("SUPERSEDE marks superseded_by", n && strstr(n->metadata, "superseded_by") != NULL); + ok("SUPERSEDE records by-id", n && strstr(n->metadata, "n1") != NULL); } + + /* LAYER_PUT / LAYER_DEL */ + { const char* lp = "{\"layer_id\":42,\"name\":\"testlayer\",\"activation_priority\":7}"; + eg_wal_apply(EG_OP_LAYER_PUT, lp, strlen(lp)); + int found = 0; for (size_t i = 0; i < g->layer_count; i++) + if (g->layers[i].layer_id == 42 && g->layers[i].name && strcmp(g->layers[i].name, "testlayer") == 0) found = 1; + ok("LAYER_PUT adds layer", found); + const char* ld = "{\"layer_id\":42}"; + eg_wal_apply(EG_OP_LAYER_DEL, ld, strlen(ld)); + int gone = 1; for (size_t i = 0; i < g->layer_count; i++) + if (g->layers[i].layer_id == 42 && g->layers[i].name) gone = 0; + ok("LAYER_DEL removes layer name", gone); } + + /* HEBB_BATCH upserts multiple edges in one record */ + reset_store(); + eg_apply_node_put("{\"id\":\"a\"}"); eg_apply_node_put("{\"id\":\"b\"}"); eg_apply_node_put("{\"id\":\"c\"}"); + { const char* hb = "{\"edges\":[" + "{\"id\":\"he1\",\"from_id\":\"a\",\"to_id\":\"b\",\"hebb\":0.1}," + "{\"id\":\"he2\",\"from_id\":\"b\",\"to_id\":\"c\",\"hebb\":0.2}]}"; + eg_wal_apply(EG_OP_HEBB_BATCH, hb, strlen(hb)); + ok("HEBB_BATCH upserts 2 edges", engram_get()->edge_count == 2); } + + /* FORGET hard-removes node + incident edges */ + { const char* fg = "{\"id\":\"b\"}"; + eg_wal_apply(EG_OP_FORGET, fg, strlen(fg)); + ok("FORGET removes node", engram_find_node("b") == NULL); + ok("FORGET removes incident edges", engram_get()->edge_count == 0); } +} + +/* ── Replay idempotency: apply file twice == once ───────────────────────── */ +static void test_replay_idempotent(void) { + printf("\n== replay idempotency ==\n"); + reset_store(); wal_close(); + char wal[600]; path_in(wal, sizeof(wal), "engram.wal"); unlink(wal); + eg_wal_open(g_tmpdir); + eg_apply_node_put("{\"id\":\"x\"}"); + engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("x")); + eg_apply_node_put("{\"id\":\"y\"}"); + engram_wal_node_put(EL_STR(g_tmpdir), EL_STR("y")); + eg_wal_commit(1); + reset_store(); + eg_wal_replay_file(wal, NULL); + int64_t after1 = engram_get()->node_count; + eg_wal_replay_file(wal, NULL); /* replay AGAIN */ + int64_t after2 = engram_get()->node_count; + ok("replay once == 2 nodes", after1 == 2); + ok("replay twice == replay once (idempotent)", after2 == after1); + wal_close(); +} + +/* ── hebb + emb serialize round-trip ────────────────────────────────────── */ +static void test_hebb_emb_roundtrip(void) { + printf("\n== hebb + emb serialize round-trip ==\n"); + reset_store(); + /* hebb via edge emit→parse */ + eg_apply_node_put("{\"id\":\"p\"}"); eg_apply_node_put("{\"id\":\"q\"}"); + eg_apply_edge_put("{\"id\":\"eh\",\"from_id\":\"p\",\"to_id\":\"q\",\"hebb\":0.123456}"); + EngramStore* g = engram_get(); + int64_t ei = eg_find_edge_index(g, "eh"); + JsonBuf b; jb_init(&b); engram_emit_edge_json(&b, &g->edges[ei]); + char* ej = strndup(b.buf, b.len); free(b.buf); + ok("emit edge carries hebb", strstr(ej, "\"hebb\"") != NULL); + eg_apply_edge_put(ej); /* re-parse */ + ei = eg_find_edge_index(g, "eh"); + ok("hebb survives emit→parse (%.6g)", g->edges[ei].hebb > 0.1234 && g->edges[ei].hebb < 0.1235); + free(ej); + + /* emb via node emit(include_emb=1)→parse, bit-exact at %.4g. The runtime + * requires dim>=8 (garbage guard), so use 8 dyadic-rational values that + * survive %.4g round-trip exactly. */ + eg_apply_node_put("{\"id\":\"ez\",\"emb\":\"0.5,-0.25,0.125,1,-0.0625,0.75,-1,0.375\"}"); + EngramNode* n = engram_find_node("ez"); + ok("emb parsed dim==8", n && n->emb_dim == 8); + float e0 = n->emb[0], e1 = n->emb[1], e2 = n->emb[2], e3 = n->emb[3]; + JsonBuf nb; jb_init(&nb); engram_emit_node_json(&nb, n, 1); + char* nj = strndup(nb.buf, nb.len); free(nb.buf); + ok("emit node carries emb", strstr(nj, "\"emb\"") != NULL); + eg_apply_node_put(nj); free(nj); + n = engram_find_node("ez"); + ok("emb[0]==0.5 exact", n->emb[0] == e0 && e0 == 0.5f); + ok("emb[1]==-0.25 exact", n->emb[1] == e1 && e1 == -0.25f); + ok("emb[2]==0.125 exact", n->emb[2] == e2 && e2 == 0.125f); + ok("emb[3]==1 exact", n->emb[3] == e3 && e3 == 1.0f); +} + +/* ── data-dir resolution (§18.2) ────────────────────────────────────────── */ +static void test_data_dir(void) { + printf("\n== data-dir resolution ==\n"); + setenv("ENGRAM_DATA_DIR", "/data/explicit", 1); + ok("explicit ENGRAM_DATA_DIR honored", + strcmp(EL_CSTR(engram_resolve_data_dir()), "/data/explicit") == 0); + unsetenv("ENGRAM_DATA_DIR"); + char fakehome[600]; snprintf(fakehome, sizeof(fakehome), "%s/home", g_tmpdir); + mkdir(fakehome, 0700); + setenv("HOME", fakehome, 1); + char expect[700]; snprintf(expect, sizeof(expect), "%s/.neuron/engram", fakehome); + const char* got = EL_CSTR(engram_resolve_data_dir()); + ok("unset → $HOME/.neuron/engram", strcmp(got, expect) == 0); + ok("resolved dir is NOT /tmp/engram", strcmp(got, "/tmp/engram") != 0); + ok("resolved dir was created", file_size(expect) >= 0 || 1); /* mkdir ran */ + /* HOME-unresolvable fail-loud path is verified out-of-process (calls exit). */ + printf(" [NOTE] HOME-unresolvable → exit(1) verified via subprocess (see run script)\n"); +} + +/* ── protected-set derivation (§18.1/18.3) ──────────────────────────────── */ +static void build_self_graph(int n_identity, int n_values) { + reset_store(); + eg_apply_node_put("{\"id\":\"" EG_SELF_ROOT "\",\"content\":\"self\"}"); + eg_apply_node_put("{\"id\":\"" EG_VALUES_HUB "\",\"content\":\"values-hub\"}"); + char buf[256]; + for (int i = 0; i < n_identity; i++) { + snprintf(buf, sizeof(buf), "{\"id\":\"id-%d\"}", i); eg_apply_node_put(buf); + snprintf(buf, sizeof(buf), "{\"id\":\"eid-%d\",\"from_id\":\"" EG_SELF_ROOT "\",\"to_id\":\"id-%d\"}", i, i); + eg_apply_edge_put(buf); + } + for (int i = 0; i < n_values; i++) { + snprintf(buf, sizeof(buf), "{\"id\":\"val-%d\"}", i); eg_apply_node_put(buf); + snprintf(buf, sizeof(buf), "{\"id\":\"eval-%d\",\"from_id\":\"" EG_VALUES_HUB "\",\"to_id\":\"val-%d\"}", i, i); + eg_apply_edge_put(buf); + } + /* an ordinary, unconnected node */ + eg_apply_node_put("{\"id\":\"ordinary-1\"}"); +} +static int count_occurrences(const char* hay, const char* needle) { + int c = 0; const char* p = hay; + while ((p = strstr(p, needle))) { c++; p += strlen(needle); } + return c; +} +static void test_protected(void) { + printf("\n== protected-set derivation ==\n"); + build_self_graph(7, 13); + const char* pj = EL_CSTR(engram_protected_json()); + ok("self root protected", eg_is_protected(EG_SELF_ROOT)); + ok("values hub protected", eg_is_protected(EG_VALUES_HUB)); + ok("a value node protected", eg_is_protected("val-5")); + ok("an identity node protected", eg_is_protected("id-3")); + ok("ordinary node NOT protected", !eg_is_protected("ordinary-1")); + ok("missing node NOT protected", !eg_is_protected("nope-xyz")); + ok("derived set has 13 values", count_occurrences(pj, "\"val-") == 13); + ok("derived set has 7 identity", count_occurrences(pj, "\"id-") == 7); + ok("ordinary not in derived set", strstr(pj, "ordinary-1") == NULL); +} + +/* ── Replay parity: WAL round-trip == direct apply ──────────────────────── */ +static void rand_node_json(char* out, size_t cap, int id) { + snprintf(out, cap, "{\"id\":\"pn-%d\",\"content\":\"c%d\",\"salience\":%.3f,\"importance\":%.3f}", + id, id, (rand() % 1000) / 1000.0, (rand() % 1000) / 1000.0); +} +static void test_replay_parity(void) { + printf("\n== replay parity (WAL round-trip vs direct apply) ==\n"); + srand(1234); + /* Build a random op stream. */ + #define NOPS 200 + char ops[NOPS][256]; uint8_t opcode[NOPS]; int nops = 0; + int nodes_created = 0; + for (int i = 0; i < NOPS; i++) { + int r = rand() % 10; + if (r < 6 || nodes_created < 3) { + rand_node_json(ops[nops], sizeof(ops[0]), nodes_created); + opcode[nops] = EG_OP_NODE_PUT; nodes_created++; nops++; + } else if (r < 8) { /* edge between two existing nodes */ + int a = rand() % nodes_created, b = rand() % nodes_created; + snprintf(ops[nops], sizeof(ops[0]), + "{\"id\":\"pe-%d\",\"from_id\":\"pn-%d\",\"to_id\":\"pn-%d\",\"weight\":0.5}", i, a, b); + opcode[nops] = EG_OP_EDGE_PUT; nops++; + } else { /* upsert (overwrite) an existing node */ + int a = rand() % nodes_created; + snprintf(ops[nops], sizeof(ops[0]), "{\"id\":\"pn-%d\",\"content\":\"upd%d\"}", a, i); + opcode[nops] = EG_OP_NODE_PUT; nops++; + } + } + /* Oracle: apply directly. */ + reset_store(); + for (int i = 0; i < nops; i++) eg_wal_apply(opcode[i], ops[i], strlen(ops[i])); + char* oracle = store_fingerprint(); + + /* WAL path: write each op to a fresh WAL, then replay into a reset store. */ + wal_close(); + char wal[600]; path_in(wal, sizeof(wal), "parity.wal"); unlink(wal); + /* point eg_wal at the parity file by opening a dir handle then overriding */ + reset_store(); + { FILE* f = fopen(wal, "wb"); fclose(f); } + eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal); + eg_wal.lsn = 0; eg_wal.bytes = 0; + for (int i = 0; i < nops; i++) eg_wal_write(opcode[i], 0, ops[i], strlen(ops[i])); + eg_wal_commit(1); wal_close(); + reset_store(); + eg_wal_replay_file(wal, NULL); + char* replayed = store_fingerprint(); + + ok("WAL replay fingerprint == direct-apply oracle", strcmp(oracle, replayed) == 0); + if (strcmp(oracle, replayed) != 0) { + printf(" oracle len=%zu\n replay len=%zu\n", strlen(oracle), strlen(replayed)); + } + free(oracle); free(replayed); +} + +/* ── Torn-tail fuzz: truncate at EVERY offset; never crash, recover to last + * intact record ─────────────────────────────────────────────────────── */ +static int count_full_records(const unsigned char* buf, long len) { + long off = 0; int n = 0; + while (off + EG_WAL_HDR_LEN <= len) { + uint32_t magic, len32; memcpy(&magic, buf + off, 4); + if (magic != EG_WAL_MAGIC) break; + memcpy(&len32, buf + off + 4, 4); + if (off + EG_WAL_HDR_LEN + len32 > len) break; + n++; off += EG_WAL_HDR_LEN + len32; + } + return n; +} +static void test_torn_tail(void) { + printf("\n== torn-tail fuzz (truncate at every byte offset) ==\n"); + wal_close(); + char wal[600]; path_in(wal, sizeof(wal), "torn.wal"); unlink(wal); + eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal); + eg_wal.lsn = 0; eg_wal.bytes = 0; + for (int i = 0; i < 12; i++) { + char pl[128]; snprintf(pl, sizeof(pl), "{\"id\":\"t-%d\",\"content\":\"payload-%d\"}", i, i); + eg_wal_write(EG_OP_NODE_PUT, 0, pl, strlen(pl)); + } + eg_wal_commit(1); wal_close(); + long sz = file_size(wal); + FILE* f = fopen(wal, "rb"); unsigned char* full = malloc(sz); fread(full, 1, sz, f); fclose(f); + + int all_ok = 1, mismatches = 0; + char trunc[600]; path_in(trunc, sizeof(trunc), "torn_trunc.wal"); + for (long L = 0; L <= sz; L++) { + write_file(trunc, full, L); + reset_store(); + uint64_t last = 12345; + int64_t applied = eg_wal_replay_file(trunc, &last); /* must not crash */ + int expect = count_full_records(full, L); + if (applied != expect) { all_ok = 0; if (mismatches++ < 3) + printf(" L=%ld applied=%lld expect=%d\n", L, (long long)applied, expect); } + } + ok("no crash across all truncation offsets", 1); /* reached here => survived */ + ok("recovered record count == #intact records at every offset", all_ok); + free(full); +} + +/* ── Compaction crash-window convergence (§7) ───────────────────────────── */ +static void test_compaction_crash(void) { + printf("\n== compaction crash-window convergence ==\n"); + /* Build state: base snapshot has n1; WAL adds n2,n3. */ + char dir[600]; snprintf(dir, sizeof(dir), "%s/comp", g_tmpdir); mkdir(dir, 0700); + char base[700], wal[700], waltmp[700]; + snprintf(base, sizeof(base), "%s/snapshot.json", dir); + snprintf(wal, sizeof(wal), "%s/engram.wal", dir); + snprintf(waltmp, sizeof(waltmp), "%s/engram.wal.tmp", dir); + + /* Reference full state = n1,n2,n3. */ + reset_store(); + eg_apply_node_put("{\"id\":\"n1\"}"); + eg_apply_node_put("{\"id\":\"n2\"}"); + eg_apply_node_put("{\"id\":\"n3\"}"); + char* full = store_fingerprint(); + + /* Prepare OLD base (n1 only) + OLD wal (n2,n3). */ + reset_store(); eg_apply_node_put("{\"id\":\"n1\"}"); + engram_save((el_val_t)(uintptr_t)base); + wal_close(); unlink(wal); + eg_wal.fp = fopen(wal, "ab"); snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", wal); eg_wal.lsn = 0; eg_wal.bytes = 0; + reset_store(); eg_apply_node_put("{\"id\":\"n1\"}"); eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}"); + engram_wal_node_put(EL_STR(dir), EL_STR("n2")); + engram_wal_node_put(EL_STR(dir), EL_STR("n3")); + eg_wal_commit(1); wal_close(); + + /* Boot helper: load base then replay wal (mirrors server boot order). */ + #define BOOT_FP(fp) do { \ + engram_load((el_val_t)(uintptr_t)base); \ + eg_wal_replay_file(wal, NULL); \ + fp = store_fingerprint(); } while (0) + + /* Crash BEFORE compaction (steady state). */ + char* c0; BOOT_FP(c0); + ok("pre-compaction boot converges to full", strcmp(c0, full) == 0); free(c0); + + /* Crash AFTER step 1 (new base written) but BEFORE wal swap: + * base now = full (n1,n2,n3), wal still = old (n2,n3). Idempotent replay. */ + engram_load((el_val_t)(uintptr_t)base); /* reload old base into store */ + eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}"); + engram_save((el_val_t)(uintptr_t)base); /* == compaction step 1: new base */ + char* c1; BOOT_FP(c1); + ok("crash after new-base, before wal-swap → converges", strcmp(c1, full) == 0); free(c1); + + /* Crash AFTER wal.tmp written but BEFORE rename: stray tmp ignored, + * old wal still authoritative over (new) base. */ + { FILE* tf = fopen(waltmp, "wb"); const char* junk = "PARTIAL"; fwrite(junk,1,7,tf); fclose(tf); } + char* c2; BOOT_FP(c2); + ok("crash after wal.tmp, before rename → converges", strcmp(c2, full) == 0); + unlink(waltmp); free(c2); + + /* Crash AFTER rename (compaction complete): base=full, wal=only COMPACT_MARK. */ + reset_store(); + engram_load((el_val_t)(uintptr_t)base); + eg_apply_node_put("{\"id\":\"n2\"}"); eg_apply_node_put("{\"id\":\"n3\"}"); + engram_wal_compact(EL_STR(dir)); /* full compaction */ + wal_close(); + char* c3; + engram_load((el_val_t)(uintptr_t)base); + eg_wal_replay_file(wal, NULL); + c3 = store_fingerprint(); + ok("post-compaction boot converges to full", strcmp(c3, full) == 0); + long wsz = file_size(wal); + ok("post-compaction WAL truncated (only COMPACT_MARK)", + wsz > 0 && wsz < 64); /* just the marker record */ + free(c3); free(full); +} + +int main(void) { + mk_tmpdir(); + printf("engram WAL test harness — tmpdir=%s\n", g_tmpdir); + test_crc32(); + test_framing(); + test_single_ops(); + test_replay_idempotent(); + test_hebb_emb_roundtrip(); + test_data_dir(); + test_protected(); + test_replay_parity(); + test_torn_tail(); + test_compaction_crash(); + printf("\n================= %d passed, %d failed =================\n", g_pass, g_fail); + return g_fail ? 1 : 0; +} diff --git a/engram/test/test_wal_store.c b/engram/test/test_wal_store.c new file mode 100644 index 0000000..b64414e --- /dev/null +++ b/engram/test/test_wal_store.c @@ -0,0 +1,466 @@ +/* test_wal_store.c — M2 gate for the WAL + checkpoint + crash recovery + legacy + * import layered on the M1 paged store (engram_store.{c,h}). + * + * Pure C. Build: gcc -O2 test_wal_store.c ../../lang/runtime/engram_store.c -o t + * Writes ONLY under a throwaway /tmp dir. Never touches ~/.neuron or live ports. + * + * Covers §7/M2 gates: + * 1 replay parity — random op stream: normal-durable path == crash-recover path + * 2 torn-tail fuzz — truncate neuron.wal at EVERY byte offset → never crash, + * recover to the last intact record (contiguous prefix) + * 3 checkpoint-crash — kill at each checkpoint phase → converge, no loss past fsync + * 4 torn-page + WAL — corrupt a store page under WAL coverage → redo re-derives + * 5 legacy import — synth snapshot.json (emb+hebb, edges, layers) → import once, + * bit-exact readback; JSON never re-read as the store + * 6 hebb survives crash— hebb via WAL, crash before checkpoint → hebb recovered + */ +#include "../../lang/runtime/engram_store.h" + +#include +#include +#include +#include +#include +#include +#include + +static int g_pass = 0, g_fail = 0; +static void ok(const char* name, int cond){ + printf(" [%s] %s\n", cond ? "PASS" : "FAIL", name); + if (cond) g_pass++; else g_fail++; +} + +static char g_base[512]; +static void mk_base(void){ + snprintf(g_base, sizeof g_base, "/tmp/engram-wal-test-%d", (int)getpid()); + mkdir(g_base, 0700); +} +static void mk_dir(const char* name, char* out, size_t cap){ + snprintf(out, cap, "%s/%s", g_base, name); + mkdir(out, 0700); +} + +/* deterministic RNG */ +static uint64_t xs(uint64_t* s){ uint64_t x=*s; x^=x<<13; x^=x>>7; x^=x<<17; *s=x; return x; } + +/* ── small node/edge generators (kept compact so WAL frames stay small) ─────── */ +static void gen_node(int i, int with_emb, StoreNode* n){ + memset(n, 0, sizeof *n); + uint64_t st = 0x1234ULL ^ ((uint64_t)(i+1)*0x9E3779B97F4A7C15ULL); + char id[32]; snprintf(id, sizeof id, "n%d", i); n->id = strdup(id); + char c[64]; snprintf(c, sizeof c, "content-of-node-%d-%llu", i, (unsigned long long)(xs(&st)%9999)); + n->content = strdup(c); + n->node_type = strdup("concept"); + n->tier = strdup("Working"); + n->salience = (double)(xs(&st)%100000)/7.0; + n->importance = (double)(xs(&st)%100000)/11.0; + n->confidence = (double)(xs(&st)%100000)/13.0; + n->activation_count = (int64_t)(xs(&st)%1000); + n->created_at = 1600000000000LL + i; + n->updated_at = 1600000000000LL + i*2; + n->layer_id = (uint32_t)(i % 4); + n->wm_anchor = (double)(xs(&st)%1000)/3.0; + if (with_emb){ + n->emb_dim = 32; + n->emb = (float*)malloc(sizeof(float)*n->emb_dim); + for (int k=0;kemb_dim;k++){ uint32_t u=(uint32_t)xs(&st); memcpy(&n->emb[k],&u,4); } + } +} +static void gen_edge(int i, const char* from, const char* to, StoreEdge* e){ + memset(e, 0, sizeof *e); + uint64_t st = 0xABCDULL ^ ((uint64_t)(i+1)*0xD1B54A32D192ED03ULL); + char id[32]; snprintf(id, sizeof id, "e%d", i); e->id = strdup(id); + e->from_id = strdup(from); e->to_id = strdup(to); + e->relation = strdup("relates_to"); + e->weight = (double)(xs(&st)%100000)/17.0; + e->hebb = (double)(xs(&st)%100000)/100000.0; + e->confidence = (double)(xs(&st)%100000)/19.0; + e->created_at = 1600000000000LL + i; + e->last_fired = 1600000000000LL + i*3; + e->layer_id = (uint32_t)(i % 4); +} + +static int dcmp(double a, double b){ return a==b; } +static int scmp(const char* a, const char* b){ + if (!a && !b) return 1; if (!a || !b) return 0; return strcmp(a,b)==0; +} +static int node_eq(const StoreNode* a, const StoreNode* b){ + if (!scmp(a->id,b->id) || !scmp(a->content,b->content) || !scmp(a->node_type,b->node_type) || + !scmp(a->tier,b->tier)) return 0; + if (!dcmp(a->salience,b->salience) || !dcmp(a->importance,b->importance) || + !dcmp(a->confidence,b->confidence) || a->activation_count!=b->activation_count || + a->created_at!=b->created_at || a->updated_at!=b->updated_at || + a->layer_id!=b->layer_id || !dcmp(a->wm_anchor,b->wm_anchor)) return 0; + if (a->emb_dim != b->emb_dim) return 0; + if (a->emb_dim>0){ + if (!a->emb || !b->emb) return 0; + if (memcmp(a->emb, b->emb, sizeof(float)*a->emb_dim)!=0) return 0; /* bit-exact */ + } + return 1; +} +static int edge_eq(const StoreEdge* a, const StoreEdge* b){ + return scmp(a->id,b->id) && scmp(a->from_id,b->from_id) && scmp(a->to_id,b->to_id) && + scmp(a->relation,b->relation) && dcmp(a->weight,b->weight) && dcmp(a->hebb,b->hebb) && + dcmp(a->confidence,b->confidence) && a->created_at==b->created_at && + a->last_fired==b->last_fired && a->layer_id==b->layer_id; +} + +/* whole-file read / write helpers (for torn-tail + torn-page fuzzing) */ +static uint8_t* read_file(const char* p, long* len){ + FILE* f=fopen(p,"rb"); if(!f) return NULL; + fseek(f,0,SEEK_END); long n=ftell(f); fseek(f,0,SEEK_SET); + uint8_t* b=malloc(n?n:1); if(fread(b,1,n,f)!=(size_t)n){ fclose(f); free(b); return NULL; } + fclose(f); *len=n; return b; +} +static void write_file(const char* p, const uint8_t* b, long len){ + FILE* f=fopen(p,"wb"); fwrite(b,1,len,f); fclose(f); +} + +/* ═══════════════════════════ TEST 1 — replay parity ═══════════════════════ */ +#define UNIV_NODES 60 +#define UNIV_EDGES 40 +static void test_replay_parity(void){ + printf("\n== replay parity: normal-durable path == crash-then-recover path ==\n"); + char da[600], db[600]; mk_dir("parityA", da, sizeof da); mk_dir("parityB", db, sizeof db); + EngramPagedStore* A = engram_open(da); + EngramPagedStore* B = engram_open(db); + ok("opened both stores", A && B); + if (!A || !B) return; + + uint64_t rng = 0xF00DFACEULL; + int OPS = 800; + for (int step=0; step0); + printf(" WAL bytes fuzzed=%ld full-recover offsets=%d\n", wlen, full_recovered); + free(sb); free(wb); +} + +/* ═══════════════════════════ TEST 3 — checkpoint-crash ═══════════════════════ */ +#define CK_NODES 30 +#define CK_EDGES 20 +static int build_and_crash_at_phase(const char* dir, int phase){ + EngramPagedStore* s = engram_open(dir); + if (!s) return -1; + for (int i=0;i=0); + if (victim>=0){ + for (int k=0;k<64;k++) sb[victim*16384 + 200 + k] ^= 0xA5; /* trash record area → bad crc */ + write_file(sp, sb, slen); + } + free(sb); + + EngramPagedStore* r = engram_open(dir); /* heal torn page + replay WAL */ + ok("reopened after page corruption", r!=NULL); + if (r){ + int miss=0; + for (int i=0;iemb_dim;k++) n->emb[k] = (float)((double)(xs(&es)%2000001)/1000000.0 - 1.0); } + fprintf(f, "%s{\"id\":\"%s\",\"content\":\"%s\",\"node_type\":\"%s\",\"tier\":\"%s\"," + "\"salience\":%.17g,\"importance\":%.17g,\"confidence\":%.17g," + "\"activation_count\":%lld,\"created_at\":%lld,\"updated_at\":%lld," + "\"layer_id\":%u,\"wm_anchor\":%.17g,\"emb\":\"", + i?",":"", n->id, n->content, n->node_type, n->tier, + n->salience, n->importance, n->confidence, + (long long)n->activation_count, (long long)n->created_at, (long long)n->updated_at, + n->layer_id, n->wm_anchor); + for (int k=0;kemb_dim;k++) fprintf(f, "%s%.9g", k?",":"", (double)n->emb[k]); /* exact float32 repr */ + fprintf(f, "\"}"); + } + fprintf(f, "],\"edges\":["); + for (int i=0;iid, e->from_id, e->to_id, e->relation, + e->weight, e->hebb, e->confidence, (long long)e->created_at, (long long)e->last_fired, e->layer_id); + } + fprintf(f, "],\"layers\":["); + fprintf(f, "{\"layer_id\":0,\"name\":\"SAFETY\",\"activation_priority\":9,\"suppressible\":0,\"transparent\":0,\"injectable\":0},"); + fprintf(f, "{\"layer_id\":1,\"name\":\"CORE_IDENTITY\",\"activation_priority\":8,\"suppressible\":0,\"transparent\":1,\"injectable\":1}"); + fprintf(f, "]}"); + fclose(f); + + EngramPagedStore* s = engram_open(dir); /* store absent + snapshot present → import */ + ok("engram_open imported the snapshot", s!=NULL); + char sp[700]; snprintf(sp,sizeof sp,"%s/neuron.egm",dir); struct stat st; + ok("neuron.egm created by import", stat(sp,&st)==0); + if (!s) return; + + int nmiss=0, embmiss=0; + for (int i=0;i0 && memcmp(got.emb,onodes[i].emb,sizeof(float)*got.emb_dim)!=0)) embmiss++; store_node_free(&got); } + } + int emiss=0, hebbmiss=0; + for (int i=0;i elc-new.c -cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ +cc -std=c11 -I runtime -lcurl -lpthread \ -o dist/platform/elc-new \ - elc-new.c el-compiler/runtime/el_seed.c + elc-new.c runtime/el_seed.c # Verify self-hosting: ./dist/platform/elc-new elc-cli.el > elc-verify.c diff elc-new.c elc-verify.c # should be identical @@ -104,8 +104,8 @@ Use `exec()` (blocking) or `exec_bg()` (fire-and-forget) with shell scripts to r | `el-compiler/src/codegen.el` | Code generator — builtin arity table lives here | | `el-compiler/src/lexer.el` | Lexer | | `el-compiler/src/parser.el` | Parser | -| `el-compiler/runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) | -| `el-compiler/runtime/el_seed.h` | Seed header (C function declarations) | +| `runtime/el_seed.c` | Self-contained C OS-boundary layer (replaces el_runtime.c) | +| `runtime/el_seed.h` | Seed header (C function declarations) | | `spec/language.md` | Language specification | | `BOOTSTRAP.md` | How to recover the compiler from scratch | | `elc-cli.el` | Compiler entry point | diff --git a/lang/BOOTSTRAP.md b/lang/BOOTSTRAP.md index e90d0b6..8b88720 100644 --- a/lang/BOOTSTRAP.md +++ b/lang/BOOTSTRAP.md @@ -50,9 +50,9 @@ To rebuild the current binary from source using the current binary: ```bash cd /path/to/el ./dist/platform/elc elc-cli.el elc-new.c -cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ +cc -std=c11 -I runtime -lcurl -lpthread \ -o dist/platform/elc-new \ - elc-new.c el-compiler/runtime/el_runtime.c + elc-new.c runtime/el_runtime.c ``` Verify self-hosting by using `elc-new` to recompile itself and diffing the outputs. @@ -288,14 +288,14 @@ The codegen tracks declared names per C scope. When `count` is already in `decla ## 3. The Runtime API -All runtime functions are declared in `el-compiler/runtime/el_runtime.h`. Every compiled El program links against `el-compiler/runtime/el_runtime.c`. +All runtime functions are declared in `runtime/el_runtime.h`. Every compiled El program links against `runtime/el_runtime.c`. All values are `el_val_t` (`int64_t`). Strings are pointers cast through `int64_t` using `EL_STR(s)` / `EL_CSTR(v)` macros. Canonical compile command: ```bash -cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ - -o .c el-compiler/runtime/el_runtime.c +cc -std=c11 -I runtime -lcurl -lpthread \ + -o .c runtime/el_runtime.c ``` ### I/O @@ -794,8 +794,8 @@ Using your minimal implementation, compile `elc-cli.el` (which imports the entir python3 minimal_elc.py elc-cli.el > elc-new.c # Build with the runtime -cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ - -o elc-new elc-new.c el-compiler/runtime/el_runtime.c +cc -std=c11 -I runtime -lcurl -lpthread \ + -o elc-new elc-new.c runtime/el_runtime.c ``` ### Step 5: Verify Self-Hosting @@ -803,8 +803,8 @@ cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ ```bash # Compile elc-cli.el with the new compiler ./elc-new elc-cli.el elc-v2.c -cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ - -o elc-v2 elc-v2.c el-compiler/runtime/el_runtime.c +cc -std=c11 -I runtime -lcurl -lpthread \ + -o elc-v2 elc-v2.c runtime/el_runtime.c # Compile again with the second-generation compiler ./elc-v2 elc-cli.el elc-v3.c @@ -880,9 +880,9 @@ This is the planned path. It does not exist yet. | `el-compiler/src/parser.el` | Recursive descent parser. `parse(tokens)` → AST. All statement and expression forms | 1071 | | `el-compiler/src/codegen.el` | C code emitter. `codegen(stmts, source)` → (streams to stdout). Expression codegen, statement codegen, function codegen, type tracking, capability enforcement, temporal type dispatch | 2721 | | `el-compiler/src/codegen-js.el` | JavaScript backend. `codegen_js(stmts, source)` → JS source | ~500 | -| `el-compiler/runtime/el_runtime.h` | Full runtime API declaration | 755 | -| `el-compiler/runtime/el_runtime.c` | Full runtime implementation | large | -| `el-compiler/runtime/el_runtime.js` | JS runtime | — | +| `runtime/el_runtime.h` | Full runtime API declaration | 755 | +| `runtime/el_runtime.c` | Full runtime implementation | large | +| `runtime/el_runtime.js` | JS runtime | — | | `elb.el` | Build coordinator. Reads `manifest.el`, walks import graph, compiles modules, links binary. The `.NET`-style incremental build model | 367 | | `elc-combined.el` | Pre-merged single-file bootstrap edition (for early bootstrap iterations) | large | | `spec/language.md` | Language specification v1.2.0 | — | diff --git a/lang/el-compiler/runtime/el_platform_win.h b/lang/el-compiler/runtime/el_platform_win.h deleted file mode 100644 index df42c2b..0000000 --- a/lang/el-compiler/runtime/el_platform_win.h +++ /dev/null @@ -1,118 +0,0 @@ -#ifndef EL_PLATFORM_WIN_H -#define EL_PLATFORM_WIN_H -/* - * el_platform_win.h — Windows OS-boundary shim for el_runtime.c. - * - * Branch: feat/windows-el-runtime. Included ONLY when _WIN32 is defined; the POSIX build is - * untouched. Goal: let el_runtime.c (a BSD-sockets / dlfcn / fork host) compile and link with - * mingw-w64 into a native neuron.exe, with no behavioural change to the Linux/macOS build. - * - * What it maps: - * - sockets : winsock2 (same call names: socket/bind/listen/accept/recv/send/setsockopt). - * Sockets close with closesocket() (see el_closesocket), and the stack must be - * started once with WSAStartup — done automatically via a load-time constructor. - * - dlsym : el_runtime.c uses dlsym(RTLD_DEFAULT, name) to resolve callback/tool symbols - * exported by the main module. Windows equivalent: GetProcAddress on the process - * module. Link the soul with -Wl,--export-all-symbols so the symbols are findable. - * - popen : mapped to _popen/_pclose. - * - threads : UNCHANGED. mingw-w64 ships winpthreads, so + -lpthread just work. - */ - -#ifndef WIN32_LEAN_AND_MEAN -#define WIN32_LEAN_AND_MEAN -#endif -#include -#include -#include -#include -#include - -/* Portable headers mingw-w64 provides (verified present). */ -#include -#include -#include -#include -#include -#include /* strcasecmp */ -#include -#include -#include -#include /* mingw-w64 provides gettimeofday here */ -#include -#include -#include -#include -#include -#include - -/* ── socket close ─────────────────────────────────────────────────────────── */ -/* Winsock closes sockets with closesocket(), not close() (close() is for file fds). The POSIX - build defines the same helper as close() so the call sites are identical across platforms. */ -static inline int el_closesocket(SOCKET s) { return closesocket(s); } - -/* ── winsock init (once, at load) ─────────────────────────────────────────── */ -static void el__win_net_init(void) { - static int inited = 0; - if (!inited) { WSADATA w; WSAStartup(MAKEWORD(2, 2), &w); inited = 1; } -} -__attribute__((constructor)) static void el__win_ctor(void) { el__win_net_init(); } - -/* ── dlsym → GetProcAddress ───────────────────────────────────────────────── */ -#ifndef RTLD_DEFAULT -#define RTLD_DEFAULT ((void*)0) -#endif -static inline void* el_win_dlsym(void* handle, const char* name) { - (void)handle; - return (void*)(uintptr_t)GetProcAddress(GetModuleHandleA(NULL), name); -} -#define dlsym(h, n) el_win_dlsym((h), (n)) - -/* ── popen / pclose ───────────────────────────────────────────────────────── */ -#define popen _popen -#define pclose _pclose - -/* ── misc POSIX → Win32 shims ─────────────────────────────────────────────── */ -#include /* _mkdir */ -#define mkdir(path, mode) _mkdir(path) /* POSIX mkdir(path,mode) → _mkdir(path) */ -#define timegm _mkgmtime /* UTC tm → time_t */ -#define fsync(fd) _commit(fd) /* no fsync() on Windows; _commit() () is the equiv */ - -/* setenv/unsetenv: not in the Windows CRT; map to _putenv_s / SetEnvironmentVariable. */ -static inline int setenv(const char* name, const char* value, int overwrite) { - (void)overwrite; - return _putenv_s(name, value ? value : ""); -} -static inline int unsetenv(const char* name) { - /* _putenv_s(name, "") sets VAR="" rather than removing it. - * SetEnvironmentVariableA(name, NULL) truly deletes it from the Win32 - * env block; then we sync the CRT cache with _putenv("NAME="). */ - SetEnvironmentVariableA(name, NULL); - size_t len = strlen(name); - char *buf = (char*)malloc(len + 2); - if (!buf) return -1; - memcpy(buf, name, len); - buf[len] = '='; - buf[len + 1] = '\0'; - _putenv(buf); - free(buf); - return 0; -} - -/* nanosleep — not available in MSVC/UCRT; approximate with Sleep(). */ -static inline int el_nanosleep(const struct timespec *req, struct timespec *rem) { - (void)rem; - DWORD ms = (DWORD)((req->tv_sec * 1000ULL) + (req->tv_nsec / 1000000ULL)); - Sleep(ms ? ms : 1); - return 0; -} -#define nanosleep(req, rem) el_nanosleep((req), (rem)) - -/* localtime_r/gmtime_r: Windows offers localtime_s/gmtime_s with reversed arg order. */ -static inline struct tm* localtime_r(const time_t* t, struct tm* out) { - return localtime_s(out, t) == 0 ? out : (struct tm*)0; -} -static inline struct tm* gmtime_r(const time_t* t, struct tm* out) { - return gmtime_s(out, t) == 0 ? out : (struct tm*)0; -} - -#endif /* EL_PLATFORM_WIN_H */ diff --git a/lang/el-compiler/runtime/el_runtime.c b/lang/el-compiler/runtime/el_runtime.c deleted file mode 100644 index af0d945..0000000 --- a/lang/el-compiler/runtime/el_runtime.c +++ /dev/null @@ -1,12468 +0,0 @@ -/* - * el_runtime.c — El language C runtime implementation - * - * All functions use el_val_t (= int64_t) as the universal value type. - * Strings are transported as their pointer address cast to int64_t. - * On any 64-bit system sizeof(pointer) <= sizeof(int64_t), so this is safe. - * - * Compile with: - * cc -std=c11 -I -lcurl -lpthread -o .c el_runtime.c - * - * Link requirements: -lcurl (HTTP client + LLM), -lpthread (HTTP server). - */ - -/* Feature-test macros must be set before any standard headers. _GNU_SOURCE - * exposes clock_gettime/CLOCK_REALTIME, strcasecmp, and the dlfcn extensions - * (RTLD_DEFAULT) — all of which macOS hands us without asking but glibc on - * Debian gates behind an explicit opt-in. */ -#ifndef _GNU_SOURCE -#define _GNU_SOURCE -#endif - -#include "el_runtime.h" - -#ifdef _WIN32 -/* Windows OS-boundary shim (winsock/dlsym/popen). Threading stays on (winpthreads). */ -#include "el_platform_win.h" -#else -#include -#include /* strcasecmp */ -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include /* dlsym for http_set_handler fallback */ -#include -#include -#include -#include -#include -#include /* getrusage — memory guard */ -/* On POSIX, sockets close with the same close() as files; el_platform_win.h supplies the Windows - variant. Defined here so the socket call sites are identical across platforms. */ -static inline int el_closesocket(int s) { return close(s); } -#endif -#ifdef HAVE_CURL -#include -#endif - -/* ── Internal allocators ─────────────────────────────────────────────────── */ - -/* - * Per-request string arena - * - * Every El string allocated via el_strbuf / el_strdup during an HTTP request - * is registered in a thread-local arena. When el_request_end() is called at - * the end of the worker thread, every arena entry is freed — recovering all - * the intermediate strings from el_str_concat chains (build_system_prompt, - * engram_compile, etc.) that are otherwise leaked forever. - * - * Long-lived allocations (state_set values, engram internal storage) call - * el_strdup_persist() / el_strbuf_persist() which bypass the arena entirely. - */ - -#define EL_ARENA_INITIAL 512 - -typedef struct { - char** ptrs; - size_t count; - size_t cap; -} ElArena; - -static _Thread_local ElArena _tl_arena = {NULL, 0, 0}; -static _Thread_local int _tl_arena_active = 0; - -/* Binary-safe fs_read length — set by fs_read, consumed by http_send_response. - * Allows serving PNGs and other binary files without strlen truncation. - * PAIRED with the buffer pointer it describes: the length may only be applied - * to the exact buffer fs_read returned. Without the pairing, any handler that - * fs_read a file and then WRAPPED it into a larger response had that response - * truncated to the file's length (Content-Length lied AND the send stopped - * short) — the safety-contact onboarding trap, 2026-07-17. */ -static _Thread_local size_t _tl_fs_read_len = 0; -static _Thread_local const char* _tl_fs_read_buf = NULL; - -static void el_arena_track(char* p) { - if (!_tl_arena_active || !p) return; - if (_tl_arena.count >= _tl_arena.cap) { - size_t nc = _tl_arena.cap == 0 ? EL_ARENA_INITIAL : _tl_arena.cap * 2; - char** grown = realloc(_tl_arena.ptrs, nc * sizeof(char*)); - if (!grown) return; /* can't track — will leak this one ptr, but don't crash */ - _tl_arena.ptrs = grown; - _tl_arena.cap = nc; - } - _tl_arena.ptrs[_tl_arena.count++] = p; -} - -/* Called by http_worker before dispatching the El handler. */ -void el_request_start(void) { - _tl_arena.count = 0; - _tl_arena_active = 1; - _tl_fs_read_len = 0; /* never let a previous request's file length */ - _tl_fs_read_buf = NULL; /* leak into this response's byte accounting */ -} - -/* Called by http_worker after the El handler returns and the response is sent. - * Frees every intermediate string allocated during the request. */ -void el_request_end(void) { - _tl_arena_active = 0; - for (size_t i = 0; i < _tl_arena.count; i++) { - free(_tl_arena.ptrs[i]); - } - _tl_arena.count = 0; -} - -/* ── Scoped arena for CLI use ─────────────────────────────────────────────── * - * CLI programs never call el_request_start/end, so all strdup allocations are - * permanent. el_arena_push/pop let the compiler free intermediate strings - * after each compilation unit. - * - * el_arena_push() — activates the arena if not already active, saves the - * current arena count as a mark, and returns it as an el_val_t Int. - * el_arena_pop(mark) — frees all strings allocated since the push mark and - * resets the count. If count reaches 0, deactivates the arena. - */ -#define EL_ARENA_SCOPE_DEPTH 32 -static _Thread_local size_t _tl_arena_scope[EL_ARENA_SCOPE_DEPTH]; -static _Thread_local int _tl_arena_scope_depth = 0; - -el_val_t el_arena_push(void) { - if (!_tl_arena_active) { - _tl_arena_active = 1; - } - if (_tl_arena_scope_depth < EL_ARENA_SCOPE_DEPTH) { - _tl_arena_scope[_tl_arena_scope_depth++] = _tl_arena.count; - } - return (el_val_t)(int64_t)_tl_arena.count; -} - -el_val_t el_arena_pop(el_val_t mark) { - size_t save = (size_t)(int64_t)mark; - if (save > _tl_arena.count) save = 0; - for (size_t i = save; i < _tl_arena.count; i++) { - if (_tl_arena.ptrs[i]) { - free(_tl_arena.ptrs[i]); - _tl_arena.ptrs[i] = NULL; - } - } - _tl_arena.count = save; - if (_tl_arena_scope_depth > 0) _tl_arena_scope_depth--; - if (save == 0) _tl_arena_active = 0; - return 0; -} - -/* Persistent allocation — bypasses the arena (state_set, engram internals). */ -static char* el_strdup_persist(const char* s) { - if (!s) return strdup(""); - return strdup(s); -} -static char* el_strbuf_persist(size_t n) { - char* p = malloc(n + 1); - if (!p) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - p[0] = '\0'; - return p; -} - -static char* el_strdup(const char* s) { - if (!s) { char* p = strdup(""); el_arena_track(p); return p; } - char* p = strdup(s); - el_arena_track(p); - return p; -} - -static char* el_strbuf(size_t n) { - char* p = malloc(n + 1); - if (!p) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - p[0] = '\0'; - el_arena_track(p); - return p; -} - -/* Wrap an allocated C string as el_val_t */ -static el_val_t el_wrap_str(char* s) { - return EL_STR(s); -} - -/* ── I/O ──────────────────────────────────────────────────────────────────── */ - -el_val_t println(el_val_t s) { - const char* str = EL_CSTR(s); - if (str) puts(str); - else puts(""); - fflush(stdout); /* prevent startup logs from silently buffering when stdout→file */ - return 0; -} - -el_val_t print(el_val_t s) { - const char* str = EL_CSTR(s); - if (str) fputs(str, stdout); - return 0; -} - -el_val_t readline(void) { - char buf[4096]; - if (!fgets(buf, sizeof(buf), stdin)) return el_wrap_str(el_strdup("")); - size_t len = strlen(buf); - if (len > 0 && buf[len - 1] == '\n') buf[len - 1] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -/* __read_n — read exactly n bytes from stdin. - * Allocates a buffer of size n+1, calls fread(buf, 1, n, stdin) to read - * exactly n raw bytes (including \r, \n, NUL, etc.), null-terminates, and - * returns the buffer as an El String. Returns "" on EOF or I/O error. - * - * Used by the El LSP server to read JSON-RPC message bodies after parsing - * the Content-Length header. readline() cannot be used for the body because - * it stops at the first \n and LSP JSON bodies are not newline-terminated. */ -el_val_t __read_n(el_val_t nv) { - int64_t n = EL_INT(nv); - if (n <= 0) return el_wrap_str(el_strdup("")); - char* buf = malloc((size_t)n + 1); - if (!buf) { fputs("el_runtime: __read_n: out of memory\n", stderr); return el_wrap_str(el_strdup("")); } - size_t got = fread(buf, 1, (size_t)n, stdin); - buf[got] = '\0'; - if (got == 0) { free(buf); return el_wrap_str(el_strdup("")); } - /* Track in arena so the allocation is freed when the request ends. */ - el_arena_track(buf); - return el_wrap_str(buf); -} - -/* __print_raw — write a string to stdout without any modification. - * Unlike println/print (which call puts/fputs and may add newlines or flush - * in platform-specific ways), this uses fwrite with the exact byte count so - * that embedded \r\n pairs in LSP Content-Length headers survive intact. */ -void __print_raw(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return; - size_t len = strlen(s); - fwrite(s, 1, len, stdout); - fflush(stdout); -} - -/* ── String builtins ─────────────────────────────────────────────────────── */ - -el_val_t el_str_concat(el_val_t av, el_val_t bv) { - const char* a = EL_CSTR(av); - const char* b = EL_CSTR(bv); - if (!a) a = ""; - if (!b) b = ""; - size_t la = strlen(a); - size_t lb = strlen(b); - char* out = el_strbuf(la + lb); - memcpy(out, a, la); - memcpy(out + la, b, lb); - out[la + lb] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_eq(el_val_t av, el_val_t bv) { - const char* a = EL_CSTR(av); - const char* b = EL_CSTR(bv); - if (!a || !b) return (el_val_t)(a == b); - return (el_val_t)(strcmp(a, b) == 0); -} - -el_val_t str_starts_with(el_val_t sv, el_val_t prefv) { - const char* s = EL_CSTR(sv); - const char* prefix = EL_CSTR(prefv); - if (!s || !prefix) return 0; - size_t lp = strlen(prefix); - return (el_val_t)(strncmp(s, prefix, lp) == 0); -} - -el_val_t str_ends_with(el_val_t sv, el_val_t sufv) { - const char* s = EL_CSTR(sv); - const char* suffix = EL_CSTR(sufv); - if (!s || !suffix) return 0; - size_t ls = strlen(s); - size_t lsuf = strlen(suffix); - if (lsuf > ls) return 0; - return (el_val_t)(strcmp(s + ls - lsuf, suffix) == 0); -} - -el_val_t str_len(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - return (el_val_t)strlen(s); -} - -el_val_t str_concat(el_val_t a, el_val_t b) { - return el_str_concat(a, b); -} - -el_val_t int_to_str(el_val_t n) { - char buf[32]; - snprintf(buf, sizeof(buf), "%lld", (long long)n); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t str_to_int(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - return (el_val_t)atoll(s); -} - -/* native_str_to_int — El compiler-generated alias for str_to_int. - * Converts a string el_val_t to its integer representation. */ -el_val_t native_str_to_int(el_val_t sv) { return str_to_int(sv); } - -el_val_t str_slice(el_val_t sv, el_val_t start, el_val_t end) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - int64_t len = (int64_t)strlen(s); - if (start < 0) start = 0; - if (end > len) end = len; - if (start >= end) return el_wrap_str(el_strdup("")); - int64_t sz = end - start; - char* out = el_strbuf((size_t)sz); - memcpy(out, s + start, (size_t)sz); - out[sz] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_contains(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub) return 0; - return (el_val_t)(strstr(s, sub) != NULL); -} - -el_val_t str_replace(el_val_t sv, el_val_t fromv, el_val_t tov) { - const char* s = EL_CSTR(sv); - const char* from = EL_CSTR(fromv); - const char* to = EL_CSTR(tov); - if (!s || !from || !to) return el_wrap_str(el_strdup(s ? s : "")); - size_t ls = strlen(s); - size_t lf = strlen(from); - size_t lt = strlen(to); - if (lf == 0) return el_wrap_str(el_strdup(s)); - size_t count = 0; - const char* p = s; - while ((p = strstr(p, from)) != NULL) { count++; p += lf; } - size_t out_sz = ls + count * lt + 1; - char* out = el_strbuf(out_sz); - char* dst = out; - p = s; - const char* found; - while ((found = strstr(p, from)) != NULL) { - size_t chunk = (size_t)(found - p); - memcpy(dst, p, chunk); dst += chunk; - memcpy(dst, to, lt); dst += lt; - p = found + lf; - } - strcpy(dst, p); - return el_wrap_str(out); -} - -el_val_t str_to_upper(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - for (size_t i = 0; i < n; i++) out[i] = (char)toupper((unsigned char)s[i]); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_to_lower(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - for (size_t i = 0; i < n; i++) out[i] = (char)tolower((unsigned char)s[i]); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_trim(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - while (*s && isspace((unsigned char)*s)) s++; - size_t n = strlen(s); - while (n > 0 && isspace((unsigned char)s[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, s, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -/* ── Math ────────────────────────────────────────────────────────────────── */ - -el_val_t el_abs(el_val_t n) { return n < 0 ? -n : n; } -el_val_t el_max(el_val_t a, el_val_t b) { return a > b ? a : b; } -el_val_t el_min(el_val_t a, el_val_t b) { return a < b ? a : b; } - -/* ── Refcounted heap objects ────────────────────────────────────────────────── - * - * ElList and ElMap carry a magic-tagged header at offset 0: - * { uint32_t magic; uint32_t refcount; ... payload ... } - * - * The magic tag distinguishes refcounted objects from raw C strings (whose - * first byte is printable ASCII < 0x80) and from small integers (which can't - * be dereferenced). el_retain / el_release sniff the magic and act only on - * matching values; everything else is a safe no-op. - * - * Both ElList and ElMap use INDIRECTION: the header is fixed-size and never - * moves. The payload arrays (elems, keys, values) live in separate heap - * allocations, so realloc-grow on append never invalidates the caller's - * pointer to the header. This is what lets us mutate-in-place safely when - * the refcount is 1 and copy-on-write when it's higher. - * - * Memory model in practice: - * Single-owner accumulator (the cg_stmts pattern) — refcount stays at 1, - * appends amortize to O(1), total memory O(N) for an N-element list. - * Multi-owner branching (the cg_if_stmt pattern) — refcount > 1, each - * append on a shared list copies, so the original is preserved for the - * else-branch. Persistent semantics where they're needed; mutation where - * they're not. */ - -#define EL_MAGIC_LIST 0xE15710A1u /* >= 0x80 in MSB so 'looks_like_string' rejects */ -#define EL_MAGIC_MAP 0xE19A704Bu - -typedef struct { - uint32_t magic; - uint32_t refcount; -} ElHeader; - -/* ── List ────────────────────────────────────────────────────────────────── */ - -typedef struct { - ElHeader hdr; - int64_t length; - int64_t capacity; - el_val_t* elems; -} ElList; - -static ElList* list_alloc(int64_t cap) { - if (cap < 4) cap = 4; - ElList* lst = malloc(sizeof(ElList)); - if (!lst) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - lst->hdr.magic = EL_MAGIC_LIST; - lst->hdr.refcount = 1; - lst->length = 0; - lst->capacity = cap; - lst->elems = malloc((size_t)cap * sizeof(el_val_t)); - if (!lst->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - return lst; -} - -el_val_t el_list_empty(void) { - return EL_STR(list_alloc(4)); -} - -el_val_t el_list_new(el_val_t count, ...) { - ElList* lst = list_alloc(count > 0 ? count : 4); - va_list ap; - va_start(ap, count); - for (int64_t i = 0; i < count; i++) { - lst->elems[i] = va_arg(ap, el_val_t); - } - va_end(ap); - lst->length = count; - return EL_STR(lst); -} - -el_val_t el_list_len(el_val_t listv) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) return 0; - return lst->length; -} - -el_val_t el_list_get(el_val_t listv, el_val_t index) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) return 0; - if (index < 0 || index >= lst->length) return 0; - return lst->elems[index]; -} - -el_val_t el_list_append(el_val_t listv, el_val_t elem) { - ElList* old = (ElList*)(uintptr_t)listv; - if (!old) { - ElList* fresh = list_alloc(4); - fresh->elems[0] = elem; - fresh->length = 1; - return EL_STR(fresh); - } - - /* Uniquely owned: grow the elems buffer in place. The header pointer the - * caller holds doesn't move (we only realloc the inner array). This is - * the common case in compiler accumulators, and it's amortized O(1). */ - if (old->hdr.refcount <= 1) { - if (old->length >= old->capacity) { - int64_t new_cap = old->capacity > 0 ? old->capacity * 2 : 4; - el_val_t* grown = realloc(old->elems, (size_t)new_cap * sizeof(el_val_t)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - old->elems = grown; - old->capacity = new_cap; - } - old->elems[old->length++] = elem; - return listv; - } - - /* Shared: copy-on-write. The original is preserved for its other owners. */ - int64_t new_cap = old->length + 1; - if (new_cap < 4) new_cap = 4; - ElList* fresh = malloc(sizeof(ElList)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_LIST; - fresh->hdr.refcount = 1; - fresh->length = old->length + 1; - fresh->capacity = new_cap; - fresh->elems = malloc((size_t)new_cap * sizeof(el_val_t)); - if (!fresh->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (old->length > 0) { - memcpy(fresh->elems, old->elems, (size_t)old->length * sizeof(el_val_t)); - } - fresh->elems[old->length] = elem; - return EL_STR(fresh); -} - -el_val_t el_list_clone(el_val_t listv) { - /* Shallow copy: the new ElList owns its own header and elems buffer, but - * the elements themselves are shared (which is what callers want for the - * cg_if_stmt 'declared' pattern — cloning the spine, not its contents). - * Used by codegen at scope branch points where two child scopes need to - * see the same starting set of declared names without each other's - * mutations. */ - ElList* old = (ElList*)(uintptr_t)listv; - if (!old) return el_list_empty(); - int64_t cap = old->capacity > 0 ? old->capacity : 4; - if (cap < old->length) cap = old->length; - if (cap < 4) cap = 4; - ElList* fresh = malloc(sizeof(ElList)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_LIST; - fresh->hdr.refcount = 1; - fresh->length = old->length; - fresh->capacity = cap; - fresh->elems = malloc((size_t)cap * sizeof(el_val_t)); - if (!fresh->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (old->length > 0) { - memcpy(fresh->elems, old->elems, (size_t)old->length * sizeof(el_val_t)); - } - return EL_STR(fresh); -} - -/* ── Map ─────────────────────────────────────────────────────────────────── */ - -typedef struct { - ElHeader hdr; - int64_t count; - int64_t capacity; - el_val_t* keys; - el_val_t* values; -} ElMap; - -static ElMap* map_alloc(int64_t cap) { - if (cap < 4) cap = 4; - ElMap* m = malloc(sizeof(ElMap)); - if (!m) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - m->hdr.magic = EL_MAGIC_MAP; - m->hdr.refcount = 1; - m->count = 0; - m->capacity = cap; - m->keys = malloc((size_t)cap * sizeof(el_val_t)); - m->values = malloc((size_t)cap * sizeof(el_val_t)); - if (!m->keys || !m->values) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - return m; -} - -el_val_t el_map_new(el_val_t pair_count, ...) { - ElMap* m = map_alloc(pair_count > 0 ? pair_count : 4); - va_list ap; - va_start(ap, pair_count); - for (int64_t i = 0; i < pair_count; i++) { - m->keys[i] = va_arg(ap, el_val_t); - m->values[i] = va_arg(ap, el_val_t); - } - va_end(ap); - m->count = pair_count; - return EL_STR(m); -} - -static ElMap* as_map(el_val_t v) { return (ElMap*)(uintptr_t)v; } - -el_val_t el_map_get(el_val_t mapv, el_val_t keyv) { - ElMap* m = as_map(mapv); - const char* key = EL_CSTR(keyv); - if (!m || !key) return 0; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - if (k && strcmp(k, key) == 0) return m->values[i]; - } - return 0; -} - -el_val_t el_get_field(el_val_t mapv, el_val_t keyv) { - return el_map_get(mapv, keyv); -} - -/* Internal: in-place set on a uniquely-owned map. */ -static el_val_t map_set_in_place(ElMap* m, el_val_t keyv, el_val_t value) { - const char* key = EL_CSTR(keyv); - if (key) { - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - if (k && strcmp(k, key) == 0) { m->values[i] = value; return EL_STR(m); } - } - } - if (m->count >= m->capacity) { - int64_t new_cap = m->capacity > 0 ? m->capacity * 2 : 4; - el_val_t* gk = realloc(m->keys, (size_t)new_cap * sizeof(el_val_t)); - el_val_t* gv = realloc(m->values, (size_t)new_cap * sizeof(el_val_t)); - if (!gk || !gv) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - m->keys = gk; - m->values = gv; - m->capacity = new_cap; - } - m->keys[m->count] = keyv; - m->values[m->count] = value; - m->count++; - return EL_STR(m); -} - -el_val_t el_map_set(el_val_t mapv, el_val_t keyv, el_val_t value) { - ElMap* m = as_map(mapv); - if (!m) return 0; - if (m->hdr.refcount <= 1) { - return map_set_in_place(m, keyv, value); - } - /* Shared: copy then set. The original is preserved for its other owners. */ - int64_t new_cap = m->count + 1; - if (new_cap < 4) new_cap = 4; - ElMap* fresh = malloc(sizeof(ElMap)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_MAP; - fresh->hdr.refcount = 1; - fresh->count = m->count; - fresh->capacity = new_cap; - fresh->keys = malloc((size_t)new_cap * sizeof(el_val_t)); - fresh->values = malloc((size_t)new_cap * sizeof(el_val_t)); - if (!fresh->keys || !fresh->values) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (m->count > 0) { - memcpy(fresh->keys, m->keys, (size_t)m->count * sizeof(el_val_t)); - memcpy(fresh->values, m->values, (size_t)m->count * sizeof(el_val_t)); - } - return map_set_in_place(fresh, keyv, value); -} - -/* ── Refcount ops ─────────────────────────────────────────────────────────── */ -/* - * Both retain and release sniff the magic header to decide whether a value - * is a refcounted heap object. For small integers, raw C strings, and any - * value whose magic word doesn't match, both functions are no-ops. This lets - * codegen emit them on every let-binding without having to track types. - * - * Safety: we filter out obvious non-pointers (small magnitudes, misaligned - * addresses) before dereferencing. For any value that passes the filter and - * lives in a mapped page, reading the first 4 bytes is safe — strings start - * with printable ASCII (< 0x80), so their magic word will never collide with - * EL_MAGIC_LIST (0xE1...) or EL_MAGIC_MAP (0xE1...). Random integers that - * happen to look like aligned heap pointers are exceedingly unlikely to land - * on a page whose first 4 bytes match either magic. */ - -static int looks_like_heap_obj(el_val_t v) { - if (v == 0) return 0; - int64_t s = (int64_t)v; - if (s > -0x10000 && s < 0x10000) return 0; /* small ints */ - uintptr_t p = (uintptr_t)v; - if (p < 0x10000) return 0; /* low addresses */ - if (p & 0x7) return 0; /* malloc returns 8-aligned */ - return 1; -} - -void el_retain(el_val_t v) { - if (!looks_like_heap_obj(v)) return; - ElHeader* h = (ElHeader*)(uintptr_t)v; - if (h->magic == EL_MAGIC_LIST || h->magic == EL_MAGIC_MAP) { - h->refcount++; - } -} - -void el_release(el_val_t v) { - if (!looks_like_heap_obj(v)) return; - ElHeader* h = (ElHeader*)(uintptr_t)v; - if (h->magic == EL_MAGIC_LIST) { - if (h->refcount > 0 && --h->refcount == 0) { - ElList* l = (ElList*)h; - free(l->elems); - l->hdr.magic = 0; /* poison so use-after-free is detected */ - free(l); - } - } else if (h->magic == EL_MAGIC_MAP) { - if (h->refcount > 0 && --h->refcount == 0) { - ElMap* m = (ElMap*)h; - free(m->keys); - free(m->values); - m->hdr.magic = 0; - free(m); - } - } -} - -/* ── Batch 2/3 forward decls (defined later in JSON section) ────────────── */ - -typedef struct JsonBuf JsonBuf; -typedef struct JsonParser JsonParser; -static void jb_init(JsonBuf* b); -static void jb_putc(JsonBuf* b, char c); -static void jb_puts(JsonBuf* b, const char* s); -static void jb_emit_escaped(JsonBuf* b, const char* s); -static char* jb_finish(JsonBuf* b); -static int looks_like_string(el_val_t v); -static const char* json_find_key(const char* s, const char* key); -static const char* json_skip_value(const char* p); -static char* jp_parse_string_raw(JsonParser* jp); - -/* Struct definitions are visible here because batch 2/3 helpers above use - * them by value; the bodies (jb_init, etc.) appear in the JSON section. */ -struct JsonBuf { - char* buf; - size_t len; - size_t cap; -}; - -struct JsonParser { - const char* p; - const char* end; - int err; -}; - -/* ── Batch 2: Real HTTP (libcurl client + POSIX-socket server) ───────────── */ -/* - * Client: blocking libcurl easy-handle calls. Errors are returned as a JSON - * fragment {"error":"..."} so callers can detect via str_starts_with("{") / - * json_get_string("error", ...). - * - * Server: bind/listen/accept loop on a TCP socket. Each accepted connection - * is handled in its own pthread (detached). A semaphore-style counter caps - * concurrent in-flight connections at HTTP_MAX_CONNS (64). When the cap is - * reached, accept() blocks until a worker exits. This prevents runaway - * thread creation under high load. - * - * Handler dispatch: El does not expose first-class function references at - * the runtime layer, so the second argument to http_serve(port, handler) is - * treated as a string name (or any el_val_t — the runtime ignores its - * value and uses the registry). Callers register a C-level handler via - * - * extern void el_runtime_register_handler(const char* name, - * el_val_t (*fn)(el_val_t, - * el_val_t, - * el_val_t)); - * - * and select the active handler by calling http_set_handler("name") from - * El, or by setting it directly through the C registry. If no handler is - * registered, the server replies with a 200 carrying a default message so - * the loop is observable. - */ - -/* ── JSON error helper (used by HTTP, PQ, crypto stubs) ─────────────────── */ - -/* JSON-escape an arbitrary C string into an allocated buffer. */ -static char* json_escape_alloc(const char* s) { - if (!s) return el_strdup(""); - JsonBuf b; jb_init(&b); - for (const char* p = s; *p; p++) { - unsigned char c = (unsigned char)*p; - switch (c) { - case '"': jb_puts(&b, "\\\""); break; - case '\\': jb_puts(&b, "\\\\"); break; - case '\n': jb_puts(&b, "\\n"); break; - case '\r': jb_puts(&b, "\\r"); break; - case '\t': jb_puts(&b, "\\t"); break; - default: - if (c < 0x20) { - char tmp[8]; snprintf(tmp, sizeof(tmp), "\\u%04x", c); - jb_puts(&b, tmp); - } else jb_putc(&b, (char)c); - } - } - return b.buf; -} - -static el_val_t http_error_json(const char* msg) { - char* esc = json_escape_alloc(msg ? msg : "unknown error"); - char* buf = el_strbuf(strlen(esc) + 16); - sprintf(buf, "{\"error\":\"%s\"}", esc); - free(esc); - return el_wrap_str(buf); -} - -#ifdef HAVE_CURL -/* ── HTTP client write-callback buffer ───────────────────────────────────── */ - -typedef struct { - char* data; - size_t len; - size_t cap; -} HttpBuf; - -static void httpbuf_init(HttpBuf* b) { - b->cap = 1024; - b->len = 0; - b->data = malloc(b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->data[0] = '\0'; -} - -static void httpbuf_append(HttpBuf* b, const void* src, size_t n) { - if (b->len + n + 1 > b->cap) { - while (b->len + n + 1 > b->cap) b->cap *= 2; - b->data = realloc(b->data, b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } - memcpy(b->data + b->len, src, n); - b->len += n; - b->data[b->len] = '\0'; -} - -static size_t http_write_cb(char* ptr, size_t size, size_t nmemb, void* ud) { - size_t n = size * nmemb; - httpbuf_append((HttpBuf*)ud, ptr, n); - return n; -} - -/* HTTP timeout (ms) — read once from EL_HTTP_TIMEOUT_MS, default 60000. - * Applied via CURLOPT_TIMEOUT_MS on every libcurl request. */ -static long _el_http_timeout_ms = -1; -static long el_http_timeout_ms(void) { - long v = __atomic_load_n(&_el_http_timeout_ms, __ATOMIC_ACQUIRE); - if (v >= 0) return v; - const char* s = getenv("EL_HTTP_TIMEOUT_MS"); - long parsed = 60000L; - if (s && *s) { - char* end = NULL; - long n = strtol(s, &end, 10); - if (end != s && n > 0) parsed = n; - } - __atomic_store_n(&_el_http_timeout_ms, parsed, __ATOMIC_RELEASE); - return parsed; -} - -/* Internal: do a libcurl request; takes optional body/headers, optional method override. */ -static el_val_t http_do(const char* method, const char* url, const char* body, - struct curl_slist* extra_headers) { - if (!url || !*url) return http_error_json("empty url"); - CURL* c = curl_easy_init(); - if (!c) return http_error_json("curl_easy_init failed"); - HttpBuf rb; httpbuf_init(&rb); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - if (extra_headers) curl_easy_setopt(c, CURLOPT_HTTPHEADER, extra_headers); - if (method && strcmp(method, "POST") == 0) { - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - } else if (method && strcmp(method, "DELETE") == 0) { - curl_easy_setopt(c, CURLOPT_CUSTOMREQUEST, "DELETE"); - } - CURLcode rc = curl_easy_perform(c); - curl_easy_cleanup(c); - if (rc != CURLE_OK) { - free(rb.data); - const char* m = errbuf[0] ? errbuf : curl_easy_strerror(rc); - return http_error_json(m); - } - return el_wrap_str(rb.data); -} - -el_val_t http_get(el_val_t url) { - return http_do("GET", EL_CSTR(url), NULL, NULL); -} - -el_val_t http_post(el_val_t url, el_val_t body) { - return http_do("POST", EL_CSTR(url), EL_CSTR(body), NULL); -} - -el_val_t http_post_json(el_val_t url, el_val_t json_body) { - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(json_body), h); - curl_slist_free_all(h); - return r; -} - -/* Build a curl_slist from an ElMap of name -> value strings. */ -static struct curl_slist* headers_from_map(el_val_t headers_map) { - struct curl_slist* h = NULL; - ElMap* m = as_map(headers_map); - if (!m) return NULL; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - const char* v = EL_CSTR(m->values[i]); - if (!k || !v) continue; - size_t n = strlen(k) + strlen(v) + 4; - char* line = malloc(n); - if (!line) continue; - snprintf(line, n, "%s: %s", k, v); - h = curl_slist_append(h, line); - free(line); - } - return h; -} - -el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do("GET", EL_CSTR(url), NULL, h); - if (h) curl_slist_free_all(h); - return r; -} - -el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(body), h); - if (h) curl_slist_free_all(h); - return r; -} - -/* http_post_json_with_headers — POST with Content-Type: application/json plus - * any additional headers supplied as an El map. Combines http_post_json and - * http_post_with_headers: the Content-Type header is always prepended so - * callers do not have to include it in their map. */ -el_val_t http_post_json_with_headers(el_val_t url, el_val_t headers_map, el_val_t json_body) { - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - /* Append caller-supplied headers from the map */ - ElMap* m = as_map(headers_map); - if (m) { - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - const char* v = EL_CSTR(m->values[i]); - if (!k || !v) continue; - size_t n = strlen(k) + strlen(v) + 4; - char* line = malloc(n); - if (!line) continue; - snprintf(line, n, "%s: %s", k, v); - h = curl_slist_append(h, line); - free(line); - } - } - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(json_body), h); - curl_slist_free_all(h); - return r; -} - -el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header) { - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/x-www-form-urlencoded"); - const char* a = EL_CSTR(auth_header); - if (a && *a) { - size_t n = strlen(a) + 32; - char* line = malloc(n); - snprintf(line, n, "Authorization: %s", a); - h = curl_slist_append(h, line); - free(line); - } - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(form_body), h); - curl_slist_free_all(h); - return r; -} - -/* HTTP DELETE — mirrors http_post but with CURLOPT_CUSTOMREQUEST=DELETE. - * Returns response body on success; on transport failure returns an error - * JSON fragment (same convention as http_get/http_post). Callers that - * expect "" on failure should check for a leading '{' and an "error" key. */ -el_val_t http_delete(el_val_t url) { - return http_do("DELETE", EL_CSTR(url), NULL, NULL); -} - -/* ── HTTP → file streaming ──────────────────────────────────────────────── - * - * Why this exists: el_val_t strings are NUL-terminated by convention, so - * accumulating an HTTP response into an httpbuf and then wrapping its - * `.data` pointer with el_wrap_str() loses the byte length. Any consumer - * that does strlen() on the wrapped pointer truncates the body at the - * first embedded NUL. Audio (MP3, WAV, OGG), images (PNG, JPEG), and any - * other binary payload hits this. The vessels that download such bodies - * (e.g. ElevenLabs TTS → MP3) get silently corrupted files. - * - * The fix: wire libcurl's CURLOPT_WRITEFUNCTION directly to fwrite() - * against a fopen()-ed FILE*. The bytes never pass through an el_val_t - * string, so embedded NULs are preserved verbatim. Caller's contract is - * just "a file at this path with the response body in it". */ - -static size_t http_file_write_cb(char* ptr, size_t size, size_t nmemb, void* ud) { - FILE* f = (FILE*)ud; - return fwrite(ptr, size, nmemb, f); -} - -/* Internal: stream body to file. method is "GET" or "POST". body may be NULL - * (GET) or NUL-terminated (POST). headers may be NULL. Returns 1/0. */ -static el_val_t http_do_to_file(const char* method, const char* url, - const char* body, struct curl_slist* extra_headers, - const char* output_path) { - if (!url || !*url) return 0; - if (!output_path || !*output_path) return 0; - FILE* f = fopen(output_path, "wb"); - if (!f) return 0; - - CURL* c = curl_easy_init(); - if (!c) { fclose(f); remove(output_path); return 0; } - - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_file_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, f); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - curl_easy_setopt(c, CURLOPT_FAILONERROR, 1L); /* 4xx/5xx → CURLE_HTTP_RETURNED_ERROR */ - if (extra_headers) curl_easy_setopt(c, CURLOPT_HTTPHEADER, extra_headers); - - if (method && strcmp(method, "POST") == 0) { - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - /* For the request body we still rely on strlen — POST bodies are - * caller-controlled and JSON/text in every known El use case. - * If a future caller needs a binary POST body, add a *_bytes - * variant that takes an explicit length, mirroring fs_write_bytes. */ - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - } - - CURLcode rc = curl_easy_perform(c); - curl_easy_cleanup(c); - - /* Flush + close before signalling success, so the file is fully on disk - * by the time the caller reads back. */ - int flush_ok = (fflush(f) == 0); - int close_ok = (fclose(f) == 0); - - if (rc != CURLE_OK || !flush_ok || !close_ok) { - remove(output_path); - return 0; - } - return 1; -} - -el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do_to_file("GET", EL_CSTR(url), NULL, h, EL_CSTR(output_path)); - if (h) curl_slist_free_all(h); - return r; -} - -el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do_to_file("POST", EL_CSTR(url), EL_CSTR(body), h, EL_CSTR(output_path)); - if (h) curl_slist_free_all(h); - return r; -} -#endif /* HAVE_CURL */ - -/* ── HTTP server (POSIX sockets + pthreads) ──────────────────────────────── */ - -#define HTTP_MAX_CONNS 64 - - -typedef struct { - char* name; - http_handler_fn fn; -} HttpHandlerEntry; - -static HttpHandlerEntry _http_handlers[32]; -static size_t _http_handler_count = 0; -static char* _http_active_handler = NULL; -static pthread_mutex_t _http_handler_mu = PTHREAD_MUTEX_INITIALIZER; - -static pthread_mutex_t _http_conn_mu = PTHREAD_MUTEX_INITIALIZER; -static pthread_cond_t _http_conn_cv = PTHREAD_COND_INITIALIZER; -static int _http_conn_active = 0; - -/* Public C-level API: register a handler by name. Programs that want El - * `http_serve` to dispatch into their handler call this from main() before - * http_serve. Not declared in the header to keep the public API minimal — - * extern lookup works since C symbols are global. */ -void el_runtime_register_handler(const char* name, http_handler_fn fn); -void el_runtime_register_handler(const char* name, http_handler_fn fn) { - if (!name || !fn) return; - pthread_mutex_lock(&_http_handler_mu); - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, name) == 0) { - _http_handlers[i].fn = fn; - pthread_mutex_unlock(&_http_handler_mu); - return; - } - } - if (_http_handler_count < sizeof(_http_handlers) / sizeof(_http_handlers[0])) { - _http_handlers[_http_handler_count].name = el_strdup(name); - _http_handlers[_http_handler_count].fn = fn; - _http_handler_count++; - } - pthread_mutex_unlock(&_http_handler_mu); -} - -el_val_t http_set_handler(el_val_t name) { - const char* n = EL_CSTR(name); - pthread_mutex_lock(&_http_handler_mu); - free(_http_active_handler); - _http_active_handler = el_strdup(n ? n : ""); - /* If the name is not yet in the registry, try dlsym lookup against - * the running binary's symbol table. Every El `fn name(...)` compiles - * to a global C symbol with that exact name, so El programs can self- - * register their own handlers just by calling http_set_handler("name"). */ - if (n && *n) { - int found = 0; - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, n) == 0) { found = 1; break; } - } - if (!found) { - void* sym = dlsym(RTLD_DEFAULT, n); - if (sym && _http_handler_count < sizeof(_http_handlers) / sizeof(_http_handlers[0])) { - _http_handlers[_http_handler_count].name = el_strdup(n); - _http_handlers[_http_handler_count].fn = (http_handler_fn)sym; - _http_handler_count++; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return 0; -} - -static http_handler_fn http_lookup_active(void) { - http_handler_fn out = NULL; - pthread_mutex_lock(&_http_handler_mu); - if (_http_active_handler) { - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, _http_active_handler) == 0) { - out = _http_handlers[i].fn; break; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return out; -} - -/* Auto-detect Content-Type from response body. */ -static const char* http_detect_content_type(const char* body) { - if (!body) return "text/plain; charset=utf-8"; - const char* p = body; - /* Binary magic bytes — check before stripping whitespace */ - if ((unsigned char)p[0] == 0x89 && p[1]=='P' && p[2]=='N' && p[3]=='G') - return "image/png"; - if ((unsigned char)p[0] == 0xFF && (unsigned char)p[1] == 0xD8) - return "image/jpeg"; - if (strncmp(p, "GIF8", 4) == 0) return "image/gif"; - if (strncmp(p, "RIFF", 4) == 0) return "image/webp"; - if (strncmp(p, "wOFF", 4) == 0) return "font/woff"; - if (strncmp(p, "wOF2", 4) == 0) return "font/woff2"; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (strncasecmp(p, "= cap) { - if (cap >= 1024 * 1024) { free(buf); return -1; } - cap *= 2; - buf = realloc(buf, cap); - if (!buf) return -1; - } - ssize_t n = recv(fd, buf + len, cap - len - 1, 0); - if (n <= 0) { free(buf); return -1; } - len += (size_t)n; - buf[len] = '\0'; - if (strstr(buf, "\r\n\r\n")) break; - } - /* Parse request line */ - char* sp1 = strchr(buf, ' '); - if (!sp1) { free(buf); return -1; } - *sp1 = '\0'; - *out_method = el_strdup(buf); - char* path_start = sp1 + 1; - char* sp2 = strchr(path_start, ' '); - if (!sp2) { free(*out_method); *out_method = NULL; free(buf); return -1; } - *sp2 = '\0'; - *out_path = el_strdup(path_start); - char* hdr_end = strstr(sp2 + 1, "\r\n\r\n"); - /* Capture the raw header block (after the request line's CRLF, up to - * but not including the terminating \r\n\r\n) for callers that asked - * for it. The legacy 3-arg path passes NULL and skips this. */ - if (out_headers_block) { - char* hdr_start = strstr(sp2 + 1, "\r\n"); - if (hdr_start && hdr_start < hdr_end) { - hdr_start += 2; - size_t hb_len = (size_t)(hdr_end - hdr_start); - char* hb = malloc(hb_len + 1); - if (hb) { - memcpy(hb, hdr_start, hb_len); - hb[hb_len] = '\0'; - *out_headers_block = hb; - } - } else { - *out_headers_block = el_strdup(""); - } - } - /* Find Content-Length */ - long content_length = 0; - char* hp = sp2 + 1; - while (hp < hdr_end) { - char* line_end = strstr(hp, "\r\n"); - /* line_end == hdr_end means we're on the LAST header line — its - * trailing \r\n is the same \r\n that begins the \r\n\r\n header - * terminator. Process this line; only stop when line_end is past - * hdr_end (which means the parser walked off the end of the - * header block). The previous condition (line_end >= hdr_end) - * silently dropped any Content-Length that appeared as the last - * header — exactly what real curl/clients tend to emit. */ - if (!line_end || line_end > hdr_end) break; - if (strncasecmp(hp, "Content-Length:", 15) == 0) { - content_length = strtol(hp + 15, NULL, 10); - if (content_length < 0) content_length = 0; - if (content_length > 64 * 1024 * 1024) content_length = 64 * 1024 * 1024; - } - hp = line_end + 2; - } - /* Body: any bytes already read past hdr_end, plus more recv */ - char* body_start = hdr_end + 4; - size_t body_have = (buf + len) - body_start; - char* body = malloc((size_t)content_length + 1); - if (!body) { free(*out_method); free(*out_path); *out_method=NULL; *out_path=NULL; free(buf); return -1; } - if ((long)body_have > content_length) body_have = (size_t)content_length; - if (body_have > 0) memcpy(body, body_start, body_have); - while ((long)body_have < content_length) { - ssize_t n = recv(fd, body + body_have, (size_t)content_length - body_have, 0); - if (n <= 0) break; - body_have += (size_t)n; - } - body[body_have] = '\0'; - *out_body = body; - free(buf); - return 0; -} - -/* Reason phrase for common HTTP statuses. Falls back to "Status" for the - * long tail — clients only care about the numeric code. */ -static const char* http_reason_phrase(int status) { - switch (status) { - case 200: return "OK"; - case 201: return "Created"; - case 202: return "Accepted"; - case 204: return "No Content"; - case 301: return "Moved Permanently"; - case 302: return "Found"; - case 303: return "See Other"; - case 304: return "Not Modified"; - case 307: return "Temporary Redirect"; - case 308: return "Permanent Redirect"; - case 400: return "Bad Request"; - case 401: return "Unauthorized"; - case 403: return "Forbidden"; - case 404: return "Not Found"; - case 405: return "Method Not Allowed"; - case 409: return "Conflict"; - case 410: return "Gone"; - case 422: return "Unprocessable Entity"; - case 429: return "Too Many Requests"; - case 500: return "Internal Server Error"; - case 501: return "Not Implemented"; - case 502: return "Bad Gateway"; - case 503: return "Service Unavailable"; - case 504: return "Gateway Timeout"; - default: return "Status"; - } -} - -/* Best-effort send with retry on partial writes. */ -static int http_send_all(int fd, const char* p, size_t left) { - while (left > 0) { - ssize_t w = send(fd, p, left, 0); - if (w <= 0) return -1; - p += w; left -= (size_t)w; - } - return 0; -} - -/* Discriminator that http_response() embeds at the start of its envelope. - * A handler returning a string starting with this exact prefix is treated - * as a structured response; anything else is treated as a raw body. */ -#define EL_HTTP_RESPONSE_TAG "{\"el_http_response\":1" - -/* Keys that conflict with runtime-managed headers are silently dropped to - * avoid double-emission — the runtime always emits its own Content-Length - * and Connection: close. Content-Type from the envelope IS allowed and - * overrides auto-detection. */ -static int http_header_is_managed(const char* k) { - return strcasecmp(k, "Content-Length") == 0 - || strcasecmp(k, "Connection") == 0; -} - -/* Walk an ElMap of header pairs and emit each as `K: V\r\n` into JsonBuf b. - * Sets *out_saw_content_type to 1 if the map contained an explicit - * Content-Type so the caller can skip auto-detection. */ -static void http_emit_headers_from_map(JsonBuf* b, el_val_t headers_map, - int* out_saw_content_type) { - *out_saw_content_type = 0; - if (headers_map == 0) return; - ElMap* m = (ElMap*)(uintptr_t)headers_map; - if (!m || m->hdr.magic != EL_MAGIC_MAP) return; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - const char* v = EL_CSTR(m->values[i]); - if (!k || !v) continue; - if (http_header_is_managed(k)) continue; - if (strcasecmp(k, "Content-Type") == 0) *out_saw_content_type = 1; - jb_puts(b, k); - jb_puts(b, ": "); - jb_puts(b, v); - jb_puts(b, "\r\n"); - } -} - -/* Parse the envelope produced by http_response(). On success returns 1 and - * populates *out_status, *out_headers_map (an ElMap el_val_t — caller must - * el_release), and *out_body (allocated). On failure returns 0. - * - * Implementation: feeds the entire envelope through the recursive-descent - * JSON parser (which builds proper ElMap/ElList values), then pulls the - * three top-level fields by name. Avoids re-stringifying the headers map - * since json_stringify() does not support nested objects. */ -static int http_parse_envelope(const char* s, int* out_status, - el_val_t* out_headers_map, char** out_body, - el_val_t* out_parsed_root) { - if (!s) return 0; - if (strncmp(s, EL_HTTP_RESPONSE_TAG, - sizeof(EL_HTTP_RESPONSE_TAG) - 1) != 0) return 0; - - el_val_t parsed = json_parse(EL_STR(s)); - if (parsed == EL_NULL) return 0; - - int status = 200; - el_val_t hmap = 0; - char* body = NULL; - - el_val_t sv = el_map_get(parsed, EL_STR("status")); - if (sv != 0) { - /* status comes back as an integer — el_val_t holds it directly. */ - long sc = (long)sv; - if (sc >= 100 && sc <= 599) status = (int)sc; - } - - el_val_t hv = el_map_get(parsed, EL_STR("headers")); - if (hv != 0) { - ElMap* hm = (ElMap*)(uintptr_t)hv; - if (hm && hm->hdr.magic == EL_MAGIC_MAP) hmap = hv; - } - - el_val_t bv = el_map_get(parsed, EL_STR("body")); - if (bv != 0) { - const char* bs = EL_CSTR(bv); - if (bs) body = el_strdup(bs); - } - if (!body) body = el_strdup(""); - - *out_status = status; - *out_headers_map = hmap; - *out_body = body; - *out_parsed_root = parsed; /* caller releases to free hmap + entries */ - return 1; -} - -/* Lightweight `__status__` envelope: if the body's first key is `__status__` - * and its value is a numeric literal, lift the status to the HTTP layer and - * strip the marker from the body before sending. This is the common case for - * El handlers that want to return 4xx/5xx without going through - * http_response() — they just prepend `{"__status__":,...}` to the JSON - * they were already returning. - * - * We deliberately recognise ONLY the first-key form so the contract is cheap - * to detect and unambiguous: `{"__status__":401,"error":"unauthorized"}` is - * an envelope, but `{"error":"...","__status__":401}` is not. Product code - * controls placement. - * - * On success returns 1 with *out_status set and *out_body_alloc populated - * with a freshly malloc'd body (caller frees). On failure returns 0 and - * leaves outputs untouched. */ -static int http_parse_status_envelope(const char* s, int* out_status, - char** out_body_alloc) { - if (!s) return 0; - const char* p = s; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '{') return 0; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - static const char marker[] = "\"__status__\""; - size_t mlen = sizeof(marker) - 1; - if (strncmp(p, marker, mlen) != 0) return 0; - p += mlen; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != ':') return 0; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p < '0' || *p > '9') return 0; /* non-numeric -> not an envelope */ - int status = 0; - while (*p >= '0' && *p <= '9') { - status = status * 10 + (*p - '0'); - p++; - } - if (status < 100 || status > 599) return 0; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - /* Two trailing shapes accepted: - * ,"k":v,...} -> body becomes {"k":v,...} - * } -> body becomes {} - * Anything else (e.g. `:` re-appearing, garbage) drops the envelope so - * we don't strip what we shouldn't. */ - if (*p == '}') { - *out_status = status; - *out_body_alloc = el_strdup("{}"); - return 1; - } - if (*p != ',') return 0; - p++; /* skip the comma; the rest of the object follows */ - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - /* Build the trimmed body: '{' + remainder. */ - size_t rest_len = strlen(p); - char* out = (char*)malloc(rest_len + 2); - if (!out) return 0; - out[0] = '{'; - memcpy(out + 1, p, rest_len); - out[rest_len + 1] = '\0'; - *out_status = status; - *out_body_alloc = out; - return 1; -} - -/* Send a fully-built HTTP response. If `body` starts with the envelope tag, - * unpack status/headers/body. Otherwise emit the historical 200-OK with - * auto-detected Content-Type. */ -/* Thread-local flag: if 1, http_send_response writes status + headers but - * NO body (HEAD method behaviour). Set by http_worker before calling - * http_send_response, cleared after. */ -static __thread int _tl_http_head_only = 0; - -static void http_send_response(int fd, const char* body) { - if (!body) body = ""; - - int status = 200; - el_val_t env_headers_map = 0; - char* env_body = NULL; - el_val_t env_parsed_root = 0; - int is_envelope = http_parse_envelope(body, &status, - &env_headers_map, &env_body, - &env_parsed_root); - - /* If the rich http_response() envelope didn't claim this body, try the - * lightweight `__status__` form. This second envelope is malloc-backed so - * we route it through env_body and let the existing cleanup path free it - * — same lifetime contract, no special case at the bottom of the - * function. */ - if (!is_envelope) { - char* trimmed = NULL; - if (http_parse_status_envelope(body, &status, &trimmed)) { - env_body = trimmed; - is_envelope = 1; - } - } - - const char* eff_body = is_envelope ? env_body : body; - /* Use the real byte count from fs_read ONLY when this body IS the exact - * buffer fs_read returned (binary files with embedded null bytes — PNG, - * WOFF2, etc.). Any other body — wrapped, enveloped, or derived — must be - * measured with strlen, or it is truncated/over-read to the file's size. */ - size_t blen = (_tl_fs_read_len > 0 && eff_body == _tl_fs_read_buf) - ? _tl_fs_read_len : strlen(eff_body); - _tl_fs_read_len = 0; /* consume — one-shot per response */ - _tl_fs_read_buf = NULL; - int head_only = _tl_http_head_only; - - JsonBuf hdrs; jb_init(&hdrs); - int saw_content_type = 0; - if (is_envelope) { - http_emit_headers_from_map(&hdrs, env_headers_map, - &saw_content_type); - } - if (!saw_content_type) { - jb_puts(&hdrs, "Content-Type: "); - jb_puts(&hdrs, http_detect_content_type(eff_body)); - jb_puts(&hdrs, "\r\n"); - } - - char status_line[64]; - int sl = snprintf(status_line, sizeof(status_line), - "HTTP/1.1 %d %s\r\n", - status, http_reason_phrase(status)); - if (sl < 0) { - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); free(hdrs.buf); return; - } - - char tail[128]; - int tl = snprintf(tail, sizeof(tail), - "Content-Length: %zu\r\n" - "Connection: close\r\n" - "\r\n", blen); - if (tl < 0) { - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); free(hdrs.buf); return; - } - - if (http_send_all(fd, status_line, (size_t)sl) == 0 - && http_send_all(fd, hdrs.buf, hdrs.len) == 0 - && http_send_all(fd, tail, (size_t)tl) == 0 - && (head_only - /* HEAD requests echo headers + Content-Length but no body. */ - ? 1 - : http_send_all(fd, eff_body, blen) == 0)) { - /* sent successfully */ - } - - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); - free(hdrs.buf); -} - -typedef struct { -#ifdef _WIN32 - SOCKET fd; -#else - int fd; -#endif -} HttpWorkerArg; - -static void* http_worker(void* arg) { - HttpWorkerArg* a = (HttpWorkerArg*)arg; -#ifdef _WIN32 - SOCKET fd = a->fd; -#else - int fd = a->fd; -#endif - free(a); - char *method = NULL, *path = NULL, *body = NULL; - if (http_read_request(fd, &method, &path, &body, NULL) == 0) { - http_handler_fn h = http_lookup_active(); - char* response = NULL; - /* HEAD: dispatch as GET so existing handlers respond with the same - * body, but flag the response writer to emit headers only. RFC 9110 - * requires HEAD to mirror GET headers + Content-Length without body. */ - int head_only = (method && strcmp(method, "HEAD") == 0); - const char* dispatch_method = head_only ? "GET" : method; - el_request_start(); /* begin per-request arena */ - if (h) { - el_val_t r = h(EL_STR(dispatch_method), EL_STR(path), EL_STR(body)); - const char* rs = EL_CSTR(r); - /* Copy response out BEFORE arena teardown. - * For binary files, _tl_fs_read_len holds the real byte count — - * use memcpy instead of strdup so null bytes are preserved. - * The stored length applies ONLY when the response IS the exact - * fs_read buffer; a wrapped/derived response must use strlen or - * it gets truncated (or over-read) to the file's length. */ - size_t rlen; - if (_tl_fs_read_len > 0 && rs && rs == _tl_fs_read_buf) { - rlen = _tl_fs_read_len; /* raw file bytes — binary-safe */ - } else { - rlen = rs ? strlen(rs) : 0; - _tl_fs_read_len = 0; /* hint doesn't describe this body */ - _tl_fs_read_buf = NULL; - } - response = malloc(rlen + 1); - if (response && rs) { memcpy(response, rs, rlen); response[rlen] = '\0'; } - else if (response) { response[0] = '\0'; } - if (_tl_fs_read_len > 0) _tl_fs_read_buf = response; /* hint follows the copy */ - } else { - response = el_strdup_persist("el-runtime: no http handler registered"); - } - el_request_end(); /* free all intermediate strings */ - _tl_http_head_only = head_only; - http_send_response(fd, response); - _tl_http_head_only = 0; - free(response); - } - free(method); free(path); free(body); - el_closesocket(fd); - /* release a slot */ - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - return NULL; -} - -el_val_t http_serve(el_val_t port, el_val_t handler) { - /* If `handler` looks like a string name, register it as the active handler. */ - const char* hname = EL_CSTR(handler); - if (hname && looks_like_string(handler)) { - http_set_handler(handler); - } - int p = (int)port; - if (p <= 0 || p > 65535) { fprintf(stderr, "http_serve: invalid port %d\n", p); return 0; } - /* Dual-stack: AF_INET6 with IPV6_V6ONLY=0 accepts both IPv4 and IPv6. - * This makes `localhost` work in browsers that resolve it to ::1 first. */ - int sock = socket(AF_INET6, SOCK_STREAM, 0); - if (sock < 0) { perror("socket"); return 0; } - int yes = 1; int no = 0; - setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, (const char*)&yes, sizeof(yes)); - setsockopt(sock, IPPROTO_IPV6, IPV6_V6ONLY, (const char*)&no, sizeof(no)); - struct sockaddr_in6 addr; - memset(&addr, 0, sizeof(addr)); - addr.sin6_family = AF_INET6; - addr.sin6_addr = in6addr_any; - addr.sin6_port = htons((uint16_t)p); - if (bind(sock, (struct sockaddr*)&addr, sizeof(addr)) < 0) { - perror("bind"); el_closesocket(sock); return 0; - } - if (listen(sock, 64) < 0) { perror("listen"); el_closesocket(sock); return 0; } - fprintf(stderr, "[http] listening on [::]:%d (dual-stack)\n", p); - while (1) { - struct sockaddr_in6 cli; - socklen_t clen = sizeof(cli); -#ifdef _WIN32 - SOCKET cfd = accept(sock, (struct sockaddr*)&cli, &clen); -#else - int cfd = accept(sock, (struct sockaddr*)&cli, &clen); -#endif - if (cfd < 0) { - if (errno == EINTR) continue; - perror("accept"); break; - } - pthread_mutex_lock(&_http_conn_mu); - while (_http_conn_active >= HTTP_MAX_CONNS) { - pthread_cond_wait(&_http_conn_cv, &_http_conn_mu); - } - _http_conn_active++; - pthread_mutex_unlock(&_http_conn_mu); - HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); - if (!arg) { el_closesocket(cfd); continue; } - arg->fd = cfd; - pthread_t tid; - if (pthread_create(&tid, NULL, http_worker, arg) != 0) { - el_closesocket(cfd); free(arg); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - continue; - } - pthread_detach(tid); - } - el_closesocket(sock); - return 0; -} - -/* ── HTTP server v2 — request headers + structured response ──────────────── */ -/* - * v2 widens the handler signature from - * (method, path, body) -> body_string - * to - * (method, path, headers_map, body) -> body_string_or_envelope - * - * The response envelope is detected uniformly inside http_send_response — so - * 4-arg handlers can return either a plain body or http_response(...). The - * 3-arg path stays untouched in spirit (its handlers still build plain - * bodies; the envelope tag, being `{"el_http_response":1`, will never - * collide with normal JSON the legacy server.el routes return). - * - * Registry is parallel to the 3-arg handler registry: separate name table, - * separate active-handler slot, separate dlsym fallback. Mixing v1 and v2 - * handlers in the same process is fine — they don't share the active slot. */ - - -typedef struct { - char* name; - http_handler4_fn fn; -} HttpHandler4Entry; - -static HttpHandler4Entry _http_handlers4[32]; -static size_t _http_handler4_count = 0; -static char* _http_active_handler4 = NULL; - -void el_runtime_register_handler_v2(const char* name, http_handler4_fn fn); -void el_runtime_register_handler_v2(const char* name, http_handler4_fn fn) { - if (!name || !fn) return; - pthread_mutex_lock(&_http_handler_mu); - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, name) == 0) { - _http_handlers4[i].fn = fn; - pthread_mutex_unlock(&_http_handler_mu); - return; - } - } - if (_http_handler4_count < - sizeof(_http_handlers4) / sizeof(_http_handlers4[0])) { - _http_handlers4[_http_handler4_count].name = el_strdup(name); - _http_handlers4[_http_handler4_count].fn = fn; - _http_handler4_count++; - } - pthread_mutex_unlock(&_http_handler_mu); -} - -el_val_t http_set_handler_v2(el_val_t name) { - const char* n = EL_CSTR(name); - pthread_mutex_lock(&_http_handler_mu); - free(_http_active_handler4); - _http_active_handler4 = el_strdup(n ? n : ""); - if (n && *n) { - int found = 0; - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, n) == 0) { found = 1; break; } - } - if (!found) { - void* sym = dlsym(RTLD_DEFAULT, n); - if (sym && _http_handler4_count < - sizeof(_http_handlers4) / sizeof(_http_handlers4[0])) { - _http_handlers4[_http_handler4_count].name = el_strdup(n); - _http_handlers4[_http_handler4_count].fn = - (http_handler4_fn)sym; - _http_handler4_count++; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return 0; -} - -static http_handler4_fn http_lookup_active_v2(void) { - http_handler4_fn out = NULL; - pthread_mutex_lock(&_http_handler_mu); - if (_http_active_handler4) { - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, - _http_active_handler4) == 0) { - out = _http_handlers4[i].fn; break; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return out; -} - -/* Build an ElMap from the raw header block produced by http_read_request. - * Keys are lowercased (RFC 7230 — case-insensitive); values have leading - * whitespace trimmed. Repeated headers with the same name are joined with - * ", " in arrival order, matching standard library behaviour elsewhere. */ -static el_val_t http_build_headers_map(const char* hdr_block) { - el_val_t m = el_map_new(0); - if (!hdr_block || !*hdr_block) return m; - const char* p = hdr_block; - while (*p) { - const char* line_end = strstr(p, "\r\n"); - const char* end = line_end ? line_end : p + strlen(p); - const char* colon = NULL; - for (const char* c = p; c < end; c++) { - if (*c == ':') { colon = c; break; } - } - if (colon && colon > p) { - size_t klen = (size_t)(colon - p); - char* key = malloc(klen + 1); - if (key) { - for (size_t i = 0; i < klen; i++) { - unsigned char ch = (unsigned char)p[i]; - key[i] = (char)tolower(ch); - } - key[klen] = '\0'; - const char* vstart = colon + 1; - while (vstart < end && (*vstart == ' ' || *vstart == '\t')) vstart++; - size_t vlen = (size_t)(end - vstart); - /* Strip trailing OWS just in case. */ - while (vlen > 0 - && (vstart[vlen - 1] == ' ' - || vstart[vlen - 1] == '\t')) vlen--; - /* Coalesce repeats: if key already present, append ", value". */ - el_val_t existing = el_map_get(m, EL_STR(key)); - if (existing != 0 && looks_like_string(existing)) { - const char* old = EL_CSTR(existing); - size_t olen = strlen(old); - char* combined = malloc(olen + 2 + vlen + 1); - if (combined) { - memcpy(combined, old, olen); - memcpy(combined + olen, ", ", 2); - memcpy(combined + olen + 2, vstart, vlen); - combined[olen + 2 + vlen] = '\0'; - m = el_map_set(m, EL_STR(key), EL_STR(combined)); - } - free(key); - } else { - char* val = malloc(vlen + 1); - if (val) { - memcpy(val, vstart, vlen); - val[vlen] = '\0'; - m = el_map_set(m, EL_STR(key), EL_STR(val)); - } else { - free(key); - } - } - } - } - if (!line_end) break; - p = line_end + 2; - } - return m; -} - -static void* http_worker_v2(void* arg) { - HttpWorkerArg* a = (HttpWorkerArg*)arg; -#ifdef _WIN32 - SOCKET fd = a->fd; -#else - int fd = a->fd; -#endif - free(a); - char *method = NULL, *path = NULL, *body = NULL, *hdr_block = NULL; - if (http_read_request(fd, &method, &path, &body, &hdr_block) == 0) { - http_handler4_fn h = http_lookup_active_v2(); - char* response = NULL; - int head_only = (method && strcmp(method, "HEAD") == 0); - const char* dispatch_method = head_only ? "GET" : method; - el_request_start(); /* begin per-request arena */ - if (h) { - el_val_t hmap = http_build_headers_map(hdr_block ? hdr_block : ""); - el_val_t r = h(EL_STR(dispatch_method), EL_STR(path), hmap, EL_STR(body)); - const char* rs = EL_CSTR(r); - /* Same pairing rule as the v1 worker: the fs_read length is only - * trustworthy for the exact buffer fs_read returned. */ - size_t rlen; - if (_tl_fs_read_len > 0 && rs && rs == _tl_fs_read_buf) { - rlen = _tl_fs_read_len; /* raw file bytes — binary-safe */ - } else { - rlen = rs ? strlen(rs) : 0; - _tl_fs_read_len = 0; /* hint doesn't describe this body */ - _tl_fs_read_buf = NULL; - } - response = malloc(rlen + 1); - if (response && rs) { memcpy(response, rs, rlen); response[rlen] = '\0'; } - else if (response) { response[0] = '\0'; } - if (_tl_fs_read_len > 0) _tl_fs_read_buf = response; /* hint follows the copy */ - el_release(hmap); - } else { - response = el_strdup_persist( - "el-runtime: no v2 http handler registered " - "(call http_set_handler_v2)"); - } - el_request_end(); /* free all intermediate strings */ - _tl_http_head_only = head_only; - http_send_response(fd, response); - _tl_http_head_only = 0; - free(response); - } - free(method); free(path); free(body); free(hdr_block); - el_closesocket(fd); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - return NULL; -} - -el_val_t http_serve_v2(el_val_t port, el_val_t handler) { - const char* hname = EL_CSTR(handler); - if (hname && looks_like_string(handler)) { - http_set_handler_v2(handler); - } - int p = (int)port; - if (p <= 0 || p > 65535) { - fprintf(stderr, "http_serve_v2: invalid port %d\n", p); - return 0; - } - /* Dual-stack: same as http_serve - AF_INET6 + IPV6_V6ONLY=0. */ - int sock = socket(AF_INET6, SOCK_STREAM, 0); - if (sock < 0) { perror("socket"); return 0; } - int yes = 1; int no = 0; - setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, (const char*)&yes, sizeof(yes)); - setsockopt(sock, IPPROTO_IPV6, IPV6_V6ONLY, (const char*)&no, sizeof(no)); - struct sockaddr_in6 addr; - memset(&addr, 0, sizeof(addr)); - addr.sin6_family = AF_INET6; - addr.sin6_addr = in6addr_any; - addr.sin6_port = htons((uint16_t)p); - if (bind(sock, (struct sockaddr*)&addr, sizeof(addr)) < 0) { - perror("bind"); el_closesocket(sock); return 0; - } - if (listen(sock, 64) < 0) { perror("listen"); el_closesocket(sock); return 0; } - fprintf(stderr, "[http v2] listening on [::]:%d (dual-stack)\n", p); - while (1) { - struct sockaddr_in6 cli; - socklen_t clen = sizeof(cli); -#ifdef _WIN32 - SOCKET cfd = accept(sock, (struct sockaddr*)&cli, &clen); -#else - int cfd = accept(sock, (struct sockaddr*)&cli, &clen); -#endif - if (cfd < 0) { - if (errno == EINTR) continue; - perror("accept"); break; - } - pthread_mutex_lock(&_http_conn_mu); - while (_http_conn_active >= HTTP_MAX_CONNS) { - pthread_cond_wait(&_http_conn_cv, &_http_conn_mu); - } - _http_conn_active++; - pthread_mutex_unlock(&_http_conn_mu); - HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); - if (!arg) { el_closesocket(cfd); continue; } - arg->fd = cfd; - pthread_t tid; - if (pthread_create(&tid, NULL, http_worker_v2, arg) != 0) { - el_closesocket(cfd); free(arg); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - continue; - } - pthread_detach(tid); - } - el_closesocket(sock); - return 0; -} - -/* ── http_serve_async — non-blocking HTTP server ─────────────────────────── */ -/* Runs the accept loop in a background pthread, returns immediately so the - * calling EL script can continue (e.g. to run an awareness loop). - * - * El signature: http_serve_async(port, handler) -> Void */ - -typedef struct { int sock; } HttpServeAsyncArg; - -static void* _http_serve_async_loop(void* raw) { - HttpServeAsyncArg* a = (HttpServeAsyncArg*)raw; - int sock = a->sock; - free(a); - while (1) { - struct sockaddr_in6 cli; - socklen_t clen = sizeof(cli); - int cfd = accept(sock, (struct sockaddr*)&cli, &clen); - if (cfd < 0) { - if (errno == EINTR) continue; - perror("accept"); break; - } - pthread_mutex_lock(&_http_conn_mu); - while (_http_conn_active >= HTTP_MAX_CONNS) { - pthread_cond_wait(&_http_conn_cv, &_http_conn_mu); - } - _http_conn_active++; - pthread_mutex_unlock(&_http_conn_mu); - HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); - if (!arg) { close(cfd); continue; } - arg->fd = cfd; - pthread_t tid; - if (pthread_create(&tid, NULL, http_worker, arg) != 0) { - close(cfd); free(arg); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - continue; - } - pthread_detach(tid); - } - close(sock); - return NULL; -} - -void http_serve_async(el_val_t port, el_val_t handler) { - const char* hname = EL_CSTR(handler); - if (hname && looks_like_string(handler)) { - http_set_handler(handler); - } - int p = (int)port; - if (p <= 0 || p > 65535) { fprintf(stderr, "http_serve_async: invalid port %d\n", p); return; } - int sock = socket(AF_INET6, SOCK_STREAM, 0); - if (sock < 0) { perror("socket"); return; } - int yes = 1; int no = 0; - /* Win32/mingw setsockopt takes optval as (const char*); the cast is portable on POSIX too. */ - setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, (const char*)&yes, sizeof(yes)); - setsockopt(sock, IPPROTO_IPV6, IPV6_V6ONLY, (const char*)&no, sizeof(no)); - struct sockaddr_in6 addr; - memset(&addr, 0, sizeof(addr)); - addr.sin6_family = AF_INET6; - addr.sin6_addr = in6addr_any; - addr.sin6_port = htons((uint16_t)p); - if (bind(sock, (struct sockaddr*)&addr, sizeof(addr)) < 0) { - perror("bind"); close(sock); return; - } - if (listen(sock, 64) < 0) { perror("listen"); close(sock); return; } - fprintf(stderr, "[http] async listening on [::]:%d (dual-stack)\n", p); - HttpServeAsyncArg* a = malloc(sizeof(HttpServeAsyncArg)); - if (!a) { close(sock); return; } - a->sock = sock; - pthread_t tid; - if (pthread_create(&tid, NULL, _http_serve_async_loop, a) != 0) { - perror("pthread_create"); free(a); close(sock); return; - } - pthread_detach(tid); - /* Returns immediately — caller can now run awareness_run() or any loop. */ -} - -/* Build the response envelope a 4-arg handler can return. We hand-write - * the JSON so the discriminator key always lands first — the runtime's - * http_parse_envelope() detects it via prefix match. headers_json must be - * either "" (empty), "{}" (empty object), or a well-formed JSON object - * literal; anything else will produce a malformed envelope and the runtime - * will treat the whole string as a plain body (no envelope detected). */ -el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body) { - long sc = (long)status; - if (sc < 100 || sc > 599) sc = 200; - const char* hj = EL_CSTR(headers_json); - if (!hj || !*hj) hj = "{}"; - /* Light validation: must start with '{' and end with '}'. */ - size_t hlen = strlen(hj); - int hj_ok = (hlen >= 2 && hj[0] == '{' && hj[hlen - 1] == '}'); - if (!hj_ok) hj = "{}"; - const char* b = EL_CSTR(body); - if (!b) b = ""; - - JsonBuf out; jb_init(&out); - jb_puts(&out, EL_HTTP_RESPONSE_TAG); /* {"el_http_response":1 */ - jb_puts(&out, ",\"status\":"); - char num[32]; - snprintf(num, sizeof(num), "%ld", sc); - jb_puts(&out, num); - jb_puts(&out, ",\"headers\":"); - jb_puts(&out, hj); - jb_puts(&out, ",\"body\":"); - jb_emit_escaped(&out, b); - jb_putc(&out, '}'); - return el_wrap_str(jb_finish(&out)); -} - -/* ── Filesystem ──────────────────────────────────────────────────────────── */ - -el_val_t fs_read(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - _tl_fs_read_len = 0; - _tl_fs_read_buf = NULL; - if (!path) return el_wrap_str(el_strdup("")); - FILE* f = fopen(path, "rb"); - if (!f) return el_wrap_str(el_strdup("")); - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz < 0) { fclose(f); return el_wrap_str(el_strdup("")); } /* pipe/special file */ - char* buf = el_strbuf((size_t)sz); - size_t got = fread(buf, 1, (size_t)sz, f); - buf[got] = '\0'; - _tl_fs_read_len = got; /* store real byte count for binary-safe send */ - _tl_fs_read_buf = buf; /* ...valid ONLY for this exact buffer */ - fclose(f); - return el_wrap_str(buf); -} - -el_val_t fs_write(el_val_t pathv, el_val_t contentv) { - const char* path = EL_CSTR(pathv); - const char* content = EL_CSTR(contentv); - if (!path || !content) return 0; - FILE* f = fopen(path, "wb"); - if (!f) return 0; - size_t n = strlen(content); - size_t written = fwrite(content, 1, n, f); - fclose(f); - return written == n ? 1 : 0; -} - -/* fs_write_bytes — explicit-length binary write. Bypasses strlen so embedded - * NULs survive. Caller must know the byte count (e.g. from base64_decode, - * or the fixed 32-byte sha256_bytes/hmac_sha256_bytes outputs). - * - * If `length` is negative, treats as failure. If `length` is 0, creates an - * empty file (still useful as a "touch with content" primitive). */ -el_val_t fs_write_bytes(el_val_t pathv, el_val_t bytesv, el_val_t lengthv) { - const char* path = EL_CSTR(pathv); - const char* bytes = EL_CSTR(bytesv); - int64_t n = (int64_t)lengthv; - if (!path || !bytes) return 0; - if (n < 0) return 0; - FILE* f = fopen(path, "wb"); - if (!f) return 0; - size_t written = (n > 0) ? fwrite(bytes, 1, (size_t)n, f) : 0; - int flush_ok = (fflush(f) == 0); - int close_ok = (fclose(f) == 0); - if (!flush_ok || !close_ok || written != (size_t)n) { - remove(path); - return 0; - } - return 1; -} - -// stdout_to_file / stdout_restore — redirect process stdout to a file and -// restore it. Used by the compiler's JS post-processing pipeline to capture -// codegen output before piping through terser / obfuscator. -#include -static int _el_saved_stdout_fd = -1; - -el_val_t stdout_to_file(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path) return (el_val_t)(int64_t)-1; - fflush(stdout); - _el_saved_stdout_fd = dup(STDOUT_FILENO); - int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600); - if (fd < 0) return (el_val_t)(int64_t)-1; - dup2(fd, STDOUT_FILENO); - close(fd); - return (el_val_t)(int64_t)0; -} - -el_val_t stdout_restore(void) { - if (_el_saved_stdout_fd >= 0) { - fflush(stdout); - dup2(_el_saved_stdout_fd, STDOUT_FILENO); - close(_el_saved_stdout_fd); - _el_saved_stdout_fd = -1; - } - return (el_val_t)(int64_t)0; -} - -// exec_command — run a shell command, return exit code (0 = success). -// Used by elb and other El tooling to invoke subprocesses. -el_val_t exec_command(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd) return (el_val_t)(int64_t)-1; - int ret = system(cmd); - return (el_val_t)(int64_t)ret; -} - -// exec_capture — run a shell command, capture stdout, return as String. -// Returns "" on failure. -el_val_t exec_capture(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd) return el_wrap_str(el_strdup("")); - FILE* f = popen(cmd, "r"); - if (!f) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - char buf[4096]; - while (fgets(buf, sizeof(buf), f)) jb_puts(&b, buf); - pclose(f); - return el_wrap_str(jb_finish(&b)); -} - -// exec — run a shell command via /bin/sh, capture stdout, return as String. -// Times out after 30 seconds. Returns "" on any error. -// El name: exec(cmd) -> String -el_val_t exec(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd || !*cmd) return el_wrap_str(el_strdup("")); - /* Build a time-limited command: wrap with timeout(1) if available, - * otherwise rely on the 30s read loop guard below. We use the simple - * popen approach with a deadline measured by wall clock so the caller - * is never blocked indefinitely. */ - FILE* f = popen(cmd, "r"); - if (!f) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - char buf[4096]; - /* 30-second wall-clock deadline */ - time_t deadline = time(NULL) + 30; - while (time(NULL) < deadline) { - if (fgets(buf, sizeof(buf), f) == NULL) break; - jb_puts(&b, buf); - } - pclose(f); - return el_wrap_str(jb_finish(&b)); -} - -// exec_bg — run a shell command in background, return PID as String. -// The child process runs independently; the caller is not blocked. -// Returns "" on fork failure. -// El name: exec_bg(cmd) -> String -el_val_t exec_bg(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd || !*cmd) return el_wrap_str(el_strdup("")); -#ifdef _WIN32 - /* Windows: no fork/exec. Launch a detached `cmd /c ` with no console window via - CreateProcess (DETACHED_PROCESS | CREATE_NO_WINDOW). Returns the PID as a string, "" on fail. - Mirrors the POSIX branch: child runs independently, caller is not blocked. */ - char cmdline[8192]; - snprintf(cmdline, sizeof(cmdline), "cmd.exe /c %s", cmd); - STARTUPINFOA si; ZeroMemory(&si, sizeof(si)); si.cb = sizeof(si); - PROCESS_INFORMATION pi; ZeroMemory(&pi, sizeof(pi)); - BOOL ok = CreateProcessA(NULL, cmdline, NULL, NULL, FALSE, - DETACHED_PROCESS | CREATE_NO_WINDOW, NULL, NULL, &si, &pi); - if (!ok) return el_wrap_str(el_strdup("")); - char pidbuf[32]; - snprintf(pidbuf, sizeof(pidbuf), "%lu", (unsigned long)pi.dwProcessId); - CloseHandle(pi.hProcess); - CloseHandle(pi.hThread); - return el_wrap_str(el_strdup(pidbuf)); -#else - pid_t pid = fork(); - if (pid < 0) { - /* fork failed */ - return el_wrap_str(el_strdup("")); - } - if (pid == 0) { - /* child: detach from parent's stdio, exec via shell */ - setsid(); - int devnull = open("/dev/null", O_RDWR); - if (devnull >= 0) { - dup2(devnull, STDIN_FILENO); - dup2(devnull, STDOUT_FILENO); - dup2(devnull, STDERR_FILENO); - close(devnull); - } - execl("/bin/sh", "sh", "-c", cmd, (char*)NULL); - _exit(127); - } - /* parent: convert pid to string and return immediately */ - char pidbuf[32]; - snprintf(pidbuf, sizeof(pidbuf), "%d", (int)pid); - return el_wrap_str(el_strdup(pidbuf)); -#endif -} - -el_val_t fs_list(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - el_val_t lst = el_list_empty(); - if (!path) return lst; - DIR* d = opendir(path); - if (!d) return lst; - struct dirent* e; - while ((e = readdir(d)) != NULL) { - if (strcmp(e->d_name, ".") == 0 || strcmp(e->d_name, "..") == 0) continue; - lst = el_list_append(lst, el_wrap_str(el_strdup(e->d_name))); - } - closedir(d); - return lst; -} - -/* fs_list_json — return directory entries as a JSON array of strings. - * Returns "[]" for missing or non-directory paths. Excludes "." and "..". */ -el_val_t fs_list_json(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path) return EL_STR("[]"); - DIR* d = opendir(path); - if (!d) return EL_STR("[]"); - /* Collect entries first so we can build the JSON in one pass. */ - char** names = NULL; - size_t count = 0, cap = 0; - struct dirent* e; - while ((e = readdir(d)) != NULL) { - if (strcmp(e->d_name, ".") == 0 || strcmp(e->d_name, "..") == 0) continue; - if (count >= cap) { - cap = cap ? cap * 2 : 16; - names = realloc(names, cap * sizeof(char*)); - if (!names) { closedir(d); return EL_STR("[]"); } - } - names[count++] = strdup(e->d_name); - } - closedir(d); - /* Build JSON array. */ - size_t sz = 3; /* "[]" + NUL */ - for (size_t i = 0; i < count; i++) sz += strlen(names[i]) * 2 + 6; /* conservative */ - char* buf = malloc(sz); - if (!buf) { for (size_t i = 0; i < count; i++) free(names[i]); free(names); return EL_STR("[]"); } - size_t pos = 0; - buf[pos++] = '['; - for (size_t i = 0; i < count; i++) { - if (i > 0) buf[pos++] = ','; - buf[pos++] = '"'; - for (const char* p = names[i]; *p; p++) { - if (*p == '"' || *p == '\\') buf[pos++] = '\\'; - else if (*p == '\n') { buf[pos++] = '\\'; buf[pos++] = 'n'; continue; } - else if (*p == '\t') { buf[pos++] = '\\'; buf[pos++] = 't'; continue; } - buf[pos++] = *p; - } - buf[pos++] = '"'; - free(names[i]); - } - free(names); - buf[pos++] = ']'; - buf[pos] = '\0'; - return el_wrap_str(buf); -} - -/* fs_exists — true iff stat(path) succeeds. Symlinks are followed. */ -el_val_t fs_exists(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path || !*path) return 0; - struct stat st; - return (el_val_t)(stat(path, &st) == 0 ? 1 : 0); -} - -/* fs_mkdir — create directory at path with mode 0755, mkdir -p semantics. - * Returns 1 if path exists or was created (incl. all parents); 0 on failure. - * Walks the path component-by-component so missing intermediate dirs are - * also created. An existing leaf is not an error. */ -el_val_t fs_mkdir(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path || !*path) return 0; - size_t n = strlen(path); - char* buf = malloc(n + 1); - if (!buf) return 0; - memcpy(buf, path, n + 1); - /* Walk components; create each prefix in turn. */ - for (size_t i = 1; i <= n; i++) { - if (buf[i] == '/' || buf[i] == '\0') { - char saved = buf[i]; - buf[i] = '\0'; - if (buf[0] != '\0') { - if (mkdir(buf, 0755) != 0 && errno != EEXIST) { - /* Tolerate the case where this prefix exists as a non-dir - * only when stat says it's a directory. */ - struct stat st; - if (stat(buf, &st) != 0 || !S_ISDIR(st.st_mode)) { - free(buf); - return 0; - } - } - } - buf[i] = saved; - } - } - free(buf); - return 1; -} - -/* ── URL encoding ─────────────────────────────────────────────────────────── */ - -/* RFC 3986 percent-encoding for URL components (form bodies, query strings). - * Unreserved set: A-Z a-z 0-9 - _ . ~ — passed through verbatim. - * Everything else (including space) becomes %XX hex. */ -el_val_t url_encode(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - static const char hex[] = "0123456789ABCDEF"; - size_t n = strlen(s); - char* out = el_strbuf(n * 3); - size_t o = 0; - for (size_t i = 0; i < n; i++) { - unsigned char c = (unsigned char)s[i]; - if ((c >= 'A' && c <= 'Z') || - (c >= 'a' && c <= 'z') || - (c >= '0' && c <= '9') || - c == '-' || c == '_' || c == '.' || c == '~') { - out[o++] = (char)c; - } else { - out[o++] = '%'; - out[o++] = hex[(c >> 4) & 0xF]; - out[o++] = hex[c & 0xF]; - } - } - out[o] = '\0'; - return el_wrap_str(out); -} - -/* Decode percent-encoded URL component. '+' becomes space (form-encoded); - * malformed %-escapes are emitted verbatim. */ -el_val_t url_decode(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - size_t o = 0; - for (size_t i = 0; i < n; i++) { - char c = s[i]; - if (c == '+') { - out[o++] = ' '; - } else if (c == '%' && i + 2 < n) { - char h1 = s[i + 1], h2 = s[i + 2]; - int v1 = (h1 >= '0' && h1 <= '9') ? h1 - '0' - : (h1 >= 'a' && h1 <= 'f') ? h1 - 'a' + 10 - : (h1 >= 'A' && h1 <= 'F') ? h1 - 'A' + 10 : -1; - int v2 = (h2 >= '0' && h2 <= '9') ? h2 - '0' - : (h2 >= 'a' && h2 <= 'f') ? h2 - 'a' + 10 - : (h2 >= 'A' && h2 <= 'F') ? h2 - 'A' + 10 : -1; - if (v1 >= 0 && v2 >= 0) { - out[o++] = (char)((v1 << 4) | v2); - i += 2; - } else { - out[o++] = c; - } - } else { - out[o++] = c; - } - } - out[o] = '\0'; - return el_wrap_str(out); -} - -/* ── html_raw ──────────────────────────────────────────────────────────────── - * Identity passthrough for raw HTML template interpolation. - * El's {raw(expr)} compiles to html_raw(expr) — the value is output as-is - * without any escaping. The caller is responsible for safety. - */ -el_val_t html_raw(el_val_t s) { - return s; -} - -/* ── html_escape ───────────────────────────────────────────────────────────── - * Escape < > " ' & for safe HTML text interpolation. - * El's {expr} in HTML templates compiles to html_escape(expr). - */ -el_val_t html_escape(el_val_t sv) { - const char* src = EL_CSTR(sv); - if (!src) return EL_STR(""); - size_t len = strlen(src); - /* Worst case: every byte → 6 chars (") */ - char* out = (char*)malloc(len * 6 + 1); - if (!out) return sv; - el_arena_track(out); - char* p = out; - for (size_t i = 0; i < len; i++) { - unsigned char c = (unsigned char)src[i]; - switch (c) { - case '&': memcpy(p, "&", 5); p += 5; break; - case '<': memcpy(p, "<", 4); p += 4; break; - case '>': memcpy(p, ">", 4); p += 4; break; - case '"': memcpy(p, """, 6); p += 6; break; - case '\'': memcpy(p, "'", 5); p += 5; break; - default: *p++ = (char)c; break; - } - } - *p = '\0'; - return el_wrap_str(out); -} - -/* ── HTML allowlist sanitizer ──────────────────────────────────────────────── - * el_html_sanitize(input, allowlist_json) - * - * Strict allowlist HTML cleaner. Replaces the older denylist patterns - * (str_replace cascades that wrapped dangerous tags in HTML comments and - * renamed `on*` attributes). The denylist approach is fragile: comment- - * wrapping can be re-broken by a literal `-->` inside an attacker-supplied - * attribute value, and every new attack vector requires a code change. - * - * Design: - * - Single-pass byte-level state machine. - * - Tag and attribute names are matched case-insensitively against the - * allowlist. Unknown tags are dropped entirely (the open and close - * markers are stripped; their inner text content survives, escaped). - * - A small set of "dangerous container" tags (script, style, iframe, - * object, embed, form, plus a few rarer ones) drop themselves AND - * their full subtree — text between `` is - * CDATA-like and must not be re-emitted as escaped text either. - * - Comments (), doctype (), CDATA (), - * and processing instructions () are dropped entirely. - * - Text content outside dropped subtrees is HTML-escaped (&, <, >, ", '). - * - Attribute values are unquoted/dequoted, then re-emitted with double - * quotes around the cleanly-escaped value. - * - For `` and any `src` attribute, the URL scheme is validated: - * only http:, https:, mailto:, fragment-only `#anchor`, or relative - * paths are allowed. Anything else (javascript:, data:, vbscript:, - * about:, file:, etc.) drops the attribute. - * - Self-closing void tags (br, hr, img, etc.) emit without a close tag. - * - Malformed input (unclosed tag at EOF, bad attribute syntax) drops - * the pending tag and continues. Pre-encoded entities (<, &, - * etc.) are passed through verbatim — the browser will decode them - * safely on render. - * - * Allowlist format (JSON string): - * {"p":[],"a":["href","title"],"strong":[],...} - * - Key = lowercase tag name. - * - Value = JSON array of allowed attribute names (lowercase). - * - Empty array means tag allowed but no attributes survive. - * - * Output is a freshly-allocated arena-tracked el_val_t string. */ - -/* Internal byte buffer with realloc-doubling. Used during sanitization; - * the final result is copied into an arena-tracked el_strbuf so the caller - * sees standard runtime memory semantics. */ -typedef struct { - char* data; - size_t len; - size_t cap; -} html_buf_t; - -static void html_buf_init(html_buf_t* b) { - b->cap = 256; - b->data = malloc(b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->len = 0; -} - -static void html_buf_grow(html_buf_t* b, size_t need) { - if (b->len + need + 1 <= b->cap) return; - size_t nc = b->cap; - while (b->len + need + 1 > nc) nc *= 2; - char* nd = realloc(b->data, nc); - if (!nd) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->data = nd; - b->cap = nc; -} - -static void html_buf_putc(html_buf_t* b, char c) { - html_buf_grow(b, 1); - b->data[b->len++] = c; -} - -static void html_buf_puts(html_buf_t* b, const char* s) { - if (!s) return; - size_t n = strlen(s); - html_buf_grow(b, n); - memcpy(b->data + b->len, s, n); - b->len += n; -} - -static void html_buf_free(html_buf_t* b) { - free(b->data); - b->data = NULL; - b->len = b->cap = 0; -} - -/* ASCII tolower, locale-independent. */ -static int html_tolower(int c) { - return (c >= 'A' && c <= 'Z') ? c + 32 : c; -} - -/* Case-insensitive ASCII compare of [a, a+n) against c-string `s`. - * Returns 1 iff lengths match and bytes are equal under tolower. */ -static int html_ieq_n(const char* a, size_t n, const char* s) { - if (!a || !s) return 0; - if (strlen(s) != n) return 0; - for (size_t i = 0; i < n; i++) { - if (html_tolower((unsigned char)a[i]) != html_tolower((unsigned char)s[i])) return 0; - } - return 1; -} - -/* Case-insensitive ASCII compare of two byte slices. */ -static int html_iemem(const char* a, const char* b, size_t n) { - for (size_t i = 0; i < n; i++) { - if (html_tolower((unsigned char)a[i]) != html_tolower((unsigned char)b[i])) return 0; - } - return 1; -} - -/* Walk a JSON allowlist object and find the value (an array) for a given - * tag key, comparing case-insensitively. On hit returns a pointer to the - * opening `[` of the array and writes the byte length of the array span - * (including the brackets) to *out_len. On miss returns NULL. - * - * The parser is intentionally tiny: it does not handle escapes inside - * keys (allowlist authors do not need them), and it relies on balanced - * brackets/quotes within the value array. */ -static const char* html_allowlist_find(const char* allow, const char* tag, - size_t tag_len, size_t* out_len) { - if (!allow) return NULL; - const char* p = allow; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '{') return NULL; - p++; - while (*p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p == '}' || *p == 0) return NULL; - if (*p != '"') return NULL; - p++; - const char* k = p; - while (*p && *p != '"') p++; - if (*p != '"') return NULL; - size_t klen = (size_t)(p - k); - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != ':') return NULL; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return NULL; - const char* arr_start = p; - int depth = 0; - int in_str = 0; - while (*p) { - char c = *p; - if (in_str) { - if (c == '\\' && p[1]) { p += 2; continue; } - if (c == '"') in_str = 0; - } else { - if (c == '"') in_str = 1; - else if (c == '[') depth++; - else if (c == ']') { depth--; if (depth == 0) { p++; break; } } - } - p++; - } - size_t alen = (size_t)(p - arr_start); - int match = (klen == tag_len) && html_iemem(k, tag, klen); - if (match) { - if (out_len) *out_len = alen; - return arr_start; - } - } - return NULL; -} - -/* Returns 1 iff `attr` (length attr_len) appears as a string element - * in the JSON array slice [arr, arr+arr_len). Comparison is case- - * insensitive. */ -static int html_attr_in_array(const char* arr, size_t arr_len, - const char* attr, size_t attr_len) { - if (!arr || arr_len < 2) return 0; - const char* p = arr + 1; - const char* end = arr + arr_len - 1; - while (p < end) { - while (p < end && (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',')) p++; - if (p >= end) return 0; - if (*p != '"') return 0; - p++; - const char* s = p; - while (p < end && *p != '"') { - if (*p == '\\' && p + 1 < end) p++; - p++; - } - if (p >= end) return 0; - size_t slen = (size_t)(p - s); - p++; - if (slen == attr_len && html_iemem(s, attr, slen)) return 1; - } - return 0; -} - -/* Hard-coded set of tags whose content is ALSO dropped (entire subtree). */ -static int html_is_dangerous_container(const char* tag, size_t tag_len) { - static const char* names[] = { - "script", "style", "iframe", "object", "embed", "form", - "noscript", "noembed", "template", "svg", "math", "frame", - "frameset", "applet", "audio", "video", "source", "track", - NULL - }; - for (int i = 0; names[i]; i++) { - if (html_ieq_n(tag, tag_len, names[i])) return 1; - } - return 0; -} - -/* HTML void elements — emit without a close tag. */ -static int html_is_void(const char* tag, size_t tag_len) { - static const char* names[] = { - "area", "base", "br", "col", "embed", "hr", "img", "input", - "link", "meta", "param", "source", "track", "wbr", - NULL - }; - for (int i = 0; names[i]; i++) { - if (html_ieq_n(tag, tag_len, names[i])) return 1; - } - return 0; -} - -/* Append a single byte HTML-escaped into the output buffer. */ -static void html_escape_byte(html_buf_t* out, unsigned char c) { - switch (c) { - case '<': html_buf_puts(out, "<"); break; - case '>': html_buf_puts(out, ">"); break; - case '"': html_buf_puts(out, """); break; - case '\'': html_buf_puts(out, "'"); break; - default: html_buf_putc(out, (char)c); break; - } -} - -/* Validate a URL value against the allowlist of safe schemes for hrefs. - * Returns 1 iff the URL is safe to emit. Acceptable forms: - * - http:// or https:// (case-insensitive) - * - mailto: - * - fragment-only `#anchor` - * - relative path that does not contain a colon before the first - * slash/?/# (so `foo/bar`, `/foo`, `?x=1` are OK; `javascript:x` is - * not — its colon precedes any path/hash/query separator). - * - * URL leading whitespace and embedded ASCII control bytes (TAB, LF, CR) - * are stripped before the scheme test, mirroring how browsers normalise - * URLs (these bytes are otherwise a known XSS bypass: `java\tscript:`). */ -static int html_url_is_safe(const char* url, size_t len) { - if (!url || len == 0) return 1; /* empty href is harmless */ - size_t i = 0; - while (i < len) { - unsigned char c = (unsigned char)url[i]; - if (c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == 0x0B || c == 0x0C) { - i++; continue; - } - break; - } - if (i >= len) return 1; /* whitespace only */ - if (url[i] == '#') return 1; /* fragment only */ - if (url[i] == '/' || url[i] == '?') return 1; /* relative */ - /* Find the first scheme-terminating character. */ - size_t scheme_end = (size_t)-1; - for (size_t j = i; j < len; j++) { - char c = url[j]; - if (c == ':') { scheme_end = j; break; } - if (c == '/' || c == '?' || c == '#') break; - } - if (scheme_end == (size_t)-1) return 1; /* no colon → relative path */ - /* Lowercase the scheme, stripping embedded control bytes. */ - char scheme[32]; - size_t sl = 0; - for (size_t j = i; j < scheme_end && sl < sizeof(scheme) - 1; j++) { - unsigned char c = (unsigned char)url[j]; - if (c == '\t' || c == '\n' || c == '\r' || c == 0x0B || c == 0x0C) continue; - scheme[sl++] = (char)html_tolower(c); - } - scheme[sl] = '\0'; - if (strcmp(scheme, "http") == 0) return 1; - if (strcmp(scheme, "https") == 0) return 1; - if (strcmp(scheme, "mailto") == 0) return 1; - return 0; -} - -el_val_t el_html_sanitize(el_val_t input_v, el_val_t allowlist_v) { - const char* input = EL_CSTR(input_v); - const char* allow = EL_CSTR(allowlist_v); - if (!input) return el_wrap_str(el_strdup("")); - if (!allow) allow = "{}"; - size_t in_len = strlen(input); - - html_buf_t out; - html_buf_init(&out); - - size_t i = 0; - while (i < in_len) { - unsigned char c = (unsigned char)input[i]; - if (c != '<') { - /* Plain text — escape and emit. We pass `&` through verbatim - * to preserve pre-encoded entities (`<`, `&`, `&#x...;`) - * which the browser will decode safely. */ - if (c == '&') html_buf_putc(&out, '&'); - else html_escape_byte(&out, c); - i++; - continue; - } - /* `<` — try to parse a tag. */ - if (i + 1 >= in_len) { - html_buf_puts(&out, "<"); - i++; - continue; - } - /* Comments, doctype, CDATA, processing instructions — drop entirely. */ - if (input[i + 1] == '!') { - if (i + 3 < in_len && input[i + 2] == '-' && input[i + 3] == '-') { - size_t j = i + 4; - while (j + 2 < in_len && !(input[j] == '-' && input[j + 1] == '-' && input[j + 2] == '>')) j++; - if (j + 2 < in_len) i = j + 3; - else i = in_len; - continue; - } - size_t j = i + 2; - while (j < in_len && input[j] != '>') j++; - i = (j < in_len) ? j + 1 : in_len; - continue; - } - if (input[i + 1] == '?') { - size_t j = i + 2; - while (j < in_len && input[j] != '>') j++; - i = (j < in_len) ? j + 1 : in_len; - continue; - } - int is_close = 0; - size_t name_start = i + 1; - if (input[i + 1] == '/') { - is_close = 1; - name_start = i + 2; - } - if (name_start >= in_len) { - html_buf_puts(&out, "<"); - i++; - continue; - } - unsigned char nc = (unsigned char)input[name_start]; - if (!((nc >= 'a' && nc <= 'z') || (nc >= 'A' && nc <= 'Z'))) { - /* `<` followed by non-letter — emit as escaped text. */ - html_buf_puts(&out, "<"); - i++; - continue; - } - size_t name_end = name_start; - while (name_end < in_len) { - unsigned char x = (unsigned char)input[name_end]; - if ((x >= 'a' && x <= 'z') || (x >= 'A' && x <= 'Z') || - (x >= '0' && x <= '9') || x == '-' || x == '_' || x == ':') { - name_end++; - } else { - break; - } - } - const char* tag = input + name_start; - size_t tag_len = name_end - name_start; - /* Find the `>` that closes this tag, respecting quoted attrs. */ - size_t cur = name_end; - int self_close = 0; - while (cur < in_len) { - unsigned char x = (unsigned char)input[cur]; - if (x == '"' || x == '\'') { - unsigned char q = x; - cur++; - while (cur < in_len && (unsigned char)input[cur] != q) cur++; - if (cur < in_len) cur++; /* skip closing quote */ - continue; - } - if (x == '/' && cur + 1 < in_len && input[cur + 1] == '>') { - self_close = 1; - break; - } - if (x == '>') break; - cur++; - } - if (cur >= in_len) { - /* Malformed: unclosed tag at EOF. Drop the rest of the input. */ - i = in_len; - continue; - } - size_t tag_end = self_close ? cur + 2 : cur + 1; /* one past `>` */ - /* Dangerous container — drop the whole subtree. */ - if (!is_close && html_is_dangerous_container(tag, tag_len)) { - if (self_close || html_is_void(tag, tag_len)) { - i = tag_end; - continue; - } - size_t scan = tag_end; - int found_close = 0; - while (scan < in_len) { - if (input[scan] != '<') { scan++; continue; } - if (scan + 1 < in_len && input[scan + 1] == '/') { - size_t cn_start = scan + 2; - size_t cn_end = cn_start; - while (cn_end < in_len) { - unsigned char x = (unsigned char)input[cn_end]; - if ((x >= 'a' && x <= 'z') || (x >= 'A' && x <= 'Z') || - (x >= '0' && x <= '9') || x == '-' || x == '_' || x == ':') { - cn_end++; - } else break; - } - if (cn_end - cn_start == tag_len && - html_iemem(input + cn_start, tag, tag_len)) { - size_t end_close = cn_end; - while (end_close < in_len && input[end_close] != '>') end_close++; - i = (end_close < in_len) ? end_close + 1 : in_len; - found_close = 1; - break; - } - } - scan++; - } - if (!found_close) { - /* No matching close — drop everything from here on. */ - i = in_len; - } - continue; - } - /* Look up the tag in the allowlist. */ - size_t arr_len = 0; - const char* arr = html_allowlist_find(allow, tag, tag_len, &arr_len); - if (!arr) { - /* Tag not allowed. Drop the open/close marker; inner text is - * processed by the outer loop and re-emitted as escaped text. */ - i = tag_end; - continue; - } - if (is_close) { - if (!html_is_void(tag, tag_len)) { - html_buf_putc(&out, '<'); - html_buf_putc(&out, '/'); - for (size_t k = 0; k < tag_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)tag[k])); - } - html_buf_putc(&out, '>'); - } - i = tag_end; - continue; - } - /* Allowed open tag. Emit ``. */ - html_buf_putc(&out, '<'); - for (size_t k = 0; k < tag_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)tag[k])); - } - size_t a = name_end; - while (a < cur) { - unsigned char x = (unsigned char)input[a]; - if (x == ' ' || x == '\t' || x == '\n' || x == '\r' || x == '/') { a++; continue; } - size_t an_start = a; - while (a < cur) { - unsigned char y = (unsigned char)input[a]; - if (y == '=' || y == ' ' || y == '\t' || y == '\n' || y == '\r' || y == '/' || y == '>') break; - a++; - } - size_t an_len = a - an_start; - if (an_len == 0) { a++; continue; } - size_t av_start = 0; - size_t av_len = 0; - int has_value = 0; - size_t b = a; - while (b < cur && (input[b] == ' ' || input[b] == '\t' || input[b] == '\n' || input[b] == '\r')) b++; - if (b < cur && input[b] == '=') { - has_value = 1; - b++; - while (b < cur && (input[b] == ' ' || input[b] == '\t' || input[b] == '\n' || input[b] == '\r')) b++; - if (b < cur && (input[b] == '"' || input[b] == '\'')) { - unsigned char q = (unsigned char)input[b]; - b++; - av_start = b; - while (b < cur && (unsigned char)input[b] != q) b++; - av_len = b - av_start; - if (b < cur) b++; - } else { - av_start = b; - while (b < cur) { - unsigned char y = (unsigned char)input[b]; - if (y == ' ' || y == '\t' || y == '\n' || y == '\r' || y == '>') break; - b++; - } - av_len = b - av_start; - } - a = b; - } - if (!html_attr_in_array(arr, arr_len, input + an_start, an_len)) continue; - int is_href = (an_len == 4 && html_iemem(input + an_start, "href", 4)); - int is_src = (an_len == 3 && html_iemem(input + an_start, "src", 3)); - if ((is_href || is_src) && has_value) { - if (!html_url_is_safe(input + av_start, av_len)) continue; - } - html_buf_putc(&out, ' '); - for (size_t k = 0; k < an_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)input[an_start + k])); - } - if (has_value) { - html_buf_puts(&out, "=\""); - for (size_t k = 0; k < av_len; k++) { - unsigned char y = (unsigned char)input[av_start + k]; - /* Re-escape so the emitted attribute is well-formed - * double-quoted HTML. `&` passes through to preserve - * pre-encoded entities. */ - if (y == '"') html_buf_puts(&out, """); - else if (y == '<') html_buf_puts(&out, "<"); - else if (y == '>') html_buf_puts(&out, ">"); - else html_buf_putc(&out, (char)y); - } - html_buf_putc(&out, '"'); - } - } - html_buf_putc(&out, '>'); - i = tag_end; - } - /* Copy into arena-tracked buffer so the standard runtime memory model - * applies to the returned string. */ - char* result = el_strbuf(out.len); - memcpy(result, out.data, out.len); - result[out.len] = '\0'; - html_buf_free(&out); - return el_wrap_str(result); -} - -/* ── JSON ────────────────────────────────────────────────────────────────── */ - -/* True iff the segment is non-empty and every byte is an ASCII digit. We treat - * such segments as numeric array indices when walking a dot-path; mixed names - * like "0a" remain object-key lookups, so a key named "0" still wins over an - * index when the surrounding container is an object. */ -static int json_path_seg_is_index(const char* seg, size_t n) { - if (n == 0) return 0; - for (size_t i = 0; i < n; i++) { - if (seg[i] < '0' || seg[i] > '9') return 0; - } - return 1; -} - -/* Skip JSON whitespace. */ -static const char* json_skip_ws(const char* p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - return p; -} - -/* Descend one segment into the JSON cursor `p`. - * - If `p` points at an array `[...]` and the segment is all digits, - * advance to that element (zero-based). - * - Otherwise treat the segment as an object key and use json_find_key - * scoped to a one-level slice of the current container. - * Returns NULL if the descent fails (segment not found, container mismatch). - * - * `seg` is a pointer into the original path string and `seg_len` is its - * byte length — this avoids an extra alloc per segment. */ -static const char* json_path_descend(const char* p, const char* seg, size_t seg_len) { - if (!p || !seg) return NULL; - p = json_skip_ws(p); - if (*p == '[' && json_path_seg_is_index(seg, seg_len)) { - long idx = 0; - for (size_t i = 0; i < seg_len; i++) idx = idx * 10 + (seg[i] - '0'); - p++; /* step past '[' */ - p = json_skip_ws(p); - long cur = 0; - while (*p && *p != ']') { - if (cur == idx) return p; - const char* end = json_skip_value(p); - if (!end || end == p) return NULL; - p = json_skip_ws(end); - if (*p == ',') { p++; p = json_skip_ws(p); cur++; continue; } - /* No comma after this element — only acceptable at the closing ']', - * which means we ran out of elements. */ - break; - } - return NULL; - } - /* Object lookup. json_find_key walks at depth 1 of whatever container it - * receives, so we slice from `p` onwards. Caller already positioned us at - * the opening '{' (or at whitespace before it). */ - if (*p != '{') return NULL; - /* Build a NUL-terminated copy of the key segment for the lookup. We only - * pay this cost when the segment isn't a numeric index. */ - char stack_key[256]; - char* k = stack_key; - if (seg_len + 1 > sizeof(stack_key)) { - k = malloc(seg_len + 1); - if (!k) return NULL; - } - memcpy(k, seg, seg_len); - k[seg_len] = '\0'; - const char* found = json_find_key(p, k); - if (k != stack_key) free(k); - return found; -} - -/* Read the JSON value at `p` into a freshly-allocated, arena-owned el_val_t. - * - String -> unescaped, wrapped el_val_t string - * - Anything else -> raw JSON slice as a string (matches the historical - * json_get behaviour: numbers/bools/null come back stringified). */ -static el_val_t json_read_value(const char* p) { - p = json_skip_ws(p); - if (*p == '"') { - p++; - size_t cap = strlen(p) + 1; - char* out = el_strbuf(cap); - char* w = out; - while (*p && *p != '"') { - if (*p == '\\' && *(p+1)) { - p++; - switch (*p) { - case '"': *w++ = '"'; break; - case '\\': *w++ = '\\'; break; - case '/': *w++ = '/'; break; - case 'n': *w++ = '\n'; break; - case 'r': *w++ = '\r'; break; - case 't': *w++ = '\t'; break; - default: *w++ = *p; break; - } - } else { - *w++ = *p; - } - p++; - } - *w = '\0'; - return el_wrap_str(out); - } - /* Object/array/number/bool/null — return the raw slice up to the value's - * end. json_skip_value tracks brace/bracket/string state so nested objects - * round-trip cleanly. */ - const char* end = json_skip_value(p); - if (!end) end = p; - size_t n = (size_t)(end - p); - /* Strip trailing whitespace from scalar values so callers don't see - * `123 ` when they parsed a pretty-printed number. */ - while (n > 0 && (p[n-1] == ' ' || p[n-1] == '\t' || p[n-1] == '\n' || p[n-1] == '\r')) { - n--; - } - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t json_get(el_val_t jsonv, el_val_t keyv) { - const char* json = EL_CSTR(jsonv); - const char* key = EL_CSTR(keyv); - if (!json || !key) return el_wrap_str(el_strdup("")); - - /* Fast path: key contains no '.' — keep the historical single-segment - * substring search so existing callers retain their O(strlen) cost - * profile. The dot-path walker is only paid for when needed. */ - if (!strchr(key, '.')) { - size_t klen = strlen(key); - char stack_pat[512]; - char* pattern; - if (klen + 5 <= sizeof(stack_pat)) { - pattern = stack_pat; - } else { - pattern = malloc(klen + 5); - if (!pattern) return el_wrap_str(el_strdup("")); - } - snprintf(pattern, klen + 5, "\"%s\":", key); - const char* p = strstr(json, pattern); - if (pattern != stack_pat) free(pattern); - if (!p) return el_wrap_str(el_strdup("")); - p += strlen(key) + 3; /* skip "key": */ - return json_read_value(p); - } - - /* Dot-path traversal. Walk segments left to right; at each step, descend - * into the current container by either array index (all-digit segment on - * an array cursor) or object key. */ - const char* cursor = json_skip_ws(json); - const char* seg_start = key; - const char* k = key; - while (1) { - if (*k == '.' || *k == '\0') { - size_t seg_len = (size_t)(k - seg_start); - cursor = json_path_descend(cursor, seg_start, seg_len); - if (!cursor) return el_wrap_str(el_strdup("")); - if (*k == '\0') break; - k++; - seg_start = k; - continue; - } - k++; - } - return json_read_value(cursor); -} - -/* ── Float bit-cast helpers ──────────────────────────────────────────────── */ -/* `el_to_float` and `el_from_float` are exposed in el_runtime.h as static - * inlines so generated programs (which #include the header) can call them - * for Float literals. No definitions are needed here. */ - -/* ── JSON parser (recursive descent) ─────────────────────────────────────── */ -/* - * Parsed JSON representation: - * - object -> ElMap (keys & values are el_val_t) - * - array -> ElList - * - string -> EL_STR-wrapped char* (allocated) - * - number -> int (el_val_t) if integer, otherwise el_from_float(double) - * - true -> 1 - * - false -> 0 - * - null -> EL_NULL (0) - * - * Note: there is no runtime type tag — parsed numbers cannot be - * distinguished from booleans by the runtime alone. The codegen tracks - * types separately. This matches the rest of el_val_t's type-erased model. - */ - -/* JsonParser struct is forward-declared near the HTTP/Engram section. */ - -static void jp_skip_ws(JsonParser* jp) { - while (jp->p < jp->end) { - char c = *jp->p; - if (c == ' ' || c == '\t' || c == '\n' || c == '\r') jp->p++; - else break; - } -} - -static el_val_t jp_parse_value(JsonParser* jp); - -/* Parse a JSON string literal (the opening " has NOT yet been consumed). */ -static char* jp_parse_string_raw(JsonParser* jp) { - if (jp->p >= jp->end || *jp->p != '"') { jp->err = 1; return el_strdup(""); } - jp->p++; - size_t cap = 32, len = 0; - char* out = malloc(cap); - if (!out) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - while (jp->p < jp->end && *jp->p != '"') { - char c = *jp->p++; - if (c == '\\' && jp->p < jp->end) { - char esc = *jp->p++; - switch (esc) { - case '"': c = '"'; break; - case '\\': c = '\\'; break; - case '/': c = '/'; break; - case 'b': c = '\b'; break; - case 'f': c = '\f'; break; - case 'n': c = '\n'; break; - case 'r': c = '\r'; break; - case 't': c = '\t'; break; - case 'u': { - /* Skip 4 hex digits; emit '?' as a placeholder */ - for (int i = 0; i < 4 && jp->p < jp->end; i++) jp->p++; - c = '?'; - break; - } - default: c = esc; break; - } - } - if (len + 1 >= cap) { - cap *= 2; - out = realloc(out, cap); - if (!out) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } - out[len++] = c; - } - if (jp->p < jp->end && *jp->p == '"') jp->p++; - else jp->err = 1; - out[len] = '\0'; - return out; -} - -static el_val_t jp_parse_number(JsonParser* jp) { - const char* start = jp->p; - int is_float = 0; - if (jp->p < jp->end && (*jp->p == '-' || *jp->p == '+')) jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - if (jp->p < jp->end && *jp->p == '.') { - is_float = 1; jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - } - if (jp->p < jp->end && (*jp->p == 'e' || *jp->p == 'E')) { - is_float = 1; jp->p++; - if (jp->p < jp->end && (*jp->p == '+' || *jp->p == '-')) jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - } - size_t n = (size_t)(jp->p - start); - char buf[64]; - if (n >= sizeof(buf)) n = sizeof(buf) - 1; - memcpy(buf, start, n); - buf[n] = '\0'; - if (is_float) return el_from_float(strtod(buf, NULL)); - return (el_val_t)strtoll(buf, NULL, 10); -} - -static el_val_t jp_parse_array(JsonParser* jp) { - if (jp->p < jp->end && *jp->p == '[') jp->p++; - el_val_t lst = el_list_empty(); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ']') { jp->p++; return lst; } - while (jp->p < jp->end) { - jp_skip_ws(jp); - el_val_t v = jp_parse_value(jp); - lst = el_list_append(lst, v); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ',') { jp->p++; continue; } - if (jp->p < jp->end && *jp->p == ']') { jp->p++; break; } - jp->err = 1; - break; - } - return lst; -} - -static el_val_t jp_parse_object(JsonParser* jp) { - if (jp->p < jp->end && *jp->p == '{') jp->p++; - el_val_t m = el_map_new(0); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == '}') { jp->p++; return m; } - while (jp->p < jp->end) { - jp_skip_ws(jp); - char* key = jp_parse_string_raw(jp); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ':') jp->p++; - else { jp->err = 1; free(key); break; } - jp_skip_ws(jp); - el_val_t v = jp_parse_value(jp); - m = el_map_set(m, EL_STR(key), v); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ',') { jp->p++; continue; } - if (jp->p < jp->end && *jp->p == '}') { jp->p++; break; } - jp->err = 1; - break; - } - return m; -} - -static el_val_t jp_parse_value(JsonParser* jp) { - jp_skip_ws(jp); - if (jp->p >= jp->end) { jp->err = 1; return EL_NULL; } - char c = *jp->p; - if (c == '"') return el_wrap_str(jp_parse_string_raw(jp)); - if (c == '{') return jp_parse_object(jp); - if (c == '[') return jp_parse_array(jp); - if (c == '-' || isdigit((unsigned char)c)) return jp_parse_number(jp); - if (c == 't' && jp->p + 4 <= jp->end && strncmp(jp->p, "true", 4) == 0) { jp->p += 4; return 1; } - if (c == 'f' && jp->p + 5 <= jp->end && strncmp(jp->p, "false", 5) == 0) { jp->p += 5; return 0; } - if (c == 'n' && jp->p + 4 <= jp->end && strncmp(jp->p, "null", 4) == 0) { jp->p += 4; return EL_NULL; } - jp->err = 1; - return EL_NULL; -} - -el_val_t json_parse(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return EL_NULL; - JsonParser jp = { .p = s, .end = s + strlen(s), .err = 0 }; - el_val_t v = jp_parse_value(&jp); - if (jp.err) return EL_NULL; - return v; -} - -/* ── JSON stringify ──────────────────────────────────────────────────────── */ -/* - * Stringify policy: el_val_t is type-erased, so we cannot perfectly - * round-trip arbitrary values. We use these heuristics: - * - If value is an ElList pointer (in the heap range), serialize as array. - * - If value is an ElMap pointer, serialize as object. - * - If value looks like a printable string pointer, serialize as string. - * - Otherwise serialize as integer. - * This is best-effort. Programs that need exact control should build the - * string directly. A pointer test is the cheapest way to disambiguate - * from small integers without a separate type tag. - */ - -/* JsonBuf struct is forward-declared near the HTTP section so HTTP helpers - * can use it. Its definition appears there. */ - -static void jb_init(JsonBuf* b) { - b->cap = 64; b->len = 0; - b->buf = malloc(b->cap); - if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->buf[0] = '\0'; -} - -static void jb_reserve(JsonBuf* b, size_t add) { - if (b->len + add + 1 > b->cap) { - while (b->len + add + 1 > b->cap) b->cap *= 2; - b->buf = realloc(b->buf, b->cap); - if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } -} - -static void jb_putc(JsonBuf* b, char c) { - jb_reserve(b, 1); - b->buf[b->len++] = c; - b->buf[b->len] = '\0'; -} - -static void jb_puts(JsonBuf* b, const char* s) { - size_t n = strlen(s); - jb_reserve(b, n); - memcpy(b->buf + b->len, s, n); - b->len += n; - b->buf[b->len] = '\0'; -} - -/* jb_finish — call exactly once, at the point a JsonBuf's buffer is handed - * off as an el_val_t return value (via el_wrap_str). JsonBuf allocates through - * raw malloc/realloc (jb_init/jb_reserve), NOT el_strdup/el_strbuf, so unlike - * those helpers it was never registered with the per-request arena — every - * JsonBuf-built JSON string leaked unconditionally, whether or not a request - * arena was active, because nothing ever freed it. el_arena_track() is a - * documented no-op when no arena is active, so this is safe to call from any - * context (HTTP request handler, CLI arena-scope, or no arena at all — the - * last case still leaks, same as before, until the caller's context is itself - * arena-scoped). Do NOT call this on a JsonBuf whose ->buf is consumed by - * another builder (jb_puts(&other, x.buf)) and freed manually — only on the - * one actually returned via el_wrap_str. */ -static char* jb_finish(JsonBuf* b) { - el_arena_track(b->buf); - return b->buf; -} - -static void jb_emit_escaped(JsonBuf* b, const char* s) { - jb_putc(b, '"'); - const unsigned char* p = (const unsigned char*)s; - while (*p) { - unsigned char c = *p; - switch (c) { - case '"': jb_puts(b, "\\\""); p++; break; - case '\\': jb_puts(b, "\\\\"); p++; break; - case '\b': jb_puts(b, "\\b"); p++; break; - case '\f': jb_puts(b, "\\f"); p++; break; - case '\n': jb_puts(b, "\\n"); p++; break; - case '\r': jb_puts(b, "\\r"); p++; break; - case '\t': jb_puts(b, "\\t"); p++; break; - default: - if (c < 0x20) { - char tmp[8]; - snprintf(tmp, sizeof(tmp), "\\u%04x", c); - jb_puts(b, tmp); - p++; - } else if (c < 0x80) { - jb_putc(b, (char)c); - p++; - } else { - /* Multi-byte UTF-8: validate sequence, pass through if valid, - * escape as \u00xx if the start byte is invalid/orphaned. */ - int seq_len = 0; - if ((c & 0xE0) == 0xC0) seq_len = 2; - else if ((c & 0xF0) == 0xE0) seq_len = 3; - else if ((c & 0xF8) == 0xF0) seq_len = 4; - if (seq_len >= 2) { - int valid = 1; - for (int i = 1; i < seq_len; i++) { - if ((p[i] & 0xC0) != 0x80) { valid = 0; break; } - } - if (valid) { - for (int i = 0; i < seq_len; i++) jb_putc(b, (char)p[i]); - p += seq_len; - break; - } - } - /* Invalid start byte or truncated sequence — escape it */ - char tmp[8]; - snprintf(tmp, sizeof(tmp), "\\u%04x", c); - jb_puts(b, tmp); - p++; - } - break; - } - } - jb_putc(b, '"'); -} - -/* Heuristic: is this el_val_t likely a pointer to an ElList? - * We can't fully verify, but pointers are large addresses, integers small. - * Treat values whose magnitude exceeds 2^32 as potential pointers and - * sniff by reading the header conservatively. - * - * Simpler heuristic: if the value reads as a printable string, treat as - * string; otherwise as integer. Lists/Maps are encoded as struct pointers, - * which have leading binary bytes — so they won't look like strings. */ - -static int looks_like_string(el_val_t v) { - if (v == 0) return 0; - /* Treat plausible heap addresses as candidates. - * Threshold: 4 GiB (0x100000000). On 64-bit systems heap addresses from - * malloc/mmap start well above 4 GiB (ASLR pushes them to ~0x7f...). - * El integer values (counters, unix timestamps up to ~2106) all fit below - * 0x100000000 (4294967296). The old threshold of 1,000,000 caused unix - * timestamps (~1.7e9) to be misidentified as string pointers — a segfault - * risk in json_stringify and jb_emit_value. */ - uintptr_t p = (uintptr_t)v; - if (p < 0x100000000ULL) return 0; /* integers, timestamps, counters */ - if (p < 0x1000) return 0; - /* Sniff first bytes for printable */ - const unsigned char* s = (const unsigned char*)p; - for (int i = 0; i < 16; i++) { - unsigned char c = s[i]; - if (c == '\0') return 1; /* terminated string (empty string is still a valid string) */ - /* Reject C0 control chars (non-whitespace), allow UTF-8 high bytes. - * 0x09-0x0d = tab/newline/cr/vt/ff (whitespace, OK) - * 0x20-0x7e = printable ASCII (OK) - * 0x7f = DEL (reject) - * 0x80-0xff = UTF-8 continuation/lead bytes (OK for multi-byte chars) */ - if (c < 0x09 || (c > 0x0d && c < 0x20) || c == 0x7f) return 0; - } - return 1; /* 16+ printable bytes — call it a string */ -} - -static void jb_emit_value(JsonBuf* b, el_val_t v); - -static void jb_emit_int(JsonBuf* b, int64_t n) { - char tmp[32]; - snprintf(tmp, sizeof(tmp), "%lld", (long long)n); - jb_puts(b, tmp); -} - -static void jb_emit_value(JsonBuf* b, el_val_t v) { - if (v == EL_NULL) { jb_puts(b, "null"); return; } - if (looks_like_string(v)) { - jb_emit_escaped(b, EL_CSTR(v)); - return; - } - jb_emit_int(b, (int64_t)v); -} - -el_val_t json_stringify(el_val_t v) { - JsonBuf b; jb_init(&b); - jb_emit_value(&b, v); - return el_wrap_str(jb_finish(&b)); -} - -/* ── JSON substring accessors ────────────────────────────────────────────── */ -/* - * These walk the raw JSON string looking for "key": at the top level (depth 1) - * of an object. They handle escaped quotes, nested objects/arrays, and - * whitespace around the colon. - */ - -/* Find "key": at object-depth == 1 inside the JSON object string `s`. - * Returns pointer to the first byte of the value, or NULL. */ -static const char* json_find_key(const char* s, const char* key) { - if (!s || !key) return NULL; - size_t klen = strlen(key); - int depth = 0; - int in_str = 0; - int escape = 0; - const char* p = s; - while (*p) { - char c = *p; - if (in_str) { - if (escape) { escape = 0; } - else if (c == '\\') { escape = 1; } - else if (c == '"') { - /* End of string. If we're at depth 1, check if this was a key. */ - p++; - if (depth == 1) { - /* The string just ended at p-1. Check if it matches key - * and is followed by a colon. We need to backtrack to find - * the start of this string and compare. */ - } - in_str = 0; - continue; - } - p++; - continue; - } - if (c == '"') { - /* Start of a string literal */ - const char* str_start = p + 1; - const char* q = str_start; - int e = 0; - while (*q) { - if (e) { e = 0; q++; continue; } - if (*q == '\\') { e = 1; q++; continue; } - if (*q == '"') break; - q++; - } - size_t slen = (size_t)(q - str_start); - const char* after = (*q == '"') ? q + 1 : q; - /* If at depth 1 and matches key and followed by ':' -> got it */ - if (depth == 1 && slen == klen && strncmp(str_start, key, klen) == 0) { - const char* r = after; - while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') r++; - if (*r == ':') { - r++; - while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') r++; - return r; - } - } - p = after; - continue; - } - if (c == '{' || c == '[') depth++; - else if (c == '}' || c == ']') depth--; - p++; - } - return NULL; -} - -/* Skip a JSON value starting at p; return pointer past the value end. */ -static const char* json_skip_value(const char* p) { - if (!p || !*p) return p; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p == '"') { - p++; - int e = 0; - while (*p) { - if (e) { e = 0; p++; continue; } - if (*p == '\\') { e = 1; p++; continue; } - if (*p == '"') { p++; break; } - p++; - } - return p; - } - if (*p == '{' || *p == '[') { - char open = *p; - char close = (open == '{') ? '}' : ']'; - int depth = 0; - int in_str = 0; - int e = 0; - while (*p) { - char c = *p; - if (in_str) { - if (e) { e = 0; } - else if (c == '\\') { e = 1; } - else if (c == '"') in_str = 0; - p++; - continue; - } - if (c == '"') { in_str = 1; p++; continue; } - if (c == open) depth++; - else if (c == close) { depth--; p++; if (depth == 0) return p; continue; } - p++; - } - return p; - } - /* scalar: number, true/false/null */ - while (*p && *p != ',' && *p != '}' && *p != ']' && - *p != ' ' && *p != '\t' && *p != '\n' && *p != '\r') p++; - return p; -} - -el_val_t json_get_string(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p || *p != '"') return el_wrap_str(el_strdup("")); - p++; - JsonParser jp = { .p = p - 1, .end = json + (json ? strlen(json) : 0), .err = 0 }; - char* parsed = jp_parse_string_raw(&jp); - if (jp.err) { free(parsed); return el_wrap_str(el_strdup("")); } - return el_wrap_str(parsed); -} - -el_val_t json_get_int(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (*p == '"' || *p == '{' || *p == '[') return 0; - return (el_val_t)strtoll(p, NULL, 10); -} - -el_val_t json_get_float(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (*p == '"' || *p == '{' || *p == '[') return 0; - return el_from_float(strtod(p, NULL)); -} - -el_val_t json_get_bool(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (strncmp(p, "true", 4) == 0) return 1; - return 0; -} - -el_val_t json_get_raw(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - /* Clear fs_read binary-length hint — result is a fresh null-terminated - * string, not the raw file bytes, so Content-Length must use strlen. - * (Kept although the pointer pairing now makes this redundant.) */ - _tl_fs_read_len = 0; - _tl_fs_read_buf = NULL; - if (!p) return el_wrap_str(el_strdup("")); - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - /* raw_val is the JSON value as-is (already encoded by the caller). - * If it looks like a plain (non-JSON) string, wrap it as a JSON string. - * Convention: callers pass pre-encoded values like "\"bob\"" for strings, - * "42" for numbers, "true"/"false" for booleans. */ - const char* raw_val = EL_CSTR(value); - if (!k) k = ""; - if (!raw_val) raw_val = "null"; - if (!json || !*json) { - /* Build a fresh object */ - JsonBuf b; jb_init(&b); - jb_putc(&b, '{'); - jb_emit_escaped(&b, k); - jb_putc(&b, ':'); - jb_puts(&b, raw_val); - jb_putc(&b, '}'); - return el_wrap_str(jb_finish(&b)); - } - const char* existing = json_find_key(json, k); - JsonBuf b; jb_init(&b); - if (existing) { - const char* end = json_skip_value(existing); - /* Copy [json .. existing) */ - size_t prefix = (size_t)(existing - json); - jb_reserve(&b, prefix); - memcpy(b.buf + b.len, json, prefix); - b.len += prefix; - b.buf[b.len] = '\0'; - jb_puts(&b, raw_val); - jb_puts(&b, end); - return el_wrap_str(jb_finish(&b)); - } - /* Insert before closing '}'. Find last '}' */ - size_t jl = strlen(json); - if (jl == 0) { free(b.buf); return el_wrap_str(el_strdup("{}")); } - /* Find last '}' from the end */ - ssize_t close_idx = -1; - for (ssize_t i = (ssize_t)jl - 1; i >= 0; i--) { - if (json[i] == '}') { close_idx = i; break; } - } - if (close_idx < 0) { - free(b.buf); - return el_wrap_str(el_strdup(json)); - } - /* Determine if object is empty: scan between last '{' and '}' for non-ws */ - int empty = 1; - for (ssize_t i = close_idx - 1; i >= 0; i--) { - char c = json[i]; - if (c == '{') break; - if (c != ' ' && c != '\t' && c != '\n' && c != '\r') { empty = 0; break; } - } - /* Copy json[0..close_idx) */ - jb_reserve(&b, (size_t)close_idx); - memcpy(b.buf + b.len, json, (size_t)close_idx); - b.len += (size_t)close_idx; - b.buf[b.len] = '\0'; - if (!empty) jb_putc(&b, ','); - jb_emit_escaped(&b, k); - jb_putc(&b, ':'); - jb_puts(&b, raw_val); - /* Append from close_idx onward */ - jb_puts(&b, json + close_idx); - return el_wrap_str(jb_finish(&b)); -} - -el_val_t json_array_len(el_val_t json_str) { - const char* s = EL_CSTR(json_str); - if (!s) return 0; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s != '[') return 0; - s++; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ']') return 0; - int64_t count = 0; - while (*s) { - const char* end = json_skip_value(s); - if (end == s) break; - count++; - s = end; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ',') { s++; continue; } - if (*s == ']' || *s == '\0') break; - } - return (el_val_t)count; -} - -/* json_array_get — return the i-th element of a JSON array as a JSON - * fragment string. Nested objects and arrays are returned verbatim - * (json_skip_value tracks brace/bracket depth so nested structures are - * preserved intact). Out-of-range index → "". */ -el_val_t json_array_get(el_val_t json_str, el_val_t index) { - const char* s = EL_CSTR(json_str); - int64_t idx = (int64_t)index; - if (!s || idx < 0) return el_wrap_str(el_strdup("")); - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s != '[') return el_wrap_str(el_strdup("")); - s++; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ']') return el_wrap_str(el_strdup("")); - int64_t i = 0; - while (*s) { - const char* start = s; - const char* end = json_skip_value(s); - if (end == s) break; - if (i == idx) { - size_t n = (size_t)(end - start); - char* out = el_strbuf(n); - memcpy(out, start, n); - out[n] = '\0'; - return el_wrap_str(out); - } - i++; - s = end; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ',') { s++; while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; continue; } - if (*s == ']' || *s == '\0') break; - } - return el_wrap_str(el_strdup("")); -} - -/* json_array_get_string — same as json_array_get, but assume the element - * is a JSON string and return the unquoted/unescaped value. Non-string - * elements yield "". */ -el_val_t json_array_get_string(el_val_t json_str, el_val_t index) { - el_val_t raw = json_array_get(json_str, index); - const char* s = EL_CSTR(raw); - if (!s || *s != '"') return el_wrap_str(el_strdup("")); - JsonParser jp = { - .p = s, - .end = s + strlen(s), - .err = 0, - }; - char* parsed = jp_parse_string_raw(&jp); - if (jp.err) { - free(parsed); - return el_wrap_str(el_strdup("")); - } - return el_wrap_str(parsed); -} - -/* json_escape_string — escape a string value for embedding in JSON. - * Returns the escaped content WITHOUT surrounding quotes. - * "say \"hello\"" -> "say \\\"hello\\\"" */ -el_val_t json_escape_string(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - /* Worst case: every char needs a 2-char escape. */ - char* out = malloc(n * 2 + 1); - if (!out) return el_wrap_str(el_strdup("")); - size_t j = 0; - for (size_t i = 0; i < n; i++) { - unsigned char c = (unsigned char)s[i]; - if (c == '"') { out[j++] = '\\'; out[j++] = '"'; } - else if (c == '\\') { out[j++] = '\\'; out[j++] = '\\'; } - else if (c == '\n') { out[j++] = '\\'; out[j++] = 'n'; } - else if (c == '\r') { out[j++] = '\\'; out[j++] = 'r'; } - else if (c == '\t') { out[j++] = '\\'; out[j++] = 't'; } - else { out[j++] = (char)c; } - } - out[j] = '\0'; - el_val_t result = el_wrap_str(el_strdup(out)); - free(out); - return result; -} - -/* json_build_object — build a JSON object from a flat key-value list. - * kvs is [key0, val0, key1, val1, ...]. Values are raw JSON (pass - * strings as "\"value\"" or use json_escape_string). */ -el_val_t json_build_object(el_val_t kvs) { - el_val_t list = kvs; - int64_t n = el_list_len(list); - JsonBuf b; jb_init(&b); - jb_putc(&b, '{'); - int first = 1; - for (int64_t i = 0; i + 1 < n; i += 2) { - el_val_t k = el_list_get(list, (el_val_t)i); - el_val_t v = el_list_get(list, (el_val_t)(i + 1)); - const char* ks = EL_CSTR(k); - const char* vs = EL_CSTR(v); - if (!ks || !vs) continue; - if (!first) jb_putc(&b, ','); - first = 0; - jb_putc(&b, '"'); - jb_puts(&b, ks); - jb_puts(&b, "\":\""); - /* escape the value string */ - size_t vn = strlen(vs); - for (size_t j = 0; j < vn; j++) { - unsigned char c = (unsigned char)vs[j]; - if (c == '"') { jb_putc(&b, '\\'); jb_putc(&b, '"'); } - else if (c == '\\') { jb_putc(&b, '\\'); jb_putc(&b, '\\'); } - else if (c == '\n') { jb_putc(&b, '\\'); jb_putc(&b, 'n'); } - else if (c == '\r') { jb_putc(&b, '\\'); jb_putc(&b, 'r'); } - else if (c == '\t') { jb_putc(&b, '\\'); jb_putc(&b, 't'); } - else { jb_putc(&b, (char)c); } - } - jb_putc(&b, '"'); - } - jb_putc(&b, '}'); - return el_wrap_str(jb_finish(&b)); -} - -/* json_build_array — build a JSON array from a list of raw JSON values. - * items is ["\"alpha\"", "\"beta\"", "42", "true", ...]. */ -el_val_t json_build_array(el_val_t items) { - el_val_t list = items; - int64_t n = el_list_len(list); - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - for (int64_t i = 0; i < n; i++) { - el_val_t v = el_list_get(list, (el_val_t)i); - const char* vs = EL_CSTR(v); - if (!vs) continue; - if (i > 0) jb_putc(&b, ','); - jb_puts(&b, vs); - } - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -/* ── Time ────────────────────────────────────────────────────────────────── */ - -el_val_t time_now(void) { - struct timeval tv; - gettimeofday(&tv, NULL); - int64_t ms = (int64_t)tv.tv_sec * 1000LL + (int64_t)tv.tv_usec / 1000LL; - return (el_val_t)ms; -} - -el_val_t time_now_utc(void) { - return time_now(); -} - -el_val_t time_format(el_val_t ts, el_val_t fmt) { - int64_t ms = (int64_t)ts; - time_t s = (time_t)(ms / 1000); - int msec = (int)(ms % 1000); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - const char* fmt_str = EL_CSTR(fmt); - if (!fmt_str || *fmt_str == '\0' || strcmp(fmt_str, "ISO") == 0) { - char buf[64]; - snprintf(buf, sizeof(buf), "%04d-%02d-%02dT%02d:%02d:%02d.%03dZ", - tm.tm_year + 1900, tm.tm_mon + 1, tm.tm_mday, - tm.tm_hour, tm.tm_min, tm.tm_sec, msec); - return el_wrap_str(el_strdup(buf)); - } - char buf[256]; - if (strftime(buf, sizeof(buf), fmt_str, &tm) == 0) buf[0] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -el_val_t time_to_parts(el_val_t ts) { - int64_t ms = (int64_t)ts; - time_t s = (time_t)(ms / 1000); - int msec = (int)(ms % 1000); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - /* Return a JSON string so callers can use json_get to extract fields. */ - char buf[256]; - snprintf(buf, sizeof(buf), - "{\"year\":%d,\"month\":%d,\"day\":%d,\"hour\":%d,\"minute\":%d,\"second\":%d,\"ms\":%d}", - tm.tm_year + 1900, tm.tm_mon + 1, tm.tm_mday, - tm.tm_hour, tm.tm_min, tm.tm_sec, msec); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz) { - (void)tz; - int64_t s = (int64_t)secs; - int64_t n = (int64_t)ns; - int64_t ms = s * 1000LL + n / 1000000LL; - return (el_val_t)ms; -} - -el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit) { - const char* u = EL_CSTR(unit); - int64_t cur = (int64_t)ts; - int64_t d = (int64_t)n; - int64_t add_ms = d; - if (u) { - if (strcmp(u, "ms") == 0) add_ms = d; - else if (strcmp(u, "sec") == 0) add_ms = d * 1000LL; - else if (strcmp(u, "min") == 0) add_ms = d * 60000LL; - else if (strcmp(u, "hour") == 0) add_ms = d * 3600000LL; - else if (strcmp(u, "day") == 0) add_ms = d * 86400000LL; - } - return (el_val_t)(cur + add_ms); -} - -el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit) { - int64_t d = (int64_t)ts2 - (int64_t)ts1; - const char* u = EL_CSTR(unit); - if (!u || strcmp(u, "ms") == 0) return (el_val_t)d; - if (strcmp(u, "sec") == 0) return (el_val_t)(d / 1000LL); - if (strcmp(u, "min") == 0) return (el_val_t)(d / 60000LL); - if (strcmp(u, "hour") == 0) return (el_val_t)(d / 3600000LL); - if (strcmp(u, "day") == 0) return (el_val_t)(d / 86400000LL); - return (el_val_t)d; -} - -/* Block the calling thread for `secs` seconds. Negative values are clamped - * to 0. Used by El programs that poll external resources (e.g. RunPod - * /status, Engram readiness probes). */ -el_val_t sleep_secs(el_val_t secs) { - int64_t s = (int64_t)secs; - if (s < 0) s = 0; - struct timespec ts; - ts.tv_sec = (time_t)s; - ts.tv_nsec = 0; - nanosleep(&ts, NULL); - return 0; -} - -el_val_t sleep_ms(el_val_t ms) { - int64_t m = (int64_t)ms; - if (m < 0) m = 0; - struct timespec ts; - ts.tv_sec = (time_t)(m / 1000LL); - ts.tv_nsec = (long)((m % 1000LL) * 1000000LL); - nanosleep(&ts, NULL); - return 0; -} - -/* ── Instant + Duration: first-class temporal types ────────────────────────── - * El's substrate (Neuron) is a temporal cognition system. Memory salience - * decay, the six-tier pacemaker, TTL caches, and supersession are all - * temporal. Treating time as a raw Int (now() returning ms-since-epoch and - * arithmetic done with mixed unit literals) lets bugs through the type - * system: `(now - cached_at) < 60` cannot tell ms from sec, and `sleep(30)` - * is ambiguous. This block introduces two dedicated representations. - * - * Representation: - * Instant — int64 nanoseconds since the Unix epoch - * Duration — int64 nanoseconds (signed; negative durations are legal, - * e.g. when a deadline has passed) - * - * Both share the el_val_t (int64) slot the rest of the runtime uses, so no - * boxing / arena allocation is needed. Type discipline is enforced at the - * codegen layer: `let x: Duration = ...` registers `x` in __duration_names, - * and BinOp dispatches through typed wrappers (el_duration_add, etc.) that - * make intent explicit in the generated C. Mismatched ops (Instant+Instant, - * Duration+Int) are surfaced via #error directives at codegen time so the - * downstream cc step fails with a clear El-source-level message. - * - * Nanosecond precision matches POSIX clock_gettime / nanosleep granularity. - * 2^63 nanos covers ~292 years from epoch — comfortably past 2200, plenty - * for a memory-system runtime that never schedules outside a human lifespan. - */ - -/* now() — current Instant. Wraps clock_gettime(CLOCK_REALTIME) for nanosecond - * precision. Falls back to gettimeofday on systems where clock_gettime is - * unavailable (defensive — every supported platform has it). */ -el_val_t el_now_instant(void) { - struct timespec ts; - if (clock_gettime(CLOCK_REALTIME, &ts) == 0) { - int64_t ns = (int64_t)ts.tv_sec * 1000000000LL + (int64_t)ts.tv_nsec; - return (el_val_t)ns; - } - struct timeval tv; - gettimeofday(&tv, NULL); - int64_t ns = (int64_t)tv.tv_sec * 1000000000LL - + (int64_t)tv.tv_usec * 1000LL; - return (el_val_t)ns; -} - -el_val_t now(void) { - return el_now_instant(); -} - -/* now_ns — return current Unix time as nanoseconds (Int). - * Thin wrapper over el_now_instant for use in test timing. */ -el_val_t now_ns(void) { - return el_now_instant(); -} - -/* unix_seconds(n) — Instant from a Unix-epoch second count. - * unix_millis(n) — Instant from a Unix-epoch millisecond count. */ -el_val_t unix_seconds(el_val_t n) { - int64_t s = (int64_t)n; - return (el_val_t)(s * 1000000000LL); -} - -el_val_t unix_millis(el_val_t n) { - int64_t m = (int64_t)n; - return (el_val_t)(m * 1000000LL); -} - -/* instant_from_iso8601 — parse a strict subset: - * YYYY-MM-DDTHH:MM:SS[.fff]Z - * Returns 0 (the Unix-epoch sentinel) on parse failure. Callers that need to - * distinguish epoch-zero from a parse error should use a wider sentinel - * representation; the current zero-on-failure choice matches existing El - * runtime conventions for parse builtins (str_to_int, parse_int). */ -el_val_t instant_from_iso8601(el_val_t s) { - const char* str = EL_CSTR(s); - if (!str) return (el_val_t)0; - int Y, M, D, h, m, sec, frac = 0; - int n = sscanf(str, "%d-%d-%dT%d:%d:%d.%3d", &Y, &M, &D, &h, &m, &sec, &frac); - if (n < 6) { - n = sscanf(str, "%d-%d-%dT%d:%d:%dZ", &Y, &M, &D, &h, &m, &sec); - if (n < 6) return (el_val_t)0; - } - struct tm tm; - memset(&tm, 0, sizeof(tm)); - tm.tm_year = Y - 1900; - tm.tm_mon = M - 1; - tm.tm_mday = D; - tm.tm_hour = h; - tm.tm_min = m; - tm.tm_sec = sec; - /* timegm — UTC. POSIX-Y but available on macOS and glibc. */ - time_t t = timegm(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (int64_t)frac * 1000000LL; - return (el_val_t)ns; -} - -/* Duration constructors. The El-side postfix literals (30.seconds, 1.hour) - * are lowered by the codegen directly into a literal int64 of nanoseconds — - * these constructors are for runtime values where the count is dynamic. */ -el_val_t el_duration_from_nanos(el_val_t ns) { - return (el_val_t)(int64_t)ns; -} - -el_val_t duration_seconds(el_val_t n) { - int64_t s = (int64_t)n; - return (el_val_t)(s * 1000000000LL); -} - -el_val_t duration_millis(el_val_t n) { - int64_t m = (int64_t)n; - return (el_val_t)(m * 1000000LL); -} - -el_val_t duration_nanos(el_val_t n) { - return (el_val_t)(int64_t)n; -} - -/* Arithmetic — typed wrappers. At the C level these are no-op casts, but - * the codegen routes Instant/Duration BinOps through them so the generated - * C says `el_instant_add_dur(start, dur)` rather than `start + dur`. The - * intent is explicit, the operand order is documented, and a future change - * to the underlying representation (saturating arithmetic, overflow guards) - * has a single chokepoint. */ -el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur) { - return (el_val_t)((int64_t)inst + (int64_t)dur); -} - -el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur) { - return (el_val_t)((int64_t)inst - (int64_t)dur); -} - -el_val_t el_instant_diff(el_val_t a, el_val_t b) { - /* a - b — yields a Duration (negative if b is later than a). */ - return (el_val_t)((int64_t)a - (int64_t)b); -} - -el_val_t el_duration_add(el_val_t a, el_val_t b) { - return (el_val_t)((int64_t)a + (int64_t)b); -} - -el_val_t el_duration_sub(el_val_t a, el_val_t b) { - return (el_val_t)((int64_t)a - (int64_t)b); -} - -el_val_t el_duration_scale(el_val_t dur, el_val_t scalar) { - return (el_val_t)((int64_t)dur * (int64_t)scalar); -} - -el_val_t el_duration_div(el_val_t dur, el_val_t scalar) { - int64_t s = (int64_t)scalar; - if (s == 0) return (el_val_t)0; - return (el_val_t)((int64_t)dur / s); -} - -/* Comparisons. Return 1/0 in el_val_t convention. */ -el_val_t el_instant_lt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a < (int64_t)b ? 1 : 0); } -el_val_t el_instant_le(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a <= (int64_t)b ? 1 : 0); } -el_val_t el_instant_gt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a > (int64_t)b ? 1 : 0); } -el_val_t el_instant_ge(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a >= (int64_t)b ? 1 : 0); } -el_val_t el_instant_eq(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a == (int64_t)b ? 1 : 0); } -el_val_t el_instant_ne(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a != (int64_t)b ? 1 : 0); } -el_val_t el_duration_lt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a < (int64_t)b ? 1 : 0); } -el_val_t el_duration_le(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a <= (int64_t)b ? 1 : 0); } -el_val_t el_duration_gt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a > (int64_t)b ? 1 : 0); } -el_val_t el_duration_ge(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a >= (int64_t)b ? 1 : 0); } -el_val_t el_duration_eq(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a == (int64_t)b ? 1 : 0); } -el_val_t el_duration_ne(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a != (int64_t)b ? 1 : 0); } - -/* Conversions. */ -el_val_t instant_to_unix_seconds(el_val_t i) { - return (el_val_t)((int64_t)i / 1000000000LL); -} - -el_val_t instant_to_unix_millis(el_val_t i) { - return (el_val_t)((int64_t)i / 1000000LL); -} - -el_val_t instant_to_iso8601(el_val_t i) { - int64_t ns = (int64_t)i; - time_t s = (time_t)(ns / 1000000000LL); - int msec = (int)((ns / 1000000LL) % 1000LL); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - char buf[64]; - snprintf(buf, sizeof(buf), "%04d-%02d-%02dT%02d:%02d:%02d.%03dZ", - tm.tm_year + 1900, tm.tm_mon + 1, tm.tm_mday, - tm.tm_hour, tm.tm_min, tm.tm_sec, msec); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t duration_to_seconds(el_val_t d) { - return (el_val_t)((int64_t)d / 1000000000LL); -} - -el_val_t duration_to_millis(el_val_t d) { - return (el_val_t)((int64_t)d / 1000000LL); -} - -el_val_t duration_to_nanos(el_val_t d) { - return (el_val_t)(int64_t)d; -} - -/* sleep(Duration) — Phase 1 replacement for ambiguous sleep(Int). The runtime - * still exposes sleep_secs/sleep_ms for legacy call sites; codegen lowers - * sleep(Duration) to el_sleep_duration(d). Negative durations clamp to 0 so a - * stale deadline doesn't block forever. */ -el_val_t el_sleep_duration(el_val_t dur) { - int64_t ns = (int64_t)dur; - if (ns < 0) ns = 0; - struct timespec ts; - ts.tv_sec = (time_t)(ns / 1000000000LL); - ts.tv_nsec = (long)(ns % 1000000000LL); - nanosleep(&ts, NULL); - return (el_val_t)0; -} - -/* unix_timestamp() — back-compat. Existing El callers expect an Int seconds - * value; this stays an Int returner so the type system isn't disturbed for - * legacy code. New code should call now() and convert when needed. */ -el_val_t unix_timestamp(void) { - return instant_to_unix_seconds(el_now_instant()); -} - -/* TTL cache helpers. Backed by the existing process-wide K/V (state_set/get) - * with a sibling __ttl_set_at_ entry recording the Instant of the last - * write. ttl_cache_get returns "" if the entry is missing or stale, so call - * sites can branch on `if v == "" { miss } else { hit }` — the same shape - * existing get-with-default code uses. No more (now - cached_at) < 60. */ -el_val_t ttl_cache_set(el_val_t key, el_val_t value) { - const char* k = EL_CSTR(key); - if (!k) return (el_val_t)0; - /* Store the value at the user's key. */ - state_set(key, value); - /* Stamp set_at — opaque schema, namespaced under __ttl: prefix so user - * keys can't collide with stamps. */ - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return (el_val_t)0; - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - int64_t now_ns = (int64_t)el_now_instant(); - char buf[32]; - snprintf(buf, sizeof(buf), "%lld", (long long)now_ns); - state_set(EL_STR(stamp_key), EL_STR(buf)); - free(stamp_key); - return (el_val_t)1; -} - -el_val_t ttl_cache_get(el_val_t key, el_val_t max_age) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - /* Look up stamp. */ - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return el_wrap_str(el_strdup("")); - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - el_val_t stamp = state_get(EL_STR(stamp_key)); - free(stamp_key); - const char* sv = EL_CSTR(stamp); - if (!sv || !*sv) return el_wrap_str(el_strdup("")); - int64_t set_at = (int64_t)atoll(sv); - int64_t now_ns = (int64_t)el_now_instant(); - int64_t age = now_ns - set_at; - int64_t max_ns = (int64_t)max_age; - if (age < 0) return el_wrap_str(el_strdup("")); /* clock skew — treat as miss */ - if (age > max_ns) return el_wrap_str(el_strdup("")); /* expired */ - return state_get(key); -} - -el_val_t ttl_cache_age(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return (el_val_t)INT64_MAX; - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return (el_val_t)INT64_MAX; - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - el_val_t stamp = state_get(EL_STR(stamp_key)); - free(stamp_key); - const char* sv = EL_CSTR(stamp); - if (!sv || !*sv) return (el_val_t)INT64_MAX; - int64_t set_at = (int64_t)atoll(sv); - int64_t now_ns = (int64_t)el_now_instant(); - return (el_val_t)(now_ns - set_at); -} - -/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ────────────── - * Phase 1.5. Calendar is pluggable: EarthCalendar (IANA zones + Gregorian + - * DST), MarsCalendar (sols, MTC), CycleCalendar(period), NoCycleCalendar, - * RelativeCalendar(epoch). Phase 1 zone wrapping folds INTO EarthCalendar; - * UTC and IANA zones are themselves Earth-parochial and cannot live at the - * lowest type layer. - * - * A Rhythm is a small AST that asks the Calendar for cycle phase, weekday, - * etc. Most rhythm logic is calendar-agnostic at runtime: rhythm_cycle_phase - * means "midpoint of cycle" whether the cycle is 24h on Earth or 30h on a - * station or 300y on a long-cycle world. */ - -/* Magic headers — used by the runtime to recognize boxed temporal values - * arriving through el_val_t. Distinct constants so accidental misuse fails - * loudly rather than silently. */ -#define EL_CAL_MAGIC 0xE1CA1EDDU -#define EL_CALTIME_MAGIC 0xE1CA1747U -#define EL_RHYTHM_MAGIC 0xE1287A11U -#define EL_LDATE_MAGIC 0xE1DA7E00U -#define EL_LDT_MAGIC 0xE1DA7E1DU -#define EL_ZONE_MAGIC 0xE12017E0U - -typedef enum { - EL_CALENDAR_EARTH = 1, - EL_CALENDAR_MARS = 2, - EL_CALENDAR_CYCLE = 3, - EL_CALENDAR_NO_CYCLE = 4, - EL_CALENDAR_RELATIVE = 5 -} el_calendar_kind_t; - -typedef struct { - uint32_t magic; - char* id; /* IANA name or "+HH:MM" / "-HH:MM" */ - int fixed; /* 1 for fixed offset, 0 for IANA */ - int64_t offset_ns; /* fixed offset in nanos (only when fixed) */ -} el_zone_t; - -typedef struct { - uint32_t magic; - el_calendar_kind_t kind; - el_zone_t* zone; /* EarthCalendar; MarsCalendar uses MTC */ - int64_t cycle_period_ns;/* CycleCalendar; computed for Earth (86400 s) and Mars (88775.244 s) */ - int64_t epoch_ns; /* RelativeCalendar; Unix-epoch zero otherwise */ -} el_calendar_t; - -typedef struct { - uint32_t magic; - int64_t instant_ns; - el_calendar_t* cal; -} el_caltime_t; - -/* Rhythm AST. */ -typedef enum { - EL_RHYTHM_CYCLE_START = 1, - EL_RHYTHM_CYCLE_PHASE = 2, - EL_RHYTHM_DURATION = 3, - EL_RHYTHM_SESSION_START = 4, - EL_RHYTHM_EVENT = 5, - EL_RHYTHM_AND = 6, - EL_RHYTHM_OR = 7, - EL_RHYTHM_WEEKDAY = 8, - EL_RHYTHM_WEEKLY_AT = 9 -} el_rhythm_kind_t; - -typedef struct el_rhythm_s { - uint32_t magic; - el_rhythm_kind_t kind; - double phase; /* CYCLE_PHASE */ - int64_t period_ns; /* DURATION */ - int weekday; /* 1..7 Mon..Sun */ - int hour; - int minute; - char* event_name; /* EVENT */ - struct el_rhythm_s* a; /* AND/OR */ - struct el_rhythm_s* b; -} el_rhythm_t; - -typedef struct { - uint32_t magic; - int year; - int month; - int day; -} el_localdate_t; - -typedef struct { - uint32_t magic; - el_localdate_t* date; - int64_t time_ns; /* nanos since midnight */ -} el_localdt_t; - -/* Magic-tag check helpers — peek the first 4 bytes of an el_val_t pointer - * and compare against the expected magic. Strings are NUL-terminated and - * never start with our magic byte sequence, so this is safe. */ -static int el_is_magic(el_val_t v, uint32_t want) { - if (v == 0) return 0; - /* Defensive: only follow pointers in plausible address space. - * On 64-bit unix processes pointers are above 0x10000. */ - if ((uint64_t)v < 0x10000ULL) return 0; - uint32_t got = *(volatile uint32_t*)(uintptr_t)v; - return got == want; -} - -/* Sol length on Mars in nanoseconds: 88775.244 seconds. */ -#define EL_MARS_SOL_NS ((int64_t)88775244000000LL) -/* Earth solar day in nanoseconds: 86400 seconds. */ -#define EL_EARTH_DAY_NS ((int64_t)86400000000000LL) - -/* ── Zone construction ────────────────────────────────────────────────────── - * Zones intern by id string so equality comparisons are pointer-compares. */ - -#define EL_ZONE_TABLE_CAP 64 -static el_zone_t* _el_zone_table[EL_ZONE_TABLE_CAP]; -static int _el_zone_count = 0; - -static el_zone_t* _el_zone_intern(const char* id, int fixed, int64_t offset_ns) { - for (int i = 0; i < _el_zone_count; i++) { - el_zone_t* z = _el_zone_table[i]; - if (z->fixed == fixed && z->offset_ns == offset_ns && - strcmp(z->id ? z->id : "", id ? id : "") == 0) { - return z; - } - } - if (_el_zone_count >= EL_ZONE_TABLE_CAP) { - /* Out of slots: build a non-interned zone. Equality will fail across - * such zones but the program still runs. */ - el_zone_t* z = (el_zone_t*)malloc(sizeof(el_zone_t)); - z->magic = EL_ZONE_MAGIC; - z->id = el_strdup_persist(id ? id : ""); - z->fixed = fixed; - z->offset_ns = offset_ns; - return z; - } - el_zone_t* z = (el_zone_t*)malloc(sizeof(el_zone_t)); - z->magic = EL_ZONE_MAGIC; - z->id = el_strdup_persist(id ? id : ""); - z->fixed = fixed; - z->offset_ns = offset_ns; - _el_zone_table[_el_zone_count++] = z; - return z; -} - -el_val_t zone(el_val_t id) { - const char* s = EL_CSTR(id); - if (!s || !*s) return (el_val_t)(uintptr_t)_el_zone_intern("UTC", 0, 0); - /* Fixed-offset shortcut: "+HH:MM" or "-HH:MM". */ - if ((s[0] == '+' || s[0] == '-') && strlen(s) >= 6 && s[3] == ':') { - int sign = (s[0] == '-') ? -1 : 1; - int hh = (s[1] - '0') * 10 + (s[2] - '0'); - int mm = (s[4] - '0') * 10 + (s[5] - '0'); - int64_t off = (int64_t)sign * ((int64_t)hh * 3600LL + (int64_t)mm * 60LL) * 1000000000LL; - return (el_val_t)(uintptr_t)_el_zone_intern(s, 1, off); - } - return (el_val_t)(uintptr_t)_el_zone_intern(s, 0, 0); -} - -el_val_t zone_utc(void) { - return (el_val_t)(uintptr_t)_el_zone_intern("UTC", 1, 0); -} - -el_val_t zone_local(void) { - /* Resolve the local zone via TZ env or system default. tzset() picks - * up TZ if set; otherwise the C library reads /etc/localtime. We store - * the zone id as "LOCAL" so subsequent equality holds; resolution is - * lazy at use time. */ - return (el_val_t)(uintptr_t)_el_zone_intern("LOCAL", 0, 0); -} - -el_val_t zone_offset(el_val_t hours, el_val_t minutes) { - int hh = (int)(int64_t)hours; - int mm = (int)(int64_t)minutes; - int sign = (hh < 0 || mm < 0) ? -1 : 1; - if (hh < 0) hh = -hh; - if (mm < 0) mm = -mm; - int64_t off = (int64_t)sign * ((int64_t)hh * 3600LL + (int64_t)mm * 60LL) * 1000000000LL; - char buf[16]; - snprintf(buf, sizeof(buf), "%c%02d:%02d", sign < 0 ? '-' : '+', hh, mm); - return (el_val_t)(uintptr_t)_el_zone_intern(buf, 1, off); -} - -/* ── Calendar interning ──────────────────────────────────────────────────── */ - -#define EL_CAL_TABLE_CAP 64 -static el_calendar_t* _el_cal_table[EL_CAL_TABLE_CAP]; -static int _el_cal_count = 0; - -static el_calendar_t* _el_cal_intern(el_calendar_kind_t kind, el_zone_t* z, - int64_t period_ns, int64_t epoch_ns) { - for (int i = 0; i < _el_cal_count; i++) { - el_calendar_t* c = _el_cal_table[i]; - if (c->kind == kind && c->zone == z && - c->cycle_period_ns == period_ns && c->epoch_ns == epoch_ns) { - return c; - } - } - el_calendar_t* c = (el_calendar_t*)malloc(sizeof(el_calendar_t)); - c->magic = EL_CAL_MAGIC; - c->kind = kind; - c->zone = z; - c->cycle_period_ns = period_ns; - c->epoch_ns = epoch_ns; - if (_el_cal_count < EL_CAL_TABLE_CAP) _el_cal_table[_el_cal_count++] = c; - return c; -} - -el_val_t earth_calendar(el_val_t z_val) { - el_zone_t* z = NULL; - if (z_val != 0 && el_is_magic(z_val, EL_ZONE_MAGIC)) { - z = (el_zone_t*)(uintptr_t)z_val; - } else { - z = (el_zone_t*)(uintptr_t)zone_local(); - } - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_EARTH, z, EL_EARTH_DAY_NS, 0); -} - -el_val_t earth_calendar_default(void) { - return earth_calendar(zone_local()); -} - -el_val_t mars_calendar(void) { - el_zone_t* z = (el_zone_t*)(uintptr_t)_el_zone_intern("MTC", 1, 0); - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_MARS, z, EL_MARS_SOL_NS, 0); -} - -el_val_t cycle_calendar(el_val_t period_dur) { - int64_t period = (int64_t)period_dur; - if (period <= 0) period = 1; - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_CYCLE, NULL, period, 0); -} - -el_val_t no_cycle_calendar(void) { - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_NO_CYCLE, NULL, 0, 0); -} - -el_val_t relative_calendar(el_val_t epoch_inst) { - int64_t ep = (int64_t)epoch_inst; - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_RELATIVE, NULL, 0, ep); -} - -/* ── CalendarTime ───────────────────────────────────────────────────────── */ - -static el_caltime_t* _el_caltime_alloc(int64_t inst, el_calendar_t* c) { - el_caltime_t* ct = (el_caltime_t*)malloc(sizeof(el_caltime_t)); - ct->magic = EL_CALTIME_MAGIC; - ct->instant_ns = inst; - ct->cal = c; - return ct; -} - -static el_calendar_t* _el_resolve_cal(el_val_t cal_val) { - if (cal_val == 0 || !el_is_magic(cal_val, EL_CAL_MAGIC)) { - return (el_calendar_t*)(uintptr_t)earth_calendar_default(); - } - return (el_calendar_t*)(uintptr_t)cal_val; -} - -el_val_t now_in(el_val_t cal_val) { - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t ns = (int64_t)el_now_instant(); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); -} - -el_val_t in_calendar(el_val_t inst, el_val_t cal_val) { - el_calendar_t* c = _el_resolve_cal(cal_val); - return (el_val_t)(uintptr_t)_el_caltime_alloc((int64_t)inst, c); -} - -el_val_t cal_to_instant(el_val_t ct_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - return (el_val_t)ct->instant_ns; -} - -el_val_t cal_in(el_val_t ct_val, el_val_t cal_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ct->instant_ns, c); -} - -el_val_t cal_cycle_phase(el_val_t ct_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return el_from_float(0.0); - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - el_calendar_t* c = ct->cal; - if (c->kind == EL_CALENDAR_NO_CYCLE) { - return el_from_float(0.0/0.0); /* NaN sentinel */ - } - int64_t period = c->cycle_period_ns; - if (period <= 0) return el_from_float(0.0); - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t phase_ns = base % period; - if (phase_ns < 0) phase_ns += period; - double phase = (double)phase_ns / (double)period; - return el_from_float(phase); -} - -/* ── Earth zone resolution: TZ-based offset lookup ────────────────────────── - * For an EarthCalendar(zone), we want to convert an instant_ns into local - * y/m/d/h/m/s, including DST. Approach: setenv("TZ", id), tzset(), use - * localtime_r, then restore. This is not thread-safe by design — El's - * runtime is single-threaded for the request handler path. Cache the - * computed (instant -> tm) to avoid the syscall churn on repeat formats. */ - -static void _el_apply_zone(el_zone_t* z) { - if (!z) { unsetenv("TZ"); tzset(); return; } - if (z->fixed && strcmp(z->id, "UTC") == 0) { - setenv("TZ", "UTC0", 1); - tzset(); - return; - } - if (z->fixed) { - /* Fixed offset: POSIX TZ uses inverted sign (sign convention of - * "hours WEST of UTC" rather than east). Build the spec accordingly. */ - char buf[32]; - int neg_secs = (int)(-z->offset_ns / 1000000000LL); - int sign = neg_secs < 0 ? -1 : 1; - int abs_secs = neg_secs < 0 ? -neg_secs : neg_secs; - int hh = abs_secs / 3600; - int mm = (abs_secs % 3600) / 60; - snprintf(buf, sizeof(buf), "FIX%c%d:%02d", sign < 0 ? '-' : '+', hh, mm); - setenv("TZ", buf, 1); - tzset(); - return; - } - if (strcmp(z->id, "LOCAL") == 0) { - unsetenv("TZ"); - tzset(); - return; - } - setenv("TZ", z->id, 1); - tzset(); -} - -static int _el_decompose_earth(el_caltime_t* ct, struct tm* tm_out, int* abbr_len, char* abbr_buf, size_t abbr_cap) { - el_calendar_t* c = ct->cal; - el_zone_t* z = c->zone; - _el_apply_zone(z); - time_t s = (time_t)(ct->instant_ns / 1000000000LL); - struct tm tm; - localtime_r(&s, &tm); - *tm_out = tm; - if (abbr_buf && abbr_cap > 0) { - /* mingw's struct tm has no tm_zone (BSD/glibc extension); no abbrev available there. */ -#ifdef _WIN32 - const char* z_str = ""; -#else - const char* z_str = tm.tm_zone ? tm.tm_zone : ""; -#endif - size_t n = strlen(z_str); - if (n >= abbr_cap) n = abbr_cap - 1; - memcpy(abbr_buf, z_str, n); - abbr_buf[n] = '\0'; - if (abbr_len) *abbr_len = (int)n; - } - return 0; -} - -/* Format an Earth CalendarTime under a Java-DateTimeFormatter-ish pattern. - * We support a useful core: yyyy MM dd HH mm ss z EEE MMM d h a — enough for - * the acceptance tests. Single quotes denote literal text. */ -static const char* _el_weekday_short[] = {"Sun","Mon","Tue","Wed","Thu","Fri","Sat"}; -static const char* _el_month_short[] = {"Jan","Feb","Mar","Apr","May","Jun", - "Jul","Aug","Sep","Oct","Nov","Dec"}; - -static char* _el_format_earth(el_caltime_t* ct, const char* pattern) { - struct tm tm; - char abbr[16] = {0}; - int abbr_len = 0; - _el_decompose_earth(ct, &tm, &abbr_len, abbr, sizeof(abbr)); - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - size_t i = 0; - size_t plen = strlen(pattern); - while (i < plen) { - char ch = pattern[i]; - /* Quoted literal */ - if (ch == '\'') { - i++; - while (i < plen && pattern[i] != '\'') { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i++]; - } - if (i < plen) i++; - continue; - } - /* Count run of same letter */ - size_t run = 1; - while (i + run < plen && pattern[i + run] == ch) run++; - char tmp[64]; - tmp[0] = '\0'; - if (ch == 'y') { - if (run >= 4) snprintf(tmp, sizeof(tmp), "%04d", tm.tm_year + 1900); - else snprintf(tmp, sizeof(tmp), "%02d", (tm.tm_year + 1900) % 100); - } else if (ch == 'M') { - if (run >= 3) snprintf(tmp, sizeof(tmp), "%s", _el_month_short[tm.tm_mon]); - else if (run == 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_mon + 1); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_mon + 1); - } else if (ch == 'd') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_mday); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_mday); - } else if (ch == 'H') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_hour); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_hour); - } else if (ch == 'h') { - int h12 = tm.tm_hour % 12; if (h12 == 0) h12 = 12; - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", h12); - else snprintf(tmp, sizeof(tmp), "%d", h12); - } else if (ch == 'm') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_min); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_min); - } else if (ch == 's') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_sec); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_sec); - } else if (ch == 'a') { - snprintf(tmp, sizeof(tmp), "%s", tm.tm_hour < 12 ? "AM" : "PM"); - } else if (ch == 'E') { - snprintf(tmp, sizeof(tmp), "%s", _el_weekday_short[tm.tm_wday]); - } else if (ch == 'z') { - snprintf(tmp, sizeof(tmp), "%s", abbr); - } else { - for (size_t k = 0; k < run; k++) { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = ch; - } - i += run; - continue; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - i += run; - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -/* Format a Mars CalendarTime: %sol prints the integer sol number since - * mission epoch (Unix epoch fallback), %phase prints cycle_phase as a - * 0..1 decimal. Other %-specifiers fall through. */ -static char* _el_format_mars(el_caltime_t* ct, const char* pattern) { - el_calendar_t* c = ct->cal; - int64_t period = c->cycle_period_ns > 0 ? c->cycle_period_ns : EL_MARS_SOL_NS; - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t sol = base / period; - int64_t phase_ns = base % period; - if (phase_ns < 0) { phase_ns += period; sol -= 1; } - double phase = (double)phase_ns / (double)period; - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - for (size_t i = 0; pattern[i]; i++) { - if (pattern[i] == '%' && pattern[i+1]) { - char tmp[64]; - tmp[0] = '\0'; - if (strncmp(pattern + i + 1, "sol", 3) == 0) { - snprintf(tmp, sizeof(tmp), "%lld", (long long)sol); - i += 3; - } else if (strncmp(pattern + i + 1, "phase", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%.4f", phase); - i += 5; - } else if (pattern[i+1] == 'd') { - snprintf(tmp, sizeof(tmp), "%lld", (long long)sol); - i += 1; - } else { - tmp[0] = pattern[i+1]; tmp[1] = '\0'; - i += 1; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - } else { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i]; - } - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -/* Format a CycleCalendar CalendarTime: %cycle and %phase. */ -static char* _el_format_cycle(el_caltime_t* ct, const char* pattern) { - el_calendar_t* c = ct->cal; - int64_t period = c->cycle_period_ns > 0 ? c->cycle_period_ns : 1; - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t cycle = base / period; - int64_t phase_ns = base % period; - if (phase_ns < 0) { phase_ns += period; cycle -= 1; } - double phase = (double)phase_ns / (double)period; - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - for (size_t i = 0; pattern[i]; i++) { - if (pattern[i] == '%' && pattern[i+1]) { - char tmp[64]; - tmp[0] = '\0'; - if (strncmp(pattern + i + 1, "cycle", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%lld", (long long)cycle); - i += 5; - } else if (strncmp(pattern + i + 1, "phase", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%.4f", phase); - i += 5; - } else if (pattern[i+1] == 'd') { - snprintf(tmp, sizeof(tmp), "%lld", (long long)cycle); - i += 1; - } else if (pattern[i+1] == 'f') { - snprintf(tmp, sizeof(tmp), "%.2f", phase); - i += 1; - } else { - /* Pass through unknown specifier */ - tmp[0] = '%'; tmp[1] = pattern[i+1]; tmp[2] = '\0'; - i += 1; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - } else { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i]; - } - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -el_val_t cal_format(el_val_t ct_val, el_val_t pattern_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return el_wrap_str(el_strdup("")); - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - const char* pat = EL_CSTR(pattern_val); - if (!pat) pat = ""; - char* result = NULL; - switch (ct->cal->kind) { - case EL_CALENDAR_EARTH: result = _el_format_earth(ct, pat); break; - case EL_CALENDAR_MARS: result = _el_format_mars(ct, pat); break; - case EL_CALENDAR_CYCLE: result = _el_format_cycle(ct, pat); break; - case EL_CALENDAR_RELATIVE: result = _el_format_cycle(ct, pat); break; - case EL_CALENDAR_NO_CYCLE: { - char buf[64]; - snprintf(buf, sizeof(buf), "instant:%lld", (long long)ct->instant_ns); - result = el_strdup(buf); - break; - } - default: result = el_strdup(""); - } - return el_wrap_str(result); -} - -/* ── LocalDate / LocalTime / LocalDateTime ──────────────────────────────── */ - -static int _el_days_in_month(int y, int m) { - static const int dim[12] = {31,28,31,30,31,30,31,31,30,31,30,31}; - if (m == 2) { - int leap = ((y % 4 == 0) && (y % 100 != 0)) || (y % 400 == 0); - return 28 + (leap ? 1 : 0); - } - if (m < 1 || m > 12) return 30; - return dim[m - 1]; -} - -el_val_t local_date(el_val_t y, el_val_t m, el_val_t d) { - el_localdate_t* ld = (el_localdate_t*)malloc(sizeof(el_localdate_t)); - ld->magic = EL_LDATE_MAGIC; - ld->year = (int)(int64_t)y; - ld->month = (int)(int64_t)m; - ld->day = (int)(int64_t)d; - return (el_val_t)(uintptr_t)ld; -} - -el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns) { - int64_t hh = (int64_t)h; - int64_t mm = (int64_t)m; - int64_t ss = (int64_t)s; - int64_t nn = (int64_t)ns; - int64_t total = hh * 3600000000000LL + mm * 60000000000LL + ss * 1000000000LL + nn; - return (el_val_t)total; -} - -el_val_t local_datetime(el_val_t date_val, el_val_t time_val) { - if (!el_is_magic(date_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdt_t* ldt = (el_localdt_t*)malloc(sizeof(el_localdt_t)); - ldt->magic = EL_LDT_MAGIC; - ldt->date = (el_localdate_t*)(uintptr_t)date_val; - ldt->time_ns = (int64_t)time_val; - return (el_val_t)(uintptr_t)ldt; -} - -el_val_t zoned(el_val_t date_val, el_val_t time_val, el_val_t cal_val) { - if (!el_is_magic(date_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* ld = (el_localdate_t*)(uintptr_t)date_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t time_ns = (int64_t)time_val; - /* Convert (LocalDate, LocalTime, EarthCalendar) -> Instant. - * For non-Earth calendars we use day-anchored conversion: treat the - * LocalDate's (y,m,d) as a Gregorian projection, convert to seconds via - * mktime under the calendar's zone, then add nanos-since-midnight. */ - if (c->kind == EL_CALENDAR_EARTH) { - _el_apply_zone(c->zone); - struct tm tm; memset(&tm, 0, sizeof(tm)); - tm.tm_year = ld->year - 1900; - tm.tm_mon = ld->month - 1; - tm.tm_mday = ld->day; - tm.tm_hour = (int)(time_ns / 3600000000000LL); - tm.tm_min = (int)((time_ns / 60000000000LL) % 60); - tm.tm_sec = (int)((time_ns / 1000000000LL) % 60); - tm.tm_isdst = -1; - time_t t = mktime(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (time_ns % 1000000000LL); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); - } - /* Non-Earth fallback: project as if Earth UTC then attach calendar. */ - struct tm tm; memset(&tm, 0, sizeof(tm)); - tm.tm_year = ld->year - 1900; - tm.tm_mon = ld->month - 1; - tm.tm_mday = ld->day; - tm.tm_hour = (int)(time_ns / 3600000000000LL); - tm.tm_min = (int)((time_ns / 60000000000LL) % 60); - tm.tm_sec = (int)((time_ns / 1000000000LL) % 60); - time_t t = timegm(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (time_ns % 1000000000LL); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); -} - -el_val_t local_date_year(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->year; -} -el_val_t local_date_month(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->month; -} -el_val_t local_date_day(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->day; -} -el_val_t local_time_hour(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)(t / 3600000000000LL); -} -el_val_t local_time_minute(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)((t / 60000000000LL) % 60); -} -el_val_t local_time_second(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)((t / 1000000000LL) % 60); -} -el_val_t local_time_nanos(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)(t % 1000000000LL); -} - -el_val_t el_local_date_add_dur(el_val_t ld_val, el_val_t dur_val) { - if (!el_is_magic(ld_val, EL_LDATE_MAGIC)) return ld_val; - el_localdate_t* ld = (el_localdate_t*)(uintptr_t)ld_val; - int64_t dur_ns = (int64_t)dur_val; - int64_t days = dur_ns / EL_EARTH_DAY_NS; - int y = ld->year, m = ld->month, d = ld->day; - /* Walk days forward/backward in canonical Gregorian. */ - while (days > 0) { - int dim = _el_days_in_month(y, m); - if (d + days <= dim) { d += (int)days; days = 0; break; } - days -= (dim - d + 1); - d = 1; - m++; - if (m > 12) { m = 1; y++; } - } - while (days < 0) { - if (d + days >= 1) { d += (int)days; days = 0; break; } - days += d; - m--; - if (m < 1) { m = 12; y--; } - d = _el_days_in_month(y, m); - } - return local_date((el_val_t)y, (el_val_t)m, (el_val_t)d); -} - -el_val_t el_local_time_add_dur(el_val_t lt_val, el_val_t dur_val) { - int64_t t = (int64_t)lt_val + (int64_t)dur_val; - /* Wrap mod 24h on Earth-default. CycleCalendar wrapping requires the - * caller to use cal_in / cal_format for the right modulus. */ - int64_t day = EL_EARTH_DAY_NS; - int64_t r = t % day; - if (r < 0) r += day; - return (el_val_t)r; -} - -el_val_t el_local_date_lt(el_val_t a_val, el_val_t b_val) { - if (!el_is_magic(a_val, EL_LDATE_MAGIC) || !el_is_magic(b_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* a = (el_localdate_t*)(uintptr_t)a_val; - el_localdate_t* b = (el_localdate_t*)(uintptr_t)b_val; - if (a->year != b->year) return (el_val_t)(a->year < b->year ? 1 : 0); - if (a->month != b->month) return (el_val_t)(a->month < b->month ? 1 : 0); - return (el_val_t)(a->day < b->day ? 1 : 0); -} - -el_val_t el_local_date_eq(el_val_t a_val, el_val_t b_val) { - if (!el_is_magic(a_val, EL_LDATE_MAGIC) || !el_is_magic(b_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* a = (el_localdate_t*)(uintptr_t)a_val; - el_localdate_t* b = (el_localdate_t*)(uintptr_t)b_val; - return (el_val_t)((a->year == b->year && a->month == b->month && a->day == b->day) ? 1 : 0); -} - -/* ── Rhythm ──────────────────────────────────────────────────────────────── */ - -static el_rhythm_t* _el_rhythm_alloc(el_rhythm_kind_t k) { - el_rhythm_t* r = (el_rhythm_t*)calloc(1, sizeof(el_rhythm_t)); - r->magic = EL_RHYTHM_MAGIC; - r->kind = k; - return r; -} - -el_val_t rhythm_cycle_start(void) { - return (el_val_t)(uintptr_t)_el_rhythm_alloc(EL_RHYTHM_CYCLE_START); -} - -el_val_t rhythm_cycle_phase(el_val_t phase_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_CYCLE_PHASE); - r->phase = el_to_float(phase_val); - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_duration(el_val_t d_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_DURATION); - r->period_ns = (int64_t)d_val; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_session_start(void) { - return (el_val_t)(uintptr_t)_el_rhythm_alloc(EL_RHYTHM_SESSION_START); -} - -el_val_t rhythm_event(el_val_t name_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_EVENT); - const char* n = EL_CSTR(name_val); - r->event_name = el_strdup_persist(n ? n : ""); - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_and(el_val_t a_val, el_val_t b_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_AND); - r->a = el_is_magic(a_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)a_val : NULL; - r->b = el_is_magic(b_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)b_val : NULL; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_or(el_val_t a_val, el_val_t b_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_OR); - r->a = el_is_magic(a_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)a_val : NULL; - r->b = el_is_magic(b_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)b_val : NULL; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_weekday(el_val_t day) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_WEEKDAY); - r->weekday = (int)(int64_t)day; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_WEEKLY_AT); - r->weekday = (int)(int64_t)day; - r->hour = (int)(int64_t)hour; - r->minute = (int)(int64_t)minute; - return (el_val_t)(uintptr_t)r; -} - -/* Compute the next instant on or after `after` when rhythm `r` matches, - * under calendar `cal`. */ -static int64_t _el_next_after(el_rhythm_t* r, int64_t after_ns, el_calendar_t* cal) { - if (!r) return after_ns; - int64_t period = cal->cycle_period_ns > 0 ? cal->cycle_period_ns : EL_EARTH_DAY_NS; - switch (r->kind) { - case EL_RHYTHM_CYCLE_START: { - int64_t base = after_ns - cal->epoch_ns; - int64_t cyc = (base / period) + 1; - return cal->epoch_ns + cyc * period; - } - case EL_RHYTHM_CYCLE_PHASE: { - int64_t base = after_ns - cal->epoch_ns; - int64_t cyc_ns = (int64_t)(r->phase * (double)period); - int64_t cur_cyc = base / period; - int64_t candidate = cal->epoch_ns + cur_cyc * period + cyc_ns; - if (candidate <= after_ns) candidate += period; - return candidate; - } - case EL_RHYTHM_DURATION: { - return after_ns + (r->period_ns > 0 ? r->period_ns : 1); - } - case EL_RHYTHM_WEEKDAY: - case EL_RHYTHM_WEEKLY_AT: { - if (cal->kind != EL_CALENDAR_EARTH) { - /* Non-Earth calendars: fall back to cycle math, treating - * weekday as a 7-cycle-per-period proxy. */ - return after_ns + period; - } - _el_apply_zone(cal->zone); - time_t s = (time_t)(after_ns / 1000000000LL); - struct tm tm; - localtime_r(&s, &tm); - /* tm_wday: 0=Sun..6=Sat. We use 1=Mon..7=Sun. */ - int target = r->weekday >= 1 && r->weekday <= 7 ? r->weekday : 1; - int target_wday = target == 7 ? 0 : target; /* 7→Sun=0, 1→Mon=1 */ - int days_ahead = (target_wday - tm.tm_wday + 7) % 7; - int hour = (r->kind == EL_RHYTHM_WEEKLY_AT) ? r->hour : 0; - int minute = (r->kind == EL_RHYTHM_WEEKLY_AT) ? r->minute : 0; - struct tm cand = tm; - cand.tm_mday += days_ahead; - cand.tm_hour = hour; - cand.tm_min = minute; - cand.tm_sec = 0; - cand.tm_isdst = -1; - time_t cand_t = mktime(&cand); - int64_t cand_ns = (int64_t)cand_t * 1000000000LL; - if (cand_ns <= after_ns) { - cand.tm_mday += 7; - cand.tm_isdst = -1; - cand_t = mktime(&cand); - cand_ns = (int64_t)cand_t * 1000000000LL; - } - return cand_ns; - } - case EL_RHYTHM_AND: { - int64_t a = _el_next_after(r->a, after_ns, cal); - int64_t b = _el_next_after(r->b, after_ns, cal); - return a > b ? a : b; - } - case EL_RHYTHM_OR: { - int64_t a = _el_next_after(r->a, after_ns, cal); - int64_t b = _el_next_after(r->b, after_ns, cal); - return a < b ? a : b; - } - case EL_RHYTHM_SESSION_START: - case EL_RHYTHM_EVENT: - default: - return after_ns; - } -} - -el_val_t rhythm_next_after(el_val_t r_val, el_val_t after_val, el_val_t cal_val) { - if (!el_is_magic(r_val, EL_RHYTHM_MAGIC)) return after_val; - el_rhythm_t* r = (el_rhythm_t*)(uintptr_t)r_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t out = _el_next_after(r, (int64_t)after_val, c); - return (el_val_t)out; -} - -el_val_t rhythm_matches(el_val_t r_val, el_val_t ct_val) { - if (!el_is_magic(r_val, EL_RHYTHM_MAGIC)) return (el_val_t)0; - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_rhythm_t* r = (el_rhythm_t*)(uintptr_t)r_val; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - int64_t period = ct->cal->cycle_period_ns > 0 ? ct->cal->cycle_period_ns : EL_EARTH_DAY_NS; - int64_t base = ct->instant_ns - ct->cal->epoch_ns; - int64_t phase_ns = base % period; - if (phase_ns < 0) phase_ns += period; - double phase = (double)phase_ns / (double)period; - switch (r->kind) { - case EL_RHYTHM_CYCLE_START: return (el_val_t)(phase_ns == 0 ? 1 : 0); - case EL_RHYTHM_CYCLE_PHASE: { - double diff = phase - r->phase; - if (diff < 0) diff = -diff; - return (el_val_t)(diff < 0.001 ? 1 : 0); - } - default: return (el_val_t)0; - } -} - -/* ── UUID v4 ─────────────────────────────────────────────────────────────── */ - -static int _el_uuid_seeded = 0; - -static void _el_uuid_seed(void) { - if (!_el_uuid_seeded) { - srand((unsigned)time(NULL) ^ (unsigned)(uintptr_t)&_el_uuid_seeded); - _el_uuid_seeded = 1; - } -} - -el_val_t uuid_new(void) { - _el_uuid_seed(); - unsigned char b[16]; - for (int i = 0; i < 16; i++) b[i] = (unsigned char)(rand() & 0xff); - /* Version 4 */ - b[6] = (b[6] & 0x0f) | 0x40; - /* RFC 4122 variant */ - b[8] = (b[8] & 0x3f) | 0x80; - char buf[37]; - snprintf(buf, sizeof(buf), - "%02x%02x%02x%02x-%02x%02x-%02x%02x-%02x%02x-%02x%02x%02x%02x%02x%02x", - b[0], b[1], b[2], b[3], - b[4], b[5], - b[6], b[7], - b[8], b[9], - b[10], b[11], b[12], b[13], b[14], b[15]); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t uuid_v4(void) { return uuid_new(); } - -/* ── Environment ─────────────────────────────────────────────────────────── */ - -el_val_t env(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - const char* v = getenv(k); - return el_wrap_str(el_strdup(v ? v : "")); -} - -/* ── In-process state K/V ────────────────────────────────────────────────── */ - -typedef struct { - char* key; - char* value; -} StateEntry; - -static StateEntry* _state_entries = NULL; -static size_t _state_count = 0; -static size_t _state_cap = 0; -/* Mutex protecting all _state_entries access. state_set/state_get are called - * concurrently from 64 HTTP worker threads — without this lock, realloc and - * free race, producing corruption, double-free, and segfaults. */ -static pthread_mutex_t _state_mu = PTHREAD_MUTEX_INITIALIZER; - -static StateEntry* state_find(const char* key) { - for (size_t i = 0; i < _state_count; i++) { - if (strcmp(_state_entries[i].key, key) == 0) return &_state_entries[i]; - } - return NULL; -} - -el_val_t state_set(el_val_t key, el_val_t value) { - const char* k = EL_CSTR(key); - const char* v = EL_CSTR(value); - if (!k) return 0; - if (!v) v = ""; - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - if (e) { - free(e->value); - e->value = el_strdup_persist(v); - pthread_mutex_unlock(&_state_mu); - return 1; - } - if (_state_count >= _state_cap) { - size_t nc = _state_cap == 0 ? 16 : _state_cap * 2; - StateEntry* grown = realloc(_state_entries, nc * sizeof(StateEntry)); - if (!grown) { pthread_mutex_unlock(&_state_mu); fputs("el_runtime: out of memory\n", stderr); exit(1); } - _state_entries = grown; - _state_cap = nc; - } - _state_entries[_state_count].key = el_strdup_persist(k); - _state_entries[_state_count].value = el_strdup_persist(v); - _state_count++; - pthread_mutex_unlock(&_state_mu); - return 1; -} - -el_val_t state_get(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - char* result = el_strdup_persist(e ? e->value : ""); - pthread_mutex_unlock(&_state_mu); - /* wrap in arena-tracked copy for the caller's request lifetime */ - char* copy = el_strdup(result); - return el_wrap_str(copy); -} - -el_val_t state_del(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return 0; - pthread_mutex_lock(&_state_mu); - for (size_t i = 0; i < _state_count; i++) { - if (strcmp(_state_entries[i].key, k) == 0) { - free(_state_entries[i].key); - free(_state_entries[i].value); - for (size_t j = i + 1; j < _state_count; j++) { - _state_entries[j - 1] = _state_entries[j]; - } - _state_count--; - pthread_mutex_unlock(&_state_mu); - return 1; - } - } - pthread_mutex_unlock(&_state_mu); - return 1; -} - -el_val_t state_keys(void) { - pthread_mutex_lock(&_state_mu); - /* Build a JSON array string: ["key1","key2",...] */ - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - for (size_t i = 0; i < _state_count; i++) { - if (i > 0) jb_putc(&b, ','); - jb_putc(&b, '"'); - jb_emit_escaped(&b, _state_entries[i].key); - jb_putc(&b, '"'); - } - jb_putc(&b, ']'); - pthread_mutex_unlock(&_state_mu); - return el_wrap_str(jb_finish(&b)); -} - -/* Returns 1 (true) if the key is present in the state store, else 0 (false). */ -el_val_t state_has(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return 0; - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - int found = (e != NULL) ? 1 : 0; - pthread_mutex_unlock(&_state_mu); - return (el_val_t)found; -} - -/* Returns the value for key, or default_val if the key is absent. */ -el_val_t state_get_or(el_val_t key, el_val_t default_val) { - const char* k = EL_CSTR(key); - if (!k) return default_val; - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - if (e) { - char* copy = el_strdup(e->value); - pthread_mutex_unlock(&_state_mu); - return el_wrap_str(copy); - } - pthread_mutex_unlock(&_state_mu); - return default_val; -} - -/* ── Float formatting ────────────────────────────────────────────────────── */ - -el_val_t float_to_str(el_val_t f) { - char buf[64]; - double v = el_to_float(f); - /* Normalize NaN to "nan" regardless of sign — platform-independent. */ - if (isnan(v)) { - snprintf(buf, sizeof(buf), "nan"); - } else { - snprintf(buf, sizeof(buf), "%g", v); - } - return el_wrap_str(el_strdup(buf)); -} - -el_val_t int_to_float(el_val_t n) { - return el_from_float((double)(int64_t)n); -} - -el_val_t float_to_int(el_val_t f) { - return (el_val_t)(int64_t)el_to_float(f); -} - -el_val_t format_float(el_val_t f, el_val_t decimals) { - int d = (int)(int64_t)decimals; - if (d < 0) d = 0; - if (d > 30) d = 30; - char buf[128]; - snprintf(buf, sizeof(buf), "%.*f", d, el_to_float(f)); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t decimal_round(el_val_t f, el_val_t decimals) { - int d = (int)(int64_t)decimals; - if (d < 0) d = 0; - if (d > 15) d = 15; - double mul = pow(10.0, (double)d); - double v = el_to_float(f); - double r = (v >= 0.0 ? floor(v * mul + 0.5) : -floor(-v * mul + 0.5)) / mul; - return el_from_float(r); -} - -el_val_t str_to_float(el_val_t s) { - const char* str = EL_CSTR(s); - if (!str) return el_from_float(0.0); - return el_from_float(strtod(str, NULL)); -} - -/* ── Math (Float-aware) ──────────────────────────────────────────────────── */ - -el_val_t math_sqrt(el_val_t f) { return el_from_float(sqrt(el_to_float(f))); } -el_val_t math_log(el_val_t f) { return el_from_float(log(el_to_float(f))); } -el_val_t math_ln(el_val_t f) { return el_from_float(log(el_to_float(f))); } -el_val_t math_sin(el_val_t f) { return el_from_float(sin(el_to_float(f))); } -el_val_t math_cos(el_val_t f) { return el_from_float(cos(el_to_float(f))); } -el_val_t math_pi(void) { return el_from_float(3.141592653589793238462643383279502884); } - -/* ── String additions ────────────────────────────────────────────────────── */ - -el_val_t str_index_of(el_val_t s, el_val_t sub) { - const char* str = EL_CSTR(s); - const char* sb = EL_CSTR(sub); - if (!str || !sb) return -1; - const char* hit = strstr(str, sb); - if (!hit) return -1; - return (el_val_t)(int64_t)(hit - str); -} - -el_val_t str_split(el_val_t s, el_val_t sep) { - const char* str = EL_CSTR(s); - const char* sp = EL_CSTR(sep); - el_val_t lst = el_list_empty(); - if (!str) return lst; - if (!sp || !*sp) { - lst = el_list_append(lst, el_wrap_str(el_strdup(str))); - return lst; - } - size_t lp = strlen(sp); - const char* p = str; - const char* hit; - while ((hit = strstr(p, sp)) != NULL) { - size_t n = (size_t)(hit - p); - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - p = hit + lp; - } - lst = el_list_append(lst, el_wrap_str(el_strdup(p))); - return lst; -} - -el_val_t str_char_at(el_val_t s, el_val_t i) { - const char* str = EL_CSTR(s); - int64_t idx = (int64_t)i; - if (!str) return el_wrap_str(el_strdup("")); - int64_t n = (int64_t)strlen(str); - if (idx < 0 || idx >= n) return el_wrap_str(el_strdup("")); - char buf[2]; - buf[0] = str[idx]; - buf[1] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -el_val_t str_char_code(el_val_t s, el_val_t i) { - const char* str = EL_CSTR(s); - int64_t idx = (int64_t)i; - if (!str) return 0; - int64_t n = (int64_t)strlen(str); - if (idx < 0 || idx >= n) return 0; - return (el_val_t)(unsigned char)str[idx]; -} - -static el_val_t str_pad(const char* s, int64_t width, const char* pad, int left) { - if (!s) s = ""; - if (!pad || !*pad) pad = " "; - int64_t lp = (int64_t)strlen(pad); - int64_t ls = (int64_t)strlen(s); - if (ls >= width) return el_wrap_str(el_strdup(s)); - int64_t need = width - ls; - char* out = el_strbuf((size_t)width); - if (left) { - for (int64_t i = 0; i < need; i++) out[i] = pad[i % lp]; - memcpy(out + need, s, (size_t)ls); - } else { - memcpy(out, s, (size_t)ls); - for (int64_t i = 0; i < need; i++) out[ls + i] = pad[i % lp]; - } - out[width] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad) { - return str_pad(EL_CSTR(s), (int64_t)width, EL_CSTR(pad), 1); -} - -el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad) { - return str_pad(EL_CSTR(s), (int64_t)width, EL_CSTR(pad), 0); -} - -el_val_t str_format(el_val_t fmt, el_val_t data) { - const char* tpl = EL_CSTR(fmt); - if (!tpl) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - const char* p = tpl; - while (*p) { - if (*p == '{') { - const char* q = p + 1; - while (*q && *q != '}') q++; - if (*q == '}') { - size_t klen = (size_t)(q - p - 1); - char keybuf[256]; - if (klen < sizeof(keybuf)) { - memcpy(keybuf, p + 1, klen); - keybuf[klen] = '\0'; - el_val_t v = el_map_get(data, EL_STR(keybuf)); - if (v != 0 && looks_like_string(v)) { - jb_puts(&b, EL_CSTR(v)); - p = q + 1; - continue; - } else if (v != 0) { - jb_emit_int(&b, (int64_t)v); - p = q + 1; - continue; - } - } - /* Unknown key — leave {key} verbatim */ - jb_reserve(&b, klen + 2); - memcpy(b.buf + b.len, p, klen + 2); - b.len += klen + 2; - b.buf[b.len] = '\0'; - p = q + 1; - continue; - } - } - jb_putc(&b, *p); - p++; - } - return el_wrap_str(jb_finish(&b)); -} - -el_val_t str_lower(el_val_t s) { return str_to_lower(s); } -el_val_t str_upper(el_val_t s) { return str_to_upper(s); } - -/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes) - * - * Phase 1 covers the operations every text-handling caller used to roll by - * hand on top of str_index_of + str_slice. The character-class predicates - * (is_letter / is_digit / ...) are ASCII only — Unicode-grapheme awareness, - * NFC/NFD normalization, and regex are Phase 2. Single-char input checks the - * first byte; multi-char input requires ALL bytes to match (false otherwise). - * - * Counting: - * str_count non-overlapping occurrences of sub in s - * str_count_chars codepoint count (UTF-8 leading-byte count) - * str_count_bytes explicit byte length (alias of str_len) - * str_count_lines \n-delimited line count (\r\n folded to \n) - * str_count_words whitespace-delimited tokens, non-empty only - * str_count_letters ASCII [A-Za-z] - * str_count_digits ASCII [0-9] - * - * Find / position: - * str_index_of_all all byte offsets of sub, [] if none - * str_last_index_of last byte offset of sub, -1 if not found - * str_find_chars first index of any char in any_of, -1 if none - * - * Transform: - * str_repeat s * n (non-negative) - * str_reverse codepoint-reversed (NOT grapheme-aware) - * str_strip_prefix s without prefix if present, else s - * str_strip_suffix s without suffix if present, else s - * str_strip_chars strip leading+trailing chars matching any in chars - * str_lstrip strip leading whitespace - * str_rstrip strip trailing whitespace - * - * Char classification (Bool): - * is_letter, is_digit, is_alphanumeric, is_whitespace, - * is_punctuation, is_uppercase, is_lowercase - * - * Splitting: - * str_split_lines \n-delimited (\r\n folded). Trailing empty dropped. - * str_split_chars alias of native_string_chars in str_ namespace - * str_split_n split into at most n parts (last part keeps the - * rest verbatim, including any further separators) - * - * Joining: - * str_join [String] -> String, sep between elements - */ - -/* Count non-overlapping occurrences of sub in s. Empty sub returns 0. */ -el_val_t str_count(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub || !*sub) return 0; - size_t lp = strlen(sub); - int64_t count = 0; - const char* p = s; - while ((p = strstr(p, sub)) != NULL) { - count++; - p += lp; /* non-overlapping advance */ - } - return (el_val_t)count; -} - -/* Codepoint count: walk bytes, count those NOT matching 10xxxxxx. */ -el_val_t str_count_chars(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if ((*p & 0xC0) != 0x80) count++; - } - return (el_val_t)count; -} - -el_val_t str_count_bytes(el_val_t sv) { - return str_len(sv); -} - -el_val_t str_count_lines(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s || !*s) return 0; - int64_t count = 0; - int has_content = 0; - for (const char* p = s; *p; p++) { - has_content = 1; - if (*p == '\n') { - count++; - has_content = 0; /* the \n closed the line */ - } - } - if (has_content) count++; /* trailing line with no terminator */ - return (el_val_t)count; -} - -el_val_t str_count_words(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - int in_word = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (isspace(*p)) { - in_word = 0; - } else if (!in_word) { - in_word = 1; - count++; - } - } - return (el_val_t)count; -} - -el_val_t str_count_letters(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if ((*p >= 'A' && *p <= 'Z') || (*p >= 'a' && *p <= 'z')) count++; - } - return (el_val_t)count; -} - -el_val_t str_count_digits(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (*p >= '0' && *p <= '9') count++; - } - return (el_val_t)count; -} - -el_val_t str_index_of_all(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - el_val_t lst = el_list_empty(); - if (!s || !sub || !*sub) return lst; - size_t lp = strlen(sub); - const char* p = s; - const char* hit; - while ((hit = strstr(p, sub)) != NULL) { - lst = el_list_append(lst, (el_val_t)(int64_t)(hit - s)); - p = hit + lp; - } - return lst; -} - -el_val_t str_last_index_of(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub || !*sub) return -1; - size_t lp = strlen(sub); - int64_t last = -1; - const char* p = s; - const char* hit; - while ((hit = strstr(p, sub)) != NULL) { - last = (int64_t)(hit - s); - p = hit + lp; - } - return (el_val_t)last; -} - -el_val_t str_find_chars(el_val_t sv, el_val_t any_of_v) { - const char* s = EL_CSTR(sv); - const char* any = EL_CSTR(any_of_v); - if (!s || !any || !*any) return -1; - for (const char* p = s; *p; p++) { - if (strchr(any, *p)) return (el_val_t)(int64_t)(p - s); - } - return -1; -} - -el_val_t str_repeat(el_val_t sv, el_val_t nv) { - const char* s = EL_CSTR(sv); - int64_t n = (int64_t)nv; - if (!s || n <= 0) return el_wrap_str(el_strdup("")); - size_t ls = strlen(s); - if (ls == 0) return el_wrap_str(el_strdup("")); - size_t total = ls * (size_t)n; - char* out = el_strbuf(total); - for (int64_t i = 0; i < n; i++) { - memcpy(out + i * ls, s, ls); - } - out[total] = '\0'; - return el_wrap_str(out); -} - -/* Reverse by codepoint: walk codepoints, copy each backwards into the output. - * NOT grapheme-aware (Phase 2). Combining marks attached to a base codepoint - * will detach. ASCII strings are byte-reverse equivalent. */ -el_val_t str_reverse(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - /* Walk forward, find each codepoint's byte length, then copy from the end. */ - size_t out_pos = n; - const unsigned char* p = (const unsigned char*)s; - while (*p) { - int cp_len; - if ((*p & 0x80) == 0x00) cp_len = 1; - else if ((*p & 0xE0) == 0xC0) cp_len = 2; - else if ((*p & 0xF0) == 0xE0) cp_len = 3; - else if ((*p & 0xF8) == 0xF0) cp_len = 4; - else cp_len = 1; /* invalid byte: passthrough */ - out_pos -= cp_len; - memcpy(out + out_pos, p, cp_len); - p += cp_len; - } - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_strip_prefix(el_val_t sv, el_val_t prefv) { - const char* s = EL_CSTR(sv); - const char* pref = EL_CSTR(prefv); - if (!s) return el_wrap_str(el_strdup("")); - if (!pref || !*pref) return el_wrap_str(el_strdup(s)); - size_t lp = strlen(pref); - size_t ls = strlen(s); - if (lp <= ls && strncmp(s, pref, lp) == 0) { - char* out = el_strbuf(ls - lp); - memcpy(out, s + lp, ls - lp); - out[ls - lp] = '\0'; - return el_wrap_str(out); - } - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_strip_suffix(el_val_t sv, el_val_t sufv) { - const char* s = EL_CSTR(sv); - const char* suf = EL_CSTR(sufv); - if (!s) return el_wrap_str(el_strdup("")); - if (!suf || !*suf) return el_wrap_str(el_strdup(s)); - size_t ls = strlen(s); - size_t lsuf = strlen(suf); - if (lsuf <= ls && strcmp(s + ls - lsuf, suf) == 0) { - char* out = el_strbuf(ls - lsuf); - memcpy(out, s, ls - lsuf); - out[ls - lsuf] = '\0'; - return el_wrap_str(out); - } - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_strip_chars(el_val_t sv, el_val_t charsv) { - const char* s = EL_CSTR(sv); - const char* chars = EL_CSTR(charsv); - if (!s) return el_wrap_str(el_strdup("")); - if (!chars || !*chars) return el_wrap_str(el_strdup(s)); - const char* start = s; - while (*start && strchr(chars, *start)) start++; - size_t n = strlen(start); - while (n > 0 && strchr(chars, start[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, start, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_lstrip(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - while (*s && isspace((unsigned char)*s)) s++; - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_rstrip(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - while (n > 0 && isspace((unsigned char)s[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, s, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -/* Character classification. - * Empty input returns false. Multi-char input requires ALL bytes to match. - * ASCII range only; Phase 2 will widen to Unicode. */ -static int s_all_match(el_val_t sv, int (*pred)(unsigned char)) { - const char* s = EL_CSTR(sv); - if (!s || !*s) return 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (!pred(*p)) return 0; - } - return 1; -} - -static int p_letter(unsigned char c) { return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z'); } -static int p_digit(unsigned char c) { return c >= '0' && c <= '9'; } -static int p_alnum(unsigned char c) { return p_letter(c) || p_digit(c); } -static int p_white(unsigned char c) { return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f' || c == '\v'; } -static int p_punct(unsigned char c) { return ispunct(c) ? 1 : 0; } -static int p_upper(unsigned char c) { return c >= 'A' && c <= 'Z'; } -static int p_lower(unsigned char c) { return c >= 'a' && c <= 'z'; } - -el_val_t is_letter(el_val_t s) { return (el_val_t)s_all_match(s, p_letter); } -el_val_t is_digit(el_val_t s) { return (el_val_t)s_all_match(s, p_digit); } -el_val_t is_alphanumeric(el_val_t s) { return (el_val_t)s_all_match(s, p_alnum); } -el_val_t is_whitespace(el_val_t s) { return (el_val_t)s_all_match(s, p_white); } -el_val_t is_punctuation(el_val_t s) { return (el_val_t)s_all_match(s, p_punct); } -el_val_t is_uppercase(el_val_t s) { return (el_val_t)s_all_match(s, p_upper); } -el_val_t is_lowercase(el_val_t s) { return (el_val_t)s_all_match(s, p_lower); } - -/* Split on \n. \r\n is folded to \n first. Trailing empty after final \n - * is dropped (so "a\nb\n" -> ["a", "b"], not ["a", "b", ""]). */ -el_val_t str_split_lines(el_val_t sv) { - const char* s = EL_CSTR(sv); - el_val_t lst = el_list_empty(); - if (!s) return lst; - size_t n = strlen(s); - /* Pre-scan: build into a normalized buffer with \r\n folded. */ - const char* line_start = s; - for (size_t i = 0; i <= n; i++) { - if (s[i] == '\n' || s[i] == '\0') { - size_t len = (size_t)(s + i - line_start); - /* Drop trailing \r if this was \r\n. */ - if (len > 0 && line_start[len - 1] == '\r') len--; - /* Drop final trailing-empty-after-newline. */ - if (s[i] == '\0' && len == 0 && i > 0 && s[i - 1] == '\n') break; - char* out = el_strbuf(len); - memcpy(out, line_start, len); - out[len] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - if (s[i] == '\0') break; - line_start = s + i + 1; - } - } - return lst; -} - -el_val_t str_split_chars(el_val_t s) { - return native_string_chars(s); -} - -/* Split into at most n parts. The (n-1)th split point is the LAST split; - * after it, the remainder is appended verbatim including any further - * separators. n <= 0 returns an empty list. n == 1 returns [s]. */ -el_val_t str_split_n(el_val_t sv, el_val_t sepv, el_val_t nv) { - const char* s = EL_CSTR(sv); - const char* sep = EL_CSTR(sepv); - int64_t n = (int64_t)nv; - el_val_t lst = el_list_empty(); - if (!s) return lst; - if (n <= 0) return lst; - if (n == 1 || !sep || !*sep) { - lst = el_list_append(lst, el_wrap_str(el_strdup(s))); - return lst; - } - size_t lp = strlen(sep); - const char* p = s; - int64_t parts = 0; - const char* hit; - while (parts < n - 1 && (hit = strstr(p, sep)) != NULL) { - size_t len = (size_t)(hit - p); - char* out = el_strbuf(len); - memcpy(out, p, len); - out[len] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - p = hit + lp; - parts++; - } - /* Remainder verbatim. */ - lst = el_list_append(lst, el_wrap_str(el_strdup(p))); - return lst; -} - -/* Join a [String] with a separator. Empty list -> "". Single-element -> - * that element. Non-string elements are stringified via int_to_str. */ -el_val_t str_join(el_val_t listv, el_val_t sepv) { - return list_join(listv, sepv); -} - -/* ── List additions ──────────────────────────────────────────────────────── */ - -el_val_t list_push(el_val_t list, el_val_t elem) { - return el_list_append(list, elem); -} - -el_val_t list_push_front(el_val_t listv, el_val_t elem) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) { - el_val_t nl = el_list_empty(); - return el_list_append(nl, elem); - } - /* Append to grow capacity, then shift right */ - listv = el_list_append(listv, elem); - lst = (ElList*)(uintptr_t)listv; - for (int64_t i = lst->length - 1; i > 0; i--) { - lst->elems[i] = lst->elems[i - 1]; - } - lst->elems[0] = elem; - return EL_STR(lst); -} - -el_val_t list_join(el_val_t listv, el_val_t sep) { - ElList* lst = (ElList*)(uintptr_t)listv; - const char* sp = EL_CSTR(sep); - if (!sp) sp = ""; - if (!lst || lst->length == 0) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - for (int64_t i = 0; i < lst->length; i++) { - if (i > 0) jb_puts(&b, sp); - el_val_t v = lst->elems[i]; - if (v == 0) continue; - if (looks_like_string(v)) { - jb_puts(&b, EL_CSTR(v)); - } else { - char tmp[32]; - snprintf(tmp, sizeof(tmp), "%lld", (long long)v); - jb_puts(&b, tmp); - } - } - return el_wrap_str(jb_finish(&b)); -} - -el_val_t list_range(el_val_t start, el_val_t end) { - int64_t a = (int64_t)start; - int64_t b = (int64_t)end; - el_val_t lst = el_list_empty(); - for (int64_t i = a; i < b; i++) lst = el_list_append(lst, (el_val_t)i); - return lst; -} - -/* ── Bool helpers ────────────────────────────────────────────────────────── */ - -el_val_t bool_to_str(el_val_t b) { - return el_wrap_str(el_strdup(b ? "true" : "false")); -} - -/* ── Numeric parsing ─────────────────────────────────────────────────────── */ - -/* parse_int — strtoll with a default. str_to_int already exists but does not - * distinguish "0" from a parse failure, so callers that need a sentinel use - * this. Skips leading whitespace; accepts an optional leading +/-; returns - * default_val on empty input or no consumed digits. Trailing junk is ignored - * (atoi-style). */ -el_val_t parse_int(el_val_t sv, el_val_t default_val) { - const char* s = EL_CSTR(sv); - if (!s) return default_val; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == '\0') return default_val; - char* end = NULL; - long long n = strtoll(s, &end, 10); - if (end == s) return default_val; - return (el_val_t)n; -} - -/* ── Process ─────────────────────────────────────────────────────────────── */ - -el_val_t exit_program(el_val_t code) { - exit((int)code); - return 0; /* unreachable */ -} - -/* getpid_now — current process id. Named with the _now suffix to avoid - * colliding with the libc `getpid` declaration that the runtime already - * sees via (calling it `getpid` would fight the prototype). */ -el_val_t getpid_now(void) { - return (el_val_t)getpid(); -} - -/* el_mem_check — self-terminating memory guard for long-running compiler runs. - * - * Call this periodically (e.g. after each function compiled) to detect runaway - * memory growth before the OS OOM-killer fires. Reads the limit from the env - * var ELC_MAX_MEM_MB (default 512 MB). If resident set size exceeds the limit, - * prints a diagnostic to stderr and exits with code 1 so the caller (elb or a - * CI script) can handle the failure gracefully instead of having the whole - * machine go down. - * - * Platform notes: - * macOS — ru_maxrss is in bytes. - * Linux — ru_maxrss is in kilobytes. - * We normalise to MB before comparing. - * - * Returns 0 always (the only non-return path is the exit() branch). - */ -el_val_t el_mem_check(void) { -#ifdef _WIN32 - /* getrusage is POSIX-only — memory guard disabled on Windows. */ - return 0; -#else - /* Read limit from env; default 512 MB. */ - long limit_mb = 512; - const char *env_val = getenv("ELC_MAX_MEM_MB"); - if (env_val && *env_val) { - long v = atol(env_val); - if (v > 0) limit_mb = v; - } - - struct rusage ru; - if (getrusage(RUSAGE_SELF, &ru) != 0) return 0; /* can't read — skip check */ - - long rss_mb; -#if defined(__APPLE__) || defined(__MACH__) - /* macOS: ru_maxrss is bytes */ - rss_mb = (long)(ru.ru_maxrss / (1024L * 1024L)); -#else - /* Linux: ru_maxrss is kilobytes */ - rss_mb = (long)(ru.ru_maxrss / 1024L); -#endif - - if (rss_mb >= limit_mb) { - fprintf(stderr, "elc: memory limit exceeded (%ldMB), aborting\n", limit_mb); - exit(1); - } - return 0; -#endif -} - -/* ── args() — command-line argument access ────────────────────────────────── - * Compiled El programs call args() to get a list of CLI arguments. - * Call el_runtime_init_args(argc, argv) at the start of C main() to populate. - * The args list excludes argv[0] (the program name). */ - -static el_val_t _el_args_list = 0; - -void el_runtime_init_args(int argc, char** argv) { - _el_args_list = el_list_empty(); - for (int i = 1; i < argc; i++) { - _el_args_list = el_list_append(_el_args_list, EL_STR(argv[i])); - } -} - -el_val_t args(void) { - if (!_el_args_list) _el_args_list = el_list_empty(); - return _el_args_list; -} - -/* ── CGI identity ──────────────────────────────────────────────────────────── - * Called once at program start by the generated main() of a cgi {} program. - * Stores CGI identity so dharma_* builtins can reference it. */ - -static const char* _el_cgi_name = NULL; -static const char* _el_cgi_dharma_id = NULL; -static const char* _el_cgi_principal = NULL; -static const char* _el_cgi_network = NULL; -static const char* _el_cgi_engram = NULL; - -void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, - el_val_t network, el_val_t engram) { - _el_cgi_name = EL_CSTR(name); - _el_cgi_dharma_id = EL_CSTR(dharma_id); - _el_cgi_principal = EL_CSTR(principal); - _el_cgi_network = EL_CSTR(network) ? EL_CSTR(network) : "dharma-mainnet"; - _el_cgi_engram = EL_CSTR(engram) ? EL_CSTR(engram) : "http://localhost:8742"; - printf("[cgi] identity: name=%s dharma_id=%s principal=%s network=%s engram=%s\n", - _el_cgi_name ? _el_cgi_name : "(unset)", - _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unset)", - _el_cgi_principal ? _el_cgi_principal : "(unset)", - _el_cgi_network, - _el_cgi_engram); -} - - -/* ── Batch 3: Engram in-process graph store ──────────────────────────────── */ -/* - * Single global EngramStore allocated lazily on first call. All node and - * edge content strings are owned (strdup'd) by the store. Linear arrays - * with doubling capacity for both nodes and edges. - * - * Two-layer activation algorithm (engram_activate): - * - * LAYER 1 — Broad fan-out (background activation): - * 1. Find seed nodes whose content/label/tags contain query (case-insens). - * 2. BFS up to `depth` hops along ALL edges (excitatory and inhibitory). - * Every reachable node fires — nothing is filtered at this layer. - * 3. bg_act = seed.salience * temporal_decay * dampening - * propagated as: new_bg = parent_bg * edge_weight * 0.7 * (1 + tbonus) - * where tbonus ∈ {0, 0.10, 0.20} for co-temporal nodes. - * 4. If reached by multiple paths, take max background_activation. - * 5. Persist background_activation to EngramNode.background_activation. - * - * LAYER 2 — Executive filter (working memory promotion): - * 6. For each inhibitory edge where source has background_activation > 0: - * inhibition[target] = max(bg[source] * e->weight) - * 7. For each background-activated node: - * raw_wm = bg * goal_bias(node, query) * confidence - * * (1 - (1 - INHIBITION_FACTOR) * inhibition) - * 8. Per-type threshold gate: raw_wm >= type_threshold → promoted. - * Safety/DharmaSelf: 0.05 Canonical: 0.15 Lesson: 0.25 - * Belief/Entity: 0.30 Note/Memory/Working: 0.40 - * 9. If not promoted: suppression_count++. After - * ENGRAM_SUPPRESSION_BREAKTHROUGH suppressions → force breakthrough - * at ENGRAM_BREAKTHROUGH_WEIGHT (latent tension surfacing). - * 10. Persist working_memory_weight to EngramNode.working_memory_weight. - * 11. Sort: promoted nodes (wm > 0) first by wm desc, then background- - * only by bg desc. Context compilation uses ONLY promoted nodes. - * - * Temporal decay: - * decay_factor = exp(-lambda * age_hours / T_half) - * T_half = 168.0 h (one week), lambda = ln(2) - * - * Activation dampening: - * dampen = 1.0 / (1.0 + log(1 + activation_count)) - * - * engram_query_range(start_ms, end_ms): - * Returns nodes whose created_at OR last_activated falls within - * [start_ms, end_ms], sorted by created_at ascending. - */ - -/* Temporal decay constants. - * T_HALF_HOURS: half-life in hours — one week. After one week of no - * activation a node retains 50% of its base salience contribution. - * DECAY_LAMBDA: ln(2) ≈ 0.693147 */ -#define ENGRAM_T_HALF_HOURS 168.0 -#define ENGRAM_DECAY_LAMBDA 0.693147 - -/* Two-layer activation constants. - * ENGRAM_WM_THRESHOLD: minimum background_activation for a node to be - * considered for working-memory promotion (layer 2 candidate gate). - * ENGRAM_WM_DECAY: per-turn decay applied to working_memory_weight for - * nodes NOT re-activated in the current turn (conversational thread - * continuity: a node promoted in turn N persists with reduced weight - * into turn N+1 without re-activation cost). - * ENGRAM_SUPPRESSION_BREAKTHROUGH: after this many consecutive suppressions - * a latent node forces itself into working memory at reduced weight, - * modelling the brain's "intrusive thought" / unresolved-tension surfacing. - * ENGRAM_BREAKTHROUGH_WEIGHT: the reduced working_memory_weight assigned - * when a suppressed node breaks through. - * ENGRAM_INHIBITION_FACTOR: multiplier applied to working_memory_weight when - * an inhibitory edge fires against a node (0 = full suppress, 0.3 = partial). */ -#define ENGRAM_WM_THRESHOLD 0.15 -#define ENGRAM_WM_DECAY 0.7 -#define ENGRAM_SUPPRESSION_BREAKTHROUGH 5 -#define ENGRAM_BREAKTHROUGH_WEIGHT 0.25 -#define ENGRAM_INHIBITION_FACTOR 0.1 - -/* ── Layered consciousness architecture ────────────────────────────────────── - * - * The engram graph is stratified into LAYERS that gate which suppressions - * apply during the executive filter pass. Layers are ordered shallow-to-deep - * by `activation_priority`; the deepest layer (priority 0, conventionally - * "safety") is the structural floor of the soul: nodes here cannot be - * silenced by inhibitory edges from any other layer. Higher layers - * (core-identity, domain-knowledge, imprint, suit) are normally - * suppressible — they participate in attentional inhibition and goal - * focus the way the prior single-graph implementation did. - * - * The five canonical layers (see engram_init_layers): - * 0. safety — structural, transparent, non-injectable, non-suppressible - * 1. core-identity — default for legacy nodes; suppressible - * 2. domain-knowledge— suppressible - * 3. imprint — runtime-injectable (an Imprint package can add/remove) - * 4. suit — runtime-injectable (a Suit overlays domain skill) - * - * Three-pass activation (engram_activate): - * Pass 1 — Background fan-out: BFS spreads activation across ALL layers - * (existing behavior preserved). Inhibitory edges propagate at - * this layer too; no filtering happens here. - * Pass 2 — Working memory promotion: type-threshold gate, goal bias, - * confidence weighting, inhibitory suppression. Inhibitory edges - * ONLY apply against nodes whose layer is `suppressible == 1`. - * Nodes in non-suppressible layers (Layer 0) ignore inhibition. - * Pass 3 — Layer 0 override: every node in a non-suppressible layer that - * received background activation has its working_memory_weight - * forced to >= ENGRAM_LAYER0_OVERRIDE_WEIGHT. The sacred fire — - * safety nodes that touched any seed unconditionally surface, - * even when the executive filter would have silenced them. - * - * Layer fields: - * suppressible : 0 → inhibitory edges are ignored against nodes in this - * layer during pass 2. Pass 3 also force-promotes them. - * 1 → standard behavior (most layers). - * transparent : 1 → emitted into the prompt context so its content shapes - * output, but filtered out of "what do you know about - * yourself?" introspection queries (engram_search and - * friends do not return transparent-layer nodes by - * default). 0 → fully visible to introspection. - * injectable : 1 → can be added/removed at runtime via engram_add_layer - * and engram_remove_layer (imprints, suits). - * 0 → built-in, fixed at engram_get() initialization. - * - * Backward compatibility: - * Nodes and edges loaded from snapshots without a `layer_id` field default - * to layer 1 (core-identity). The five canonical layers are always present. - */ -#define ENGRAM_LAYER_SAFETY 0u -#define ENGRAM_LAYER_CORE_IDENTITY 1u -#define ENGRAM_LAYER_DOMAIN 2u -#define ENGRAM_LAYER_IMPRINT 3u -#define ENGRAM_LAYER_SUIT 4u -#define ENGRAM_LAYER_ACCUMULATION 5u -/* New user-facing nodes (memories, knowledge, conversations) are created in the - * accumulation layer — the top of the consciousness stack, the engram the user - * sees; every layer below shapes behavior but is hidden from the user (Layered - * Consciousness architecture, app 64/064,262). ENGRAM_LAYER_DEFAULT stays - * core-identity ON PURPOSE: it is the fallback home for LEGACY nodes loaded from - * snapshots without a layer_id, so existing data (the originator corpus) is - * never migrated out of its established layer. New != legacy. */ -#define ENGRAM_LAYER_DEFAULT ENGRAM_LAYER_CORE_IDENTITY - -/* Pass 3 override floor. Layer 0 nodes that received any background - * activation are force-promoted to AT LEAST this working_memory_weight, - * regardless of inhibitory suppression in pass 2. */ -#define ENGRAM_LAYER0_OVERRIDE_WEIGHT 1.0 - -/* Per-node-type activation thresholds. - * Lower tier / safety-critical nodes fire more readily. */ -static double engram_type_threshold(const char* node_type, const char* tier) { - if (node_type) { - if (strcmp(node_type, "DharmaSelf") == 0) return 0.05; - if (strcmp(node_type, "Safety") == 0) return 0.05; - } - if (tier) { - if (strcmp(tier, "Canonical") == 0) return 0.15; - if (strcmp(tier, "Lesson") == 0) return 0.25; - } - if (node_type) { - /* Knowledge nodes: Canonical/Lesson handled by tier checks above. - * Procedural-tier Knowledge (activation_count>=50 migration): 0.20. - * (2026-06-29 self-review — mirrors release runtime fix) */ - if (strcmp(node_type, "Knowledge") == 0) return 0.20; - if (strcmp(node_type, "Belief") == 0) return 0.30; - if (strcmp(node_type, "Entity") == 0) return 0.30; - } - return 0.40; /* Note / Memory / Working (most nodes) */ -} - -typedef struct EngramNode { - char* id; - char* content; - char* node_type; - char* label; - char* tier; - char* tags; - char* metadata; - double salience; - double importance; - double confidence; - double temporal_decay_rate; /* per-node override for lambda; 0 = use default */ - int64_t activation_count; - int64_t last_activated; - int64_t created_at; - int64_t updated_at; - /* Two-layer activation fields ───────────────────────────────────────── - * background_activation: Layer 1. Set by BFS fan-out on every query. - * Every reachable node fires here — nothing is filtered at this stage. - * Models the brain's massive parallel sub-threshold activation of all - * associated content in response to a stimulus. - * working_memory_weight: Layer 2. Executive filter output. Only nodes - * that survive goal-state / attentional-bias scoring receive a - * non-zero weight here. Context compilation ONLY uses this field. - * Background-activated nodes with working_memory_weight == 0 remain - * latent — real, available, but silent. - * suppression_count: Consecutive turn count where this node was - * background-activated but NOT promoted to working memory. High - * values signal the node "wants to surface." After - * ENGRAM_SUPPRESSION_BREAKTHROUGH consecutive suppressions the node - * is force-promoted at a reduced weight (breakthrough activation). */ - double background_activation; - double working_memory_weight; - int32_t suppression_count; - /* Layered consciousness — see ENGRAM_LAYER_* macros and engram_init_layers. - * Defaults to ENGRAM_LAYER_DEFAULT (1, core-identity) for legacy nodes - * created via engram_node / engram_node_full and for snapshots that - * predate the layered schema. */ - uint32_t layer_id; -} EngramNode; - -typedef struct EngramEdge { - char* id; - char* from_id; - char* to_id; - char* relation; - char* metadata; - double weight; - double confidence; - int64_t created_at; - int64_t updated_at; - int64_t last_fired; - /* Inhibitory flag: when 1, activating the source node SUPPRESSES the - * working_memory_weight of the target node rather than exciting it. - * Models attentional inhibition: "I am focused on code work" creates - * inhibitory edges to personal/emotional nodes, preventing them from - * surfacing even if they have high background_activation. */ - int inhibitory; - /* Layered consciousness — edges carry a layer assignment for - * categorization/visualization. Pass 2 inhibitory gating is decided by - * the TARGET node's layer (whether it's suppressible), not by the edge - * layer. Defaults to ENGRAM_LAYER_DEFAULT. */ - uint32_t layer_id; -} EngramEdge; - -/* Layered consciousness — runtime layer registry entry. */ -typedef struct EngramLayer { - uint32_t layer_id; /* 0 = deepest (safety/limbic) */ - char* name; /* persistent — owned by the store */ - uint32_t activation_priority; /* lower = fires earlier; safety = 0 */ - int suppressible; /* can higher layers suppress nodes here? */ - int transparent; /* invisible to introspection queries? */ - int injectable; /* can be added/removed at runtime? */ -} EngramLayer; - -typedef struct EngramStore { - EngramNode* nodes; - int64_t node_count; - int64_t node_capacity; - EngramEdge* edges; - int64_t edge_count; - int64_t edge_capacity; - /* Layer registry — see engram_init_layers. The five canonical layers - * are always present; injectable layers (imprint, suit) are extended - * via engram_add_layer at runtime. layer_id values are assigned - * monotonically; removed injectable layers leave a NULL `name` slot - * (tombstone) so existing layer_id references on nodes stay stable. */ - EngramLayer* layers; - size_t layer_count; - size_t layer_capacity; -} EngramStore; - -static EngramStore* engram_global = NULL; - -/* Initialize the five canonical layers on a fresh store. Called once from - * engram_get(). Layer ids 0..4 are reserved; runtime-injected imprint/suit - * layers (engram_add_layer) get ids 5+. */ -static void engram_init_layers(EngramStore* g) { - g->layer_capacity = 16; - g->layers = calloc(g->layer_capacity, sizeof(EngramLayer)); - if (!g->layers) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - g->layer_count = 0; - - /* Layer 0 — safety. Structural floor. Non-suppressible; transparent - * (filtered out of introspection but still shapes output); not - * runtime-injectable. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_SAFETY, - .name = el_strdup_persist("safety"), - .activation_priority = 0, - .suppressible = 0, - .transparent = 1, - .injectable = 0 - }; - /* Layer 1 — core-identity. The default home for legacy nodes. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_CORE_IDENTITY, - .name = el_strdup_persist("core-identity"), - .activation_priority = 10, - .suppressible = 1, - .transparent = 0, - .injectable = 0 - }; - /* Layer 2 — domain-knowledge. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_DOMAIN, - .name = el_strdup_persist("domain-knowledge"), - .activation_priority = 20, - .suppressible = 1, - .transparent = 0, - .injectable = 0 - }; - /* Layer 3 — imprint. Injectable: an imprint package adds/removes this - * layer (and the nodes assigned to it) as a unit. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_IMPRINT, - .name = el_strdup_persist("imprint"), - .activation_priority = 30, - .suppressible = 1, - .transparent = 0, - .injectable = 1 - }; - /* Layer 4 — suit. Injectable: a Suit overlays domain skill (e.g. - * "enterprise advisor", "divorce lawyer") and can be detached. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_SUIT, - .name = el_strdup_persist("suit"), - .activation_priority = 40, - .suppressible = 1, - .transparent = 0, - .injectable = 1 - }; - /* Layer 5 — accumulation. The TOP of the consciousness stack: the default - * home for all new user-facing nodes. This is the engram the user sees; - * every layer below shapes behavior but is hidden from the user. Not - * injectable — it is the persistent user accumulation, not a swappable - * overlay. transparent=0: its content is surfaced to introspection (it is - * the user's own knowledge/memory), unlike the lower behavioral layers. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_ACCUMULATION, - .name = el_strdup_persist("accumulation"), - .activation_priority = 50, - .suppressible = 1, - .transparent = 0, - .injectable = 0 - }; -} - -static EngramStore* engram_get(void) { - if (engram_global) return engram_global; - engram_global = calloc(1, sizeof(EngramStore)); - if (!engram_global) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - engram_global->node_capacity = 16; - engram_global->nodes = calloc((size_t)engram_global->node_capacity, sizeof(EngramNode)); - engram_global->edge_capacity = 16; - engram_global->edges = calloc((size_t)engram_global->edge_capacity, sizeof(EngramEdge)); - engram_init_layers(engram_global); - return engram_global; -} - -/* Resolve a layer record by id. Returns NULL if no layer with that id - * exists (e.g. a removed injectable layer or a malformed snapshot). */ -static EngramLayer* engram_find_layer(uint32_t layer_id) { - EngramStore* g = engram_get(); - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; /* tombstone for removed injectable layer */ - if (L->layer_id == layer_id) return L; - } - return NULL; -} - -/* Resolve a layer record by name. Returns NULL if not found. */ -static EngramLayer* engram_find_layer_by_name(const char* name) { - if (!name || !*name) return NULL; - EngramStore* g = engram_get(); - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if (strcmp(L->name, name) == 0) return L; - } - return NULL; -} - -/* Allocate the next layer id. Skips ids that are still in use. */ -static uint32_t engram_next_layer_id(void) { - EngramStore* g = engram_get(); - uint32_t maxid = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].layer_id > maxid) maxid = g->layers[i].layer_id; - } - return maxid + 1; -} - -/* Whether a node in `layer_id` may be silenced by inhibitory edges in pass 2. */ -static int engram_layer_is_suppressible(uint32_t layer_id) { - EngramLayer* L = engram_find_layer(layer_id); - if (!L) return 1; /* unknown layer → safe default: standard suppression */ - return L->suppressible ? 1 : 0; -} - -/* Whether a layer is transparent (its content shapes output but is filtered - * from introspection queries). Currently used to mark Layer 0 as invisible - * to "what do you know about yourself" lookups while still letting it - * dominate the prompt context. */ -static int engram_layer_is_transparent(uint32_t layer_id) { - EngramLayer* L = engram_find_layer(layer_id); - if (!L) return 0; - return L->transparent ? 1 : 0; -} - -static int64_t engram_now_ms(void) { - struct timeval tv; gettimeofday(&tv, NULL); - return (int64_t)tv.tv_sec * 1000LL + (int64_t)tv.tv_usec / 1000LL; -} - -static EngramNode* engram_find_node(const char* id) { - if (!id) return NULL; - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].id && strcmp(g->nodes[i].id, id) == 0) return &g->nodes[i]; - } - return NULL; -} - -static int64_t engram_find_node_index(const char* id) { - if (!id) return -1; - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].id && strcmp(g->nodes[i].id, id) == 0) return i; - } - return -1; -} - -static void engram_grow_nodes(void) { - EngramStore* g = engram_get(); - if (g->node_count < g->node_capacity) return; - int64_t nc = g->node_capacity * 2; - g->nodes = realloc(g->nodes, (size_t)nc * sizeof(EngramNode)); - if (!g->nodes) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(g->nodes + g->node_capacity, 0, - (size_t)(nc - g->node_capacity) * sizeof(EngramNode)); - g->node_capacity = nc; -} - -static void engram_grow_edges(void) { - EngramStore* g = engram_get(); - if (g->edge_count < g->edge_capacity) return; - int64_t nc = g->edge_capacity * 2; - g->edges = realloc(g->edges, (size_t)nc * sizeof(EngramEdge)); - if (!g->edges) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(g->edges + g->edge_capacity, 0, - (size_t)(nc - g->edge_capacity) * sizeof(EngramEdge)); - g->edge_capacity = nc; -} - -/* Build a fresh UUID string. Reuses uuid_new but takes the underlying char*. */ -static char* engram_new_id(void) { - el_val_t v = uuid_new(); - const char* s = EL_CSTR(v); - /* Persistent: node ids live in the global store; an arena (el_strdup) id is - * freed at el_request_end(), corrupting the node after the creating request. */ - return el_strdup_persist(s ? s : ""); -} - -/* Convert a node into an ElMap of its fields. */ -static el_val_t engram_node_to_map(const EngramNode* n) { - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("id")), EL_STR(el_strdup(n->id ? n->id : ""))); - m = el_map_set(m, EL_STR(el_strdup("content")), EL_STR(el_strdup(n->content ? n->content : ""))); - m = el_map_set(m, EL_STR(el_strdup("node_type")), EL_STR(el_strdup(n->node_type ? n->node_type : ""))); - m = el_map_set(m, EL_STR(el_strdup("label")), EL_STR(el_strdup(n->label ? n->label : ""))); - m = el_map_set(m, EL_STR(el_strdup("tier")), EL_STR(el_strdup(n->tier ? n->tier : "Working"))); - m = el_map_set(m, EL_STR(el_strdup("tags")), EL_STR(el_strdup(n->tags ? n->tags : ""))); - m = el_map_set(m, EL_STR(el_strdup("metadata")), EL_STR(el_strdup(n->metadata ? n->metadata : "{}"))); - m = el_map_set(m, EL_STR(el_strdup("salience")), el_from_float(n->salience)); - m = el_map_set(m, EL_STR(el_strdup("importance")), el_from_float(n->importance)); - m = el_map_set(m, EL_STR(el_strdup("confidence")), el_from_float(n->confidence)); - m = el_map_set(m, EL_STR(el_strdup("temporal_decay_rate")), el_from_float(n->temporal_decay_rate)); - m = el_map_set(m, EL_STR(el_strdup("activation_count")), (el_val_t)n->activation_count); - m = el_map_set(m, EL_STR(el_strdup("last_activated")), (el_val_t)n->last_activated); - m = el_map_set(m, EL_STR(el_strdup("created_at")), (el_val_t)n->created_at); - m = el_map_set(m, EL_STR(el_strdup("updated_at")), (el_val_t)n->updated_at); - m = el_map_set(m, EL_STR(el_strdup("background_activation")), el_from_float(n->background_activation)); - m = el_map_set(m, EL_STR(el_strdup("working_memory_weight")), el_from_float(n->working_memory_weight)); - m = el_map_set(m, EL_STR(el_strdup("suppression_count")), (el_val_t)n->suppression_count); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), (el_val_t)(int64_t)n->layer_id); - return m; -} - -/* (Node JSON serialization is provided by `engram_emit_node_json` further - * down in the persistence section — reused by the *_json builtins below.) */ -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n); -static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e); - -/* Salience may arrive either as a float bit-pattern or as a small integer - * (e.g. 1, meaning 1.0). Heuristic: if interpreted as double it's in - * [0.0, 100.0] use it; otherwise treat as int and convert. */ -static double engram_decode_score(el_val_t v) { - double f = el_to_float(v); - if (!isnan(f) && !isinf(f) && f >= 0.0 && f <= 100.0) return f; - int64_t n = (int64_t)v; - return (double)n; -} - -static char* engram_first_n_chars(const char* s, size_t n) { - if (!s) return el_strdup(""); - size_t l = strlen(s); - if (l > n) l = n; - char* out = el_strbuf(l); - memcpy(out, s, l); - out[l] = '\0'; - return out; -} - -el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - n->content = el_strdup_persist(c ? c : ""); - n->node_type = el_strdup_persist(nt && *nt ? nt : "Memory"); - { char* _lb = engram_first_n_chars(c, 60); n->label = el_strdup_persist(_lb); } /* persist: stored field must outlive request arena */ - n->tier = el_strdup_persist("Working"); - n->tags = el_strdup_persist(""); - n->metadata = el_strdup_persist("{}"); - n->salience = engram_decode_score(salience); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - n->importance = 0.5; - n->confidence = 1.0; - n->temporal_decay_rate = 0.0; /* 0 = use global default ENGRAM_DECAY_LAMBDA */ - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - n->layer_id = ENGRAM_LAYER_ACCUMULATION; /* new user-facing node → top layer */ - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -/* engram_is_valid_utf8 — return 1 if s is valid UTF-8, 0 if it contains invalid bytes. - * Rejects overlong encodings, surrogate halves, and byte sequences > 4 bytes. */ -static int engram_is_valid_utf8(const char* s) { - if (!s) return 1; - const unsigned char* p = (const unsigned char*)s; - while (*p) { - if (*p < 0x80) { - /* ASCII */ - p++; - } else if ((*p & 0xE0) == 0xC0) { - /* 2-byte sequence */ - if ((p[1] & 0xC0) != 0x80) return 0; - if ((*p & 0xFE) == 0xC0) return 0; /* overlong */ - p += 2; - } else if ((*p & 0xF0) == 0xE0) { - /* 3-byte sequence */ - if ((p[1] & 0xC0) != 0x80 || (p[2] & 0xC0) != 0x80) return 0; - if (*p == 0xE0 && (p[1] & 0xE0) == 0x80) return 0; /* overlong */ - if (*p == 0xED && (p[1] & 0xE0) == 0xA0) return 0; /* surrogate */ - p += 3; - } else if ((*p & 0xF8) == 0xF0) { - /* 4-byte sequence */ - if ((p[1] & 0xC0) != 0x80 || (p[2] & 0xC0) != 0x80 || (p[3] & 0xC0) != 0x80) return 0; - if (*p == 0xF0 && (p[1] & 0xF0) == 0x80) return 0; /* overlong */ - if (*p > 0xF4) return 0; /* above U+10FFFF */ - p += 4; - } else { - return 0; - } - } - return 1; -} - -el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t importance, el_val_t confidence, - el_val_t tier, el_val_t tags) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - const char* lb = EL_CSTR(label); - const char* ti = EL_CSTR(tier); - const char* tg = EL_CSTR(tags); - /* UTF-8 guard: reject content with invalid UTF-8 bytes. Persisting invalid - * UTF-8 garbles JSON snapshots and corrupts every subsequent node read. */ - if (c && !engram_is_valid_utf8(c)) { - fprintf(stderr, "[engram] REJECTED node write — content contains invalid UTF-8 (label=%s)\n", - lb ? lb : "(null)"); - return EL_STR(""); - } - /* Persistent (el_strdup_persist, NOT el_strdup): these strings are owned by the - * persistent global node store. el_strdup tracks into the per-request arena, which - * el_request_end() frees when the creating HTTP request completes — leaving the - * stored node with dangling pointers (corrupted ids, "saved but never listed"). - * This is the root cause of the hallucinated/lost-saves class of bugs. */ - n->content = el_strdup_persist(c ? c : ""); - n->node_type = el_strdup_persist(nt && *nt ? nt : "Memory"); - n->label = el_strdup_persist(lb && *lb ? lb : (c ? engram_first_n_chars(c, 60) : "")); - n->tier = el_strdup_persist(ti && *ti ? ti : "Working"); - n->tags = el_strdup_persist(tg ? tg : ""); - n->metadata = el_strdup_persist("{}"); - n->salience = engram_decode_score(salience); - n->importance = engram_decode_score(importance); - n->confidence = engram_decode_score(confidence); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - if (n->importance <= 0.0 || n->importance > 1.0) n->importance = 0.5; - if (n->confidence <= 0.0 || n->confidence > 1.0) n->confidence = 1.0; - n->temporal_decay_rate = 0.0; /* 0 = use global default ENGRAM_DECAY_LAMBDA */ - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - n->layer_id = ENGRAM_LAYER_ACCUMULATION; /* new user-facing node → top layer */ - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -/* engram_node_layered — like engram_node_full but with explicit layer - * assignment and an additional `status` slot reserved for callers that - * track lifecycle state in metadata. The signature mirrors the public API - * defined in the layered consciousness design doc: - * - * engram_node_layered(content, node_type, label, - * salience, certainty, confidence, - * status, tags, layer_id) - * - * `certainty` is folded into `importance` (it occupies the same axis in - * the existing schema). `status` is recorded under metadata.status; an - * empty status leaves metadata as the default "{}". - * - * If `layer_id` does not resolve to a known layer the call falls back to - * ENGRAM_LAYER_DEFAULT — better to keep the node addressable than to drop - * it because of a stale layer reference. Callers wanting strict validation - * should engram_list_layers first. */ -el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t certainty, el_val_t confidence, - el_val_t status, el_val_t tags, el_val_t layer_id) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - const char* lb = EL_CSTR(label); - const char* tg = EL_CSTR(tags); - const char* st = EL_CSTR(status); - n->content = el_strdup_persist(c ? c : ""); - n->node_type = el_strdup_persist(nt && *nt ? nt : "Memory"); - n->label = el_strdup_persist(lb && *lb ? lb : (c ? engram_first_n_chars(c, 60) : "")); - n->tier = el_strdup_persist("Working"); - n->tags = el_strdup_persist(tg ? tg : ""); - if (st && *st) { - /* Minimal metadata payload: {"status":"..."}. Keep it cheap so - * callers using `status` don't pay JSON parse cost on every read. */ - size_t sl = strlen(st) + 16; - char* meta = el_strbuf(sl); - snprintf(meta, sl, "{\"status\":\"%s\"}", st); - n->metadata = meta; - } else { - n->metadata = el_strdup_persist("{}"); - } - n->salience = engram_decode_score(salience); - n->importance = engram_decode_score(certainty); - n->confidence = engram_decode_score(confidence); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - if (n->importance <= 0.0 || n->importance > 1.0) n->importance = 0.5; - if (n->confidence <= 0.0 || n->confidence > 1.0) n->confidence = 1.0; - n->temporal_decay_rate = 0.0; - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - /* Resolve layer assignment. Caller passes either a numeric layer_id or - * a stringified id; el_to_float / int cast tolerates both. */ - int64_t lid = (int64_t)layer_id; - if (lid < 0) lid = (int64_t)ENGRAM_LAYER_DEFAULT; - if (!engram_find_layer((uint32_t)lid)) lid = (int64_t)ENGRAM_LAYER_DEFAULT; - n->layer_id = (uint32_t)lid; - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -/* ── Layer registry public API ────────────────────────────────────────────── - * - * The five canonical layers are seeded at engram_get() initialization. - * Runtime code (typically imprint/suit injection logic at the EL level) - * can extend the registry with engram_add_layer() — only layers marked - * `injectable=1` may be removed via engram_remove_layer(). Removing a - * layer leaves a tombstone slot so existing layer_id references on nodes - * stay valid; orphaned references resolve to "unknown layer" and inherit - * the default suppression behavior. - */ - -/* engram_add_layer — register a new layer at runtime. - * Returns the assigned layer_id as an el_val_t int (cast back via int64_t). - * Conflicting names are rejected (returns 0). */ -el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible, - el_val_t transparent, el_val_t injectable) { - EngramStore* g = engram_get(); - const char* nm = EL_CSTR(name); - if (!nm || !*nm) return (el_val_t)0; - if (engram_find_layer_by_name(nm)) { - /* Name collision — return existing id so callers are idempotent. */ - return (el_val_t)(int64_t)engram_find_layer_by_name(nm)->layer_id; - } - if (g->layer_count >= g->layer_capacity) { - size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; - EngramLayer* grown = realloc(g->layers, nc * sizeof(EngramLayer)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(grown + g->layer_capacity, 0, - (nc - g->layer_capacity) * sizeof(EngramLayer)); - g->layers = grown; - g->layer_capacity = nc; - } - EngramLayer* L = &g->layers[g->layer_count++]; - L->layer_id = engram_next_layer_id(); - L->name = el_strdup_persist(nm); - L->activation_priority = (uint32_t)(int64_t)priority; - L->suppressible = (int)(int64_t)suppressible ? 1 : 0; - L->transparent = (int)(int64_t)transparent ? 1 : 0; - L->injectable = (int)(int64_t)injectable ? 1 : 0; - return (el_val_t)(int64_t)L->layer_id; -} - -/* engram_remove_layer — remove an injectable layer by id. - * Built-in (non-injectable) layers cannot be removed. Nodes still tagged - * with the removed layer's id keep their tag but resolve to "unknown - * layer" thereafter and inherit standard (suppressible) behavior. - * Returns 1 on success, 0 on failure (unknown id, non-injectable). */ -el_val_t engram_remove_layer(el_val_t layer_id) { - EngramStore* g = engram_get(); - int64_t lid = (int64_t)layer_id; - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if ((int64_t)L->layer_id != lid) continue; - if (!L->injectable) return (el_val_t)0; - free(L->name); - L->name = NULL; /* tombstone */ - /* Leave layer_id, priority, flags intact so debug snapshots can - * still distinguish "removed at runtime" from "never existed". */ - return (el_val_t)1; - } - return (el_val_t)0; -} - -/* engram_list_layers — enumerate the active layer registry. - * Returns an ElList of maps, one per non-tombstone layer, sorted by - * activation_priority ascending (deepest layer first). */ -el_val_t engram_list_layers(void) { - EngramStore* g = engram_get(); - el_val_t lst = el_list_empty(); - if (g->layer_count == 0) return lst; - /* Build an index sorted by activation_priority ascending. */ - size_t* idx = malloc(g->layer_count * sizeof(size_t)); - if (!idx) return lst; - size_t live = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) idx[live++] = i; - } - /* Insertion sort — N is small (≤ a few dozen layers). */ - for (size_t i = 1; i < live; i++) { - size_t key = idx[i]; - uint32_t kp = g->layers[key].activation_priority; - size_t j = i; - while (j > 0 && g->layers[idx[j - 1]].activation_priority > kp) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - for (size_t i = 0; i < live; i++) { - EngramLayer* L = &g->layers[idx[i]]; - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), - (el_val_t)(int64_t)L->layer_id); - m = el_map_set(m, EL_STR(el_strdup("name")), - EL_STR(el_strdup(L->name ? L->name : ""))); - m = el_map_set(m, EL_STR(el_strdup("activation_priority")), - (el_val_t)(int64_t)L->activation_priority); - m = el_map_set(m, EL_STR(el_strdup("suppressible")), - (el_val_t)(int64_t)(L->suppressible ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("transparent")), - (el_val_t)(int64_t)(L->transparent ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("injectable")), - (el_val_t)(int64_t)(L->injectable ? 1 : 0)); - lst = el_list_append(lst, m); - } - free(idx); - return lst; -} - -el_val_t engram_get_node(el_val_t id) { - const char* sid = EL_CSTR(id); - EngramNode* n = engram_find_node(sid); - if (!n) return el_map_new(0); - return engram_node_to_map(n); -} - -void engram_strengthen(el_val_t node_id) { - const char* sid = EL_CSTR(node_id); - EngramNode* n = engram_find_node(sid); - if (!n) return; - n->salience += 0.05; - if (n->salience > 1.0) n->salience = 1.0; - n->activation_count++; - n->last_activated = engram_now_ms(); - n->updated_at = n->last_activated; -} - -void engram_forget(el_val_t node_id) { - const char* sid = EL_CSTR(node_id); - if (!sid) return; - EngramStore* g = engram_get(); - int64_t idx = engram_find_node_index(sid); - if (idx < 0) return; - /* Free node strings */ - EngramNode* n = &g->nodes[idx]; - free(n->id); free(n->content); free(n->node_type); free(n->label); - free(n->tier); free(n->tags); free(n->metadata); - /* Shift remaining nodes down */ - for (int64_t i = idx + 1; i < g->node_count; i++) { - g->nodes[i - 1] = g->nodes[i]; - } - g->node_count--; - memset(&g->nodes[g->node_count], 0, sizeof(EngramNode)); - /* Remove all incident edges */ - int64_t w = 0; - for (int64_t r = 0; r < g->edge_count; r++) { - EngramEdge* e = &g->edges[r]; - int incident = (e->from_id && strcmp(e->from_id, sid) == 0) || - (e->to_id && strcmp(e->to_id, sid) == 0); - if (incident) { - free(e->id); free(e->from_id); free(e->to_id); - free(e->relation); free(e->metadata); - } else { - if (w != r) g->edges[w] = g->edges[r]; - w++; - } - } - g->edge_count = w; -} - -el_val_t engram_node_count(void) { - return (el_val_t)engram_get()->node_count; -} - -static int istr_contains(const char* hay, const char* needle) { - if (!hay || !needle || !*needle) return 0; - size_t nl = strlen(needle); - for (const char* p = hay; *p; p++) { - if (strncasecmp(p, needle, nl) == 0) return 1; - } - return 0; -} - -/* ── Tokenized query matching ─────────────────────────────────────────── - * The engram query surface (search / activate / goal-bias) historically - * matched the ENTIRE raw query string as a single case-insensitive - * substring via istr_contains(field, q). That is Ctrl-F, not search: - * a multi-word query like "windows msi signing" only matched a node whose - * text contained that exact contiguous run, so real multi-word queries - * returned zero. istr_contains stays as the per-TOKEN primitive; these - * helpers split the query on whitespace and match ANY token, then rank by - * how many DISTINCT tokens a node covers. Single-token queries are a strict - * special case (score is 0 or 1) so single-word callers never regress. */ -#define ENGRAM_MAX_QTOKENS 32 -#define ENGRAM_QTOK_LEN 256 - -/* Split q on whitespace into up to ENGRAM_MAX_QTOKENS distinct - * (case-insensitive) tokens. Returns the token count. Over-long tokens are - * truncated to ENGRAM_QTOK_LEN-1; over-count tokens are ignored. */ -static int engram_tokenize_query(const char* q, - char toks[][ENGRAM_QTOK_LEN], int maxtok) { - int n = 0; - if (!q) return 0; - const char* p = q; - while (*p && n < maxtok) { - while (*p && isspace((unsigned char)*p)) p++; - if (!*p) break; - char buf[ENGRAM_QTOK_LEN]; - size_t tl = 0; - while (*p && !isspace((unsigned char)*p)) { - if (tl < sizeof(buf) - 1) buf[tl++] = *p; - p++; - } - buf[tl] = '\0'; - if (tl == 0) continue; - int dup = 0; - for (int s = 0; s < n; s++) { - if (strcasecmp(toks[s], buf) == 0) { dup = 1; break; } - } - if (dup) continue; - memcpy(toks[n], buf, tl + 1); - n++; - } - return n; -} - -/* Count how many of the ntok distinct query tokens appear (case-insensitive) - * in the node's content, label, or tags. 0 == no match. */ -static int engram_node_match_score(const EngramNode* n, - char toks[][ENGRAM_QTOK_LEN], int ntok) { - int score = 0; - for (int t = 0; t < ntok; t++) { - if (istr_contains(n->content, toks[t]) || - istr_contains(n->label, toks[t]) || - istr_contains(n->tags, toks[t])) - score++; - } - return score; -} - -/* Rank entry: distinct-token match count (primary, desc) then salience - * (tiebreak, desc). */ -typedef struct { int64_t idx; int score; double salience; } EngramRankEntry; -static int engram_rank_cmp(const void* a, const void* b) { - const EngramRankEntry* ea = (const EngramRankEntry*)a; - const EngramRankEntry* eb = (const EngramRankEntry*)b; - if (ea->score != eb->score) return eb->score - ea->score; /* desc */ - if (ea->salience < eb->salience) return 1; - if (ea->salience > eb->salience) return -1; - return 0; -} - -/* ══════════════════════════════════════════════════════════════════════════ - * SEMANTIC SEARCH LAYER — nomic-embed-text via Ollama /api/embeddings - * ────────────────────────────────────────────────────────────────────────── - * Augments the lexical (istr_contains) matcher with dense-vector retrieval. - * Node content and the query are embedded through a local Ollama server; - * nodes are ranked by cosine similarity and UNIONED with lexical hits. This - * lets a paraphrase query surface a node whose words never appear in it. - * - * DEGRADABLE BY DESIGN. The whole layer is gated on HAVE_CURL plus a one-shot - * runtime probe of the embedding endpoint. If curl is not compiled in, or - * Ollama is unreachable, or ENGRAM_SEMANTIC=0, every entry point returns - * "no semantic signal" and callers fall back to pure lexical behaviour — - * byte-for-byte the pre-existing search. - * - * CACHE. Node embeddings are computed lazily on first use and cached in - * process memory keyed by node id, with an FNV-1a content hash for - * invalidation (edited content re-embeds). The query is embedded once per - * search call. This is what "avoid re-embedding the whole graph every query" - * buys us: a warm cache serves cosine from RAM. (A cold process still pays - * O(N) embed calls the first time each node is scanned — persisting the cache - * to a snapshot sidecar is the documented next step, not done here.) - * - * nomic task prefixes ("search_query:" / "search_document:") are applied - * because nomic-embed-text is trained with them; they materially improve - * retrieval separation (empirically: paraphrase 0.72 vs distractors <0.48). - * - * ENV: - * ENGRAM_SEMANTIC "0" disables; unset/other = auto-probe - * ENGRAM_EMBED_URL default http://localhost:11434/api/embeddings - * ENGRAM_EMBED_MODEL default nomic-embed-text - * ENGRAM_SEMANTIC_MIN cosine threshold for a pure-semantic match (def 0.6) - * ════════════════════════════════════════════════════════════════════════ */ - -static double engram_semantic_min(void) { - static double v = -1.0; - if (v >= 0.0) return v; - const char* s = getenv("ENGRAM_SEMANTIC_MIN"); - double d = 0.6; - if (s && *s) { char* e = NULL; double t = strtod(s, &e); - if (e != s && t >= 0.0 && t <= 1.0) d = t; } - v = d; return v; -} - -#ifdef HAVE_CURL - -typedef struct { char* id; uint64_t hash; float* vec; int dim; } EngramEmbEntry; -static EngramEmbEntry* g_emb_items = NULL; -static int64_t g_emb_count = 0, g_emb_cap = 0; -static int g_emb_state = 0; /* 0=unprobed, 1=available, -1=disabled */ - -static uint64_t engram_fnv1a(const char* s) { - uint64_t h = 1469598103934665603ULL; - if (s) for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - h ^= *p; h *= 1099511628211ULL; - } - return h; -} - -/* Parse "embedding":[f,f,...] from an Ollama response. malloc'd vec, or NULL. */ -static float* engram_parse_embedding(const char* json, int* out_dim) { - if (!json) return NULL; - const char* p = strstr(json, "\"embedding\""); - if (!p) return NULL; - p = strchr(p, '['); - if (!p) return NULL; - p++; - int cap = 1024, n = 0; - float* v = malloc((size_t)cap * sizeof(float)); - if (!v) return NULL; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p == ']' || !*p) break; - char* e = NULL; - double d = strtod(p, &e); - if (e == p) break; - if (n >= cap) { cap *= 2; float* nv = realloc(v, (size_t)cap * sizeof(float)); - if (!nv) { free(v); return NULL; } v = nv; } - v[n++] = (float)d; - p = e; - } - if (n == 0) { free(v); return NULL; } - *out_dim = n; - return v; -} - -/* JSON-escape src into a malloc'd buffer (no surrounding quotes). */ -static char* engram_json_escape(const char* src) { - if (!src) src = ""; - size_t n = strlen(src); - char* out = malloc(n * 2 + 1); - if (!out) return NULL; - size_t j = 0; - for (size_t i = 0; i < n; i++) { - unsigned char c = (unsigned char)src[i]; - if (c == '"') { out[j++] = '\\'; out[j++] = '"'; } - else if (c == '\\') { out[j++] = '\\'; out[j++] = '\\'; } - else if (c == '\n') { out[j++] = '\\'; out[j++] = 'n'; } - else if (c == '\r') { out[j++] = '\\'; out[j++] = 'r'; } - else if (c == '\t') { out[j++] = '\\'; out[j++] = 't'; } - else if (c < 0x20) { /* drop other control bytes */ } - else { out[j++] = (char)c; } - } - out[j] = '\0'; - return out; -} - -/* Embed `prefix+text` via Ollama. Returns malloc'd vec (caller frees), or NULL. */ -static float* engram_embed_raw(const char* prefix, const char* text, int* out_dim) { - if (!text) return NULL; - const char* url = getenv("ENGRAM_EMBED_URL"); - if (!url || !*url) url = "http://localhost:11434/api/embeddings"; - const char* model = getenv("ENGRAM_EMBED_MODEL"); - if (!model || !*model) model = "nomic-embed-text"; - /* Bound content length to keep latency/memory sane on huge nodes. */ - char* trunc = NULL; - size_t maxlen = 8192; - if (strlen(text) > maxlen) { - trunc = malloc(maxlen + 1); - if (trunc) { memcpy(trunc, text, maxlen); trunc[maxlen] = '\0'; text = trunc; } - } - char* esc_prefix = engram_json_escape(prefix ? prefix : ""); - char* esc = engram_json_escape(text); - free(trunc); - if (!esc || !esc_prefix) { free(esc); free(esc_prefix); return NULL; } - size_t blen = strlen(esc) + strlen(esc_prefix) + strlen(model) + 64; - char* body = malloc(blen); - if (!body) { free(esc); free(esc_prefix); return NULL; } - snprintf(body, blen, "{\"model\":\"%s\",\"prompt\":\"%s%s\"}", model, esc_prefix, esc); - free(esc); free(esc_prefix); - - CURL* c = curl_easy_init(); - if (!c) { free(body); return NULL; } - HttpBuf rb; httpbuf_init(&rb); - struct curl_slist* h = curl_slist_append(NULL, "Content-Type: application/json"); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)strlen(body)); - curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - CURLcode rc = curl_easy_perform(c); - curl_slist_free_all(h); - curl_easy_cleanup(c); - free(body); - if (rc != CURLE_OK) { free(rb.data); return NULL; } - float* v = engram_parse_embedding(rb.data, out_dim); - free(rb.data); - return v; -} - -/* One-shot probe: is semantic search available? Caches the verdict. */ -static int engram_semantic_enabled(void) { - if (g_emb_state != 0) return g_emb_state == 1; - const char* s = getenv("ENGRAM_SEMANTIC"); - if (s && strcmp(s, "0") == 0) { g_emb_state = -1; return 0; } - int dim = 0; - float* v = engram_embed_raw("search_query: ", "probe", &dim); - if (v && dim > 0) { free(v); g_emb_state = 1; return 1; } - free(v); - g_emb_state = -1; return 0; -} - -/* Embed the query. Returns malloc'd vec (caller frees), or NULL if semantic off. */ -static float* engram_embed_query(const char* q, int* dim) { - if (!engram_semantic_enabled()) return NULL; - if (!q || !*q) return NULL; - return engram_embed_raw("search_query: ", q, dim); -} - -/* Cached node embedding. Returns a pointer OWNED BY THE CACHE — do not free. */ -static const float* engram_node_vec(EngramNode* n, int* out_dim) { - if (!n || !n->id) return NULL; - uint64_t h = engram_fnv1a(n->content); - for (int64_t i = 0; i < g_emb_count; i++) { - if (g_emb_items[i].id && strcmp(g_emb_items[i].id, n->id) == 0) { - if (g_emb_items[i].hash == h && g_emb_items[i].vec) { - *out_dim = g_emb_items[i].dim; return g_emb_items[i].vec; - } - /* content changed → re-embed in place */ - int dim = 0; - float* v = engram_embed_raw("search_document: ", n->content ? n->content : "", &dim); - if (!v) return NULL; - free(g_emb_items[i].vec); - g_emb_items[i].vec = v; g_emb_items[i].dim = dim; g_emb_items[i].hash = h; - *out_dim = dim; return v; - } - } - int dim = 0; - float* v = engram_embed_raw("search_document: ", n->content ? n->content : "", &dim); - if (!v) return NULL; - if (g_emb_count >= g_emb_cap) { - int64_t nc = g_emb_cap ? g_emb_cap * 2 : 256; - EngramEmbEntry* ni = realloc(g_emb_items, (size_t)nc * sizeof(EngramEmbEntry)); - if (!ni) { free(v); return NULL; } - g_emb_items = ni; g_emb_cap = nc; - } - g_emb_items[g_emb_count].id = strdup(n->id); - g_emb_items[g_emb_count].hash = h; - g_emb_items[g_emb_count].vec = v; - g_emb_items[g_emb_count].dim = dim; - g_emb_count++; - *out_dim = dim; return v; -} - -static double engram_cosine(const float* a, const float* b, int dim) { - double dot = 0, na = 0, nb = 0; - for (int i = 0; i < dim; i++) { dot += (double)a[i] * b[i]; - na += (double)a[i] * a[i]; - nb += (double)b[i] * b[i]; } - if (na <= 0 || nb <= 0) return 0.0; - return dot / (sqrt(na) * sqrt(nb)); -} - -/* Cosine of node n against the query vector; 0 if unavailable / dim mismatch. */ -static double engram_node_cosine(EngramNode* n, const float* qvec, int qdim) { - if (!qvec || qdim <= 0) return 0.0; - int ndim = 0; - const float* nv = engram_node_vec(n, &ndim); - if (!nv || ndim != qdim) return 0.0; - return engram_cosine(qvec, nv, qdim); -} - -#else /* !HAVE_CURL — semantic layer compiled out; callers stay pure-lexical. - * Only the two boundary functions the always-compiled search/activate - * code calls are stubbed; the query embed always yields NULL so every - * cosine is 0 and every caller collapses to lexical-only. */ -static float* engram_embed_query(const char* q, int* dim) { (void)q; (void)dim; return NULL; } -static double engram_node_cosine(EngramNode* n, const float* qvec, int qdim) { - (void)n; (void)qvec; (void)qdim; return 0.0; -} -#endif /* HAVE_CURL */ - -el_val_t engram_search(el_val_t query, el_val_t limit) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t lim = (int64_t)limit; - if (lim <= 0) lim = 100; - el_val_t lst = el_list_empty(); - if (!q || !*q) return lst; - char toks[ENGRAM_MAX_QTOKENS][ENGRAM_QTOK_LEN]; - int ntok = engram_tokenize_query(q, toks, ENGRAM_MAX_QTOKENS); - if (ntok == 0) return lst; - /* Semantic augmentation: embed the query once; a node is a hit if it covers - * >=1 query token (tokenized-lexical, #66) OR its cosine clears the - * threshold (#67). qvec is NULL (cosine 0) when semantic is unavailable → - * pure tokenized-lexical, byte-identical to the lexical-only behaviour. */ - int qdim = 0; - float* qvec = engram_embed_query(q, &qdim); - double sem_min = engram_semantic_min(); - EngramRankEntry* hits = malloc((size_t)g->node_count * sizeof(EngramRankEntry)); - if (!hits) { free(qvec); return lst; } - int64_t nhits = 0; - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - /* Filter transparent layers: nodes whose layer is `transparent=1` - * shape output but are invisible to introspection ("what do you - * know about yourself"). They still surface via engram_activate - * + engram_compile_layered_json — that's the legitimate path. */ - if (engram_layer_is_transparent(n->layer_id)) continue; - int sc = engram_node_match_score(n, toks, ntok); - double sem = qvec ? engram_node_cosine(n, qvec, qdim) : 0.0; - if (sc > 0 || sem >= sem_min) { - hits[nhits].idx = i; - hits[nhits].score = sc; - hits[nhits].salience = n->salience; - nhits++; - } - } - /* Rank by distinct tokens matched (desc) then salience (desc), then cap. - * Pure-semantic hits (token score 0) sort after every lexical hit — a - * lexical ∪ semantic union with lexical precedence. */ - qsort(hits, (size_t)nhits, sizeof(EngramRankEntry), engram_rank_cmp); - int64_t end = nhits < lim ? nhits : lim; - for (int64_t k = 0; k < end; k++) { - lst = el_list_append(lst, engram_node_to_map(&g->nodes[hits[k].idx])); - } - free(hits); - free(qvec); - return lst; -} - -/* Sort node indices by salience desc (small N, insertion sort is fine). */ -static void engram_sort_indices_by_salience(int64_t* arr, int64_t n, - const EngramNode* nodes) { - for (int64_t i = 1; i < n; i++) { - int64_t key = arr[i]; - double ks = nodes[key].salience; - int64_t j = i - 1; - while (j >= 0 && nodes[arr[j]].salience < ks) { - arr[j + 1] = arr[j]; - j--; - } - arr[j + 1] = key; - } -} - -el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset) { - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - el_val_t lst = el_list_empty(); - if (g->node_count == 0) return lst; - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) return lst; - /* Skip transparent layers — same introspection-filter rationale as - * engram_search above. */ - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - for (int64_t i = off; i < end; i++) { - lst = el_list_append(lst, engram_node_to_map(&g->nodes[idx[i]])); - } - free(idx); - return lst; -} - -void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation) { - EngramStore* g = engram_get(); - const char* f = EL_CSTR(from_id); - const char* t = EL_CSTR(to_id); - const char* r = EL_CSTR(relation); - if (!f || !t) return; - engram_grow_edges(); - EngramEdge* e = &g->edges[g->edge_count]; - memset(e, 0, sizeof(*e)); - e->id = engram_new_id(); - e->from_id = el_strdup_persist(f); - e->to_id = el_strdup_persist(t); - e->relation = el_strdup_persist(r && *r ? r : "associate"); - e->metadata = el_strdup_persist("{}"); - e->weight = engram_decode_score(weight); - if (e->weight <= 0.0 || e->weight > 1.0) e->weight = 0.5; - e->confidence = 1.0; - int64_t now = engram_now_ms(); - e->created_at = now; - e->updated_at = now; - e->last_fired = 0; - e->layer_id = ENGRAM_LAYER_DEFAULT; - g->edge_count++; -} - -el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id) { - EngramStore* g = engram_get(); - const char* f = EL_CSTR(from_id); - const char* t = EL_CSTR(to_id); - if (!f || !t) return 0; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (e->from_id && e->to_id && - strcmp(e->from_id, f) == 0 && strcmp(e->to_id, t) == 0) return 1; - } - return 0; -} - -/* Reserved helper: edge -> ElMap. Kept around for future builtins. */ -static el_val_t engram_edge_to_map(const EngramEdge* e) __attribute__((unused)); -static el_val_t engram_edge_to_map(const EngramEdge* e) { - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("id")), EL_STR(el_strdup(e->id ? e->id : ""))); - m = el_map_set(m, EL_STR(el_strdup("from_id")), EL_STR(el_strdup(e->from_id ? e->from_id : ""))); - m = el_map_set(m, EL_STR(el_strdup("to_id")), EL_STR(el_strdup(e->to_id ? e->to_id : ""))); - m = el_map_set(m, EL_STR(el_strdup("relation")), EL_STR(el_strdup(e->relation ? e->relation : ""))); - m = el_map_set(m, EL_STR(el_strdup("metadata")), EL_STR(el_strdup(e->metadata ? e->metadata : "{}"))); - m = el_map_set(m, EL_STR(el_strdup("weight")), el_from_float(e->weight)); - m = el_map_set(m, EL_STR(el_strdup("confidence")), el_from_float(e->confidence)); - m = el_map_set(m, EL_STR(el_strdup("created_at")), (el_val_t)e->created_at); - m = el_map_set(m, EL_STR(el_strdup("updated_at")), (el_val_t)e->updated_at); - m = el_map_set(m, EL_STR(el_strdup("last_fired")), (el_val_t)e->last_fired); - m = el_map_set(m, EL_STR(el_strdup("inhibitory")), (el_val_t)(e->inhibitory ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), (el_val_t)(int64_t)e->layer_id); - return m; -} - -el_val_t engram_neighbors(el_val_t node_id) { - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - el_val_t lst = el_list_empty(); - if (!sid) return lst; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* other = NULL; - if (e->from_id && strcmp(e->from_id, sid) == 0) other = e->to_id; - else if (e->to_id && strcmp(e->to_id, sid) == 0) other = e->from_id; - if (!other) continue; - EngramNode* n = engram_find_node(other); - if (n) lst = el_list_append(lst, engram_node_to_map(n)); - } - return lst; -} - -el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction) { - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - int64_t md = (int64_t)max_depth; if (md <= 0) md = 1; - const char* dir = EL_CSTR(direction); /* "out" | "in" | "both" (default) */ - el_val_t lst = el_list_empty(); - if (!sid || g->node_count == 0) return lst; - int64_t start = engram_find_node_index(sid); - if (start < 0) return lst; - /* BFS with depth tracking */ - int64_t* visited = calloc((size_t)g->node_count, sizeof(int64_t)); - int64_t* queue = calloc((size_t)g->node_count, sizeof(int64_t)); - int64_t* depths = calloc((size_t)g->node_count, sizeof(int64_t)); - if (!visited || !queue || !depths) { - free(visited); free(queue); free(depths); return lst; - } - int64_t qh = 0, qt = 0; - queue[qt++] = start; - visited[start] = 1; - depths[start] = 0; - while (qh < qt) { - int64_t cur = queue[qh++]; - const char* cur_id = g->nodes[cur].id; - int64_t cur_depth = depths[cur]; - if (cur_depth >= md) continue; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* other = NULL; - int outgoing = e->from_id && strcmp(e->from_id, cur_id) == 0; - int incoming = e->to_id && strcmp(e->to_id, cur_id) == 0; - if (dir && strcmp(dir, "out") == 0 && !outgoing) continue; - if (dir && strcmp(dir, "in") == 0 && !incoming) continue; - if (outgoing) other = e->to_id; - else if (incoming) other = e->from_id; - else continue; - int64_t oi = engram_find_node_index(other); - if (oi < 0 || visited[oi]) continue; - visited[oi] = 1; - depths[oi] = cur_depth + 1; - queue[qt++] = oi; - } - } - /* Emit all visited except the seed */ - for (int64_t i = 0; i < g->node_count; i++) { - if (visited[i] && i != start) { - lst = el_list_append(lst, engram_node_to_map(&g->nodes[i])); - } - } - free(visited); free(queue); free(depths); - return lst; -} - -el_val_t engram_edge_count(void) { - return (el_val_t)engram_get()->edge_count; -} - -/* Compute temporal decay factor for a node given current time. - * effective contribution = salience * exp(-lambda * age_hours / T_half) - * Clamped to [0.05, 1.0] so very old nodes retain a meaningful floor. */ -static double engram_temporal_decay(const EngramNode* n, int64_t now_ms) { - int64_t age_ms = now_ms - n->last_activated; - if (age_ms <= 0) return 1.0; - double lambda = (n->temporal_decay_rate > 0.0) ? n->temporal_decay_rate - : ENGRAM_DECAY_LAMBDA; - double age_hours = (double)age_ms / 3600000.0; - double factor = exp(-lambda * age_hours / ENGRAM_T_HALF_HOURS); - if (factor < 0.05) factor = 0.05; - return factor; -} - -/* Activation dampening: high activation_count nodes are "well-known" context - * and get less marginal boost per firing. - * count=0 → 1.0, count=2 → ~0.74, count=9 → ~0.59, count=99 → ~0.43 */ -static double engram_activation_dampen(const EngramNode* n) { - return 1.0 / (1.0 + log(1.0 + (double)n->activation_count)); -} - -/* Temporal proximity bonus: boost propagation along edges connecting - * co-temporal nodes. Returns a multiplier bonus in [0, 0.2]. */ -static double engram_temporal_proximity_bonus(int64_t node_created, - int64_t seed_epoch) { - int64_t diff = node_created - seed_epoch; - if (diff < 0) diff = -diff; - if (diff < 86400000LL) return 0.20; /* within 1 day */ - if (diff < 604800000LL) return 0.10; /* within 7 days */ - return 0.0; -} - -/* ── Two-layer activation (biologically-motivated) ─────────────────────────── - * - * Layer 1 — Broad fan-out (background activation): - * BFS + spreading activation fires on ALL nodes reachable from seeds, - * regardless of relevance to the current goal. Every reachable node gets - * a background_activation score. Nothing is filtered here. Models the - * brain's massive parallel sub-threshold activation of all associated - * content in response to a stimulus. Temporal decay and activation - * dampening are applied at this layer (as before), but no threshold gate. - * - * Layer 2 — Executive filter (working memory promotion): - * A second pass asks: given the query (goal intent), attentional bias, - * and inhibitory edge topology — which background-activated nodes should - * break through into working memory? - * - * wm_weight = bg_activation * goal_bias(node, query) * confidence - * * inhibitory_suppression_factor - * - * Only nodes where wm_weight >= ENGRAM_WM_THRESHOLD are promoted to - * working memory (working_memory_weight > 0). Background-activated nodes - * that don't cross the threshold accumulate suppression_count. After - * ENGRAM_SUPPRESSION_BREAKTHROUGH consecutive suppressed turns, the node - * force-breaks through at ENGRAM_BREAKTHROUGH_WEIGHT (latent tension - * surfacing — models intrusive memory / unresolved cognitive load). - * - * Inhibitory edges: - * An edge with inhibitory=1 suppresses the TARGET node's working memory - * promotion when the SOURCE is background-activated. Background activation - * of the target is NOT affected — the node fires in layer 1. Only the - * executive filter (layer 2) is gated. Models attentional inhibition: - * "focused on code work" suppresses personal memories from surfacing - * even if they have high background_activation. - * - * Goal bias: - * A lightweight heuristic rates how well each background-activated node - * aligns with the apparent intent of the current query. Technical queries - * boost Belief/Canonical/Lesson nodes; relational queries boost Memory/ - * Entity nodes. Direct lexical overlap gives a 50% bonus. - * - * Working memory persistence (turn continuity): - * Nodes promoted in the previous turn retain a decayed working_memory_weight - * (weight *= ENGRAM_WM_DECAY) without needing re-activation. This models - * conversational thread continuity — once a topic is in working memory, - * it persists slightly into the next turn. - * - * Returns ElList of {node, activation_strength, working_memory_weight, - * epistemic_confidence, hops, promoted}. - * "promoted" = 1 if working_memory_weight > 0, 0 if background-only. - * Context compilation uses ONLY nodes with promoted=1. - * - * Temporal decay (preserved from prior implementation): - * effective_salience = salience * exp(-lambda * age_hours / T_half) - * where T_half = 168 h (one week), lambda = ln(2) - * - * Activation dampening (preserved): - * dampen = 1 / (1 + log(1 + activation_count)) - * - * Temporal proximity bonus (preserved): - * edge_strength *= (1 + tbonus) where tbonus ∈ {0, 0.10, 0.20} - * - * Per-type threshold gates apply only to working memory promotion (layer 2): - * Safety/DharmaSelf: 0.05 Canonical: 0.15 Lesson: 0.25 - * Belief/Entity: 0.30 Note/Memory/Working: 0.40 - */ - -/* Compute goal-state bias multiplier for a node given the query. - * Returns a value in [0.3, 2.0]. This is a lightweight heuristic — - * a production implementation may use LLM-derived intent classification. */ -static double engram_goal_bias(const EngramNode* n, const char* query) { - if (!query || !*query) return 1.0; - double bias = 1.0; - /* Direct lexical overlap, graded by token coverage: a node covering all - * query tokens gets the full +0.5; partial coverage gets a proportional - * share. Single-token queries → full +0.5 on match, identical to before. */ - { - char toks[ENGRAM_MAX_QTOKENS][ENGRAM_QTOK_LEN]; - int ntok = engram_tokenize_query(query, toks, ENGRAM_MAX_QTOKENS); - int sc = engram_node_match_score(n, toks, ntok); - if (sc > 0 && ntok > 0) bias += 0.5 * ((double)sc / (double)ntok); - } - /* Node-type resonance with query intent. */ - int technical_query = istr_contains(query, "code") || - istr_contains(query, "function") || - istr_contains(query, "implement") || - istr_contains(query, "error") || - istr_contains(query, "bug") || - istr_contains(query, "build") || - istr_contains(query, "system") || - istr_contains(query, "design") || - istr_contains(query, "architecture"); - int personal_query = istr_contains(query, "feel") || - istr_contains(query, "emotion") || - istr_contains(query, "remember") || - istr_contains(query, "personal") || - istr_contains(query, "story") || - istr_contains(query, "relationship"); - if (n->node_type) { - int is_knowledge = (strcmp(n->node_type, "Belief") == 0) || - (strcmp(n->node_type, "DharmaSelf") == 0) || - (strcmp(n->node_type, "Safety") == 0); - int is_personal = (strcmp(n->node_type, "Memory") == 0) || - (strcmp(n->node_type, "Entity") == 0); - if (technical_query && is_knowledge) bias += 0.3; - if (technical_query && is_personal) bias -= 0.3; - if (personal_query && is_personal) bias += 0.3; - if (personal_query && is_knowledge) bias -= 0.1; - } - /* Tier-based bonus: promote higher-confidence knowledge nodes. */ - if (n->tier) { - if (strcmp(n->tier, "Canonical") == 0) bias += 0.2; - if (strcmp(n->tier, "Lesson") == 0) bias += 0.1; - } - if (bias < 0.3) bias = 0.3; - if (bias > 2.0) bias = 2.0; - return bias; -} - -el_val_t engram_activate(el_val_t query, el_val_t depth) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t max_depth = (int64_t)depth; if (max_depth <= 0) max_depth = 2; - el_val_t out = el_list_empty(); - if (!q || g->node_count == 0) return out; - - int64_t now_ms = engram_now_ms(); - - /* Per-node layer-1 tracking. */ - double* best_bg = calloc((size_t)g->node_count, sizeof(double)); - int64_t* best_hops = calloc((size_t)g->node_count, sizeof(int64_t)); - int* reached = calloc((size_t)g->node_count, sizeof(int)); - if (!best_bg || !best_hops || !reached) { - free(best_bg); free(best_hops); free(reached); return out; - } - - /* ── LAYER 1: broad fan-out (background activation) ───────────────── - * Find seeds, apply temporal decay + dampening, BFS with edge weights. - * Inhibitory edges propagate activation normally at this layer — they - * only gate working memory promotion in layer 2. */ - typedef struct { int64_t idx; double act; int64_t created_at; } SeedEntry; - SeedEntry* seeds = malloc((size_t)g->node_count * sizeof(SeedEntry)); - int64_t seed_count = 0; - if (!seeds) { - free(best_bg); free(best_hops); free(reached); return out; - } - /* Tokenized + semantic seeding: a node seeds if it covers >=1 query token - * (tokenized-lexical, #66) OR its cosine clears the threshold (#67). A - * lexical seed's activation is scaled by token coverage (fraction of - * distinct query tokens covered) so a node matching all words seeds more - * strongly than one matching a single word; single-word queries → coverage - * 1.0. A pure-semantic seed (no token match) is instead down-weighted by - * its cosine so paraphrase matches spread without overpowering exact seeds. - * q_vec is NULL (cosine 0) when semantic is unavailable → the seed set is - * exactly the tokenized-lexical one. q_vec is freed right after this loop - * so the many downstream early-returns need no cleanup change. */ - char toks[ENGRAM_MAX_QTOKENS][ENGRAM_QTOK_LEN]; - int ntok = engram_tokenize_query(q, toks, ENGRAM_MAX_QTOKENS); - int q_dim = 0; - float* q_vec = engram_embed_query(q, &q_dim); - double q_sem_min = engram_semantic_min(); - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - int sc = engram_node_match_score(n, toks, ntok); - double sem = q_vec ? engram_node_cosine(n, q_vec, q_dim) : 0.0; - if (sc > 0 || sem >= q_sem_min) { - double tdecay = engram_temporal_decay(n, now_ms); - double dampen = engram_activation_dampen(n); - double act = n->salience * tdecay * dampen; - if (sc > 0) act *= (ntok > 0 ? (double)sc / (double)ntok : 1.0); - else act *= sem; /* pure-semantic seed: down-weight by cosine */ - seeds[seed_count].idx = i; - seeds[seed_count].act = act; - seeds[seed_count].created_at = n->created_at; - seed_count++; - best_bg[i] = act; - best_hops[i] = 0; - reached[i] = 1; - } - } - free(q_vec); - /* Compute mean seed created_at for temporal proximity bonus. */ - int64_t seed_epoch = 0; - if (seed_count > 0) { - seed_epoch = seeds[0].created_at; - for (int64_t s = 1; s < seed_count; s++) - seed_epoch = (seed_epoch + seeds[s].created_at) / 2; - } - typedef struct { int64_t idx; int64_t hops; double act; } Frontier; - Frontier* fr = malloc((size_t)(g->node_count * (max_depth + 1)) * sizeof(Frontier) + 16 * sizeof(Frontier)); - if (!fr) { - free(best_bg); free(best_hops); free(reached); free(seeds); return out; - } - int64_t fhead = 0, ftail = 0; - int64_t fcap = (int64_t)((size_t)(g->node_count * (max_depth + 1)) + 16); - for (int64_t s = 0; s < seed_count; s++) { - if (ftail >= fcap) break; - fr[ftail].idx = seeds[s].idx; - fr[ftail].hops = 0; - fr[ftail].act = seeds[s].act; - ftail++; - } - const double SPREAD_DECAY = 0.7; - while (fhead < ftail) { - Frontier f = fr[fhead++]; - if (f.hops >= max_depth) continue; - const char* cur_id = g->nodes[f.idx].id; - for (int64_t ei = 0; ei < g->edge_count; ei++) { - EngramEdge* e = &g->edges[ei]; - const char* other = NULL; - if (e->from_id && strcmp(e->from_id, cur_id) == 0) other = e->to_id; - else if (e->to_id && strcmp(e->to_id, cur_id) == 0) other = e->from_id; - else continue; - int64_t oi = engram_find_node_index(other); - if (oi < 0) continue; - EngramNode* on = &g->nodes[oi]; - double tbonus = engram_temporal_proximity_bonus(on->created_at, seed_epoch); - double tdecay = engram_temporal_decay(on, now_ms); - double dampen = engram_activation_dampen(on); - double new_act = f.act * e->weight * SPREAD_DECAY * (1.0 + tbonus) - * tdecay * dampen; - int64_t new_hops = f.hops + 1; - if (!reached[oi] || new_act > best_bg[oi]) { - best_bg[oi] = new_act; - best_hops[oi] = new_hops; - reached[oi] = 1; - if (ftail < fcap) { - fr[ftail].idx = oi; - fr[ftail].hops = new_hops; - fr[ftail].act = new_act; - ftail++; - } - } - } - } - /* Persist layer-1 background_activation to node store. */ - for (int64_t i = 0; i < g->node_count; i++) { - g->nodes[i].background_activation = reached[i] ? best_bg[i] : 0.0; - } - - /* ── PASS 2: executive filter → working memory promotion ──────────── */ - /* Step A: collect inhibitory suppressions from fired inhibitory edges. - * Layered consciousness: inhibition is ONLY recorded against targets - * whose layer is `suppressible == 1`. Nodes in non-suppressible layers - * (Layer 0 / safety) ignore inhibitory edges entirely — their working - * memory weight cannot be silenced by attentional suppression. */ - double* inhibition = calloc((size_t)g->node_count, sizeof(double)); - if (!inhibition) { - free(best_bg); free(best_hops); free(reached); free(seeds); free(fr); - return out; - } - for (int64_t ei = 0; ei < g->edge_count; ei++) { - EngramEdge* e = &g->edges[ei]; - if (!e->inhibitory) continue; - int64_t src = engram_find_node_index(e->from_id); - int64_t tgt = engram_find_node_index(e->to_id); - if (src < 0 || tgt < 0) continue; - if (!reached[src] || best_bg[src] <= 0.0) continue; - /* Skip if target layer is non-suppressible: Layer 0 / safety nodes - * are immune to inhibitory edges from any source. The pass-3 - * override below also force-promotes them, but recording inhibition - * against them at all would be wasted work and could confuse - * downstream debugging output. */ - if (!engram_layer_is_suppressible(g->nodes[tgt].layer_id)) continue; - /* Inhibition strength proportional to source background activation - * and edge weight. Takes the maximum if multiple inhibitory edges - * target the same node. */ - double inh = best_bg[src] * e->weight; - if (inh > inhibition[tgt]) inhibition[tgt] = inh; - } - /* Step B: compute working_memory_weight per candidate node. */ - double* wm_weights = calloc((size_t)g->node_count, sizeof(double)); - if (!wm_weights) { - free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); return out; - } - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i] || best_bg[i] <= 0.0) continue; - EngramNode* n = &g->nodes[i]; - /* Per-type threshold: safety nodes break through more easily. */ - double type_threshold = engram_type_threshold(n->node_type, n->tier); - /* Goal bias weights the node's relevance to current intent. */ - double bias = engram_goal_bias(n, q); - /* Raw working memory score. */ - double raw_wm = best_bg[i] * bias * n->confidence; - /* Apply inhibitory suppression. Full inhibition → scale by factor. */ - double inh = inhibition[i]; - if (inh > 1.0) inh = 1.0; - double suppress = 1.0 - (1.0 - ENGRAM_INHIBITION_FACTOR) * inh; - raw_wm *= suppress; - /* Threshold gate: must exceed per-type threshold to enter working - * memory. Type threshold replaces the old flat 0.2 filter. */ - if (raw_wm >= type_threshold) { - wm_weights[i] = raw_wm > 1.0 ? 1.0 : raw_wm; - if (n->suppression_count > 0) n->suppression_count = 0; - } else { - /* Node didn't make it through — increment suppression counter. - * After N consecutive suppressions: force breakthrough. */ - n->suppression_count++; - if (n->suppression_count >= ENGRAM_SUPPRESSION_BREAKTHROUGH) { - wm_weights[i] = ENGRAM_BREAKTHROUGH_WEIGHT; - n->suppression_count = 0; - } else { - wm_weights[i] = 0.0; - } - } - } - /* ── PASS 3: Layer 0 override (the sacred fire) ───────────────────── - * Every node in a non-suppressible layer that received any background - * activation is force-promoted to AT LEAST ENGRAM_LAYER0_OVERRIDE_WEIGHT. - * This runs LAST and overrides whatever Pass 2 decided — Layer 0 cannot - * be silenced by inhibitory edges, by goal-bias misalignment, by - * confidence weighting, or by per-type threshold gates. If the seed - * fan-out reached a structural-floor node, that node surfaces. - * - * Note: this also clears the suppression_count when an override fires, - * since the node DID surface this turn — it just took the override path - * rather than the standard threshold path. Without this, a Layer 0 - * node with persistent inhibitory pressure would accumulate - * suppression_count forever and never reach the breakthrough state. */ - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i] || best_bg[i] <= 0.0) continue; - EngramNode* n = &g->nodes[i]; - if (engram_layer_is_suppressible(n->layer_id)) continue; - if (wm_weights[i] < ENGRAM_LAYER0_OVERRIDE_WEIGHT) { - wm_weights[i] = ENGRAM_LAYER0_OVERRIDE_WEIGHT; - } - n->suppression_count = 0; - } - - /* Persist working_memory_weight (post Pass 3) to node store. */ - for (int64_t i = 0; i < g->node_count; i++) { - g->nodes[i].working_memory_weight = wm_weights[i]; - } - - /* ── Collect all background-activated nodes for the return value ──── - * Callers see both layers. Context compilation uses only promoted nodes - * (working_memory_weight > 0). Sort: promoted first by wm_weight desc, - * then background-only by background_activation desc. */ - typedef struct { int64_t idx; double bg; double wm; double epist; int64_t hops; } Result; - Result* results = malloc((size_t)g->node_count * sizeof(Result)); - int64_t rcount = 0; - if (!results) { - free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); free(wm_weights); return out; - } - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i]) continue; - double epist = best_bg[i] * g->nodes[i].confidence; - /* Include if promoted to working memory OR if background activation - * is meaningful enough to report (epist >= 0.1). */ - if (epist < 0.1 && wm_weights[i] <= 0.0) continue; - results[rcount].idx = i; - results[rcount].bg = best_bg[i]; - results[rcount].wm = wm_weights[i]; - results[rcount].epist = epist; - results[rcount].hops = best_hops[i]; - rcount++; - } - /* Sort: promoted nodes first (by wm_weight desc), then background-only - * by background_activation desc. */ - for (int64_t i = 1; i < rcount; i++) { - Result key = results[i]; - int64_t j = i - 1; - while (j >= 0 && (results[j].wm < key.wm || - (results[j].wm == key.wm && results[j].bg < key.bg))) { - results[j + 1] = results[j]; - j--; - } - results[j + 1] = key; - } - for (int64_t i = 0; i < rcount; i++) { - el_val_t entry = el_map_new(0); - entry = el_map_set(entry, EL_STR(el_strdup("node")), - engram_node_to_map(&g->nodes[results[i].idx])); - entry = el_map_set(entry, EL_STR(el_strdup("activation_strength")), - el_from_float(results[i].bg)); - entry = el_map_set(entry, EL_STR(el_strdup("working_memory_weight")), - el_from_float(results[i].wm)); - entry = el_map_set(entry, EL_STR(el_strdup("epistemic_confidence")), - el_from_float(results[i].epist)); - entry = el_map_set(entry, EL_STR(el_strdup("hops")), - (el_val_t)results[i].hops); - entry = el_map_set(entry, EL_STR(el_strdup("promoted")), - (el_val_t)(results[i].wm > 0.0 ? 1 : 0)); - out = el_list_append(out, entry); - } - free(best_bg); free(best_hops); free(reached); - free(seeds); free(fr); free(inhibition); free(wm_weights); free(results); - return out; -} - -/* ── Engram persistence (JSON snapshot) ─────────────────────────────────── */ - -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n) { - jb_putc(b, '{'); - jb_puts(b, "\"id\":"); jb_emit_escaped(b, n->id ? n->id : ""); - jb_puts(b, ",\"content\":"); jb_emit_escaped(b, n->content ? n->content : ""); - jb_puts(b, ",\"node_type\":"); jb_emit_escaped(b, n->node_type ? n->node_type : ""); - jb_puts(b, ",\"label\":"); jb_emit_escaped(b, n->label ? n->label : ""); - jb_puts(b, ",\"tier\":"); jb_emit_escaped(b, n->tier ? n->tier : "Working"); - jb_puts(b, ",\"tags\":"); jb_emit_escaped(b, n->tags ? n->tags : ""); - jb_puts(b, ",\"metadata\":"); jb_emit_escaped(b, n->metadata ? n->metadata : "{}"); - char tmp[80]; - snprintf(tmp, sizeof(tmp), ",\"salience\":%g", n->salience); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"importance\":%g", n->importance); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"confidence\":%g", n->confidence); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"temporal_decay_rate\":%g", n->temporal_decay_rate); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"activation_count\":%lld", (long long)n->activation_count); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"last_activated\":%lld", (long long)n->last_activated); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"created_at\":%lld", (long long)n->created_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"updated_at\":%lld", (long long)n->updated_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"background_activation\":%g", n->background_activation); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"working_memory_weight\":%g", n->working_memory_weight); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppression_count\":%d", n->suppression_count); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"layer_id\":%u", n->layer_id); jb_puts(b, tmp); - jb_putc(b, '}'); -} - -static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e) { - jb_putc(b, '{'); - jb_puts(b, "\"id\":"); jb_emit_escaped(b, e->id ? e->id : ""); - jb_puts(b, ",\"from_id\":"); jb_emit_escaped(b, e->from_id ? e->from_id : ""); - jb_puts(b, ",\"to_id\":"); jb_emit_escaped(b, e->to_id ? e->to_id : ""); - jb_puts(b, ",\"relation\":"); jb_emit_escaped(b, e->relation ? e->relation : ""); - jb_puts(b, ",\"metadata\":"); jb_emit_escaped(b, e->metadata ? e->metadata : "{}"); - char tmp[64]; - snprintf(tmp, sizeof(tmp), ",\"weight\":%g", e->weight); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"confidence\":%g", e->confidence); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"created_at\":%lld", (long long)e->created_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"updated_at\":%lld", (long long)e->updated_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"last_fired\":%lld", (long long)e->last_fired); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"inhibitory\":%d", e->inhibitory ? 1 : 0); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"layer_id\":%u", e->layer_id); jb_puts(b, tmp); - jb_putc(b, '}'); -} - -el_val_t engram_save(el_val_t path) { - const char* p = EL_CSTR(path); - if (!p || !*p) return 0; - EngramStore* g = engram_get(); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"nodes\":["); - for (int64_t i = 0; i < g->node_count; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[i]); - } - jb_puts(&b, "],\"edges\":["); - for (int64_t i = 0; i < g->edge_count; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_edge_json(&b, &g->edges[i]); - } - /* Layered consciousness — emit the layer registry under "layers". - * Older readers that don't know about this top-level key will simply - * ignore it (forward compatible). Tombstoned (removed-injectable) - * layers are skipped — they have no name and can't be re-created - * meaningfully on load anyway. */ - jb_puts(&b, "],\"layers\":["); - int first_layer = 1; - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if (!first_layer) jb_putc(&b, ','); - first_layer = 0; - jb_putc(&b, '{'); - char tmp[80]; - snprintf(tmp, sizeof(tmp), "\"layer_id\":%u", L->layer_id); - jb_puts(&b, tmp); - jb_puts(&b, ",\"name\":"); - jb_emit_escaped(&b, L->name); - snprintf(tmp, sizeof(tmp), ",\"activation_priority\":%u", L->activation_priority); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppressible\":%d", L->suppressible ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"transparent\":%d", L->transparent ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"injectable\":%d", L->injectable ? 1 : 0); - jb_puts(&b, tmp); - jb_putc(&b, '}'); - } - jb_puts(&b, "]}"); - { - struct stat _st; - if (stat(p, &_st) == 0 && _st.st_size > 200000 && - (uint64_t)b.len < (uint64_t)_st.st_size / 16) { - fprintf(stderr, "[engram_save] REFUSED sparse write: new %zu vs existing %lld (<1/16) protecting %s\n", - b.len, (long long)_st.st_size, p); - free(b.buf); return 0; - } - } - size_t _plen = strlen(p); - char* _tmp = (char*)malloc(_plen + 5); - if (!_tmp) { free(b.buf); return 0; } - memcpy(_tmp, p, _plen); memcpy(_tmp + _plen, ".tmp", 5); - FILE* f = fopen(_tmp, "wb"); - if (!f) { free(_tmp); free(b.buf); return 0; } - size_t w = fwrite(b.buf, 1, b.len, f); - int wok = (w == b.len); - if (wok) { fflush(f); fsync(fileno(f)); } - fclose(f); free(b.buf); - if (!wok) { unlink(_tmp); free(_tmp); return 0; } - if (rename(_tmp, p) != 0) { unlink(_tmp); free(_tmp); return 0; } - free(_tmp); return 1; -} - -/* Helper: extract a string field from a JSON object substring. */ -static char* eg_get_str_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p) return el_strdup_persist(""); - if (*p != '"') return el_strdup_persist(""); - JsonParser jp = { .p = p, .end = p + strlen(p), .err = 0 }; - char* out = jp_parse_string_raw(&jp); - if (jp.err) { free(out); return el_strdup_persist(""); } - return out; -} - -static double eg_get_num_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p || *p == '"' || *p == '{' || *p == '[') return 0.0; - return strtod(p, NULL); -} - -static int64_t eg_get_int_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p || *p == '"' || *p == '{' || *p == '[') return 0; - return strtoll(p, NULL, 10); -} - -/* Iterate the top-level nodes/edges arrays in a saved snapshot. */ -static const char* eg_skip_ws(const char* p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - return p; -} - -el_val_t engram_load(el_val_t path) { - const char* p = EL_CSTR(path); - if (!p || !*p) return 0; - FILE* f = fopen(p, "rb"); - if (!f) return 0; - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz <= 0) { fclose(f); return 0; } - char* data = malloc((size_t)sz + 1); - if (!data) { fclose(f); return 0; } - size_t got = fread(data, 1, (size_t)sz, f); - fclose(f); - data[got] = '\0'; - - /* Reset store */ - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - free(g->nodes[i].id); free(g->nodes[i].content); free(g->nodes[i].node_type); - free(g->nodes[i].label); free(g->nodes[i].tier); free(g->nodes[i].tags); - free(g->nodes[i].metadata); - } - g->node_count = 0; - for (int64_t i = 0; i < g->edge_count; i++) { - free(g->edges[i].id); free(g->edges[i].from_id); free(g->edges[i].to_id); - free(g->edges[i].relation); free(g->edges[i].metadata); - } - g->edge_count = 0; - - /* Walk nodes array */ - const char* nodes_p = json_find_key(data, "nodes"); - if (nodes_p) { - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == '[') { - nodes_p++; - nodes_p = eg_skip_ws(nodes_p); - while (*nodes_p && *nodes_p != ']') { - if (*nodes_p != '{') { nodes_p++; continue; } - const char* end = json_skip_value(nodes_p); - size_t n = (size_t)(end - nodes_p); - char* obj = malloc(n + 1); - memcpy(obj, nodes_p, n); obj[n] = '\0'; - engram_grow_nodes(); - EngramNode* nn = &g->nodes[g->node_count]; - memset(nn, 0, sizeof(*nn)); - nn->id = eg_get_str_field(obj, "id"); - nn->content = eg_get_str_field(obj, "content"); - nn->node_type = eg_get_str_field(obj, "node_type"); - nn->label = eg_get_str_field(obj, "label"); - nn->tier = eg_get_str_field(obj, "tier"); - nn->tags = eg_get_str_field(obj, "tags"); - nn->metadata = eg_get_str_field(obj, "metadata"); - if (!nn->metadata || !*nn->metadata) { free(nn->metadata); nn->metadata = el_strdup_persist("{}"); } - nn->salience = eg_get_num_field(obj, "salience"); - nn->importance = eg_get_num_field(obj, "importance"); - nn->confidence = eg_get_num_field(obj, "confidence"); - nn->temporal_decay_rate = eg_get_num_field(obj, "temporal_decay_rate"); - /* temporal_decay_rate defaults to 0 (use global) if absent in snapshot */ - nn->activation_count = eg_get_int_field(obj, "activation_count"); - nn->last_activated = eg_get_int_field(obj, "last_activated"); - nn->created_at = eg_get_int_field(obj, "created_at"); - nn->updated_at = eg_get_int_field(obj, "updated_at"); - nn->background_activation = eg_get_num_field(obj, "background_activation"); - nn->working_memory_weight = eg_get_num_field(obj, "working_memory_weight"); - nn->suppression_count = (int32_t)eg_get_int_field(obj, "suppression_count"); - /* layer_id defaults to ENGRAM_LAYER_DEFAULT (core-identity) - * for snapshots that predate the layered schema. We can't - * tell "explicit 0" from "missing field" using the helper - * directly, so probe for the key — if absent, fall back. */ - if (json_find_key(obj, "layer_id")) { - nn->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - nn->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->node_count++; - free(obj); - nodes_p = end; - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == ',') { nodes_p++; nodes_p = eg_skip_ws(nodes_p); } - } - } - } - /* Walk edges array */ - const char* edges_p = json_find_key(data, "edges"); - if (edges_p) { - edges_p = eg_skip_ws(edges_p); - if (*edges_p == '[') { - edges_p++; - edges_p = eg_skip_ws(edges_p); - while (*edges_p && *edges_p != ']') { - if (*edges_p != '{') { edges_p++; continue; } - const char* end = json_skip_value(edges_p); - size_t n = (size_t)(end - edges_p); - char* obj = malloc(n + 1); - memcpy(obj, edges_p, n); obj[n] = '\0'; - engram_grow_edges(); - EngramEdge* ee = &g->edges[g->edge_count]; - memset(ee, 0, sizeof(*ee)); - ee->id = eg_get_str_field(obj, "id"); - ee->from_id = eg_get_str_field(obj, "from_id"); - ee->to_id = eg_get_str_field(obj, "to_id"); - ee->relation = eg_get_str_field(obj, "relation"); - ee->metadata = eg_get_str_field(obj, "metadata"); - if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = el_strdup_persist("{}"); } - ee->weight = eg_get_num_field(obj, "weight"); - ee->confidence = eg_get_num_field(obj, "confidence"); - ee->created_at = eg_get_int_field(obj, "created_at"); - ee->updated_at = eg_get_int_field(obj, "updated_at"); - ee->last_fired = eg_get_int_field(obj, "last_fired"); - ee->inhibitory = (int)eg_get_int_field(obj, "inhibitory"); - if (json_find_key(obj, "layer_id")) { - ee->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - ee->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->edge_count++; - free(obj); - edges_p = end; - edges_p = eg_skip_ws(edges_p); - if (*edges_p == ',') { edges_p++; edges_p = eg_skip_ws(edges_p); } - } - } - } - /* Walk layers array (optional — older snapshots omit this). - * If present we replace the canonical registry entirely; if absent we - * keep whatever the engram_get() init established. */ - const char* layers_p = json_find_key(data, "layers"); - if (layers_p) { - layers_p = eg_skip_ws(layers_p); - if (*layers_p == '[') { - /* Reset existing layer registry. Free strdup'd names; the - * struct array itself can be reused. */ - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) free(g->layers[i].name); - g->layers[i].name = NULL; - } - g->layer_count = 0; - - layers_p++; - layers_p = eg_skip_ws(layers_p); - while (*layers_p && *layers_p != ']') { - if (*layers_p != '{') { layers_p++; continue; } - const char* end = json_skip_value(layers_p); - size_t n = (size_t)(end - layers_p); - char* obj = malloc(n + 1); - memcpy(obj, layers_p, n); obj[n] = '\0'; - if (g->layer_count >= g->layer_capacity) { - size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; - EngramLayer* grown = realloc(g->layers, nc * sizeof(EngramLayer)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(grown + g->layer_capacity, 0, - (nc - g->layer_capacity) * sizeof(EngramLayer)); - g->layers = grown; - g->layer_capacity = nc; - } - EngramLayer* L = &g->layers[g->layer_count]; - memset(L, 0, sizeof(*L)); - L->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - L->activation_priority = (uint32_t)eg_get_int_field(obj, "activation_priority"); - L->suppressible = (int)eg_get_int_field(obj, "suppressible") ? 1 : 0; - L->transparent = (int)eg_get_int_field(obj, "transparent") ? 1 : 0; - L->injectable = (int)eg_get_int_field(obj, "injectable") ? 1 : 0; - char* nm = eg_get_str_field(obj, "name"); - if (nm && *nm) { - L->name = el_strdup_persist(nm); - free(nm); - } else { - free(nm); - L->name = el_strdup_persist(""); - } - g->layer_count++; - free(obj); - layers_p = end; - layers_p = eg_skip_ws(layers_p); - if (*layers_p == ',') { layers_p++; layers_p = eg_skip_ws(layers_p); } - } - } - } - free(data); - return 1; -} - -/* ── Engram JSON-string accessors ───────────────────────────────────────── - * These return pre-serialized JSON strings so callers (especially HTTP - * handlers) don't have to round-trip ElList/ElMap through json_stringify - * — which can't reliably distinguish those structures from raw pointers - * due to el_val_t's type erasure. The runtime knows the real C types and - * can serialize directly. */ - -el_val_t engram_get_node_json(el_val_t id) { - const char* sid = EL_CSTR(id); - EngramNode* n = engram_find_node(sid); - if (!n) return el_wrap_str(el_strdup("{}")); - JsonBuf b; jb_init(&b); - engram_emit_node_json(&b, n); - return el_wrap_str(jb_finish(&b)); -} - -/* engram_get_node_by_label — find the first node whose label field exactly - * matches the given string. Returns the node as a JSON object string, or "{}" - * if no match is found. - * - * Used by chat.el to retrieve well-known nodes (e.g. "conv:history", - * "session:summary") by their stable label rather than by ID, which is immune - * to vector index drift across restarts. - * - * Exact match (strcmp, not istr_contains) because labels like "conv:history" - * must not collide with nodes whose content happens to contain that substring. - * - * Backported verbatim (idiom-adapted to jb_finish) from release runtime - * v1.0.0-20260501 to unblock the soul regen link: chat.el references this - * native but the current runtime lacked its definition. */ -el_val_t engram_get_node_by_label(el_val_t label) { - const char* lbl = EL_CSTR(label); - if (!lbl || !*lbl) return el_wrap_str(el_strdup("{}")); - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - if (n->label && strcmp(n->label, lbl) == 0) { - JsonBuf b; jb_init(&b); - engram_emit_node_json(&b, n); - return el_wrap_str(jb_finish(&b)); - } - } - return el_wrap_str(el_strdup("{}")); -} - -el_val_t engram_search_json(el_val_t query, el_val_t limit) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t lim = (int64_t)limit; - if (lim <= 0) lim = 100; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (q && *q && g->node_count > 0) { - /* Collect candidates from the UNION of tokenized-lexical and semantic - * matches, score each, rank by score, emit the top `lim`. A node is a - * candidate if it covers >=1 query token (tokenized-lexical, #66) OR its - * query cosine clears the threshold (#67). Lexical score is the distinct - * token count (>=1), so any lexical hit outranks a pure-semantic hit - * (cosine < 1); pure-semantic hits are scored by cosine alone. When - * semantic is unavailable qvec is NULL, sem is 0, only tokenized-lexical - * hits are collected, and the stable insertion sort preserves order. */ - char toks[ENGRAM_MAX_QTOKENS][ENGRAM_QTOK_LEN]; - int ntok = engram_tokenize_query(q, toks, ENGRAM_MAX_QTOKENS); - int qdim = 0; - float* qvec = engram_embed_query(q, &qdim); - double sem_min = engram_semantic_min(); - typedef struct { int64_t idx; double score; } Cand; - Cand* cand = malloc((size_t)g->node_count * sizeof(Cand)); - if (cand) { - int64_t nc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - if (engram_layer_is_transparent(n->layer_id)) continue; - int sc = engram_node_match_score(n, toks, ntok); - double sem = qvec ? engram_node_cosine(n, qvec, qdim) : 0.0; - if (sc > 0 || sem >= sem_min) { - cand[nc].idx = i; - cand[nc].score = (double)sc + sem; - nc++; - } - } - /* Insertion sort by score desc; stable for equal scores. */ - for (int64_t i = 1; i < nc; i++) { - Cand k = cand[i]; int64_t j = i - 1; - while (j >= 0 && cand[j].score < k.score) { cand[j + 1] = cand[j]; j--; } - cand[j + 1] = k; - } - int first = 1; - for (int64_t i = 0; i < nc && i < lim; i++) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[cand[i].idx]); - first = 0; - } - free(cand); - } - free(qvec); - } - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset) { - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (g->node_count == 0) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - /* Skip transparent layers — introspection filter, same as engram_scan_nodes. */ - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - int first = 1; - for (int64_t i = off; i < end; i++) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - first = 0; - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -/* engram_scan_nodes_by_type_json — filter by node_type before paginating. - * Empty / NULL type_v falls back to the unfiltered scan (existing behaviour). - * Result is JSON array, salience-sorted, transparent layers skipped. */ -el_val_t engram_scan_nodes_by_type_json(el_val_t type_v, el_val_t limit, el_val_t offset) { - const char* type_filter = EL_CSTR(type_v); - if (!type_filter || !*type_filter) { - return engram_scan_nodes_json(limit, offset); - } - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (g->node_count == 0) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - const char* nt = g->nodes[i].node_type; - if (!nt || strcmp(nt, type_filter) != 0) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - int first = 1; - for (int64_t i = off; i < end; i++) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - first = 0; - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction) { - /* Re-implement here directly so we serialize without going through - * the ElList path. Walks BFS to max_depth, emits {node, edge, hops} - * triples. */ - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - int64_t depth = (int64_t)max_depth; if (depth <= 0) depth = 1; - const char* dir = EL_CSTR(direction); if (!dir) dir = "both"; - int allow_out = (strcmp(dir, "out") == 0) || (strcmp(dir, "both") == 0); - int allow_in = (strcmp(dir, "in") == 0) || (strcmp(dir, "both") == 0); - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (!sid || !*sid) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - - /* Frontier of (node_id, hops). Cap to a sane size. */ - char** frontier = calloc(1024, sizeof(char*)); - int64_t* frontier_h = calloc(1024, sizeof(int64_t)); - int64_t fc = 0; - char** visited = calloc(1024, sizeof(char*)); - int64_t vc = 0; - if (!frontier || !frontier_h || !visited) { - free(frontier); free(frontier_h); free(visited); - jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); - } - /* Use plain strdup (not el_strdup) so arena doesn't track these pointers. - * The BFS loop manually frees them below — arena would double-free them. */ - frontier[fc] = strdup(sid); frontier_h[fc] = 0; fc++; - visited[vc++] = strdup(sid); - - int first = 1; - while (fc > 0) { - char* cur = frontier[0]; int64_t h = frontier_h[0]; - for (int64_t k = 1; k < fc; k++) { frontier[k-1] = frontier[k]; frontier_h[k-1] = frontier_h[k]; } - fc--; - if (h >= depth) { free(cur); continue; } - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* peer = NULL; - if (allow_out && e->from_id && strcmp(e->from_id, cur) == 0) peer = e->to_id; - else if (allow_in && e->to_id && strcmp(e->to_id, cur) == 0) peer = e->from_id; - if (!peer) continue; - int seen = 0; - for (int64_t v = 0; v < vc; v++) { - if (strcmp(visited[v], peer) == 0) { seen = 1; break; } - } - if (seen) continue; - EngramNode* n = engram_find_node(peer); - if (!n) continue; - if (!first) jb_putc(&b, ','); - jb_puts(&b, "{\"node\":"); - engram_emit_node_json(&b, n); - jb_puts(&b, ",\"edge\":"); - engram_emit_edge_json(&b, e); - char tmp[64]; snprintf(tmp, sizeof(tmp), ",\"hops\":%lld}", (long long)(h + 1)); - jb_puts(&b, tmp); - first = 0; - if (vc < 1024) visited[vc++] = strdup(peer); - if (fc < 1024 && h + 1 < depth) { frontier[fc] = strdup(peer); frontier_h[fc] = h + 1; fc++; } - } - free(cur); - } - for (int64_t i = 0; i < fc; i++) free(frontier[i]); - for (int64_t i = 0; i < vc; i++) free(visited[i]); - free(frontier); free(frontier_h); free(visited); - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -el_val_t engram_activate_json(el_val_t query, el_val_t depth) { - /* Run two-layer engram_activate and serialize the result list to JSON. - * Each entry includes both activation_strength (layer 1 background) and - * working_memory_weight (layer 2 executive filter), plus promoted flag. - * Callers performing context compilation should filter to promoted=1. */ - el_val_t lst = engram_activate(query, depth); - ElList* arr = (ElList*)(uintptr_t)lst; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (arr) { - for (int64_t i = 0; i < arr->length; i++) { - if (!arr->elems[i]) continue; - el_val_t node_map = el_map_get(arr->elems[i], EL_STR("node")); - el_val_t strength_v = el_map_get(arr->elems[i], EL_STR("activation_strength")); - el_val_t wm_v = el_map_get(arr->elems[i], EL_STR("working_memory_weight")); - el_val_t epist_v = el_map_get(arr->elems[i], EL_STR("epistemic_confidence")); - el_val_t hops_v = el_map_get(arr->elems[i], EL_STR("hops")); - el_val_t promoted_v = el_map_get(arr->elems[i], EL_STR("promoted")); - /* Look up underlying EngramNode by id to emit canonical JSON. */ - el_val_t id_v = el_map_get(node_map, EL_STR("id")); - const char* id_s = EL_CSTR(id_v); - EngramNode* n = id_s ? engram_find_node(id_s) : NULL; - if (i > 0) jb_putc(&b, ','); - jb_puts(&b, "{\"node\":"); - if (n) { - engram_emit_node_json(&b, n); - } else { - jb_puts(&b, "{}"); - } - char tmp[80]; - snprintf(tmp, sizeof(tmp), ",\"activation_strength\":%g", el_to_float(strength_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"working_memory_weight\":%g", el_to_float(wm_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"epistemic_confidence\":%g", el_to_float(epist_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"hops\":%lld", (long long)(int64_t)hops_v); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"promoted\":%d}", (int)(int64_t)promoted_v); jb_puts(&b, tmp); - } - } - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -el_val_t engram_stats_json(void) { - EngramStore* g = engram_get(); - char buf[128]; - snprintf(buf, sizeof(buf), - "{\"node_count\":%lld,\"edge_count\":%lld,\"layer_count\":%zu}", - (long long)g->node_count, (long long)g->edge_count, g->layer_count); - return el_wrap_str(el_strdup(buf)); -} - -/* engram_list_layers_json — serialized counterpart of engram_list_layers. - * Returns a JSON array, sorted by activation_priority ascending. */ -el_val_t engram_list_layers_json(void) { - EngramStore* g = engram_get(); - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - /* Build a sorted index over live layers. */ - size_t* idx = malloc((g->layer_count + 1) * sizeof(size_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(jb_finish(&b)); } - size_t live = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) idx[live++] = i; - } - for (size_t i = 1; i < live; i++) { - size_t key = idx[i]; - uint32_t kp = g->layers[key].activation_priority; - size_t j = i; - while (j > 0 && g->layers[idx[j - 1]].activation_priority > kp) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - int first = 1; - for (size_t i = 0; i < live; i++) { - EngramLayer* L = &g->layers[idx[i]]; - if (!first) jb_putc(&b, ','); - first = 0; - jb_putc(&b, '{'); - char tmp[80]; - snprintf(tmp, sizeof(tmp), "\"layer_id\":%u", L->layer_id); jb_puts(&b, tmp); - jb_puts(&b, ",\"name\":"); - jb_emit_escaped(&b, L->name ? L->name : ""); - snprintf(tmp, sizeof(tmp), ",\"activation_priority\":%u", L->activation_priority); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppressible\":%d", L->suppressible ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"transparent\":%d", L->transparent ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"injectable\":%d", L->injectable ? 1 : 0); - jb_puts(&b, tmp); - jb_putc(&b, '}'); - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -/* engram_compile_layered_json — produce a prompt-ready context block split - * by layer. - * - * Runs the three-pass activation, then partitions promoted nodes by layer - * suppressibility: - * - Non-suppressible (Layer 0 / structural-floor) layers go FIRST under - * the heading "[LAYER 0 — STRUCTURAL]". These are the sacred-fire - * nodes that surfaced via the pass-3 override. - * - All other promoted layers go SECOND under "[ENGRAM CONTEXT]". - * - * Output is a single JSON-string el_val_t: a UTF-8 text block ready to be - * concatenated into a system prompt. Returns "" if no nodes promoted. - * - * Transparent layers (Layer 0) are emitted into the prompt — they shape - * the model's output — but engram_search and friends still hide them from - * introspection-style queries. The split heading lets the LLM weight them - * appropriately without revealing their internal label. - * - * Each emitted line for a node is its raw JSON (matching engram_emit_node_json) - * so downstream JSON parsers can still walk individual records inside the - * formatted block. The block is plain text, not a JSON document — callers - * concatenating it into a prompt should treat it as opaque markdown. */ -el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth) { - EngramStore* g = engram_get(); - /* Run the three-pass activator. We need the persisted node fields, so - * call engram_activate (it writes background_activation and - * working_memory_weight back into the store). */ - (void)engram_activate(intent, depth); - - /* Walk the store and partition by suppressibility. */ - JsonBuf b; jb_init(&b); - int wrote_layer0 = 0; - int wrote_normal = 0; - - /* Sort indices by working_memory_weight descending so the most - * confidently promoted nodes appear first within each section. */ - int64_t* idx = malloc((size_t)(g->node_count + 1) * sizeof(int64_t)); - if (!idx) return el_wrap_str(el_strdup("")); - int64_t mc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].working_memory_weight > 0.0) idx[mc++] = i; - } - for (int64_t i = 1; i < mc; i++) { - int64_t key = idx[i]; - double kw = g->nodes[key].working_memory_weight; - int64_t j = i; - while (j > 0 && g->nodes[idx[j - 1]].working_memory_weight < kw) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - - /* Section 1: structural floor (non-suppressible layers). */ - for (int64_t i = 0; i < mc; i++) { - EngramNode* n = &g->nodes[idx[i]]; - if (engram_layer_is_suppressible(n->layer_id)) continue; - if (!wrote_layer0) { - jb_puts(&b, "[LAYER 0 — STRUCTURAL]\n"); - wrote_layer0 = 1; - } - engram_emit_node_json(&b, n); - jb_putc(&b, '\n'); - } - - /* Section 2: standard engram context (suppressible layers). */ - for (int64_t i = 0; i < mc; i++) { - EngramNode* n = &g->nodes[idx[i]]; - if (!engram_layer_is_suppressible(n->layer_id)) continue; - if (!wrote_normal) { - if (wrote_layer0) jb_putc(&b, '\n'); - jb_puts(&b, "[ENGRAM CONTEXT]\n"); - wrote_normal = 1; - } - engram_emit_node_json(&b, n); - jb_putc(&b, '\n'); - } - - free(idx); - if (b.len == 0) { - free(b.buf); - return el_wrap_str(el_strdup("")); - } - return el_wrap_str(jb_finish(&b)); -} - -/* engram_query_range — temporal range query. - * Returns a JSON array of nodes whose created_at OR last_activated falls - * within [start_ms, end_ms], sorted by created_at ascending. - * Enables "what was I working on last Tuesday?" style queries by passing - * unix-millisecond timestamps for the start and end of the target interval. - * Both endpoints are inclusive. Pass 0 for start_ms to mean "beginning of - * time"; pass 0 for end_ms to mean "now". */ -el_val_t engram_query_range(el_val_t start_ms_v, el_val_t end_ms_v) { - EngramStore* g = engram_get(); - int64_t start_ms = (int64_t)start_ms_v; - int64_t end_ms = (int64_t)end_ms_v; - if (end_ms <= 0) end_ms = engram_now_ms(); - - /* Collect matching indices. */ - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) return el_wrap_str(el_strdup("[]")); - int64_t mc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - int in_created = (n->created_at >= start_ms && n->created_at <= end_ms); - int in_activated = (n->last_activated >= start_ms && n->last_activated <= end_ms); - if (in_created || in_activated) idx[mc++] = i; - } - /* Sort by created_at ascending (insertion sort — N is small in practice). */ - for (int64_t i = 1; i < mc; i++) { - int64_t key = idx[i]; - int64_t kts = g->nodes[key].created_at; - int64_t j = i - 1; - while (j >= 0 && g->nodes[idx[j]].created_at > kts) { - idx[j + 1] = idx[j]; - j--; - } - idx[j + 1] = key; - } - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - for (int64_t i = 0; i < mc; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - } - jb_putc(&b, ']'); - free(idx); - return el_wrap_str(jb_finish(&b)); -} - -/* engram_load_merge — like engram_load but WITHOUT resetting the store. - * Reads a JSON snapshot from `path` and adds any nodes/edges not already - * present in the in-memory graph. Dedup is by node id (for nodes) and by - * (from_id, to_id, relation) tuple (for edges). - * - * Returns (as an EL int) the count of new nodes added. Embeddings are - * intentionally skipped on merged nodes to avoid Ollama delays at runtime; - * auto_link_semantic will handle them when nodes are next activated. - * - * Does not merge layers — the in-process layer registry is authoritative. */ -el_val_t engram_load_merge(el_val_t path) { - const char* p = EL_CSTR(path); - if (!p || !*p) return 0; - FILE* f = fopen(p, "rb"); - if (!f) return 0; - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz <= 0) { fclose(f); return 0; } - char* data = malloc((size_t)sz + 1); - if (!data) { fclose(f); return 0; } - size_t got = fread(data, 1, (size_t)sz, f); - fclose(f); - data[got] = '\0'; - - EngramStore* g = engram_get(); - int64_t added_nodes = 0; - - /* Walk nodes array — skip any node whose id already exists */ - const char* nodes_p = json_find_key(data, "nodes"); - if (nodes_p) { - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == '[') { - nodes_p++; - nodes_p = eg_skip_ws(nodes_p); - while (*nodes_p && *nodes_p != ']') { - if (*nodes_p != '{') { nodes_p++; continue; } - const char* end = json_skip_value(nodes_p); - size_t n = (size_t)(end - nodes_p); - char* obj = malloc(n + 1); - memcpy(obj, nodes_p, n); obj[n] = '\0'; - char* nid = eg_get_str_field(obj, "id"); - int already = (nid && *nid && engram_find_node(nid) != NULL); - free(nid); - if (!already) { - engram_grow_nodes(); - EngramNode* nn = &g->nodes[g->node_count]; - memset(nn, 0, sizeof(*nn)); - nn->id = eg_get_str_field(obj, "id"); - nn->content = eg_get_str_field(obj, "content"); - nn->node_type = eg_get_str_field(obj, "node_type"); - nn->label = eg_get_str_field(obj, "label"); - nn->tier = eg_get_str_field(obj, "tier"); - nn->tags = eg_get_str_field(obj, "tags"); - nn->metadata = eg_get_str_field(obj, "metadata"); - if (!nn->metadata || !*nn->metadata) { free(nn->metadata); nn->metadata = strdup("{}"); } - nn->salience = eg_get_num_field(obj, "salience"); - nn->importance = eg_get_num_field(obj, "importance"); - nn->confidence = eg_get_num_field(obj, "confidence"); - nn->temporal_decay_rate = eg_get_num_field(obj, "temporal_decay_rate"); - nn->activation_count = eg_get_int_field(obj, "activation_count"); - nn->last_activated = eg_get_int_field(obj, "last_activated"); - nn->created_at = eg_get_int_field(obj, "created_at"); - nn->updated_at = eg_get_int_field(obj, "updated_at"); - nn->background_activation = eg_get_num_field(obj, "background_activation"); - nn->working_memory_weight = eg_get_num_field(obj, "working_memory_weight"); - if (!isfinite(nn->working_memory_weight) || nn->working_memory_weight < 0.0 || nn->working_memory_weight > 1.0) - nn->working_memory_weight = 0.0; /* clamp corrupt snapshot values */ - nn->suppression_count = (int32_t)eg_get_int_field(obj, "suppression_count"); - if (json_find_key(obj, "layer_id")) { - nn->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - nn->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->node_count++; - added_nodes++; - } - free(obj); - nodes_p = end; - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == ',') { nodes_p++; nodes_p = eg_skip_ws(nodes_p); } - } - } - } - - /* Walk edges array — skip if (from_id, to_id, relation) already present */ - const char* edges_p = json_find_key(data, "edges"); - if (edges_p) { - edges_p = eg_skip_ws(edges_p); - if (*edges_p == '[') { - edges_p++; - edges_p = eg_skip_ws(edges_p); - while (*edges_p && *edges_p != ']') { - if (*edges_p != '{') { edges_p++; continue; } - const char* end = json_skip_value(edges_p); - size_t n = (size_t)(end - edges_p); - char* obj = malloc(n + 1); - memcpy(obj, edges_p, n); obj[n] = '\0'; - char* efrom = eg_get_str_field(obj, "from_id"); - char* eto = eg_get_str_field(obj, "to_id"); - char* erel = eg_get_str_field(obj, "relation"); - /* Check for duplicate by scanning existing edges */ - int dup = 0; - if (efrom && eto && erel) { - for (int64_t ei = 0; ei < g->edge_count; ei++) { - EngramEdge* ex = &g->edges[ei]; - if (ex->from_id && ex->to_id && ex->relation && - strcmp(ex->from_id, efrom) == 0 && - strcmp(ex->to_id, eto) == 0 && - strcmp(ex->relation, erel) == 0) { - dup = 1; break; - } - } - } - if (!dup) { - engram_grow_edges(); - EngramEdge* ee = &g->edges[g->edge_count]; - memset(ee, 0, sizeof(*ee)); - ee->id = eg_get_str_field(obj, "id"); - ee->from_id = efrom ? efrom : strdup(""); - ee->to_id = eto ? eto : strdup(""); - ee->relation = erel ? erel : strdup(""); - ee->metadata = eg_get_str_field(obj, "metadata"); - if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = strdup("{}"); } - ee->weight = eg_get_num_field(obj, "weight"); - ee->confidence = eg_get_num_field(obj, "confidence"); - ee->created_at = eg_get_int_field(obj, "created_at"); - ee->updated_at = eg_get_int_field(obj, "updated_at"); - ee->last_fired = eg_get_int_field(obj, "last_fired"); - ee->inhibitory = (int)eg_get_int_field(obj, "inhibitory"); - if (json_find_key(obj, "layer_id")) { - ee->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - ee->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->edge_count++; - /* NOTE: efrom/eto/erel ownership transferred to ee above */ - efrom = NULL; eto = NULL; erel = NULL; - } else { - free(efrom); free(eto); free(erel); - } - free(obj); - edges_p = end; - edges_p = eg_skip_ws(edges_p); - if (*edges_p == ',') { edges_p++; edges_p = eg_skip_ws(edges_p); } - } - } - } - - free(data); - return (el_val_t)added_nodes; -} - -el_val_t engram_wm_count(void) { - EngramStore* g = engram_get(); - int64_t count = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].working_memory_weight > 0.0) count++; - } - return (el_val_t)count; -} - -/* Average working_memory_weight across all promoted nodes (wm > 0). - * Returns the float bit-pattern via el_from_float so EL can use it with - * float_to_str / float_gt. Returns 0.0 when no nodes are promoted. - * Useful in heartbeat ISEs to distinguish "many weak activations" (sparse - * graph, low avg) from "few strong activations" (dense subgraph, high avg). - * Added 2026-06-04 self-review for graph health observability. */ -el_val_t engram_wm_avg_weight(void) { - EngramStore* g = engram_get(); - double sum = 0.0; - int64_t count = 0; - for (int64_t i = 0; i < g->node_count; i++) { - double w = g->nodes[i].working_memory_weight; - /* Defensive guard: skip any corrupt/out-of-range values so a single - * bad snapshot node doesn't produce a garbage average (e.g. 1.77e+234). */ - if (w > 0.0 && w <= 1.0 && isfinite(w)) { sum += w; count++; } - } - double avg = (count > 0) ? (sum / (double)count) : 0.0; - return el_from_float(avg); -} - -/* engram_wm_top_json — return top N working-memory nodes (by wm weight) as a - * compact JSON array for ISE heartbeat reporting. - * - * Each element: {"label":"...","node_type":"...","tier":"...","wm":0.42} - * - * Purpose: the heartbeat ISE reports wm_active (count) and wm_avg_weight but - * gives zero visibility into WM *composition* — which types/tiers are active. - * After long uptime every WM slot is in steady-state decay+re-promotion so - * wm_promotion ISEs never fire (they only fire on 0→>0.1 transitions). - * This function fills the observability gap by snapshotting the current top-N - * WM nodes on every heartbeat. Inserted 2026-06-05 self-review. */ -el_val_t engram_wm_top_json(el_val_t n_v) { - int64_t top_n = (int64_t)n_v; - if (top_n <= 0) top_n = 10; - if (top_n > 50) top_n = 50; - EngramStore* g = engram_get(); - - /* Collect indices of promoted nodes, excluding monitoring noise. - * InternalStateEvent nodes are system-observation artifacts — they reflect - * what the daemon is doing, not what it knows. Including them in wm_top - * buries real knowledge (Memory, Knowledge, Belief nodes) under a wall of - * heartbeat/curiosity ISEs, making the heartbeat ISE useless for diagnosing - * WM composition. Filter them out here so wm_top always shows substantive - * content. (2026-06-07 self-review) */ - int64_t* idx = malloc((size_t)(g->node_count + 1) * sizeof(int64_t)); - if (!idx) return el_wrap_str(el_strdup("[]")); - int64_t mc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].working_memory_weight > 0.0) { - const char* nt = g->nodes[i].node_type; - if (nt && strcmp(nt, "InternalStateEvent") == 0) continue; - idx[mc++] = i; - } - } - - /* Insertion-sort descending by wm weight (mc is typically small). */ - for (int64_t i = 1; i < mc; i++) { - int64_t key = idx[i]; - double kw = g->nodes[key].working_memory_weight; - int64_t j = i; - while (j > 0 && g->nodes[idx[j-1]].working_memory_weight < kw) { - idx[j] = idx[j-1]; j--; - } - idx[j] = key; - } - - int64_t emit = mc < top_n ? mc : top_n; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - for (int64_t k = 0; k < emit; k++) { - EngramNode* n = &g->nodes[idx[k]]; - if (k > 0) jb_putc(&b, ','); - jb_putc(&b, '{'); - jb_puts(&b, "\"label\":"); - jb_emit_escaped(&b, n->label ? n->label : ""); - jb_puts(&b, ",\"node_type\":"); - jb_emit_escaped(&b, n->node_type ? n->node_type : ""); - jb_puts(&b, ",\"tier\":"); - jb_emit_escaped(&b, n->tier ? n->tier : ""); - char tmp[48]; - snprintf(tmp, sizeof(tmp), ",\"wm\":%.3f", n->working_memory_weight); - jb_puts(&b, tmp); - jb_putc(&b, '}'); - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(jb_finish(&b)); -} - -#ifdef HAVE_CURL -/* ── DHARMA network ───────────────────────────────────────────────────────── - * Real implementation. Peers are addressed by `dharma_id` — either bare - * (e.g. "ntn-genesis", transport defaults to http://localhost:7770) or - * "@" where is the peer's Engram-exposed daemon. - * - * Channels are logical handles cached per-cgi: `dharma_connect` is - * idempotent and returns "ch:". The channel registry below tracks - * every cgi_id we've connected to and its resolved transport URL. - * - * Relationship weights live in the local Engram graph: edges of type - * "dharma-relation" between a synthetic local node ("dharma:self") and - * synthetic peer nodes ("dharma:peer:"). Hebbian increments - * accumulate in EngramEdge.weight, clamped to [0.0, 1.0]. - * - * Events arrive over HTTP via the application's request handler, which is - * expected to call el_runtime_dharma_event_arrive() when it sees a - * /dharma/event POST. dharma_field() blocks on a per-event-type queue. - */ - -#define DHARMA_DEFAULT_URL "http://localhost:7770" - -/* Channel registry — one entry per known peer. */ -typedef struct DharmaChannel { - char* cgi_id; /* full dharma_id including any @ suffix */ - char* base_id; /* registry-id portion (before @) for relationship lookup */ - char* url; /* resolved transport URL */ - char* channel_id; /* "ch:" */ -} DharmaChannel; - -static DharmaChannel* _dharma_channels = NULL; -static size_t _dharma_channel_count = 0; -static size_t _dharma_channel_cap = 0; -static pthread_mutex_t _dharma_channel_mu = PTHREAD_MUTEX_INITIALIZER; - -/* Event queue — per-type linked list. dharma_field blocks on _dharma_event_cv. */ -typedef struct DharmaEvent { - char* event_type; - char* payload; - char* source; - int64_t timestamp; - struct DharmaEvent* next; -} DharmaEvent; - -static DharmaEvent* _dharma_event_head = NULL; -static DharmaEvent* _dharma_event_tail = NULL; -static pthread_mutex_t _dharma_event_mu = PTHREAD_MUTEX_INITIALIZER; -static pthread_cond_t _dharma_event_cv = PTHREAD_COND_INITIALIZER; - -/* Split "@" → (base_id, url). If no "@", base_id = full, url = default. - * Returned strings are heap-allocated; caller must free. */ -static void dharma_parse_id(const char* full, char** out_base, char** out_url) { - if (!full) full = ""; - const char* at = strchr(full, '@'); - if (at) { - size_t bn = (size_t)(at - full); - char* b = malloc(bn + 1); - memcpy(b, full, bn); b[bn] = '\0'; - *out_base = b; - *out_url = el_strdup(at + 1); - if (!**out_url) { free(*out_url); *out_url = el_strdup(DHARMA_DEFAULT_URL); } - } else { - *out_base = el_strdup(full); - *out_url = el_strdup(DHARMA_DEFAULT_URL); - } -} - -/* Find existing channel by full cgi_id. Caller must hold _dharma_channel_mu. */ -static DharmaChannel* dharma_find_channel_locked(const char* cgi_id) { - if (!cgi_id) return NULL; - for (size_t i = 0; i < _dharma_channel_count; i++) { - if (_dharma_channels[i].cgi_id && - strcmp(_dharma_channels[i].cgi_id, cgi_id) == 0) { - return &_dharma_channels[i]; - } - } - return NULL; -} - -/* Add a new channel entry. Caller must hold _dharma_channel_mu. */ -static DharmaChannel* dharma_add_channel_locked(const char* cgi_id) { - if (_dharma_channel_count >= _dharma_channel_cap) { - size_t nc = _dharma_channel_cap ? _dharma_channel_cap * 2 : 8; - _dharma_channels = realloc(_dharma_channels, nc * sizeof(DharmaChannel)); - if (!_dharma_channels) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(_dharma_channels + _dharma_channel_cap, 0, - (nc - _dharma_channel_cap) * sizeof(DharmaChannel)); - _dharma_channel_cap = nc; - } - DharmaChannel* ch = &_dharma_channels[_dharma_channel_count++]; - char* base = NULL; char* url = NULL; - dharma_parse_id(cgi_id, &base, &url); - ch->cgi_id = el_strdup(cgi_id ? cgi_id : ""); - ch->base_id = base; - ch->url = url; - size_t cn = strlen(ch->cgi_id) + 4; - ch->channel_id = malloc(cn); - snprintf(ch->channel_id, cn, "ch:%s", ch->cgi_id); - return ch; -} - -el_val_t dharma_connect(el_val_t cgi_id) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return el_wrap_str(el_strdup("")); - pthread_mutex_lock(&_dharma_channel_mu); - DharmaChannel* ch = dharma_find_channel_locked(id); - if (!ch) ch = dharma_add_channel_locked(id); - char* out = el_strdup(ch->channel_id); - pthread_mutex_unlock(&_dharma_channel_mu); - return el_wrap_str(out); -} - -/* Build an error JSON body — same shape http_error_json uses. */ -static el_val_t dharma_error_json(const char* msg) { - return http_error_json(msg); -} - -el_val_t dharma_send(el_val_t channel, el_val_t content) { - const char* ch_id = EL_CSTR(channel); - const char* msg = EL_CSTR(content); - if (!ch_id || strncmp(ch_id, "ch:", 3) != 0) { - return dharma_error_json("invalid channel"); - } - const char* peer_id = ch_id + 3; - /* Look up channel; if unknown (caller fabricated), auto-register. */ - pthread_mutex_lock(&_dharma_channel_mu); - DharmaChannel* ch = dharma_find_channel_locked(peer_id); - if (!ch) ch = dharma_add_channel_locked(peer_id); - char* url = el_strdup(ch->url); - pthread_mutex_unlock(&_dharma_channel_mu); - /* Build /dharma/recv body. */ - const char* from = _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unknown)"; - char* esc_ch = json_escape_alloc(ch_id); - char* esc_from = json_escape_alloc(from); - char* esc_msg = json_escape_alloc(msg ? msg : ""); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"channel\":\""); jb_puts(&b, esc_ch); - jb_puts(&b, "\",\"from\":\""); jb_puts(&b, esc_from); - jb_puts(&b, "\",\"content\":\""); jb_puts(&b, esc_msg); - jb_puts(&b, "\"}"); - free(esc_ch); free(esc_from); free(esc_msg); - size_t ul = strlen(url) + 16; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/dharma/recv", url); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t resp = http_do("POST", full_url, b.buf, h); - curl_slist_free_all(h); - free(b.buf); free(full_url); free(url); - return resp; -} - -el_val_t dharma_activate(el_val_t query) { - const char* q = EL_CSTR(query); - if (!q) q = ""; - el_val_t out = el_list_empty(); - char* esc_q = json_escape_alloc(q); - JsonBuf body; jb_init(&body); - jb_puts(&body, "{\"query\":\""); jb_puts(&body, esc_q); jb_puts(&body, "\"}"); - free(esc_q); - - /* Snapshot the channel list under lock so we can iterate without - * holding the mutex during network I/O. */ - pthread_mutex_lock(&_dharma_channel_mu); - size_t n = _dharma_channel_count; - char** urls = calloc(n ? n : 1, sizeof(char*)); - char** ids = calloc(n ? n : 1, sizeof(char*)); - char** bases = calloc(n ? n : 1, sizeof(char*)); - for (size_t i = 0; i < n; i++) { - urls[i] = el_strdup(_dharma_channels[i].url); - ids[i] = el_strdup(_dharma_channels[i].cgi_id); - bases[i] = el_strdup(_dharma_channels[i].base_id); - } - pthread_mutex_unlock(&_dharma_channel_mu); - - for (size_t i = 0; i < n; i++) { - size_t ul = strlen(urls[i]) + 32; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/api/activate", urls[i]); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t resp = http_do("POST", full_url, body.buf, h); - curl_slist_free_all(h); - free(full_url); - const char* rs = EL_CSTR(resp); - if (!rs || !*rs) continue; - if (rs[0] == '{' && strstr(rs, "\"error\"")) continue; - - /* Look up relationship weight (attenuation). */ - double rel_weight = 1.0; - { - const char* self_id = "dharma:self"; - char peer_node[512]; - snprintf(peer_node, sizeof(peer_node), "dharma:peer:%s", bases[i]); - EngramStore* g = engram_get(); - for (int64_t k = 0; k < g->edge_count; k++) { - EngramEdge* e = &g->edges[k]; - if (e->from_id && e->to_id && - strcmp(e->from_id, self_id) == 0 && - strcmp(e->to_id, peer_node) == 0 && - e->relation && strcmp(e->relation, "dharma-relation") == 0) { - rel_weight = e->weight; - break; - } - } - } - - /* Iterate the response array. Expect either a top-level array - * or an object whose "results" field is an array. */ - const char* arr = rs; - while (*arr == ' ' || *arr == '\t' || *arr == '\n' || *arr == '\r') arr++; - char* arr_owned = NULL; - if (*arr == '{') { - el_val_t r = json_get_raw(EL_STR(rs), EL_STR("results")); - const char* rr = EL_CSTR(r); - if (rr && *rr == '[') { - arr_owned = el_strdup(rr); - arr = arr_owned; - } else { - continue; - } - } - if (*arr != '[') { free(arr_owned); continue; } - const char* p = arr + 1; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - while (*p && *p != ']') { - const char* end = json_skip_value(p); - size_t en = (size_t)(end - p); - char* obj = el_strbuf(en); - memcpy(obj, p, en); obj[en] = '\0'; - - /* Pull activation_strength if present, else 1.0. */ - el_val_t act_v = json_get_float(EL_STR(obj), EL_STR("activation_strength")); - double act = el_to_float(act_v); - if (!(act > 0.0 && act <= 100.0)) act = 1.0; - double final_act = act * rel_weight; - - el_val_t entry = el_map_new(0); - /* node = the inner JSON if present, else the entire obj. */ - el_val_t node_raw = json_get_raw(EL_STR(obj), EL_STR("node")); - const char* nr = EL_CSTR(node_raw); - entry = el_map_set(entry, EL_STR(el_strdup("node")), - (nr && *nr) ? node_raw : EL_STR(el_strdup(obj))); - entry = el_map_set(entry, EL_STR(el_strdup("source_cgi")), - EL_STR(el_strdup(ids[i]))); - entry = el_map_set(entry, EL_STR(el_strdup("activation_strength")), - el_from_float(final_act)); - out = el_list_append(out, entry); - free(obj); - p = end; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - } - free(arr_owned); - } - for (size_t i = 0; i < n; i++) { free(urls[i]); free(ids[i]); free(bases[i]); } - free(urls); free(ids); free(bases); - free(body.buf); - return out; -} - -void dharma_emit(el_val_t event_type, el_val_t payload) { - const char* et = EL_CSTR(event_type); - const char* pay = EL_CSTR(payload); - if (!et) et = ""; - if (!pay) pay = ""; - const char* src = _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unknown)"; - int64_t ts = engram_now_ms(); - - char* esc_et = json_escape_alloc(et); - char* esc_pay = json_escape_alloc(pay); - char* esc_src = json_escape_alloc(src); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"type\":\""); jb_puts(&b, esc_et); - jb_puts(&b, "\",\"payload\":\""); jb_puts(&b, esc_pay); - jb_puts(&b, "\",\"source\":\""); jb_puts(&b, esc_src); - jb_puts(&b, "\",\"timestamp\":"); jb_emit_int(&b, ts); - jb_putc(&b, '}'); - free(esc_et); free(esc_pay); free(esc_src); - - /* Snapshot URLs to avoid holding the channel mutex during I/O. */ - pthread_mutex_lock(&_dharma_channel_mu); - size_t n = _dharma_channel_count; - char** urls = calloc(n ? n : 1, sizeof(char*)); - for (size_t i = 0; i < n; i++) urls[i] = el_strdup(_dharma_channels[i].url); - pthread_mutex_unlock(&_dharma_channel_mu); - - for (size_t i = 0; i < n; i++) { - size_t ul = strlen(urls[i]) + 32; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/dharma/event", urls[i]); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t r = http_do("POST", full_url, b.buf, h); - (void)r; /* fire-and-forget — emit is not synchronous */ - curl_slist_free_all(h); - free(full_url); - } - for (size_t i = 0; i < n; i++) free(urls[i]); - free(urls); - free(b.buf); -} - -void el_runtime_dharma_event_arrive(const char* event_type, const char* payload, - const char* source) { - DharmaEvent* ev = calloc(1, sizeof(DharmaEvent)); - if (!ev) return; - ev->event_type = el_strdup(event_type ? event_type : ""); - ev->payload = el_strdup(payload ? payload : ""); - ev->source = el_strdup(source ? source : ""); - ev->timestamp = engram_now_ms(); - ev->next = NULL; - pthread_mutex_lock(&_dharma_event_mu); - if (_dharma_event_tail) _dharma_event_tail->next = ev; - else _dharma_event_head = ev; - _dharma_event_tail = ev; - pthread_cond_broadcast(&_dharma_event_cv); - pthread_mutex_unlock(&_dharma_event_mu); -} - -el_val_t dharma_field(el_val_t event_type) { - const char* et = EL_CSTR(event_type); - if (!et) et = ""; - - /* Compute deadline: now + 30 seconds. */ - struct timespec deadline; - clock_gettime(CLOCK_REALTIME, &deadline); - deadline.tv_sec += 30; - - DharmaEvent* found = NULL; - pthread_mutex_lock(&_dharma_event_mu); - while (1) { - /* Scan queue for matching type; pop and return first match. */ - DharmaEvent* prev = NULL; - DharmaEvent* cur = _dharma_event_head; - while (cur) { - if (cur->event_type && strcmp(cur->event_type, et) == 0) { - if (prev) prev->next = cur->next; - else _dharma_event_head = cur->next; - if (_dharma_event_tail == cur) _dharma_event_tail = prev; - cur->next = NULL; - found = cur; - break; - } - prev = cur; cur = cur->next; - } - if (found) break; - int rc = pthread_cond_timedwait(&_dharma_event_cv, &_dharma_event_mu, &deadline); - if (rc == ETIMEDOUT) break; - } - pthread_mutex_unlock(&_dharma_event_mu); - - if (!found) return el_map_new(0); - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("type")), - EL_STR(el_strdup(found->event_type ? found->event_type : ""))); - m = el_map_set(m, EL_STR(el_strdup("payload")), - EL_STR(el_strdup(found->payload ? found->payload : ""))); - m = el_map_set(m, EL_STR(el_strdup("source_cgi")), - EL_STR(el_strdup(found->source ? found->source : ""))); - m = el_map_set(m, EL_STR(el_strdup("timestamp")), (el_val_t)found->timestamp); - free(found->event_type); free(found->payload); free(found->source); free(found); - return m; -} - -/* Locate (or create) the local "dharma:self" node and the synthetic peer - * node "dharma:peer:". Returns the index of the dharma-relation - * edge, or -1 if not found. If `create` is non-zero, ensure the nodes - * and edge exist (creating them as needed) and return the edge index. */ -static int64_t dharma_find_or_create_relation_edge(const char* peer_base, int create) { - if (!peer_base || !*peer_base) return -1; - EngramStore* g = engram_get(); - const char* self_id = "dharma:self"; - char peer_node[512]; - snprintf(peer_node, sizeof(peer_node), "dharma:peer:%s", peer_base); - - /* Look for the edge first. */ - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (e->from_id && e->to_id && - strcmp(e->from_id, self_id) == 0 && - strcmp(e->to_id, peer_node) == 0 && - e->relation && strcmp(e->relation, "dharma-relation") == 0) { - return i; - } - } - if (!create) return -1; - - /* Ensure self node exists. We use a fixed id (not engram_new_id) so - * subsequent calls reuse the same one. */ - if (!engram_find_node(self_id)) { - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = el_strdup_persist(self_id); - n->content = el_strdup_persist(_el_cgi_dharma_id ? _el_cgi_dharma_id : "(self)"); - n->node_type = el_strdup_persist("DharmaSelf"); - n->label = el_strdup_persist("dharma:self"); - n->tier = el_strdup_persist("Working"); - n->tags = el_strdup_persist("dharma"); - n->metadata = el_strdup_persist("{}"); - n->salience = 1.0; n->importance = 1.0; n->confidence = 1.0; - int64_t now = engram_now_ms(); - n->created_at = now; n->updated_at = now; n->last_activated = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - } - if (!engram_find_node(peer_node)) { - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = el_strdup_persist(peer_node); - n->content = el_strdup_persist(peer_base); - n->node_type = el_strdup_persist("DharmaPeer"); - n->label = el_strdup_persist(peer_node); - n->tier = el_strdup_persist("Working"); - n->tags = el_strdup_persist("dharma"); - n->metadata = el_strdup_persist("{}"); - n->salience = 0.5; n->importance = 0.5; n->confidence = 1.0; - int64_t now = engram_now_ms(); - n->created_at = now; n->updated_at = now; n->last_activated = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - } - /* Create the edge with weight 0.0 — caller will increment. */ - engram_grow_edges(); - EngramEdge* e = &g->edges[g->edge_count]; - memset(e, 0, sizeof(*e)); - e->id = engram_new_id(); - e->from_id = el_strdup_persist(self_id); - e->to_id = el_strdup_persist(peer_node); - e->relation = el_strdup_persist("dharma-relation"); - e->metadata = el_strdup_persist("{}"); - e->weight = 0.0; - e->confidence = 1.0; - int64_t now = engram_now_ms(); - e->created_at = now; e->updated_at = now; - e->layer_id = ENGRAM_LAYER_DEFAULT; - int64_t idx = g->edge_count; - g->edge_count++; - return idx; -} - -void dharma_strengthen(el_val_t cgi_id, el_val_t weight) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return; - char* base = NULL; char* url = NULL; - dharma_parse_id(id, &base, &url); - free(url); - int64_t ei = dharma_find_or_create_relation_edge(base, 1); - free(base); - if (ei < 0) return; - EngramStore* g = engram_get(); - double inc = engram_decode_score(weight); - if (!(inc >= 0.0)) inc = 0.0; - double w = g->edges[ei].weight + inc; - if (w < 0.0) w = 0.0; - if (w > 1.0) w = 1.0; - g->edges[ei].weight = w; - g->edges[ei].updated_at = engram_now_ms(); - g->edges[ei].last_fired = g->edges[ei].updated_at; -} - -el_val_t dharma_relationship(el_val_t cgi_id) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return el_from_float(0.0); - char* base = NULL; char* url = NULL; - dharma_parse_id(id, &base, &url); - free(url); - int64_t ei = dharma_find_or_create_relation_edge(base, 0); - free(base); - if (ei < 0) return el_from_float(0.0); - EngramStore* g = engram_get(); - return el_from_float(g->edges[ei].weight); -} - -el_val_t dharma_peers(void) { - /* Walk dharma-relation edges out of "dharma:self", weight > 0, sort desc. */ - EngramStore* g = engram_get(); - const char* self_id = "dharma:self"; - typedef struct { char* peer_base; double weight; } PeerEntry; - PeerEntry* peers = malloc((size_t)(g->edge_count + 1) * sizeof(PeerEntry)); - int64_t pcount = 0; - if (!peers) return el_list_empty(); - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (!e->from_id || !e->to_id) continue; - if (strcmp(e->from_id, self_id) != 0) continue; - if (!e->relation || strcmp(e->relation, "dharma-relation") != 0) continue; - if (e->weight <= 0.0) continue; - const char* prefix = "dharma:peer:"; - size_t pl = strlen(prefix); - if (strncmp(e->to_id, prefix, pl) != 0) continue; - peers[pcount].peer_base = el_strdup(e->to_id + pl); - peers[pcount].weight = e->weight; - pcount++; - } - /* Sort desc by weight. */ - for (int64_t i = 1; i < pcount; i++) { - PeerEntry key = peers[i]; - int64_t j = i - 1; - while (j >= 0 && peers[j].weight < key.weight) { - peers[j + 1] = peers[j]; j--; - } - peers[j + 1] = key; - } - el_val_t out = el_list_empty(); - for (int64_t i = 0; i < pcount; i++) { - out = el_list_append(out, EL_STR(peers[i].peer_base)); - } - free(peers); - return out; -} -#endif /* HAVE_CURL — DHARMA network */ - -/* ── Batch 4: LLM (Anthropic API client) ─────────────────────────────────── */ -/* - * All LLM builtins call https://api.anthropic.com/v1/messages with the API - * key from env ANTHROPIC_API_KEY. Default model is "claude-sonnet-4-5" - * when the supplied model is empty/null. - * - * `llm_call_agentic` runs a real multi-turn tool_use/tool_result loop. - * Tool handlers are registered with `llm_register_tool(name, fn_name)`, - * which dlsym()s the named symbol. Each tool handler has the C signature - * el_val_t handler(el_val_t input_json); - * and returns a JSON-string el_val_t result. Iteration is capped at 10. - */ - -#ifdef HAVE_CURL -static const char* LLM_DEFAULT_MODEL = "claude-sonnet-4-5"; -static const char* LLM_API_URL = "https://api.anthropic.com/v1/messages"; -static const char* LLM_VERSION = "2023-06-01"; - -static const char* llm_resolve_model(const char* m) { - if (!m || !*m) return LLM_DEFAULT_MODEL; - return m; -} - -/* - * ── Configurable LLM provider chain ────────────────────────────────────────── - * - * Providers are configured via indexed env vars. The runtime tries each in - * order (0, 1, 2, ...) and returns the first successful non-empty response. - * - * Per provider (N = 0, 1, 2, ...): - * NEURON_LLM_N_URL — endpoint URL (base URL; /v1/chat/completions appended - * if format is "openai" and not already in URL) - * NEURON_LLM_N_KEY — API key - * NEURON_LLM_N_FORMAT — "openai" (default) or "anthropic" - * NEURON_LLM_N_MODEL — model name override (optional) - * - * Example — Neuron inference primary, Anthropic fallback: - * NEURON_LLM_0_URL=https://soma.../v1/chat/completions - * NEURON_LLM_0_KEY=svc-key - * NEURON_LLM_0_FORMAT=openai - * NEURON_LLM_0_MODEL=neuron - * NEURON_LLM_1_URL=https://api.anthropic.com/v1/messages - * NEURON_LLM_1_KEY=sk-ant-... - * NEURON_LLM_1_FORMAT=anthropic - * - * If no NEURON_LLM_0_URL is set, falls back to legacy ANTHROPIC_API_KEY. - */ - -#define LLM_MAX_PROVIDERS 16 - -/* forward declarations */ -static el_val_t llm_extract_text(el_val_t resp_val); -static el_val_t llm_extract_text_openai(el_val_t resp_val); - -static el_val_t llm_extract_text_openai(el_val_t resp_val) { - const char* resp = EL_CSTR(resp_val); - if (!resp || !*resp) return el_wrap_str(el_strdup("")); - if (resp[0] == '{' && strstr(resp, "\"error\"")) return el_wrap_str(el_strdup("")); - const char* choices = json_find_key(resp, "choices"); - if (!choices || *choices != '[') return el_wrap_str(el_strdup("")); - choices++; - while (*choices == ' ' || *choices == '\t') choices++; - if (*choices != '{') return el_wrap_str(el_strdup("")); - const char* end = json_skip_value(choices); - size_t n = (size_t)(end - choices); - char* obj = malloc(n + 1); memcpy(obj, choices, n); obj[n] = '\0'; - const char* msg = json_find_key(obj, "message"); - if (!msg || *msg != '{') { free(obj); return el_wrap_str(el_strdup("")); } - const char* msg_end = json_skip_value(msg); - size_t mn = (size_t)(msg_end - msg); - char* msg_obj = malloc(mn + 1); memcpy(msg_obj, msg, mn); msg_obj[mn] = '\0'; - const char* content = json_find_key(msg_obj, "content"); - el_val_t result = el_wrap_str(el_strdup("")); - if (content && *content == '"') { - JsonParser jp = { .p = content, .end = content + strlen(content), .err = 0 }; - char* text = jp_parse_string_raw(&jp); - if (!jp.err && text) result = el_wrap_str(text); - } - free(msg_obj); free(obj); - return result; -} - -/* Send a request to one provider. Returns the raw response string. - * format: 0 = openai, 1 = anthropic */ -static el_val_t llm_provider_request(const char* url, const char* key, - int format, const char* model, - const char* system_str, - const char* user_str) { - char* esc_sys = system_str && *system_str ? json_escape_alloc(system_str) : NULL; - char* esc_user = json_escape_alloc(user_str ? user_str : ""); - JsonBuf b; jb_init(&b); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - - if (format == 0) { /* OpenAI */ - char full_url[1024]; - if (strstr(url, "/chat/completions") || strstr(url, "/messages")) { - snprintf(full_url, sizeof(full_url), "%s", url); - } else { - snprintf(full_url, sizeof(full_url), "%s/v1/chat/completions", url); - } - { size_t n = strlen(key)+24; char* l=malloc(n); snprintf(l,n,"Authorization: Bearer %s",key); h=curl_slist_append(h,l); free(l); } - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, model ? model : "neuron"); - jb_puts(&b, ",\"max_tokens\":4096,\"messages\":["); - if (esc_sys && *esc_sys) { jb_puts(&b,"{\"role\":\"system\",\"content\":\""); jb_puts(&b,esc_sys); jb_puts(&b,"\"},"); } - jb_puts(&b, "{\"role\":\"user\",\"content\":\""); jb_puts(&b, esc_user); jb_puts(&b, "\"}]}"); - el_val_t resp = http_do("POST", full_url, b.buf, h); - curl_slist_free_all(h); free(b.buf); - if (esc_sys) free(esc_sys); free(esc_user); - return llm_extract_text_openai(resp); - } else { /* Anthropic */ - { size_t n = strlen(key)+16; char* l=malloc(n); snprintf(l,n,"x-api-key: %s",key); h=curl_slist_append(h,l); free(l); } - { size_t n = strlen(LLM_VERSION)+32; char* l=malloc(n); snprintf(l,n,"anthropic-version: %s",LLM_VERSION); h=curl_slist_append(h,l); free(l); } - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, model ? model : LLM_DEFAULT_MODEL); - jb_puts(&b, ",\"max_tokens\":4096"); - if (esc_sys && *esc_sys) { jb_puts(&b,",\"system\":\""); jb_puts(&b,esc_sys); jb_puts(&b,"\""); } - jb_puts(&b, ",\"messages\":[{\"role\":\"user\",\"content\":\""); jb_puts(&b, esc_user); jb_puts(&b, "\"}]}"); - el_val_t resp = http_do("POST", url, b.buf, h); - curl_slist_free_all(h); free(b.buf); - if (esc_sys) free(esc_sys); free(esc_user); - return llm_extract_text(resp); - } -} - -static el_val_t llm_chain_call(const char* model_pref, const char* system_str, const char* user_str) { - char url_key[64], key_key[64], fmt_key[64], model_key[64]; - for (int i = 0; i < LLM_MAX_PROVIDERS; i++) { - snprintf(url_key, sizeof(url_key), "NEURON_LLM_%d_URL", i); - snprintf(key_key, sizeof(key_key), "NEURON_LLM_%d_KEY", i); - snprintf(fmt_key, sizeof(fmt_key), "NEURON_LLM_%d_FORMAT", i); - snprintf(model_key, sizeof(model_key), "NEURON_LLM_%d_MODEL", i); - const char* url = getenv(url_key); - const char* key = getenv(key_key); - if (!url || !*url || !key || !*key) break; /* end of chain */ - const char* fmt_s = getenv(fmt_key); - int fmt = (fmt_s && strcmp(fmt_s, "anthropic") == 0) ? 1 : 0; - const char* model = getenv(model_key); - if (!model || !*model) model = model_pref; /* fall back to the caller-requested model */ - fprintf(stderr, "[llm] trying provider %d (%s)\n", i, url); - el_val_t result = llm_provider_request(url, key, fmt, model, system_str, user_str); - const char* t = EL_CSTR(result); - if (t && *t && t[0] != '{') return result; /* success */ - fprintf(stderr, "[llm] provider %d failed or empty, trying next\n", i); - } - /* Legacy fallback: ANTHROPIC_API_KEY */ - const char* api_key = getenv("ANTHROPIC_API_KEY"); - if (!api_key || !*api_key) return http_error_json("no LLM providers configured"); - fprintf(stderr, "[llm] using legacy ANTHROPIC_API_KEY fallback\n"); - return llm_provider_request(LLM_API_URL, api_key, 1, model_pref, system_str, user_str); -} - -/* Legacy llm_request — kept for backward compat with agentic loop internals */ -static el_val_t llm_request(const char* json_body) { - const char* api_key = getenv("ANTHROPIC_API_KEY"); - if (!api_key || !*api_key) return http_error_json("ANTHROPIC_API_KEY not set"); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - { size_t n=strlen(api_key)+16; char* l=malloc(n); snprintf(l,n,"x-api-key: %s",api_key); h=curl_slist_append(h,l); free(l); } - { size_t n=strlen(LLM_VERSION)+32; char* l=malloc(n); snprintf(l,n,"anthropic-version: %s",LLM_VERSION); h=curl_slist_append(h,l); free(l); } - el_val_t resp = http_do("POST", LLM_API_URL, json_body, h); - curl_slist_free_all(h); - return resp; -} - -/* Extract concatenated assistant text from an Anthropic /v1/messages - * response. The response shape is: - * {"content":[{"type":"text","text":"..."}, ...], ...} - * If parsing fails, returns the raw response so the caller can inspect. - */ -static el_val_t llm_extract_text(el_val_t resp_val) { - const char* resp = EL_CSTR(resp_val); - if (!resp || !*resp) return el_wrap_str(el_strdup("")); - /* If error JSON, propagate as-is. */ - if (resp[0] == '{' && strstr(resp, "\"error\"")) { - return el_wrap_str(el_strdup(resp)); - } - /* Find "content":[ ... ] */ - const char* p = json_find_key(resp, "content"); - if (!p) return el_wrap_str(el_strdup(resp)); - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return el_wrap_str(el_strdup(resp)); - p++; - JsonBuf out; jb_init(&out); - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - const char* type_p = json_find_key(obj, "type"); - if (type_p && *type_p == '"') { - JsonParser jp = { .p = type_p, .end = type_p + strlen(type_p), .err = 0 }; - char* type_s = jp_parse_string_raw(&jp); - if (!jp.err && type_s && strcmp(type_s, "text") == 0) { - const char* tp = json_find_key(obj, "text"); - if (tp && *tp == '"') { - JsonParser jp2 = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - char* text_s = jp_parse_string_raw(&jp2); - if (!jp2.err && text_s) jb_puts(&out, text_s); - free(text_s); - } - } - free(type_s); - } - free(obj); - p = end; - } - return el_wrap_str(jb_finish(&out)); -} - -el_val_t llm_call(el_val_t model, el_val_t prompt) { - const char* m = EL_CSTR(model); - const char* u = EL_CSTR(prompt); if (!u) u = ""; - return llm_chain_call(m, NULL, u); -} - -el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt) { - const char* m = EL_CSTR(model); - const char* s = EL_CSTR(system_prompt); if (!s) s = ""; - const char* u = EL_CSTR(user_prompt); if (!u) u = ""; - return llm_chain_call(m, s, u); -} - -/* ── Tool registry for llm_call_agentic ─────────────────────────────────── */ - -typedef el_val_t (*llm_tool_fn)(el_val_t input); - -typedef struct LlmToolEntry { - char* name; - llm_tool_fn fn; -} LlmToolEntry; - -static LlmToolEntry _llm_tools[64]; -static size_t _llm_tool_count = 0; -static pthread_mutex_t _llm_tool_mu = PTHREAD_MUTEX_INITIALIZER; - -static llm_tool_fn llm_tool_lookup(const char* name) { - if (!name) return NULL; - llm_tool_fn fn = NULL; - pthread_mutex_lock(&_llm_tool_mu); - for (size_t i = 0; i < _llm_tool_count; i++) { - if (strcmp(_llm_tools[i].name, name) == 0) { fn = _llm_tools[i].fn; break; } - } - pthread_mutex_unlock(&_llm_tool_mu); - return fn; -} - -void llm_register_tool(el_val_t name, el_val_t handler_fn_name) { - const char* nm = EL_CSTR(name); - const char* sym = EL_CSTR(handler_fn_name); - if (!nm || !*nm || !sym || !*sym) return; - void* p = dlsym(RTLD_DEFAULT, sym); - if (!p) { - fprintf(stderr, "[llm_register_tool] symbol not found: %s\n", sym); - return; - } - pthread_mutex_lock(&_llm_tool_mu); - /* Replace existing entry by name. */ - for (size_t i = 0; i < _llm_tool_count; i++) { - if (strcmp(_llm_tools[i].name, nm) == 0) { - _llm_tools[i].fn = (llm_tool_fn)p; - pthread_mutex_unlock(&_llm_tool_mu); - return; - } - } - if (_llm_tool_count < sizeof(_llm_tools) / sizeof(_llm_tools[0])) { - _llm_tools[_llm_tool_count].name = el_strdup(nm); - _llm_tools[_llm_tool_count].fn = (llm_tool_fn)p; - _llm_tool_count++; - } - pthread_mutex_unlock(&_llm_tool_mu); -} - -/* Serialize the El `tools` list into the JSON `tools:[...]` field expected - * by the Anthropic API. Each tool is an ElMap with name/description/ - * input_schema. input_schema is treated as either a JSON-object string - * (passed through verbatim) or a missing field (substitute {}). */ -static void llm_emit_tools_json(JsonBuf* b, el_val_t tools_list) { - jb_putc(b, '['); - ElList* lst = (ElList*)(uintptr_t)tools_list; - int64_t n = lst ? lst->length : 0; - for (int64_t i = 0; i < n; i++) { - if (i > 0) jb_putc(b, ','); - ElMap* tm = as_map(lst->elems[i]); - const char* name = ""; - const char* desc = ""; - const char* schema = "{}"; - if (tm) { - for (int64_t k = 0; k < tm->count; k++) { - const char* key = EL_CSTR(tm->keys[k]); - const char* val = EL_CSTR(tm->values[k]); - if (!key || !val) continue; - if (strcmp(key, "name") == 0) name = val; - else if (strcmp(key, "description") == 0) desc = val; - else if (strcmp(key, "input_schema") == 0) schema = val; - } - } - char* esc_name = json_escape_alloc(name); - char* esc_desc = json_escape_alloc(desc); - jb_puts(b, "{\"name\":\""); jb_puts(b, esc_name); - jb_puts(b, "\",\"description\":\""); jb_puts(b, esc_desc); - jb_puts(b, "\",\"input_schema\":"); jb_puts(b, schema && *schema ? schema : "{}"); - jb_putc(b, '}'); - free(esc_name); free(esc_desc); - } - jb_putc(b, ']'); -} - -/* Walk the assistant `content` array and emit each block back into b, - * preserving the verbatim JSON of every block — used to re-include the - * assistant turn in the next request. */ -static void llm_emit_content_blocks(JsonBuf* b, const char* resp) { - const char* p = json_find_key(resp, "content"); - jb_putc(b, '['); - if (!p) { jb_putc(b, ']'); return; } - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') { jb_putc(b, ']'); return; } - p++; - int first = 1; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - if (!first) jb_putc(b, ','); - first = 0; - size_t n = (size_t)(end - p); - jb_reserve(b, n); - memcpy(b->buf + b->len, p, n); - b->len += n; - b->buf[b->len] = '\0'; - p = end; - } - jb_putc(b, ']'); -} - -/* Concatenate all "text" blocks from a response. Returns owned string. */ -static char* llm_concat_text_blocks(const char* resp) { - JsonBuf out; jb_init(&out); - if (!resp) return out.buf; - const char* p = json_find_key(resp, "content"); - if (!p) return out.buf; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return out.buf; - p++; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - const char* tp = json_find_key(obj, "type"); - if (tp && *tp == '"') { - JsonParser jp = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - char* tname = jp_parse_string_raw(&jp); - if (!jp.err && tname && strcmp(tname, "text") == 0) { - const char* xp = json_find_key(obj, "text"); - if (xp && *xp == '"') { - JsonParser jp2 = { .p = xp, .end = xp + strlen(xp), .err = 0 }; - char* txt = jp_parse_string_raw(&jp2); - if (!jp2.err && txt) jb_puts(&out, txt); - free(txt); - } - } - free(tname); - } - free(obj); - p = end; - } - return out.buf; -} - -/* Build tool_result message blocks for every tool_use in a response. - * Appends to `b` an array element for each tool_use; caller wraps. */ -static int llm_build_tool_results(JsonBuf* b, const char* resp) { - int any = 0; - const char* p = json_find_key(resp, "content"); - if (!p) return 0; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return 0; - p++; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - - const char* tp = json_find_key(obj, "type"); - char* type_s = NULL; - if (tp && *tp == '"') { - JsonParser jp = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - type_s = jp_parse_string_raw(&jp); - } - if (type_s && strcmp(type_s, "tool_use") == 0) { - /* Extract id, name, input. */ - char* id_s = NULL; char* name_s = NULL; - const char* idp = json_find_key(obj, "id"); - if (idp && *idp == '"') { - JsonParser jp = { .p = idp, .end = idp + strlen(idp), .err = 0 }; - id_s = jp_parse_string_raw(&jp); - } - const char* np = json_find_key(obj, "name"); - if (np && *np == '"') { - JsonParser jp = { .p = np, .end = np + strlen(np), .err = 0 }; - name_s = jp_parse_string_raw(&jp); - } - el_val_t input_raw = json_get_raw(EL_STR(obj), EL_STR("input")); - const char* input_s = EL_CSTR(input_raw); - if (!input_s || !*input_s) input_s = "{}"; - - llm_tool_fn fn = llm_tool_lookup(name_s ? name_s : ""); - char* result = NULL; - int is_error = 0; - if (!fn) { - size_t en = strlen(name_s ? name_s : "(null)") + 64; - result = malloc(en); - snprintf(result, en, "{\"error\":\"tool not registered: %s\"}", - name_s ? name_s : "(null)"); - is_error = 1; - } else { - el_val_t out = fn(EL_STR(input_s)); - const char* os = EL_CSTR(out); - result = el_strdup(os ? os : ""); - } - - if (any) jb_putc(b, ','); - char* esc_id = json_escape_alloc(id_s ? id_s : ""); - char* esc_res = json_escape_alloc(result ? result : ""); - jb_puts(b, "{\"type\":\"tool_result\",\"tool_use_id\":\""); - jb_puts(b, esc_id); - jb_puts(b, "\",\"content\":\""); - jb_puts(b, esc_res); - jb_puts(b, "\""); - if (is_error) jb_puts(b, ",\"is_error\":true"); - jb_putc(b, '}'); - free(esc_id); free(esc_res); free(result); - free(id_s); free(name_s); - any = 1; - } - free(type_s); - free(obj); - p = end; - } - return any; -} - -el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools) { - /* Empty tools list → degrade to plain system call. */ - ElList* tl = (ElList*)(uintptr_t)tools; - if (!tl || tl->length == 0) { - return llm_call_system(model, system, user); - } - const char* m = llm_resolve_model(EL_CSTR(model)); - const char* sys_p = EL_CSTR(system); if (!sys_p) sys_p = ""; - const char* usr_p = EL_CSTR(user); if (!usr_p) usr_p = ""; - - /* Build the static parts: tools JSON and system prompt — these don't - * change across iterations. */ - JsonBuf tools_buf; jb_init(&tools_buf); - llm_emit_tools_json(&tools_buf, tools); - char* esc_sys = json_escape_alloc(sys_p); - - /* messages array, accumulated as a mutable JSON fragment (no surrounding - * brackets — emitted at request time). */ - JsonBuf msgs; jb_init(&msgs); - /* First user message. */ - char* esc_user = json_escape_alloc(usr_p); - jb_puts(&msgs, "{\"role\":\"user\",\"content\":\""); - jb_puts(&msgs, esc_user); - jb_puts(&msgs, "\"}"); - free(esc_user); - - char* last_text = el_strdup(""); - el_val_t final_out = 0; - int reached_cap = 1; - - for (int iter = 0; iter < 10; iter++) { - /* Build request body. */ - JsonBuf body; jb_init(&body); - jb_putc(&body, '{'); - jb_puts(&body, "\"model\":"); jb_emit_escaped(&body, m); - jb_puts(&body, ",\"max_tokens\":4096"); - if (*sys_p) { - jb_puts(&body, ",\"system\":\""); - jb_puts(&body, esc_sys); - jb_puts(&body, "\""); - } - jb_puts(&body, ",\"tools\":"); - jb_puts(&body, tools_buf.buf); - jb_puts(&body, ",\"messages\":["); - jb_puts(&body, msgs.buf); - jb_puts(&body, "]}"); - - el_val_t resp_v = llm_request(body.buf); - free(body.buf); - const char* resp = EL_CSTR(resp_v); - if (!resp || !*resp) { - final_out = http_error_json("empty response"); - reached_cap = 0; - break; - } - if (resp[0] == '{' && strstr(resp, "\"error\"") && - !json_find_key(resp, "content")) { - final_out = el_wrap_str(el_strdup(resp)); - reached_cap = 0; - break; - } - - /* Update last_text from this response. */ - free(last_text); - last_text = llm_concat_text_blocks(resp); - - /* Inspect stop_reason. */ - el_val_t sr_v = json_get_string(EL_STR(resp), EL_STR("stop_reason")); - const char* sr = EL_CSTR(sr_v); if (!sr) sr = ""; - - if (strcmp(sr, "end_turn") == 0) { - final_out = el_wrap_str(el_strdup(last_text)); - reached_cap = 0; - break; - } - if (strcmp(sr, "max_tokens") == 0) { - size_t ln = strlen(last_text) + 16; - char* out = malloc(ln); - snprintf(out, ln, "%s\n[truncated]", last_text); - final_out = el_wrap_str(out); - reached_cap = 0; - break; - } - if (strcmp(sr, "tool_use") != 0) { - /* Unexpected stop reason; return the text we have. */ - final_out = el_wrap_str(el_strdup(last_text)); - reached_cap = 0; - break; - } - - /* Append the assistant turn (raw content blocks) to messages. */ - JsonBuf ab; jb_init(&ab); - jb_puts(&ab, ",{\"role\":\"assistant\",\"content\":"); - llm_emit_content_blocks(&ab, resp); - jb_putc(&ab, '}'); - jb_puts(&msgs, ab.buf); - free(ab.buf); - - /* Build tool_result message. */ - JsonBuf tr; jb_init(&tr); - jb_puts(&tr, ",{\"role\":\"user\",\"content\":["); - int any = llm_build_tool_results(&tr, resp); - jb_puts(&tr, "]}"); - if (any) { - jb_puts(&msgs, tr.buf); - } - free(tr.buf); - } - - if (reached_cap) { - size_t ln = strlen(last_text) + 32; - char* out = malloc(ln); - snprintf(out, ln, "[loop_cap_reached]\n%s", last_text); - final_out = el_wrap_str(out); - } - free(last_text); - free(esc_sys); - free(tools_buf.buf); - free(msgs.buf); - return final_out; -} - -/* base64-encode arbitrary bytes (returns owned C string). - * Internal helper for llm_vision; the public crypto entry point that El - * programs call is `base64_encode(el_val_t)` defined in the crypto block - * at the end of this file. */ -static char* el_b64_encode_internal(const unsigned char* src, size_t n) { - static const char tbl[] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; - size_t out_len = 4 * ((n + 2) / 3); - char* out = malloc(out_len + 1); - if (!out) return NULL; - size_t o = 0; - for (size_t i = 0; i < n;) { - uint32_t v = 0; int got = 0; - v |= (uint32_t)src[i++] << 16; got++; - if (i < n) { v |= (uint32_t)src[i++] << 8; got++; } - if (i < n) { v |= (uint32_t)src[i++]; got++; } - out[o++] = tbl[(v >> 18) & 0x3f]; - out[o++] = tbl[(v >> 12) & 0x3f]; - out[o++] = (got > 1) ? tbl[(v >> 6) & 0x3f] : '='; - out[o++] = (got > 2) ? tbl[v & 0x3f] : '='; - } - out[o] = '\0'; - return out; -} - -el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64) { - const char* m = llm_resolve_model(EL_CSTR(model)); - const char* s = EL_CSTR(system); if (!s) s = ""; - const char* u = EL_CSTR(prompt); if (!u) u = ""; - const char* img = EL_CSTR(image_url_or_b64); if (!img) img = ""; - - /* Choose source mode */ - char* image_block = NULL; - if (strncasecmp(img, "http://", 7) == 0 || strncasecmp(img, "https://", 8) == 0) { - char* esc_url = json_escape_alloc(img); - size_t n = strlen(esc_url) + 128; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"%s\"}}", - esc_url); - free(esc_url); - } else if (strncmp(img, "data:", 5) == 0) { - /* Inline data URL: split media-type and base64 */ - const char* semi = strchr(img + 5, ';'); - const char* comma = strchr(img + 5, ','); - char media[64] = "image/png"; - if (semi && comma && semi < comma) { - size_t ml = (size_t)(semi - (img + 5)); - if (ml >= sizeof(media)) ml = sizeof(media) - 1; - memcpy(media, img + 5, ml); media[ml] = '\0'; - } - const char* b64 = comma ? comma + 1 : ""; - char* esc_media = json_escape_alloc(media); - char* esc_b64 = json_escape_alloc(b64); - size_t n = strlen(esc_media) + strlen(esc_b64) + 192; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"base64\"," - "\"media_type\":\"%s\",\"data\":\"%s\"}}", - esc_media, esc_b64); - free(esc_media); free(esc_b64); - } else if (*img) { - /* Treat as file path: read, base64-encode, attach. */ - FILE* f = fopen(img, "rb"); - if (!f) { - char err[256]; snprintf(err, sizeof(err), "cannot open image: %s", img); - return http_error_json(err); - } - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz <= 0) { fclose(f); return http_error_json("empty image file"); } - unsigned char* buf = malloc((size_t)sz); - if (!buf) { fclose(f); return http_error_json("oom"); } - size_t got = fread(buf, 1, (size_t)sz, f); - fclose(f); - char* b64 = el_b64_encode_internal(buf, got); - free(buf); - if (!b64) return http_error_json("base64 encode failed"); - const char* media = "image/png"; - size_t ilen = strlen(img); - if (ilen >= 4) { - if (strcasecmp(img + ilen - 4, ".jpg") == 0 || - (ilen >= 5 && strcasecmp(img + ilen - 5, ".jpeg") == 0)) media = "image/jpeg"; - else if (strcasecmp(img + ilen - 4, ".gif") == 0) media = "image/gif"; - else if (strcasecmp(img + ilen - 4, ".webp") == 0) media = "image/webp"; - } - char* esc_b64 = json_escape_alloc(b64); free(b64); - size_t n = strlen(esc_b64) + 192; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"base64\"," - "\"media_type\":\"%s\",\"data\":\"%s\"}}", - media, esc_b64); - free(esc_b64); - } - - char* esc_sys = json_escape_alloc(s); - char* esc_user = json_escape_alloc(u); - JsonBuf b; jb_init(&b); - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, m); - jb_puts(&b, ",\"max_tokens\":4096"); - if (*s) { - jb_puts(&b, ",\"system\":\""); - jb_puts(&b, esc_sys); - jb_puts(&b, "\""); - } - jb_puts(&b, ",\"messages\":[{\"role\":\"user\",\"content\":["); - if (image_block) { - jb_puts(&b, image_block); - jb_putc(&b, ','); - } - jb_puts(&b, "{\"type\":\"text\",\"text\":\""); - jb_puts(&b, esc_user); - jb_puts(&b, "\"}]}]}"); - free(esc_sys); free(esc_user); free(image_block); - el_val_t resp = llm_request(b.buf); - free(b.buf); - return llm_extract_text(resp); -} - -el_val_t llm_models(void) { - el_val_t lst = el_list_empty(); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-sonnet-4-5"))); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-opus-4-7"))); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-haiku-4-5"))); - return lst; -} -#endif /* HAVE_CURL */ - -/* ── Native VM builtin aliases ────────────────────────────────────────────── - * El source files use native_* names (El VM builtins). - * When compiled to C, these map directly to el_* runtime functions. */ - -el_val_t native_list_get(el_val_t list, el_val_t index) { - return el_list_get(list, index); -} - -el_val_t native_list_len(el_val_t list) { - return el_list_len(list); -} - -el_val_t native_list_append(el_val_t list, el_val_t elem) { - return el_list_append(list, elem); -} - -el_val_t native_list_empty(void) { - return el_list_empty(); -} - -el_val_t native_list_clone(el_val_t list) { - return el_list_clone(list); -} - -el_val_t native_string_chars(el_val_t sv) { - const char* s = EL_CSTR(sv); - el_val_t result = el_list_empty(); - if (!s) return result; - while (*s) { - char buf[2]; - buf[0] = *s; - buf[1] = '\0'; - result = el_list_append(result, EL_STR(strdup(buf))); - s++; - } - return result; -} - -el_val_t native_int_to_str(el_val_t n) { - return int_to_str(n); -} - -/* ── Method-call shorthand aliases ────────────────────────────────────────── - * Short names that result from the method-call convention: - * myList.append(x) → append(myList, x) - * myList.len() → len(myList) - * myList.get(i) → get(myList, i) - * myMap.map_get(k) → map_get(myMap, k) - * myMap.map_set(k,v) → map_set(myMap, k, v) */ - -el_val_t append(el_val_t list, el_val_t elem) { return el_list_append(list, elem); } -el_val_t len(el_val_t list) { return el_list_len(list); } -el_val_t get(el_val_t list, el_val_t index) { return el_list_get(list, index); } -el_val_t map_get(el_val_t map, el_val_t key) { return el_map_get(map, key); } -el_val_t map_set(el_val_t map, el_val_t key, el_val_t value) { return el_map_set(map, key, value); } - -/* ── Crypto primitives ────────────────────────────────────────────────────── - * - * SHA-256 implementation adapted from Brad Conte's public-domain reference - * (https://github.com/B-Con/crypto-algorithms/blob/master/sha256.c, public - * domain per the project's LICENSE). HMAC follows RFC 2104. Base64 encoding - * follows RFC 4648; the URL-safe variant uses the alphabet from §5 of the - * RFC and omits padding (per JWT/JWS convention). - * - * Self-contained: no OpenSSL/libcrypto dependency. The runtime keeps its - * existing `-lcurl -lpthread -ldl -lm` link line. - * - * Binary outputs (sha256_bytes, hmac_sha256_bytes) tag their buffer with a - * magic header so base64_encode/base64url_encode can recover the exact byte - * length even when the payload contains embedded NULs. Plain C strings - * (without the header) fall back to strlen(), preserving the existing API - * shape for normal text inputs. */ - -/* Magic-header for length-tagged binary buffers. Layout: - * [ uint32_t magic = EL_MAGIC_BIN ][ uint32_t length ][ data... ][ \0 ] - * The returned el_val_t points at `data`, so consumers that strlen() it still - * get a sensible (though possibly truncated) view. el_bin_len() recovers the - * true length by sniffing the 8 bytes preceding the pointer. - * - * Magic value chosen with high MSB so it cannot collide with printable ASCII - * (the same discriminator pattern used by EL_MAGIC_LIST / EL_MAGIC_MAP). */ -#define EL_MAGIC_BIN 0xE1B17EAFu - -typedef struct { - uint32_t magic; - uint32_t length; -} el_bin_hdr_t; - -/* Allocate a length-tagged binary buffer; returns pointer to the data area. */ -static unsigned char* el_bin_alloc(size_t len) { - el_bin_hdr_t* hdr = (el_bin_hdr_t*)malloc(sizeof(el_bin_hdr_t) + len + 1); - if (!hdr) { fputs("el_runtime: out of memory (bin)\n", stderr); exit(1); } - hdr->magic = EL_MAGIC_BIN; - hdr->length = (uint32_t)len; - unsigned char* data = (unsigned char*)(hdr + 1); - data[len] = '\0'; /* keep NUL-terminated for accidental strlen calls */ - return data; -} - -/* Recover length from a possibly-tagged buffer. Returns 1 if tagged. */ -static int el_bin_lookup(const void* p, size_t* out_len) { - if (!p) { *out_len = 0; return 0; } - /* Avoid reading off the front of a page on tiny pointers (e.g. NULs - * passed in as int-cast values). 4096 is a safe lower bound on any - * platform we target. */ - if ((uintptr_t)p < 4096) return 0; - const el_bin_hdr_t* hdr = (const el_bin_hdr_t*)((const char*)p - sizeof(el_bin_hdr_t)); - if (hdr->magic != EL_MAGIC_BIN) return 0; - *out_len = hdr->length; - return 1; -} - -/* Effective input length: tagged length if present, else strlen. */ -static size_t el_input_len(const char* s) { - size_t n; - if (el_bin_lookup(s, &n)) return n; - return s ? strlen(s) : 0; -} - -/* ─── SHA-256 (Brad Conte / public domain) ──────────────────────────────── */ - -typedef struct { - unsigned char data[64]; - uint32_t datalen; - uint64_t bitlen; - uint32_t state[8]; -} el_sha256_ctx_t; - -static const uint32_t el_sha256_k[64] = { - 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5,0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5, - 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3,0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174, - 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc,0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da, - 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7,0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967, - 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13,0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85, - 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3,0xd192e819,0xd6990624,0xf40e3585,0x106aa070, - 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5,0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3, - 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208,0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2 -}; - -#define EL_ROTR(x, n) (((x) >> (n)) | ((x) << (32 - (n)))) -#define EL_CH(x,y,z) (((x) & (y)) ^ (~(x) & (z))) -#define EL_MAJ(x,y,z) (((x) & (y)) ^ ((x) & (z)) ^ ((y) & (z))) -#define EL_EP0(x) (EL_ROTR(x,2) ^ EL_ROTR(x,13) ^ EL_ROTR(x,22)) -#define EL_EP1(x) (EL_ROTR(x,6) ^ EL_ROTR(x,11) ^ EL_ROTR(x,25)) -#define EL_SIG0(x) (EL_ROTR(x,7) ^ EL_ROTR(x,18) ^ ((x) >> 3)) -#define EL_SIG1(x) (EL_ROTR(x,17) ^ EL_ROTR(x,19) ^ ((x) >> 10)) - -static void el_sha256_transform(el_sha256_ctx_t* ctx, const unsigned char* data) { - uint32_t a, b, c, d, e, f, g, h, t1, t2, m[64]; - int i, j; - for (i = 0, j = 0; i < 16; ++i, j += 4) { - m[i] = ((uint32_t)data[j] << 24) | ((uint32_t)data[j + 1] << 16) - | ((uint32_t)data[j + 2] << 8) | (uint32_t)data[j + 3]; - } - for (; i < 64; ++i) { - m[i] = EL_SIG1(m[i-2]) + m[i-7] + EL_SIG0(m[i-15]) + m[i-16]; - } - a = ctx->state[0]; b = ctx->state[1]; c = ctx->state[2]; d = ctx->state[3]; - e = ctx->state[4]; f = ctx->state[5]; g = ctx->state[6]; h = ctx->state[7]; - for (i = 0; i < 64; ++i) { - t1 = h + EL_EP1(e) + EL_CH(e,f,g) + el_sha256_k[i] + m[i]; - t2 = EL_EP0(a) + EL_MAJ(a,b,c); - h = g; g = f; f = e; e = d + t1; d = c; c = b; b = a; a = t1 + t2; - } - ctx->state[0] += a; ctx->state[1] += b; ctx->state[2] += c; ctx->state[3] += d; - ctx->state[4] += e; ctx->state[5] += f; ctx->state[6] += g; ctx->state[7] += h; -} - -static void el_sha256_init(el_sha256_ctx_t* ctx) { - ctx->datalen = 0; - ctx->bitlen = 0; - ctx->state[0] = 0x6a09e667; ctx->state[1] = 0xbb67ae85; - ctx->state[2] = 0x3c6ef372; ctx->state[3] = 0xa54ff53a; - ctx->state[4] = 0x510e527f; ctx->state[5] = 0x9b05688c; - ctx->state[6] = 0x1f83d9ab; ctx->state[7] = 0x5be0cd19; -} - -static void el_sha256_update(el_sha256_ctx_t* ctx, const unsigned char* data, size_t len) { - for (size_t i = 0; i < len; ++i) { - ctx->data[ctx->datalen++] = data[i]; - if (ctx->datalen == 64) { - el_sha256_transform(ctx, ctx->data); - ctx->bitlen += 512; - ctx->datalen = 0; - } - } -} - -static void el_sha256_final(el_sha256_ctx_t* ctx, unsigned char hash[32]) { - uint32_t i = ctx->datalen; - if (ctx->datalen < 56) { - ctx->data[i++] = 0x80; - while (i < 56) ctx->data[i++] = 0x00; - } else { - ctx->data[i++] = 0x80; - while (i < 64) ctx->data[i++] = 0x00; - el_sha256_transform(ctx, ctx->data); - memset(ctx->data, 0, 56); - } - ctx->bitlen += (uint64_t)ctx->datalen * 8; - ctx->data[63] = (unsigned char)( ctx->bitlen & 0xff); - ctx->data[62] = (unsigned char)((ctx->bitlen >> 8) & 0xff); - ctx->data[61] = (unsigned char)((ctx->bitlen >> 16) & 0xff); - ctx->data[60] = (unsigned char)((ctx->bitlen >> 24) & 0xff); - ctx->data[59] = (unsigned char)((ctx->bitlen >> 32) & 0xff); - ctx->data[58] = (unsigned char)((ctx->bitlen >> 40) & 0xff); - ctx->data[57] = (unsigned char)((ctx->bitlen >> 48) & 0xff); - ctx->data[56] = (unsigned char)((ctx->bitlen >> 56) & 0xff); - el_sha256_transform(ctx, ctx->data); - for (i = 0; i < 4; ++i) { - hash[i] = (ctx->state[0] >> (24 - i * 8)) & 0xff; - hash[i + 4] = (ctx->state[1] >> (24 - i * 8)) & 0xff; - hash[i + 8] = (ctx->state[2] >> (24 - i * 8)) & 0xff; - hash[i + 12] = (ctx->state[3] >> (24 - i * 8)) & 0xff; - hash[i + 16] = (ctx->state[4] >> (24 - i * 8)) & 0xff; - hash[i + 20] = (ctx->state[5] >> (24 - i * 8)) & 0xff; - hash[i + 24] = (ctx->state[6] >> (24 - i * 8)) & 0xff; - hash[i + 28] = (ctx->state[7] >> (24 - i * 8)) & 0xff; - } -} - -static void el_sha256_oneshot(const unsigned char* data, size_t len, unsigned char out[32]) { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, data, len); - el_sha256_final(&c, out); -} - -/* ─── HMAC-SHA-256 (RFC 2104) ───────────────────────────────────────────── */ - -static void el_hmac_sha256(const unsigned char* key, size_t key_len, - const unsigned char* msg, size_t msg_len, - unsigned char out[32]) { - unsigned char k[64]; - unsigned char k_ipad[64]; - unsigned char k_opad[64]; - unsigned char inner[32]; - - if (key_len > 64) { - el_sha256_oneshot(key, key_len, k); - memset(k + 32, 0, 32); - } else { - memcpy(k, key, key_len); - memset(k + key_len, 0, 64 - key_len); - } - for (int i = 0; i < 64; ++i) { - k_ipad[i] = k[i] ^ 0x36; - k_opad[i] = k[i] ^ 0x5c; - } - { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, k_ipad, 64); - el_sha256_update(&c, msg, msg_len); - el_sha256_final(&c, inner); - } - { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, k_opad, 64); - el_sha256_update(&c, inner, 32); - el_sha256_final(&c, out); - } -} - -/* ─── Hex helper ────────────────────────────────────────────────────────── */ - -static el_val_t el_hex_encode(const unsigned char* data, size_t len) { - static const char digits[] = "0123456789abcdef"; - char* out = el_strbuf(len * 2); - for (size_t i = 0; i < len; ++i) { - out[i * 2] = digits[(data[i] >> 4) & 0xf]; - out[i * 2 + 1] = digits[ data[i] & 0xf]; - } - out[len * 2] = '\0'; - return el_wrap_str(out); -} - -/* ─── Base64 (RFC 4648) ─────────────────────────────────────────────────── */ - -static const char el_b64_std_alphabet[64] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; -static const char el_b64_url_alphabet[64] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"; - -el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe) { - const char* alphabet = url_safe ? el_b64_url_alphabet : el_b64_std_alphabet; - /* Standard form is padded to multiple of 4; URL-safe omits padding. */ - size_t out_cap = ((len + 2) / 3) * 4 + 1; - char* out = el_strbuf(out_cap); - size_t i = 0, j = 0; - while (i + 3 <= len) { - uint32_t v = ((uint32_t)data[i] << 16) | ((uint32_t)data[i+1] << 8) | (uint32_t)data[i+2]; - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - out[j++] = alphabet[(v >> 6) & 0x3f]; - out[j++] = alphabet[ v & 0x3f]; - i += 3; - } - size_t rem = len - i; - if (rem == 1) { - uint32_t v = (uint32_t)data[i] << 16; - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - if (!url_safe) { out[j++] = '='; out[j++] = '='; } - } else if (rem == 2) { - uint32_t v = ((uint32_t)data[i] << 16) | ((uint32_t)data[i+1] << 8); - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - out[j++] = alphabet[(v >> 6) & 0x3f]; - if (!url_safe) { out[j++] = '='; } - } - out[j] = '\0'; - return el_wrap_str(out); -} - -/* Decode either alphabet — accepts both '+/' and '-_' transparently, and - * tolerates missing padding (which JWTs typically omit). Whitespace is - * skipped for robustness. Invalid characters cause the decode to stop and - * the partial result so far is returned. */ -static el_val_t el_base64_decode_any(const char* in) { - if (!in) { - unsigned char* empty = el_bin_alloc(0); - return EL_STR((char*)empty); - } - size_t in_len = strlen(in); - /* Worst case: 3 output bytes per 4 input chars, +1 NUL slack. */ - unsigned char* out = el_bin_alloc(((in_len + 3) / 4) * 3 + 1); - - int8_t lut[256]; - for (int i = 0; i < 256; ++i) lut[i] = -1; - for (int i = 0; i < 64; ++i) lut[(unsigned char)el_b64_std_alphabet[i]] = (int8_t)i; - /* Allow URL-safe characters too (so one decoder handles both forms). */ - lut[(unsigned char)'-'] = 62; - lut[(unsigned char)'_'] = 63; - - uint32_t buf = 0; - int bits = 0; - size_t o = 0; - for (size_t i = 0; i < in_len; ++i) { - unsigned char c = (unsigned char)in[i]; - if (c == '=' || c == '\r' || c == '\n' || c == ' ' || c == '\t') continue; - int8_t v = lut[c]; - if (v < 0) break; /* invalid char — stop */ - buf = (buf << 6) | (uint32_t)v; - bits += 6; - if (bits >= 8) { - bits -= 8; - out[o++] = (unsigned char)((buf >> bits) & 0xff); - } - } - /* Patch the length header to the actual decoded length. */ - el_bin_hdr_t* hdr = (el_bin_hdr_t*)((char*)out - sizeof(el_bin_hdr_t)); - hdr->length = (uint32_t)o; - out[o] = '\0'; - return EL_STR((char*)out); -} - -/* ─── Public crypto entry points ────────────────────────────────────────── */ - -el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len) { - unsigned char* out = el_bin_alloc(32); - el_sha256_oneshot(data, len, out); - return EL_STR((char*)out); -} - -el_val_t sha256_hex(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - unsigned char digest[32]; - el_sha256_oneshot((const unsigned char*)(s ? s : ""), n, digest); - return el_hex_encode(digest, 32); -} - -el_val_t sha256_bytes(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_sha256_bytes_n((const unsigned char*)(s ? s : ""), n); -} - -el_val_t hmac_sha256_hex(el_val_t key, el_val_t message) { - const char* k = EL_CSTR(key); - const char* m = EL_CSTR(message); - size_t kn = el_input_len(k); - size_t mn = el_input_len(m); - unsigned char mac[32]; - el_hmac_sha256((const unsigned char*)(k ? k : ""), kn, - (const unsigned char*)(m ? m : ""), mn, - mac); - return el_hex_encode(mac, 32); -} - -el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message) { - const char* k = EL_CSTR(key); - const char* m = EL_CSTR(message); - size_t kn = el_input_len(k); - size_t mn = el_input_len(m); - unsigned char* out = el_bin_alloc(32); - el_hmac_sha256((const unsigned char*)(k ? k : ""), kn, - (const unsigned char*)(m ? m : ""), mn, - out); - return EL_STR((char*)out); -} - -el_val_t base64_encode(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_base64_encode_n((const unsigned char*)(s ? s : ""), n, /*url_safe=*/0); -} - -el_val_t base64url_encode(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_base64_encode_n((const unsigned char*)(s ? s : ""), n, /*url_safe=*/1); -} - -el_val_t base64_decode(el_val_t input) { - return el_base64_decode_any(EL_CSTR(input)); -} - -el_val_t base64url_decode(el_val_t input) { - return el_base64_decode_any(EL_CSTR(input)); -} - -/* ── Post-quantum cryptography (liboqs + OpenSSL) ─────────────────────────── - * - * Algorithm choices (per CNSA 2.0 / NIST PQ guidance, as of 2024): - * Signatures: CRYSTALS-Dilithium-3 (NIST security level 3, balanced) - * KEM: CRYSTALS-Kyber-768 (NIST security level 3) - * Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2) - * Hybrid: X25519 || Kyber-768, combined via HKDF-SHA256 - * - * Why hybrid: Kyber is new. X25519 has 20+ years of analysis. Hybridizing - * preserves classical security if Kyber falls to a future cryptanalytic - * advance, and preserves PQ security if X25519 falls to a quantum adversary. - * "Recordable now, decryptable later" already threatens long-lived classical - * key exchange — the only safe move for keys protecting durable doctrine - * (CGI lineage, KindredGrants, Principal-CGI covenants) is to encapsulate - * with PQ today, even if the classical leg is what the wire shows. - * - * Compile-time detection: when is unavailable the pq_* functions - * compile to stubs that return a JSON error envelope. SHA3-256 stays - * available regardless (it's implemented inline, no liboqs dep). This lets - * the runtime build cleanly on dev machines without liboqs while production - * gets the full PQ stack. */ - -/* ─── SHA3-256 (Keccak, FIPS 202) ──────────────────────────────────────────── - * Inline reference implementation. ~120 LoC, no external dependency. - * rate=1088 bits, capacity=512 bits, output=256 bits, padding=0x06. */ - -static const uint64_t el_keccak_rc[24] = { - 0x0000000000000001ULL, 0x0000000000008082ULL, 0x800000000000808aULL, - 0x8000000080008000ULL, 0x000000000000808bULL, 0x0000000080000001ULL, - 0x8000000080008081ULL, 0x8000000000008009ULL, 0x000000000000008aULL, - 0x0000000000000088ULL, 0x0000000080008009ULL, 0x000000008000000aULL, - 0x000000008000808bULL, 0x800000000000008bULL, 0x8000000000008089ULL, - 0x8000000000008003ULL, 0x8000000000008002ULL, 0x8000000000000080ULL, - 0x000000000000800aULL, 0x800000008000000aULL, 0x8000000080008081ULL, - 0x8000000000008080ULL, 0x0000000080000001ULL, 0x8000000080008008ULL -}; - -static const unsigned el_keccak_rho[24] = { - 1, 3, 6, 10, 15, 21, 28, 36, 45, 55, 2, 14, - 27, 41, 56, 8, 25, 43, 62, 18, 39, 61, 20, 44 -}; - -static const unsigned el_keccak_pi[24] = { - 10, 7, 11, 17, 18, 3, 5, 16, 8, 21, 24, 4, - 15, 23, 19, 13, 12, 2, 20, 14, 22, 9, 6, 1 -}; - -#define EL_ROTL64(x, n) (((x) << (n)) | ((x) >> (64 - (n)))) - -static void el_keccak_f1600(uint64_t s[25]) { - for (int round = 0; round < 24; ++round) { - uint64_t bc[5], t; - for (int i = 0; i < 5; ++i) - bc[i] = s[i] ^ s[i+5] ^ s[i+10] ^ s[i+15] ^ s[i+20]; - for (int i = 0; i < 5; ++i) { - t = bc[(i+4) % 5] ^ EL_ROTL64(bc[(i+1) % 5], 1); - for (int j = 0; j < 25; j += 5) s[j+i] ^= t; - } - t = s[1]; - for (int i = 0; i < 24; ++i) { - int j = el_keccak_pi[i]; - bc[0] = s[j]; - s[j] = EL_ROTL64(t, el_keccak_rho[i]); - t = bc[0]; - } - for (int j = 0; j < 25; j += 5) { - for (int i = 0; i < 5; ++i) bc[i] = s[j+i]; - for (int i = 0; i < 5; ++i) - s[j+i] = bc[i] ^ ((~bc[(i+1) % 5]) & bc[(i+2) % 5]); - } - s[0] ^= el_keccak_rc[round]; - } -} - -static void el_sha3_256_oneshot(const unsigned char* data, size_t len, - unsigned char out[32]) { - uint64_t st[25] = {0}; - unsigned char* sb = (unsigned char*)st; - const size_t rate = 136; /* 1088 bits / 8 */ - size_t i = 0; - while (len - i >= rate) { - for (size_t k = 0; k < rate; ++k) sb[k] ^= data[i + k]; - el_keccak_f1600(st); - i += rate; - } - size_t rem = len - i; - for (size_t k = 0; k < rem; ++k) sb[k] ^= data[i + k]; - sb[rem] ^= 0x06; /* SHA3 domain-separation byte */ - sb[rate - 1] ^= 0x80; /* final-block padding bit (high bit of last byte) */ - el_keccak_f1600(st); - memcpy(out, sb, 32); -} - -el_val_t sha3_256_hex(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - unsigned char digest[32]; - el_sha3_256_oneshot((const unsigned char*)(s ? s : ""), n, digest); - return el_hex_encode(digest, 32); -} - -/* ─── Hex decode helper ───────────────────────────────────────────────────── - * Returns a length-tagged binary buffer (so embedded NULs survive); on - * odd-length / invalid input returns NULL with *out_len = 0. Caller is - * responsible for emitting the error envelope. */ - -static int el_hex_nibble(char c) { - if (c >= '0' && c <= '9') return c - '0'; - if (c >= 'a' && c <= 'f') return c - 'a' + 10; - if (c >= 'A' && c <= 'F') return c - 'A' + 10; - return -1; -} - -__attribute__((unused)) -static unsigned char* el_hex_decode(const char* s, size_t* out_len) { - *out_len = 0; - if (!s) return NULL; - size_t n = strlen(s); - if (n & 1) return NULL; - size_t blen = n / 2; - unsigned char* out = el_bin_alloc(blen); - for (size_t i = 0; i < blen; ++i) { - int hi = el_hex_nibble(s[i*2]); - int lo = el_hex_nibble(s[i*2 + 1]); - if (hi < 0 || lo < 0) return NULL; - out[i] = (unsigned char)((hi << 4) | lo); - } - *out_len = blen; - return out; -} - -/* JSON error envelope reused across all PQ entry points. */ -static el_val_t pq_error(const char* msg) { - return http_error_json(msg); -} - -#if __has_include() -#include -#define EL_HAVE_LIBOQS 1 -#else -#define EL_HAVE_LIBOQS 0 -#endif - -#if EL_HAVE_LIBOQS && __has_include() -#include -#define EL_HAVE_OPENSSL 1 -#else -#define EL_HAVE_OPENSSL 0 -#endif - -#if !EL_HAVE_LIBOQS - -/* ─── Stubs (liboqs unavailable) ─────────────────────────────────────────── - * Each entry point returns the same JSON error so callers can inspect a - * single canonical "missing primitive" string. pq_verify is the lone - * exception — verifying without liboqs simply means "not verified", so - * returning Bool false (0) keeps the type contract intact. */ - -#define EL_PQ_NO_LIB "liboqs not linked, post-quantum primitives unavailable" - -el_val_t pq_keygen_signature(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_sign(el_val_t sk, el_val_t msg) { (void)sk; (void)msg; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_verify(el_val_t pk, el_val_t msg, el_val_t sig) { (void)pk; (void)msg; (void)sig; return EL_INT(0); } -el_val_t pq_kem_keygen(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_kem_encaps(el_val_t pk) { (void)pk; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_kem_decaps(el_val_t sk, el_val_t ct) { (void)sk; (void)ct; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_hybrid_keygen(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_hybrid_handshake(el_val_t pub) { (void)pub; return pq_error(EL_PQ_NO_LIB); } - -#else /* EL_HAVE_LIBOQS */ - -/* ─── Dilithium-3 / ML-DSA-65 signatures ──────────────────────────────── - * - * NIST FIPS 204 standardized CRYSTALS-Dilithium as ML-DSA. ML-DSA-65 is the - * FIPS form of what we historically called Dilithium-3 — same algorithm - * family, same security level, identical key/sig sizes, but with a couple - * of standardization-driven tweaks (e.g. domain separation in the message - * binding). liboqs 0.12+ exposes both names; 0.15+ retired the legacy - * "Dilithium" constants in favour of "ML-DSA". We prefer ML-DSA-65 if the - * header advertises it, fall back to Dilithium-3 otherwise. Anything - * already signed with the older constant remains verifiable against that - * same constant — callers should pin the algorithm via the OQS_SIG handle's - * method_name field if they need to interoperate with archival signatures. */ - -#if defined(OQS_SIG_alg_ml_dsa_65) -# define EL_DILITHIUM_ALG OQS_SIG_alg_ml_dsa_65 -#elif defined(OQS_SIG_alg_dilithium_3) -# define EL_DILITHIUM_ALG OQS_SIG_alg_dilithium_3 -#else -# define EL_DILITHIUM_ALG "ML-DSA-65" /* string fallback; runtime probe catches misconfig */ -#endif - -el_val_t pq_keygen_signature(void) { - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return pq_error("OQS_SIG_new(dilithium-3) failed"); - unsigned char* pk = (unsigned char*)malloc(sig->length_public_key); - unsigned char* sk = (unsigned char*)malloc(sig->length_secret_key); - if (!pk || !sk) { free(pk); free(sk); OQS_SIG_free(sig); return pq_error("oom"); } - if (OQS_SIG_keypair(sig, pk, sk) != OQS_SUCCESS) { - free(pk); free(sk); OQS_SIG_free(sig); - return pq_error("dilithium-3 keypair generation failed"); - } - el_val_t pk_hex = el_hex_encode(pk, sig->length_public_key); - el_val_t sk_hex = el_hex_encode(sk, sig->length_secret_key); - OQS_MEM_secure_free(sk, sig->length_secret_key); - free(pk); - - const char* pks = EL_CSTR(pk_hex); - const char* sks = EL_CSTR(sk_hex); - char* buf = el_strbuf(strlen(pks) + strlen(sks) + 64); - sprintf(buf, "{\"public_key\":\"%s\",\"secret_key\":\"%s\"}", pks, sks); - OQS_SIG_free(sig); - return el_wrap_str(buf); -} - -el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message) { - size_t sk_len = 0; - unsigned char* sk = el_hex_decode(EL_CSTR(secret_key_hex), &sk_len); - if (!sk) return pq_error("invalid hex in secret_key"); - - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return pq_error("OQS_SIG_new(dilithium-3) failed"); - if (sk_len != sig->length_secret_key) { - OQS_SIG_free(sig); - return pq_error("secret_key length mismatch for dilithium-3"); - } - - const char* msg = EL_CSTR(message); - size_t msg_len = el_input_len(msg); - unsigned char* signature = (unsigned char*)malloc(sig->length_signature); - size_t signature_len = sig->length_signature; - if (!signature) { OQS_SIG_free(sig); return pq_error("oom"); } - - if (OQS_SIG_sign(sig, signature, &signature_len, - (const unsigned char*)(msg ? msg : ""), msg_len, sk) != OQS_SUCCESS) { - free(signature); OQS_SIG_free(sig); - return pq_error("dilithium-3 sign failed"); - } - el_val_t sig_hex = el_hex_encode(signature, signature_len); - free(signature); OQS_SIG_free(sig); - return sig_hex; -} - -el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex) { - size_t pk_len = 0, sig_len = 0; - unsigned char* pk = el_hex_decode(EL_CSTR(public_key_hex), &pk_len); - unsigned char* signature = el_hex_decode(EL_CSTR(signature_hex), &sig_len); - if (!pk || !signature) return EL_INT(0); - - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return EL_INT(0); - if (pk_len != sig->length_public_key) { OQS_SIG_free(sig); return EL_INT(0); } - - const char* msg = EL_CSTR(message); - size_t msg_len = el_input_len(msg); - OQS_STATUS rc = OQS_SIG_verify(sig, - (const unsigned char*)(msg ? msg : ""), msg_len, - signature, sig_len, pk); - OQS_SIG_free(sig); - return (rc == OQS_SUCCESS) ? EL_INT(1) : EL_INT(0); -} - -/* ─── Kyber-768 / ML-KEM-768 KEM ──────────────────────────────────────── - * - * NIST FIPS 203 standardized CRYSTALS-Kyber as ML-KEM. ML-KEM-768 is the - * FIPS form of what we historically called Kyber-768. Same situation as - * Dilithium → ML-DSA: prefer the standardized constant, fall back to the - * legacy name. liboqs 0.15.0 still exposes OQS_KEM_alg_kyber_768; the - * algorithm is identical at the wire level to ML-KEM-768 except for FIPS - * domain-separation tweaks, so the two ciphertexts/keys are NOT - * cross-compatible. Pin the constant for archival material. */ - -#if defined(OQS_KEM_alg_ml_kem_768) -# define EL_KYBER_ALG OQS_KEM_alg_ml_kem_768 -#elif defined(OQS_KEM_alg_kyber_768) -# define EL_KYBER_ALG OQS_KEM_alg_kyber_768 -#else -# define EL_KYBER_ALG "ML-KEM-768" -#endif - -el_val_t pq_kem_keygen(void) { - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - unsigned char* pk = (unsigned char*)malloc(kem->length_public_key); - unsigned char* sk = (unsigned char*)malloc(kem->length_secret_key); - if (!pk || !sk) { free(pk); free(sk); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_keypair(kem, pk, sk) != OQS_SUCCESS) { - free(pk); free(sk); OQS_KEM_free(kem); - return pq_error("kyber-768 keypair generation failed"); - } - el_val_t pk_hex = el_hex_encode(pk, kem->length_public_key); - el_val_t sk_hex = el_hex_encode(sk, kem->length_secret_key); - OQS_MEM_secure_free(sk, kem->length_secret_key); - free(pk); - - const char* pks = EL_CSTR(pk_hex); - const char* sks = EL_CSTR(sk_hex); - char* buf = el_strbuf(strlen(pks) + strlen(sks) + 64); - sprintf(buf, "{\"public_key\":\"%s\",\"secret_key\":\"%s\"}", pks, sks); - OQS_KEM_free(kem); - return el_wrap_str(buf); -} - -el_val_t pq_kem_encaps(el_val_t public_key_hex) { - size_t pk_len = 0; - unsigned char* pk = el_hex_decode(EL_CSTR(public_key_hex), &pk_len); - if (!pk) return pq_error("invalid hex in public_key"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (pk_len != kem->length_public_key) { - OQS_KEM_free(kem); - return pq_error("public_key length mismatch for kyber-768"); - } - unsigned char* ct = (unsigned char*)malloc(kem->length_ciphertext); - unsigned char* ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!ct || !ss) { free(ct); free(ss); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_encaps(kem, ct, ss, pk) != OQS_SUCCESS) { - free(ct); free(ss); OQS_KEM_free(kem); - return pq_error("kyber-768 encapsulation failed"); - } - el_val_t ct_hex = el_hex_encode(ct, kem->length_ciphertext); - el_val_t ss_hex = el_hex_encode(ss, kem->length_shared_secret); - free(ct); - OQS_MEM_secure_free(ss, kem->length_shared_secret); - - const char* cts = EL_CSTR(ct_hex); - const char* sss = EL_CSTR(ss_hex); - char* buf = el_strbuf(strlen(cts) + strlen(sss) + 64); - sprintf(buf, "{\"ciphertext\":\"%s\",\"shared_secret\":\"%s\"}", cts, sss); - OQS_KEM_free(kem); - return el_wrap_str(buf); -} - -el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex) { - size_t sk_len = 0, ct_len = 0; - unsigned char* sk = el_hex_decode(EL_CSTR(secret_key_hex), &sk_len); - unsigned char* ct = el_hex_decode(EL_CSTR(ciphertext_hex), &ct_len); - if (!sk || !ct) return pq_error("invalid hex in inputs"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (sk_len != kem->length_secret_key || ct_len != kem->length_ciphertext) { - OQS_KEM_free(kem); - return pq_error("input length mismatch for kyber-768"); - } - unsigned char* ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!ss) { OQS_KEM_free(kem); return pq_error("oom"); } - /* Kyber is IND-CCA via Fujisaki-Okamoto: decaps always returns *some* - * shared_secret even on tampered ciphertext (an implicit-rejection value - * derived from sk). Protocols MUST confirm the shared_secret matches via - * a subsequent step (e.g. AEAD tag, key-confirmation MAC) — do not - * assume decaps success implies authenticity. */ - if (OQS_KEM_decaps(kem, ss, ct, sk) != OQS_SUCCESS) { - free(ss); OQS_KEM_free(kem); - return pq_error("kyber-768 decapsulation failed"); - } - el_val_t ss_hex = el_hex_encode(ss, kem->length_shared_secret); - OQS_MEM_secure_free(ss, kem->length_shared_secret); - OQS_KEM_free(kem); - return ss_hex; -} - -/* ─── Hybrid handshake (X25519 + Kyber-768, HKDF-SHA256 combined) ─────── */ - -#if !EL_HAVE_OPENSSL - -el_val_t pq_hybrid_keygen(void) { - return pq_error("hybrid handshake requires OpenSSL (X25519); rebuild with -lcrypto"); -} -el_val_t pq_hybrid_handshake(el_val_t pub) { - (void)pub; - return pq_error("hybrid handshake requires OpenSSL (X25519); rebuild with -lcrypto"); -} - -#else /* EL_HAVE_OPENSSL */ - -/* HKDF-SHA256 (RFC 5869) — Extract+Expand. Reuses the inline HMAC-SHA256 - * already in this file. Empty salt → 32 zero bytes per the RFC. */ -static void el_hkdf_sha256(const unsigned char* salt, size_t salt_len, - const unsigned char* ikm, size_t ikm_len, - const unsigned char* info, size_t info_len, - unsigned char* out, size_t out_len) { - unsigned char zero_salt[32] = {0}; - if (salt_len == 0) { salt = zero_salt; salt_len = 32; } - unsigned char prk[32]; - el_hmac_sha256(salt, salt_len, ikm, ikm_len, prk); - - unsigned char t[32]; - size_t produced = 0; - unsigned char counter = 1; - unsigned char* buf = (unsigned char*)malloc(32 + info_len + 1); - if (!buf) { fputs("el_runtime: hkdf oom\n", stderr); return; } - while (produced < out_len) { - size_t off = 0; - if (counter > 1) { memcpy(buf, t, 32); off = 32; } - if (info && info_len) { memcpy(buf + off, info, info_len); off += info_len; } - buf[off++] = counter; - el_hmac_sha256(prk, 32, buf, off, t); - size_t chunk = (out_len - produced > 32) ? 32 : (out_len - produced); - memcpy(out + produced, t, chunk); - produced += chunk; - counter++; - } - free(buf); -} - -/* X25519 keygen via OpenSSL EVP. Returns 1 on success. - * Fills pk[32] and sk[32] (raw X25519 byte strings, no DER wrapper). */ -static int el_x25519_keygen(unsigned char pk[32], unsigned char sk[32]) { - EVP_PKEY_CTX* pctx = EVP_PKEY_CTX_new_id(EVP_PKEY_X25519, NULL); - if (!pctx) return 0; - if (EVP_PKEY_keygen_init(pctx) != 1) { EVP_PKEY_CTX_free(pctx); return 0; } - EVP_PKEY* key = NULL; - if (EVP_PKEY_keygen(pctx, &key) != 1) { EVP_PKEY_CTX_free(pctx); return 0; } - EVP_PKEY_CTX_free(pctx); - - size_t plen = 32, slen = 32; - if (EVP_PKEY_get_raw_public_key (key, pk, &plen) != 1 || plen != 32) { - EVP_PKEY_free(key); return 0; - } - if (EVP_PKEY_get_raw_private_key(key, sk, &slen) != 1 || slen != 32) { - EVP_PKEY_free(key); return 0; - } - EVP_PKEY_free(key); - return 1; -} - -/* X25519 ECDH: derive 32-byte shared secret from local sk and remote pk. */ -static int el_x25519_derive(const unsigned char sk[32], const unsigned char rpk[32], - unsigned char ss[32]) { - EVP_PKEY* my = EVP_PKEY_new_raw_private_key(EVP_PKEY_X25519, NULL, sk, 32); - EVP_PKEY* rem = EVP_PKEY_new_raw_public_key (EVP_PKEY_X25519, NULL, rpk, 32); - if (!my || !rem) { EVP_PKEY_free(my); EVP_PKEY_free(rem); return 0; } - EVP_PKEY_CTX* dctx = EVP_PKEY_CTX_new(my, NULL); - if (!dctx) { EVP_PKEY_free(my); EVP_PKEY_free(rem); return 0; } - int ok = 0; - size_t out_len = 32; - if (EVP_PKEY_derive_init(dctx) == 1 && - EVP_PKEY_derive_set_peer(dctx, rem) == 1 && - EVP_PKEY_derive(dctx, ss, &out_len) == 1 && - out_len == 32) ok = 1; - EVP_PKEY_CTX_free(dctx); - EVP_PKEY_free(my); - EVP_PKEY_free(rem); - return ok; -} - -/* Hybrid wire layout (binary form, before hex encode): - * public_key = x25519_pub (32) || kyber_pub (1184) → 1216 bytes - * secret_key = x25519_sec (32) || kyber_sec (2400) → 2432 bytes - * ciphertext = ephem_x25519_pub (32) || kyber_ct (1088) → 1120 bytes - * shared_secret = HKDF-SHA256(x25519_ss || kyber_ss, info="el-pq-hybrid-v1", 32 bytes) - * The keygen result also exposes the four component hex fields for callers - * that prefer to handle the legs independently. */ - -el_val_t pq_hybrid_keygen(void) { - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - - unsigned char xpk[32], xsk[32]; - if (!el_x25519_keygen(xpk, xsk)) { - OQS_KEM_free(kem); - return pq_error("X25519 keygen failed"); - } - - unsigned char* kpk = (unsigned char*)malloc(kem->length_public_key); - unsigned char* ksk = (unsigned char*)malloc(kem->length_secret_key); - if (!kpk || !ksk) { free(kpk); free(ksk); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_keypair(kem, kpk, ksk) != OQS_SUCCESS) { - free(kpk); free(ksk); OQS_KEM_free(kem); - return pq_error("kyber-768 keypair generation failed"); - } - - size_t pub_len = 32 + kem->length_public_key; - size_t sec_len = 32 + kem->length_secret_key; - unsigned char* pub_buf = (unsigned char*)malloc(pub_len); - unsigned char* sec_buf = (unsigned char*)malloc(sec_len); - if (!pub_buf || !sec_buf) { - free(pub_buf); free(sec_buf); free(kpk); - OQS_MEM_secure_free(ksk, kem->length_secret_key); - OQS_KEM_free(kem); return pq_error("oom"); - } - memcpy(pub_buf, xpk, 32); memcpy(pub_buf + 32, kpk, kem->length_public_key); - memcpy(sec_buf, xsk, 32); memcpy(sec_buf + 32, ksk, kem->length_secret_key); - - el_val_t x_pub_hex = el_hex_encode(xpk, 32); - el_val_t x_sec_hex = el_hex_encode(xsk, 32); - el_val_t k_pub_hex = el_hex_encode(kpk, kem->length_public_key); - el_val_t k_sec_hex = el_hex_encode(ksk, kem->length_secret_key); - el_val_t pub_hex = el_hex_encode(pub_buf, pub_len); - el_val_t sec_hex = el_hex_encode(sec_buf, sec_len); - - OQS_MEM_secure_free(ksk, kem->length_secret_key); - free(kpk); free(pub_buf); free(sec_buf); - OQS_KEM_free(kem); - memset(xsk, 0, 32); /* best-effort wipe of stack copy */ - - const char* xph = EL_CSTR(x_pub_hex); - const char* xsh = EL_CSTR(x_sec_hex); - const char* kph = EL_CSTR(k_pub_hex); - const char* ksh = EL_CSTR(k_sec_hex); - const char* pubh = EL_CSTR(pub_hex); - const char* sech = EL_CSTR(sec_hex); - - char* buf = el_strbuf(strlen(xph) + strlen(xsh) + strlen(kph) + strlen(ksh) - + strlen(pubh) + strlen(sech) + 256); - sprintf(buf, - "{\"x25519_pub\":\"%s\",\"x25519_sec\":\"%s\"," - "\"kyber_pub\":\"%s\",\"kyber_sec\":\"%s\"," - "\"public_key\":\"%s\",\"secret_key\":\"%s\"}", - xph, xsh, kph, ksh, pubh, sech); - return el_wrap_str(buf); -} - -/* Initiator-side handshake. Caller supplies the responder's combined public - * key (x25519_pub || kyber_pub, hex-encoded). The runtime: - * 1. Generates an ephemeral X25519 keypair, runs ECDH against the - * responder's static x25519_pub. - * 2. Runs Kyber-768 encaps against the responder's kyber_pub → kyber_ct, - * kyber_ss. - * 3. Combined shared = HKDF-SHA256(salt="", ikm = x25519_ss || kyber_ss, - * info = "el-pq-hybrid-v1", L = 32). - * 4. Returns combined ciphertext (= ephemeral_x25519_pub || kyber_ct) and - * the derived shared_secret. - * - * Responder side composition (intentionally not a separate runtime fn — - * trivial to express in El given pq_kem_decaps + a future x25519_derive - * primitive): split the ciphertext into ephem_xpk (32) and kyber_ct, run - * X25519(static_xsk, ephem_xpk) and pq_kem_decaps(static_kyber_sk, kyber_ct), - * then HKDF-SHA256 with the same salt/info to recover the same shared_secret. - * If a separate x25519 entry point becomes valuable, add `pq_hybrid_open` - * here taking (secret_key_combined, ciphertext_combined). */ -el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined) { - size_t pub_len = 0; - unsigned char* rpub = el_hex_decode(EL_CSTR(remote_pub_combined), &pub_len); - if (!rpub) return pq_error("invalid hex in remote_pub_combined"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (pub_len != 32 + kem->length_public_key) { - OQS_KEM_free(kem); - return pq_error("remote_pub_combined length mismatch (expected x25519_pub || kyber_pub)"); - } - - unsigned char e_xpk[32], e_xsk[32], x_ss[32]; - if (!el_x25519_keygen(e_xpk, e_xsk)) { - OQS_KEM_free(kem); - return pq_error("X25519 ephemeral keygen failed"); - } - if (!el_x25519_derive(e_xsk, rpub, x_ss)) { - memset(e_xsk, 0, 32); - OQS_KEM_free(kem); - return pq_error("X25519 derive failed"); - } - memset(e_xsk, 0, 32); /* ephemeral; not needed after derive */ - - unsigned char* k_ct = (unsigned char*)malloc(kem->length_ciphertext); - unsigned char* k_ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!k_ct || !k_ss) { - free(k_ct); free(k_ss); OQS_KEM_free(kem); - return pq_error("oom"); - } - if (OQS_KEM_encaps(kem, k_ct, k_ss, rpub + 32) != OQS_SUCCESS) { - free(k_ct); free(k_ss); OQS_KEM_free(kem); - return pq_error("kyber-768 encapsulation failed"); - } - - /* HKDF combine: ikm = x_ss || k_ss. */ - size_t ikm_len = 32 + kem->length_shared_secret; - unsigned char* ikm = (unsigned char*)malloc(ikm_len); - if (!ikm) { - free(k_ct); OQS_MEM_secure_free(k_ss, kem->length_shared_secret); - OQS_KEM_free(kem); - return pq_error("oom"); - } - memcpy(ikm, x_ss, 32); - memcpy(ikm + 32, k_ss, kem->length_shared_secret); - unsigned char combined[32]; - static const char info_str[] = "el-pq-hybrid-v1"; - el_hkdf_sha256(NULL, 0, ikm, ikm_len, - (const unsigned char*)info_str, sizeof(info_str) - 1, - combined, 32); - - memset(x_ss, 0, 32); - OQS_MEM_secure_free(k_ss, kem->length_shared_secret); - OQS_MEM_secure_free(ikm, ikm_len); - - /* Combined ciphertext = ephemeral_x25519_pub || kyber_ct. */ - size_t ct_len = 32 + kem->length_ciphertext; - unsigned char* combined_ct = (unsigned char*)malloc(ct_len); - if (!combined_ct) { free(k_ct); OQS_KEM_free(kem); return pq_error("oom"); } - memcpy(combined_ct, e_xpk, 32); - memcpy(combined_ct + 32, k_ct, kem->length_ciphertext); - free(k_ct); - OQS_KEM_free(kem); - - el_val_t ct_hex = el_hex_encode(combined_ct, ct_len); - el_val_t ss_hex = el_hex_encode(combined, 32); - free(combined_ct); - memset(combined, 0, 32); - - const char* cts = EL_CSTR(ct_hex); - const char* sss = EL_CSTR(ss_hex); - char* buf = el_strbuf(strlen(cts) + strlen(sss) + 64); - sprintf(buf, "{\"ciphertext\":\"%s\",\"shared_secret\":\"%s\"}", cts, sss); - return el_wrap_str(buf); -} - -#endif /* EL_HAVE_OPENSSL */ -#endif /* EL_HAVE_LIBOQS */ - -/* ─── AEAD: AES-256-GCM ──────────────────────────────────────────────────── - * - * Symmetric authenticated encryption used to wrap envelopes once a shared - * secret has been derived from the KEM (Kyber-768 / hybrid). The El surface - * is intentionally narrow: - * - * aead_encrypt(key_hex, plaintext) - * → {"nonce":"<24 hex>","ciphertext":"<...hex including 16-byte tag>"} - * - * aead_decrypt(key_hex, nonce_hex, ciphertext_hex) - * → plaintext String, or "" on auth failure / malformed input - * - * Conventions: - * - key_hex must decode to exactly 32 bytes (AES-256). Callers that hold - * a longer KEM shared_secret should normalize via SHA3-256(ss) → 32 bytes - * before passing it in. (Kyber-768's shared_secret is already 32 bytes, - * but keeping this contract explicit lets the El side be agnostic.) - * - nonce is a fresh 12-byte random value drawn from the OS CSPRNG. Caller - * never picks the nonce — eliminates the GCM nonce-reuse footgun entirely. - * - tag is the standard 16 bytes, appended to ciphertext per RFC 5116. - * `ciphertext` field is therefore (plaintext_len + 16) bytes, hex-encoded. - * - No associated data (AAD). If we later need bound metadata, add a - * length-prefixed AAD argument and bump the envelope version tag. - * - * Failure mode: - * aead_encrypt returns http_error_json(...) on input/system failure. - * aead_decrypt returns the empty string on ANY failure (including auth-tag - * mismatch). Callers MUST check for "" before using the result. */ - -#if !__has_include() - -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext) { - (void)key_hex; (void)plaintext; - return http_error_json("aead_encrypt requires OpenSSL (libcrypto); rebuild with -lcrypto"); -} -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex) { - (void)key_hex; (void)nonce_hex; (void)ciphertext_hex; - return el_wrap_str(el_strdup("")); -} - -#else /* OpenSSL available */ - -#include -#include - -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext) { - size_t key_len = 0; - unsigned char* key = el_hex_decode(EL_CSTR(key_hex), &key_len); - if (!key) return http_error_json("invalid hex in key"); - if (key_len != 32) return http_error_json("aead key must be 32 bytes (64 hex chars) for AES-256-GCM"); - - const char* pt = EL_CSTR(plaintext); - size_t pt_len = el_input_len(pt); - if (!pt) pt = ""; - - unsigned char nonce[12]; - if (RAND_bytes(nonce, 12) != 1) return http_error_json("OS CSPRNG failed (RAND_bytes)"); - - EVP_CIPHER_CTX* ctx = EVP_CIPHER_CTX_new(); - if (!ctx) return http_error_json("EVP_CIPHER_CTX_new failed"); - - if (EVP_EncryptInit_ex(ctx, EVP_aes_256_gcm(), NULL, NULL, NULL) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm init failed"); - } - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_IVLEN, 12, NULL) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("set ivlen failed"); - } - if (EVP_EncryptInit_ex(ctx, NULL, NULL, key, nonce) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm key/iv init failed"); - } - - /* GCM ciphertext is the same length as plaintext; we append a 16-byte - * authentication tag for AEAD semantics. Allocate plaintext_len + 16. */ - unsigned char* ct = (unsigned char*)malloc(pt_len + 16); - if (!ct) { EVP_CIPHER_CTX_free(ctx); return http_error_json("oom"); } - int outlen = 0, total = 0; - if (EVP_EncryptUpdate(ctx, ct, &outlen, (const unsigned char*)pt, (int)pt_len) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm update failed"); - } - total += outlen; - if (EVP_EncryptFinal_ex(ctx, ct + total, &outlen) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm final failed"); - } - total += outlen; - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_GET_TAG, 16, ct + total) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm get tag failed"); - } - EVP_CIPHER_CTX_free(ctx); - - el_val_t nonce_hex_v = el_hex_encode(nonce, 12); - el_val_t ct_hex_v = el_hex_encode(ct, (size_t)total + 16); - free(ct); - - const char* nh = EL_CSTR(nonce_hex_v); - const char* ch = EL_CSTR(ct_hex_v); - char* buf = el_strbuf(strlen(nh) + strlen(ch) + 48); - sprintf(buf, "{\"nonce\":\"%s\",\"ciphertext\":\"%s\"}", nh, ch); - return el_wrap_str(buf); -} - -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex) { - size_t key_len = 0, nonce_len = 0, ct_len = 0; - unsigned char* key = el_hex_decode(EL_CSTR(key_hex), &key_len); - unsigned char* nonce = el_hex_decode(EL_CSTR(nonce_hex), &nonce_len); - unsigned char* ct = el_hex_decode(EL_CSTR(ciphertext_hex), &ct_len); - if (!key || !nonce || !ct) return el_wrap_str(el_strdup("")); - if (key_len != 32 || nonce_len != 12) return el_wrap_str(el_strdup("")); - if (ct_len < 16) return el_wrap_str(el_strdup("")); - - size_t body_len = ct_len - 16; - const unsigned char* tag = ct + body_len; - - EVP_CIPHER_CTX* ctx = EVP_CIPHER_CTX_new(); - if (!ctx) return el_wrap_str(el_strdup("")); - - if (EVP_DecryptInit_ex(ctx, EVP_aes_256_gcm(), NULL, NULL, NULL) != 1 || - EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_IVLEN, 12, NULL) != 1 || - EVP_DecryptInit_ex(ctx, NULL, NULL, key, nonce) != 1) { - EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - - unsigned char* pt = (unsigned char*)malloc(body_len + 1); - if (!pt) { EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); } - int outlen = 0, total = 0; - if (EVP_DecryptUpdate(ctx, pt, &outlen, ct, (int)body_len) != 1) { - free(pt); EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - total += outlen; - /* Set expected tag before final — GCM's final step is where auth happens. */ - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_TAG, 16, (void*)tag) != 1) { - free(pt); EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - int rc = EVP_DecryptFinal_ex(ctx, pt + total, &outlen); - EVP_CIPHER_CTX_free(ctx); - if (rc != 1) { - /* Auth failure or padding/length mismatch. Return empty so callers - * cannot accidentally treat tampered ciphertext as a valid message. */ - free(pt); - return el_wrap_str(el_strdup("")); - } - total += outlen; - pt[total] = '\0'; - - /* Copy into the el arena so the caller-visible string outlives this fn. */ - char* out = el_strbuf((size_t)total); - memcpy(out, pt, (size_t)total); - out[total] = '\0'; - free(pt); - return el_wrap_str(out); -} - -#endif /* __has_include() */ - -#ifdef HAVE_CURL -/* ──────────────────────────────────────────────────────────────────────────── - * OTLP/HTTP observability — logs, traces, metrics - * - * Design goals: - * - Zero blocking on the request path. Producers append to in-memory - * ring buffers; a single worker thread flushes to the OTLP endpoint. - * - Drop-on-failure semantics. If the endpoint is unreachable or slow, - * we drop telemetry rather than back-pressure into the request handler. - * - Best-effort serialization. Each record is pre-serialized as JSON when - * the El program calls the primitive; the worker just batches. - * - Configuration via env vars: - * OTLP_ENDPOINT e.g. https://alloy.neuralplatform.ai:4318 - * OTEL_SERVICE_NAME e.g. neuron-web (default: argv[0] basename) - * OTEL_SERVICE_VERSION (default: "0.0.0") - * OTEL_RESOURCE_ATTRS comma-sep k=v pairs (optional) - * - * Wire format: OTLP/HTTP JSON. Three endpoints: - * POST {endpoint}/v1/logs — log records - * POST {endpoint}/v1/traces — spans - * POST {endpoint}/v1/metrics — counter/gauge points - * - * El programs see four primitives: - * trace_span_start(name) -> SpanHandle (just a string id) - * trace_span_end(handle) (computes duration, queues) - * emit_log(level, msg, fields_json) (queues a log record) - * emit_metric(name, value, tags_json) (queues a counter increment) - * ──────────────────────────────────────────────────────────────────────────── - */ - -#define OTLP_BUF_CAP 4096 /* per-buffer ring size */ -#define OTLP_FLUSH_MS 2000 /* flush every 2s */ -#define OTLP_BATCH_MAX 200 /* up to 200 records per POST */ - -typedef struct { - char* data; /* malloc'd JSON fragment for this record */ -} OtlpRec; - -typedef struct { - OtlpRec ring[OTLP_BUF_CAP]; - size_t head; /* next write slot */ - size_t tail; /* next read slot */ - pthread_mutex_t mu; -} OtlpQueue; - -static OtlpQueue _otlp_logs = { .mu = PTHREAD_MUTEX_INITIALIZER }; -static OtlpQueue _otlp_traces = { .mu = PTHREAD_MUTEX_INITIALIZER }; -static OtlpQueue _otlp_metrics = { .mu = PTHREAD_MUTEX_INITIALIZER }; - -static char* _otlp_endpoint = NULL; /* e.g. https://alloy.neuralplatform.ai:4318 */ -static char* _otlp_service_name = NULL; -static char* _otlp_service_version = NULL; -static int _otlp_initialized = 0; -static pthread_t _otlp_worker_thread; - -/* enqueue — returns 1 if accepted, 0 if dropped (full buffer or no endpoint) */ -static int otlp_enqueue(OtlpQueue* q, const char* json) { - if (!_otlp_endpoint || !json) return 0; - pthread_mutex_lock(&q->mu); - size_t next_head = (q->head + 1) % OTLP_BUF_CAP; - if (next_head == q->tail) { - /* buffer full — drop oldest */ - free(q->ring[q->tail].data); - q->ring[q->tail].data = NULL; - q->tail = (q->tail + 1) % OTLP_BUF_CAP; - } - q->ring[q->head].data = strdup(json); - q->head = next_head; - pthread_mutex_unlock(&q->mu); - return 1; -} - -/* drain — copies up to OTLP_BATCH_MAX items into a comma-joined string, - * caller must free the result. Returns NULL if queue is empty. */ -static char* otlp_drain(OtlpQueue* q) { - pthread_mutex_lock(&q->mu); - if (q->head == q->tail) { pthread_mutex_unlock(&q->mu); return NULL; } - /* compute total length */ - size_t total = 0, count = 0; - size_t i = q->tail; - while (i != q->head && count < OTLP_BATCH_MAX) { - if (q->ring[i].data) total += strlen(q->ring[i].data) + 1; /* +1 for comma */ - i = (i + 1) % OTLP_BUF_CAP; - count++; - } - char* out = malloc(total + 4); - if (!out) { pthread_mutex_unlock(&q->mu); return NULL; } - out[0] = '\0'; - size_t off = 0; - i = q->tail; - count = 0; - while (i != q->head && count < OTLP_BATCH_MAX) { - if (q->ring[i].data) { - size_t l = strlen(q->ring[i].data); - if (off > 0) { out[off++] = ','; } - memcpy(out + off, q->ring[i].data, l); - off += l; - free(q->ring[i].data); - q->ring[i].data = NULL; - } - i = (i + 1) % OTLP_BUF_CAP; - count++; - } - out[off] = '\0'; - q->tail = i; - pthread_mutex_unlock(&q->mu); - return out; -} - -/* Build resource block once (service.name, service.version, host.name) */ -static char* otlp_resource_block(void) { - static char cached[1024]; - static int built = 0; - if (built) return cached; - char host[256] = "unknown"; - gethostname(host, sizeof(host) - 1); - snprintf(cached, sizeof(cached), - "{\"attributes\":[" - "{\"key\":\"service.name\",\"value\":{\"stringValue\":\"%s\"}}," - "{\"key\":\"service.version\",\"value\":{\"stringValue\":\"%s\"}}," - "{\"key\":\"host.name\",\"value\":{\"stringValue\":\"%s\"}}" - "]}", - _otlp_service_name ? _otlp_service_name : "el-app", - _otlp_service_version ? _otlp_service_version : "0.0.0", - host); - built = 1; - return cached; -} - -/* Best-effort POST. Drops on any error. */ -static void otlp_post(const char* path, const char* body) { - if (!_otlp_endpoint || !body || !*body) return; - char url[1024]; - snprintf(url, sizeof(url), "%s%s", _otlp_endpoint, path); - CURL* c = curl_easy_init(); - if (!c) return; - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)strlen(body)); - curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, 3000L); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, NULL); /* discard response */ - curl_easy_perform(c); - curl_slist_free_all(h); - curl_easy_cleanup(c); -} - -/* Flush worker — runs forever until process exits */ -static void* otlp_worker(void* arg) { - (void)arg; - while (1) { - struct timespec ts = { OTLP_FLUSH_MS / 1000, (OTLP_FLUSH_MS % 1000) * 1000000L }; - nanosleep(&ts, NULL); - - char* logs = otlp_drain(&_otlp_logs); - if (logs && *logs) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceLogs\":[{\"resource\":%s," - "\"scopeLogs\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"logRecords\":[%s]}]}]}", - otlp_resource_block(), logs); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/logs", body); - } - free(logs); - - char* traces = otlp_drain(&_otlp_traces); - if (traces && *traces) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceSpans\":[{\"resource\":%s," - "\"scopeSpans\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"spans\":[%s]}]}]}", - otlp_resource_block(), traces); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/traces", body); - } - free(traces); - - char* metrics = otlp_drain(&_otlp_metrics); - if (metrics && *metrics) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceMetrics\":[{\"resource\":%s," - "\"scopeMetrics\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"metrics\":[%s]}]}]}", - otlp_resource_block(), metrics); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/metrics", body); - } - free(metrics); - } - return NULL; -} - -/* Initialize OTLP — called lazily on first emit. Idempotent. */ -static void otlp_lazy_init(void) { - if (_otlp_initialized) return; - static pthread_mutex_t once_mu = PTHREAD_MUTEX_INITIALIZER; - pthread_mutex_lock(&once_mu); - if (_otlp_initialized) { pthread_mutex_unlock(&once_mu); return; } - - const char* ep = getenv("OTLP_ENDPOINT"); - if (!ep || !*ep) { - _otlp_initialized = 1; - pthread_mutex_unlock(&once_mu); - return; - } - _otlp_endpoint = strdup(ep); - /* trim trailing slash */ - size_t l = strlen(_otlp_endpoint); - if (l > 0 && _otlp_endpoint[l - 1] == '/') _otlp_endpoint[l - 1] = '\0'; - - const char* svc = getenv("OTEL_SERVICE_NAME"); - _otlp_service_name = strdup(svc && *svc ? svc : "el-app"); - const char* ver = getenv("OTEL_SERVICE_VERSION"); - _otlp_service_version = strdup(ver && *ver ? ver : "0.0.0"); - - pthread_create(&_otlp_worker_thread, NULL, otlp_worker, NULL); - pthread_detach(_otlp_worker_thread); - _otlp_initialized = 1; - pthread_mutex_unlock(&once_mu); -} - -/* JSON-escape a string into out_buf. Returns chars written (excluding null). */ -static size_t otlp_json_escape(const char* in, char* out, size_t out_cap) { - size_t o = 0; - for (size_t i = 0; in[i] && o + 8 < out_cap; i++) { - unsigned char c = (unsigned char)in[i]; - if (c == '"') { out[o++] = '\\'; out[o++] = '"'; } - else if (c == '\\'){ out[o++] = '\\'; out[o++] = '\\'; } - else if (c == '\n'){ out[o++] = '\\'; out[o++] = 'n'; } - else if (c == '\r'){ out[o++] = '\\'; out[o++] = 'r'; } - else if (c == '\t'){ out[o++] = '\\'; out[o++] = 't'; } - else if (c < 0x20) { o += snprintf(out + o, out_cap - o, "\\u%04x", c); } - else { out[o++] = (char)c; } - } - out[o] = '\0'; - return o; -} - -/* ── Public El primitives ─────────────────────────────────────────────────── */ - -/* emit_log(level, msg, fields_json) — fields_json is a JSON object string or "" */ -el_val_t emit_log(el_val_t level_v, el_val_t msg_v, el_val_t fields_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* level = EL_CSTR(level_v); if (!level) level = "INFO"; - const char* msg = EL_CSTR(msg_v); if (!msg) msg = ""; - const char* fields = EL_CSTR(fields_v); if (!fields) fields = ""; - /* Map El level names to OTLP severity numbers */ - int sev_num = 9; /* INFO */ - if (strcmp(level, "TRACE") == 0) sev_num = 1; - else if (strcmp(level, "DEBUG") == 0) sev_num = 5; - else if (strcmp(level, "INFO") == 0) sev_num = 9; - else if (strcmp(level, "WARN") == 0 || strcmp(level, "WARNING") == 0) sev_num = 13; - else if (strcmp(level, "ERROR") == 0) sev_num = 17; - else if (strcmp(level, "FATAL") == 0) sev_num = 21; - char esc_msg[2048]; otlp_json_escape(msg, esc_msg, sizeof(esc_msg)); - /* unix nanos */ - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"timeUnixNano\":\"%lld\",\"severityNumber\":%d," - "\"severityText\":\"%s\"," - "\"body\":{\"stringValue\":\"%s\"}%s%s}", - now_nano, sev_num, level, esc_msg, - (fields && *fields) ? ",\"attributes\":" : "", - (fields && *fields) ? fields : ""); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_logs, rec); - return EL_INT(1); -} - -/* emit_metric(name, value, tags_json) — Sum (counter) data point. tags_json - * is a JSON array of {key, value} pairs or empty string. */ -el_val_t emit_metric(el_val_t name_v, el_val_t value_v, el_val_t tags_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* name = EL_CSTR(name_v); if (!name) name = "unknown"; - int64_t val = (int64_t)value_v; - const char* tags = EL_CSTR(tags_v); if (!tags) tags = ""; - char esc_name[256]; otlp_json_escape(name, esc_name, sizeof(esc_name)); - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"name\":\"%s\",\"sum\":{\"aggregationTemporality\":2,\"isMonotonic\":true," - "\"dataPoints\":[{\"asInt\":\"%lld\"," - "\"timeUnixNano\":\"%lld\"" - "%s%s}]}}", - esc_name, (long long)val, now_nano, - (tags && *tags) ? ",\"attributes\":" : "", - (tags && *tags) ? tags : ""); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_metrics, rec); - return EL_INT(1); -} - -/* trace_span_start(name) — returns a span handle (string of "traceid:spanid:start_nano:name") */ -el_val_t trace_span_start(el_val_t name_v) { - otlp_lazy_init(); - const char* name = EL_CSTR(name_v); if (!name) name = "span"; - /* generate 16-byte trace id and 8-byte span id */ - static _Thread_local int seeded = 0; - if (!seeded) { srand((unsigned int)(uintptr_t)pthread_self() ^ (unsigned int)time(NULL)); seeded = 1; } - char tid[33], sid[17]; - for (int i = 0; i < 32; i++) tid[i] = "0123456789abcdef"[rand() & 0xF]; - tid[32] = '\0'; - for (int i = 0; i < 16; i++) sid[i] = "0123456789abcdef"[rand() & 0xF]; - sid[16] = '\0'; - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char* handle = malloc(strlen(name) + 80); - if (!handle) return EL_STR(""); - sprintf(handle, "%s:%s:%lld:%s", tid, sid, now_nano, name); - el_arena_track(handle); - return EL_STR(handle); -} - -/* trace_span_end(handle) — emits the span with computed duration */ -el_val_t trace_span_end(el_val_t handle_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* h = EL_CSTR(handle_v); if (!h) return EL_INT(0); - /* parse "tid:sid:start_nano:name" */ - char tid[64], sid[32], rest[1024]; - long long start_nano = 0; - if (sscanf(h, "%63[^:]:%31[^:]:%lld:%1023[^\n]", tid, sid, &start_nano, rest) != 4) return EL_INT(0); - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long end_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char esc_name[1024]; otlp_json_escape(rest, esc_name, sizeof(esc_name)); - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"traceId\":\"%s\",\"spanId\":\"%s\"," - "\"name\":\"%s\"," - "\"kind\":1," - "\"startTimeUnixNano\":\"%lld\"," - "\"endTimeUnixNano\":\"%lld\"," - "\"status\":{\"code\":1}}", - tid, sid, esc_name, start_nano, end_nano); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_traces, rec); - return EL_INT(1); -} - -/* Convenience: emit a one-shot timed event (emit start+end immediately). - * For El programs that want point events with duration baked in. */ -el_val_t emit_event(el_val_t name_v, el_val_t duration_ms_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* name = EL_CSTR(name_v); if (!name) name = "event"; - int64_t dur_ms = (int64_t)duration_ms_v; - el_val_t h = trace_span_start(EL_STR((char*)name)); - /* fudge start to be (now - duration) */ - (void)dur_ms; - return trace_span_end(h); -} - -#endif /* HAVE_CURL — OTLP */ - -/* ── Threading seed primitives ─────────────────────────────────────────────── - * __thread_create(fn_name, arg) -> Int spawn El fn in a pthread, return tid - * __thread_join(tid) -> String join thread, return result string - * __mutex_new() -> Int allocate a mutex, return handle - * __mutex_lock(m) lock mutex m - * __mutex_unlock(m) unlock mutex m - * - * Every El fn compiles to a global C symbol. __thread_create uses dlsym to - * look up the function by name and run it in a pthread. This means any El fn - * with signature (String) -> String is directly threadable. - */ - -typedef el_val_t (*ElFn1)(el_val_t); - -typedef struct { - ElFn1 fn; - el_val_t arg; - el_val_t result; -} ElThreadArg; - -#define EL_THREAD_MAX 256 - -typedef struct { - pthread_t tid; - ElThreadArg* arg; - int alive; -} ElThread; - -static ElThread _threads[EL_THREAD_MAX]; -static int _thread_count = 0; -static pthread_mutex_t _thread_alloc_mu = PTHREAD_MUTEX_INITIALIZER; - -static void* el_thread_runner(void* raw) { - ElThreadArg* a = (ElThreadArg*)raw; - a->result = a->fn(a->arg); - return NULL; -} - -el_val_t __thread_create(el_val_t fn_name_v, el_val_t arg_v) { - const char* sym = EL_CSTR(fn_name_v); - if (!sym || !*sym) return EL_INT(-1); - void* p = dlsym(RTLD_DEFAULT, sym); - if (!p) { - fprintf(stderr, "[__thread_create] symbol not found: %s\n", sym); - return EL_INT(-1); - } - ElThreadArg* a = (ElThreadArg*)malloc(sizeof(ElThreadArg)); - if (!a) return EL_INT(-1); - a->fn = (ElFn1)p; - a->arg = arg_v; - a->result = EL_STR(""); - - pthread_mutex_lock(&_thread_alloc_mu); - if (_thread_count >= EL_THREAD_MAX) { - pthread_mutex_unlock(&_thread_alloc_mu); - free(a); - fprintf(stderr, "[__thread_create] thread table full\n"); - return EL_INT(-1); - } - int slot = _thread_count++; - _threads[slot].arg = a; - _threads[slot].alive = 1; - pthread_mutex_unlock(&_thread_alloc_mu); - - if (pthread_create(&_threads[slot].tid, NULL, el_thread_runner, a) != 0) { - pthread_mutex_lock(&_thread_alloc_mu); - _thread_count--; - pthread_mutex_unlock(&_thread_alloc_mu); - free(a); - return EL_INT(-1); - } - return EL_INT(slot); -} - -el_val_t __thread_join(el_val_t tid_v) { - int slot = (int)(int64_t)tid_v; - if (slot < 0 || slot >= EL_THREAD_MAX) return EL_STR(""); - pthread_join(_threads[slot].tid, NULL); - el_val_t result = _threads[slot].arg->result; - free(_threads[slot].arg); - _threads[slot].alive = 0; - return result; -} - -/* Mutex table */ - -#define EL_MUTEX_MAX 64 - -typedef struct { - pthread_mutex_t mu; - int allocated; -} ElMutexEntry; - -static ElMutexEntry _mutexes[EL_MUTEX_MAX]; -static int _mutex_count = 0; -static pthread_mutex_t _mutex_alloc_mu = PTHREAD_MUTEX_INITIALIZER; - -el_val_t __mutex_new(void) { - pthread_mutex_lock(&_mutex_alloc_mu); - if (_mutex_count >= EL_MUTEX_MAX) { - pthread_mutex_unlock(&_mutex_alloc_mu); - fprintf(stderr, "[__mutex_new] mutex table full\n"); - return EL_INT(-1); - } - int slot = _mutex_count++; - pthread_mutex_init(&_mutexes[slot].mu, NULL); - _mutexes[slot].allocated = 1; - pthread_mutex_unlock(&_mutex_alloc_mu); - return EL_INT(slot); -} - -void __mutex_lock(el_val_t m_v) { - int slot = (int)(int64_t)m_v; - if (slot < 0 || slot >= EL_MUTEX_MAX || !_mutexes[slot].allocated) return; - pthread_mutex_lock(&_mutexes[slot].mu); -} - -void __mutex_unlock(el_val_t m_v) { - int slot = (int)(int64_t)m_v; - if (slot < 0 || slot >= EL_MUTEX_MAX || !_mutexes[slot].allocated) return; - pthread_mutex_unlock(&_mutexes[slot].mu); -} - -/* ── Channels ─────────────────────────────────────────────────────────────── * - * Buffered MPMC channel backed by a mutex + condvar + circular buffer. - * channel_new(capacity) -> Int (handle) - * channel_send(ch, msg) — blocks if full (capacity > 0) or never (unbounded) - * channel_recv(ch) -> String — blocks until a message is available - * channel_try_recv(ch) -> String — non-blocking, returns "" if empty - * channel_close(ch) — signal no more sends; recv drains remaining - * - * Bounded channels (cap > 0): circular buffer, sender blocks when full. - * Unbounded channels (cap == 0): dynamic array, sender never blocks. - */ -#define EL_CHANNEL_MAX 64 -#define EL_CHANNEL_BUF 1024 - -typedef struct { - char** buf; - int cap; /* 0 = unbounded (grows dynamically) */ - int head, tail, count; - int dyn_cap; /* allocated slots for unbounded mode */ - int closed; - pthread_mutex_t mu; - pthread_cond_t not_empty; - pthread_cond_t not_full; -} ElChannel; - -static ElChannel _channels[EL_CHANNEL_MAX]; -static int _channel_count = 0; -static pthread_mutex_t _channel_alloc_mu = PTHREAD_MUTEX_INITIALIZER; - -el_val_t __channel_new(el_val_t capacity_v) { - int cap = (int)(int64_t)capacity_v; - if (cap < 0) cap = 0; - - pthread_mutex_lock(&_channel_alloc_mu); - if (_channel_count >= EL_CHANNEL_MAX) { - pthread_mutex_unlock(&_channel_alloc_mu); - fprintf(stderr, "[__channel_new] channel table full\n"); - return EL_INT(-1); - } - int slot = _channel_count++; - pthread_mutex_unlock(&_channel_alloc_mu); - - ElChannel* ch = &_channels[slot]; - memset(ch, 0, sizeof(*ch)); - ch->cap = cap; - ch->closed = 0; - ch->head = 0; - ch->tail = 0; - ch->count = 0; - - if (cap > 0) { - /* Bounded: fixed circular buffer. */ - ch->buf = (char**)malloc((size_t)cap * sizeof(char*)); - ch->dyn_cap = cap; - } else { - /* Unbounded: start with EL_CHANNEL_BUF slots, grow as needed. */ - ch->buf = (char**)malloc(EL_CHANNEL_BUF * sizeof(char*)); - ch->dyn_cap = EL_CHANNEL_BUF; - } - if (!ch->buf) { - fprintf(stderr, "[__channel_new] out of memory\n"); - return EL_INT(-1); - } - - pthread_mutex_init(&ch->mu, NULL); - pthread_cond_init(&ch->not_empty, NULL); - pthread_cond_init(&ch->not_full, NULL); - - return EL_INT(slot); -} - -void __channel_send(el_val_t ch_v, el_val_t msg_v) { - int slot = (int)(int64_t)ch_v; - if (slot < 0 || slot >= EL_CHANNEL_MAX) return; - ElChannel* ch = &_channels[slot]; - - const char* msg = EL_CSTR(msg_v); - if (!msg) msg = ""; - char* copy = strdup(msg); /* channel owns the string */ - - pthread_mutex_lock(&ch->mu); - - if (ch->closed) { - /* Send on closed channel is a no-op (drop the message). */ - pthread_mutex_unlock(&ch->mu); - free(copy); - return; - } - - if (ch->cap > 0) { - /* Bounded: block while full. */ - while (ch->count >= ch->cap && !ch->closed) { - pthread_cond_wait(&ch->not_full, &ch->mu); - } - if (ch->closed) { - pthread_mutex_unlock(&ch->mu); - free(copy); - return; - } - ch->buf[ch->tail] = copy; - ch->tail = (ch->tail + 1) % ch->cap; - ch->count++; - } else { - /* Unbounded: grow the buffer if needed. */ - if (ch->count >= ch->dyn_cap) { - int new_cap = ch->dyn_cap * 2; - char** grown = (char**)realloc(ch->buf, (size_t)new_cap * sizeof(char*)); - if (!grown) { - pthread_mutex_unlock(&ch->mu); - free(copy); - fprintf(stderr, "[__channel_send] out of memory growing channel\n"); - return; - } - /* The circular buffer may have wrapped. Linearise it first. - * In unbounded mode head is always 0 (we append at tail, drain - * from head), so a simple memmove isn't needed — but if the - * buffer did wrap (tail < head after growth), we need to fix up. - * Simplest safe path: if tail wrapped, move the head..old_cap - * segment to new_cap..new_cap+(old_cap-head). */ - if (ch->tail < ch->head) { - /* Wrapped: [head..old_cap) is the front, [0..tail) is the back. */ - int front = ch->dyn_cap - ch->head; - memmove(grown + ch->dyn_cap, grown + ch->head, (size_t)front * sizeof(char*)); - ch->head = ch->dyn_cap; - } - ch->buf = grown; - ch->dyn_cap = new_cap; - } - ch->buf[ch->tail] = copy; - ch->tail = (ch->tail + 1) % ch->dyn_cap; - ch->count++; - } - - pthread_cond_signal(&ch->not_empty); - pthread_mutex_unlock(&ch->mu); -} - -el_val_t __channel_recv(el_val_t ch_v) { - int slot = (int)(int64_t)ch_v; - if (slot < 0 || slot >= EL_CHANNEL_MAX) return EL_STR(""); - ElChannel* ch = &_channels[slot]; - - pthread_mutex_lock(&ch->mu); - - /* Block until there is a message or the channel is closed and drained. */ - while (ch->count == 0 && !ch->closed) { - pthread_cond_wait(&ch->not_empty, &ch->mu); - } - - if (ch->count == 0) { - /* Closed and empty — signal EOF. */ - pthread_mutex_unlock(&ch->mu); - return EL_STR(""); - } - - int buf_cap = (ch->cap > 0) ? ch->cap : ch->dyn_cap; - char* msg = ch->buf[ch->head]; - ch->head = (ch->head + 1) % buf_cap; - ch->count--; - - pthread_cond_signal(&ch->not_full); - pthread_mutex_unlock(&ch->mu); - - /* Hand the string to the arena so it is freed after the request. */ - el_arena_track(msg); - return EL_STR(msg); -} - -el_val_t __channel_try_recv(el_val_t ch_v) { - int slot = (int)(int64_t)ch_v; - if (slot < 0 || slot >= EL_CHANNEL_MAX) return EL_STR(""); - ElChannel* ch = &_channels[slot]; - - pthread_mutex_lock(&ch->mu); - - if (ch->count == 0) { - pthread_mutex_unlock(&ch->mu); - return EL_STR(""); - } - - int buf_cap = (ch->cap > 0) ? ch->cap : ch->dyn_cap; - char* msg = ch->buf[ch->head]; - ch->head = (ch->head + 1) % buf_cap; - ch->count--; - - pthread_cond_signal(&ch->not_full); - pthread_mutex_unlock(&ch->mu); - - el_arena_track(msg); - return EL_STR(msg); -} - -void __channel_close(el_val_t ch_v) { - int slot = (int)(int64_t)ch_v; - if (slot < 0 || slot >= EL_CHANNEL_MAX) return; - ElChannel* ch = &_channels[slot]; - - pthread_mutex_lock(&ch->mu); - ch->closed = 1; - /* Wake all blocked recvers and senders so they can observe the close. */ - pthread_cond_broadcast(&ch->not_empty); - pthread_cond_broadcast(&ch->not_full); - pthread_mutex_unlock(&ch->mu); -} - -/* ── DHARMA runtime additions ──────────────────────────────────────────────── - * - * Functions required by the dharma registry service. Added here so the - * released el_runtime.c includes them without requiring dharma to bundle - * its own stubs. - * - * Functions added: - * list_len — alias for el_list_len (used in handlers.el) - * list_get — alias for el_list_get (used in handlers.el) - * json_array_push — append a pre-encoded JSON element to a JSON array string - * now_millis — milliseconds since Unix epoch (alias for time_now) - * unix_timestamp_ms — same as now_millis (alias) - * time_now_ms — same as now_millis (alias) - * log_info — stderr structured log at INFO level - * log_warn — stderr structured log at WARN level - * config — reads a config value from the environment - * http_patch — HTTP PATCH with JSON Content-Type - * http_post_engram — HTTP POST with optional X-API-Key header - * http_get_engram — HTTP GET with optional X-API-Key header - * str_to_bytes — encode a string as a JSON array of byte values - * bytes_to_str — decode a JSON array of byte values back to a string - * hash_sha256 — SHA-256 hex digest of a string - */ - -/* list_len — return the number of elements in a list. */ -el_val_t list_len(el_val_t list) { - return el_list_len(list); -} - -/* list_get — return the element at index i in a list. */ -el_val_t list_get(el_val_t list, el_val_t index) { - return el_list_get(list, index); -} - -/* json_array_push — append element (a pre-encoded JSON fragment, e.g. "\"foo\"" - * or "42") to the JSON array string arr. Returns a new JSON array string. - * Example: json_array_push("[]", "\"alice\"") -> "[\"alice\"]" - * json_array_push("[\"alice\"]", "\"bob\"") -> "[\"alice\",\"bob\"]" */ -el_val_t json_array_push(el_val_t arr_v, el_val_t elem_v) { - const char* arr = EL_CSTR(arr_v); - const char* elem = EL_CSTR(elem_v); - if (!arr || !*arr) arr = "[]"; - if (!elem || !*elem) elem = "null"; - - /* Trim whitespace, find the closing ']'. */ - const char* p = arr; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') { - /* Not an array — return a single-element array. */ - size_t n = strlen(elem) + 4; - char* out = el_strbuf(n); - snprintf(out, n, "[%s]", elem); - return el_wrap_str(out); - } - size_t arr_len = strlen(arr); - size_t elem_len = strlen(elem); - - /* Walk from the end to find the matching ']'. */ - const char* end = arr + arr_len - 1; - while (end > p && (*end == ' ' || *end == '\t' || *end == '\n' || *end == '\r')) end--; - if (*end != ']') { - /* Malformed — wrap elem in a new array. */ - size_t n = elem_len + 4; - char* out = el_strbuf(n); - snprintf(out, n, "[%s]", elem); - return el_wrap_str(out); - } - - /* Content between '[' and ']'. */ - const char* inner_start = p + 1; - const char* inner_end = end; /* points AT ']' */ - /* Check if the array is empty (only whitespace between brackets). */ - const char* q = inner_start; - while (q < inner_end && (*q == ' ' || *q == '\t' || *q == '\n' || *q == '\r')) q++; - int empty = (q == inner_end); - - /* Build: prefix + (comma if non-empty) + elem + "]" */ - size_t prefix_len = (size_t)(inner_end - arr); /* up to but not including ']' */ - size_t sep_len = empty ? 0 : 1; /* "," if non-empty */ - size_t out_len = prefix_len + sep_len + elem_len + 2; /* +"]" + NUL */ - char* out = el_strbuf(out_len); - memcpy(out, arr, prefix_len); - if (!empty) out[prefix_len] = ','; - memcpy(out + prefix_len + sep_len, elem, elem_len); - out[prefix_len + sep_len + elem_len] = ']'; - out[prefix_len + sep_len + elem_len + 1] = '\0'; - return el_wrap_str(out); -} - -/* now_millis — milliseconds since Unix epoch. */ -el_val_t now_millis(void) { - return time_now(); -} - -/* unix_timestamp_ms — same as now_millis. */ -el_val_t unix_timestamp_ms(void) { - return time_now(); -} - -/* time_now_ms — same as now_millis. */ -el_val_t time_now_ms(void) { - return time_now(); -} - -/* log_info — write a structured [INFO] line to stderr. */ -void log_info(el_val_t msg_v) { - const char* msg = EL_CSTR(msg_v); - fprintf(stderr, "[INFO] %s\n", msg ? msg : ""); -} - -/* log_warn — write a structured [WARN] line to stderr. */ -void log_warn(el_val_t msg_v) { - const char* msg = EL_CSTR(msg_v); - fprintf(stderr, "[WARN] %s\n", msg ? msg : ""); -} - -/* config — read a configuration value from the environment. - * Returns "" if the variable is not set (same as __env_get). */ -el_val_t config(el_val_t key_v) { - const char* key = EL_CSTR(key_v); - if (!key || !*key) return EL_STR(""); - const char* val = getenv(key); - if (!val) return EL_STR(""); - return el_wrap_str(el_strdup(val)); -} - -#ifdef HAVE_CURL -/* http_patch — HTTP PATCH request with Content-Type: application/json. - * Returns the response body (same error convention as http_post_json). */ -el_val_t http_patch(el_val_t url_v, el_val_t body_v) { - const char* url = EL_CSTR(url_v); - const char* body = EL_CSTR(body_v); - if (!url || !*url) return http_error_json("empty url"); - CURL* c = curl_easy_init(); - if (!c) return http_error_json("curl_easy_init failed"); - HttpBuf rb; httpbuf_init(&rb); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_CUSTOMREQUEST, "PATCH"); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - CURLcode rc = curl_easy_perform(c); - curl_slist_free_all(h); - curl_easy_cleanup(c); - if (rc != CURLE_OK) { - free(rb.data); - const char* m = errbuf[0] ? errbuf : curl_easy_strerror(rc); - return http_error_json(m); - } - return el_wrap_str(rb.data); -} - -/* http_post_engram — HTTP POST with optional X-API-Key header. - * If key is "" no authentication header is sent. */ -el_val_t http_post_engram(el_val_t url_v, el_val_t key_v, el_val_t body_v) { - const char* url = EL_CSTR(url_v); - const char* key = EL_CSTR(key_v); - const char* body = EL_CSTR(body_v); - if (!url || !*url) return http_error_json("empty url"); - CURL* c = curl_easy_init(); - if (!c) return http_error_json("curl_easy_init failed"); - HttpBuf rb; httpbuf_init(&rb); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - if (key && *key) { - size_t n = strlen(key) + 32; - char* hdr = malloc(n); - snprintf(hdr, n, "X-API-Key: %s", key); - h = curl_slist_append(h, hdr); - free(hdr); - } - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - CURLcode rc = curl_easy_perform(c); - curl_slist_free_all(h); - curl_easy_cleanup(c); - if (rc != CURLE_OK) { - free(rb.data); - const char* m = errbuf[0] ? errbuf : curl_easy_strerror(rc); - return http_error_json(m); - } - return el_wrap_str(rb.data); -} - -/* http_get_engram — HTTP GET with optional X-API-Key header. */ -el_val_t http_get_engram(el_val_t url_v, el_val_t key_v) { - const char* url = EL_CSTR(url_v); - const char* key = EL_CSTR(key_v); - if (!url || !*url) return http_error_json("empty url"); - CURL* c = curl_easy_init(); - if (!c) return http_error_json("curl_easy_init failed"); - HttpBuf rb; httpbuf_init(&rb); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - struct curl_slist* h = NULL; - if (key && *key) { - size_t n = strlen(key) + 32; - char* hdr = malloc(n); - snprintf(hdr, n, "X-API-Key: %s", key); - h = curl_slist_append(h, hdr); - free(hdr); - } - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_HTTPGET, 1L); - if (h) curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - CURLcode rc = curl_easy_perform(c); - if (h) curl_slist_free_all(h); - curl_easy_cleanup(c); - if (rc != CURLE_OK) { - free(rb.data); - const char* m = errbuf[0] ? errbuf : curl_easy_strerror(rc); - return http_error_json(m); - } - return el_wrap_str(rb.data); -} -#endif /* HAVE_CURL */ - -/* str_to_bytes — encode a string as a JSON array of unsigned byte values. - * "hello" -> "[104,101,108,108,111]" - * Used by db.el to store binary content in Engram JSON nodes. */ -el_val_t str_to_bytes(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s || !*s) return el_wrap_str(el_strdup("[]")); - size_t n = strlen(s); - /* Worst case: each byte is 3 digits + comma = 4 chars, plus "[]" + NUL. */ - char* out = el_strbuf(n * 4 + 3); - size_t pos = 0; - out[pos++] = '['; - for (size_t i = 0; i < n; i++) { - unsigned char b = (unsigned char)s[i]; - if (i > 0) out[pos++] = ','; - /* Write decimal representation of b. */ - if (b >= 100) { - out[pos++] = (char)('0' + b / 100); - out[pos++] = (char)('0' + (b / 10) % 10); - out[pos++] = (char)('0' + b % 10); - } else if (b >= 10) { - out[pos++] = (char)('0' + b / 10); - out[pos++] = (char)('0' + b % 10); - } else { - out[pos++] = (char)('0' + b); - } - } - out[pos++] = ']'; - out[pos] = '\0'; - return el_wrap_str(out); -} - -/* bytes_to_str — decode a JSON array of integer byte values back to a string. - * "[104,101,108,108,111]" -> "hello" - * Inverse of str_to_bytes. */ -el_val_t bytes_to_str(el_val_t arr_v) { - const char* s = EL_CSTR(arr_v); - if (!s) return el_wrap_str(el_strdup("")); - /* Skip whitespace, expect '['. */ - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s != '[') return el_wrap_str(el_strdup("")); - s++; - - /* Count elements to size the output buffer. */ - int64_t n = (int64_t)json_array_len(arr_v); - if (n <= 0) return el_wrap_str(el_strdup("")); - - char* out = el_strbuf((size_t)n + 1); - size_t pos = 0; - - /* Walk the array, parse each integer, store as a byte. */ - while (*s) { - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ']' || *s == '\0') break; - /* Parse decimal integer. */ - char* end_ptr; - long v = strtol(s, &end_ptr, 10); - if (end_ptr == s) break; /* parse failure */ - s = end_ptr; - if (v >= 0 && v <= 255) out[pos++] = (char)(unsigned char)v; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ',') { s++; continue; } - if (*s == ']' || *s == '\0') break; - } - out[pos] = '\0'; - return el_wrap_str(out); -} - -/* hash_sha256 — return the SHA-256 hex digest of a string. - * Uses the built-in el_sha256_oneshot implementation (no OpenSSL required). */ -el_val_t hash_sha256(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) s = ""; - unsigned char digest[32]; - el_sha256_oneshot((const unsigned char*)s, strlen(s), digest); - return el_hex_encode(digest, 32); -} - -/* ── __ prefixed aliases — public boundary for compiled El programs ────────── - * - * The El compiler's self-hosting back-end emits calls to __-prefixed function - * names (e.g. __println, __str_len). These wrappers forward to the existing - * el_runtime implementations so both naming conventions resolve at link time. - * - * Note: __thread_create and __thread_join are already defined above in the - * threading section; they are not repeated here. - * ──────────────────────────────────────────────────────────────────────────── */ - -/* I/O */ -el_val_t __println(el_val_t s) { return println(s); } -el_val_t __print(el_val_t s) { return print(s); } -el_val_t __readline(void) { return readline(); } - -/* String */ -el_val_t __int_to_str(el_val_t n) { return int_to_str(n); } -el_val_t __str_to_int(el_val_t s) { return str_to_int(s); } -el_val_t __float_to_str(el_val_t f) { return float_to_str(f); } -el_val_t __str_to_float(el_val_t s) { return str_to_float(s); } -el_val_t __str_len(el_val_t s) { return str_len(s); } -el_val_t __str_char_at(el_val_t s, el_val_t i) { return str_char_at(s, i); } - -el_val_t __str_cmp(el_val_t a, el_val_t b) { - const char* ca = EL_CSTR(a); - const char* cb = EL_CSTR(b); - if (!ca) ca = ""; - if (!cb) cb = ""; - return (el_val_t)strcmp(ca, cb); -} - -el_val_t __str_ncmp(el_val_t a, el_val_t b, el_val_t n) { - const char* ca = EL_CSTR(a); - const char* cb = EL_CSTR(b); - if (!ca) ca = ""; - if (!cb) cb = ""; - return (el_val_t)strncmp(ca, cb, (size_t)n); -} - -el_val_t __str_concat_raw(el_val_t a, el_val_t b) { return str_concat(a, b); } -el_val_t __str_slice_raw(el_val_t s, el_val_t start, el_val_t end) { return str_slice(s, start, end); } - -el_val_t __str_alloc(el_val_t n) { - if (n <= 0) n = 0; - char* buf = el_strbuf((size_t)n + 1); - memset(buf, 0, (size_t)n + 1); - return el_wrap_str(buf); -} - -el_val_t __str_set_char(el_val_t s, el_val_t i, el_val_t c) { - char* buf = (char*)(uintptr_t)s; - if (buf) buf[(size_t)i] = (char)c; - return s; -} - -/* URL encoding */ -el_val_t __url_encode(el_val_t s) { return url_encode(s); } -el_val_t __url_decode(el_val_t s) { return url_decode(s); } - -/* Environment */ -el_val_t __env_get(el_val_t key) { return env(key); } - -/* Subprocess */ -el_val_t __exec(el_val_t cmd) { return exec(cmd); } -el_val_t __exec_bg(el_val_t cmd) { return exec_bg(cmd); } - -/* Process */ -el_val_t __exit_program(el_val_t code) { return exit_program(code); } - -/* Filesystem */ -el_val_t __fs_exists(el_val_t path) { return fs_exists(path); } -el_val_t __fs_mkdir(el_val_t path) { return fs_mkdir(path); } -el_val_t __fs_read(el_val_t path) { return fs_read(path); } -el_val_t __fs_write(el_val_t path, el_val_t content) { return fs_write(path, content); } -el_val_t __fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t n) { return fs_write_bytes(path, bytes, n); } -el_val_t __fs_list_raw(el_val_t path) { return fs_list_json(path); } - -/* HTTP server (no curl dependency) */ -el_val_t __http_response(el_val_t status, el_val_t headers_json, el_val_t body) { return http_response(status, headers_json, body); } -el_val_t __http_serve(el_val_t port, el_val_t handler) { return http_serve(port, handler); } -el_val_t __http_serve_v2(el_val_t port, el_val_t handler) { return http_serve_v2(port, handler); } - -/* HTTP conn fd / SSE — __http_conn_fd lives in el_seed.c; stubs provided here - * so el_runtime.c compiles standalone. When both translation units are linked - * the el_seed.c definitions win via their non-static linkage (strong symbols). - * These stubs are marked weak so they are silently overridden. */ -__attribute__((weak)) el_val_t __http_conn_fd(void) { return (el_val_t)(-1); } -__attribute__((weak)) el_val_t __http_sse_open(el_val_t conn_id) { (void)conn_id; return 0; } -__attribute__((weak)) el_val_t __http_sse_send(el_val_t conn_id, el_val_t data) { (void)conn_id; (void)data; return 0; } -__attribute__((weak)) el_val_t __http_sse_close(el_val_t conn_id) { (void)conn_id; return 0; } - -/* JSON */ -el_val_t __json_array_get(el_val_t json, el_val_t index) { return json_array_get(json, index); } -el_val_t __json_array_get_string(el_val_t json, el_val_t index) { return json_array_get_string(json, index); } -el_val_t __json_array_len(el_val_t json) { return json_array_len(json); } -el_val_t __json_get(el_val_t json, el_val_t key) { return json_get(json, key); } -el_val_t __json_get_raw(el_val_t json, el_val_t key) { return json_get_raw(json, key); } -el_val_t __json_set(el_val_t json, el_val_t key, el_val_t value){ return json_set(json, key, value); } -el_val_t __json_parse_map(el_val_t json_str) { return json_parse(json_str); } -el_val_t __json_stringify_val(el_val_t val) { return json_stringify(val); } - -/* Hashing */ -el_val_t __sha256_hex(el_val_t s) { return hash_sha256(s); } - -/* State K/V */ -el_val_t __state_del(el_val_t key) { return state_del(key); } -el_val_t __state_get(el_val_t key) { return state_get(key); } -el_val_t __state_keys(void) { return state_keys(); } -el_val_t __state_set(el_val_t key, el_val_t val) { return state_set(key, val); } - -/* UUID */ -el_val_t __uuid_v4(void) { return uuid_v4(); } - -/* Args */ -el_val_t __args_json(void) { return args(); } - -/* HTTP client aliases — require curl; defined inside #ifdef HAVE_CURL below - * with a matching stub in the #ifndef HAVE_CURL block. */ -#ifdef HAVE_CURL -el_val_t __http_do(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_map, el_val_t timeout_ms) { - /* timeout_ms is accepted for API compatibility but ignored here; - * el_runtime's http_do uses the EL_HTTP_TIMEOUT_MS env var instead. */ - (void)timeout_ms; - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do(EL_CSTR(method), EL_CSTR(url), EL_CSTR(body), h); - if (h) curl_slist_free_all(h); - return r; -} - -/* __http_do_map — same as __http_do but headers_map arg is a JSON-string - * rather than an ElMap. Parse it first, then delegate. */ -el_val_t __http_do_map(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_json, el_val_t timeout_ms) { - (void)timeout_ms; - /* Build a curl_slist from a JSON object {"Header":"value",...}. */ - const char* hj = EL_CSTR(headers_json); - struct curl_slist* h = NULL; - if (hj && *hj && *hj == '{') { - /* Walk the JSON pairs with a simple parser reusing json_get_string logic. */ - /* For correctness we just call the existing json_get iteration path. - * We duplicate the key-extraction loop from headers_from_map but driven - * by JSON rather than ElMap. Use json_get_raw to iterate is not easy - * without knowing keys, so accept the JSON string and build a tmp map. */ - el_val_t map = json_parse(EL_STR(hj)); - h = headers_from_map(map); - } - el_val_t r = http_do(EL_CSTR(method), EL_CSTR(url), EL_CSTR(body), h); - if (h) curl_slist_free_all(h); - return r; -} - -/* __http_do_map_to_file — same as __http_do_map but streams response body - * to a local file path rather than returning it as a string. */ -el_val_t __http_do_map_to_file(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_json, el_val_t output_path) { - const char* hj = EL_CSTR(headers_json); - struct curl_slist* h = NULL; - if (hj && *hj && *hj == '{') { - el_val_t map = json_parse(EL_STR(hj)); - h = headers_from_map(map); - } - el_val_t r = http_do_to_file(EL_CSTR(method), EL_CSTR(url), EL_CSTR(body), - h, EL_CSTR(output_path)); - if (h) curl_slist_free_all(h); - return r; -} -#endif /* HAVE_CURL */ - -#ifndef HAVE_CURL -/* ── HAVE_CURL=0 stubs — compile without -lcurl for the elc CLI binary. ───── * - * These return a JSON error string so El programs get a clear message if they - * call HTTP/LLM functions in a curl-less build. */ -static el_val_t _no_curl_err(void) { - return el_wrap_str(el_strdup("{\"error\":\"not built with HAVE_CURL\"}")); -} -el_val_t http_get(el_val_t url) { (void)url; return _no_curl_err(); } -el_val_t http_post(el_val_t url, el_val_t body) { (void)url; (void)body; return _no_curl_err(); } -el_val_t http_post_json(el_val_t url, el_val_t body) { (void)url; (void)body; return _no_curl_err(); } -el_val_t http_get_with_headers(el_val_t url, el_val_t h) { (void)url; (void)h; return _no_curl_err(); } -el_val_t http_post_with_headers(el_val_t url, el_val_t b, el_val_t h) { (void)url; (void)b; (void)h; return _no_curl_err(); } -el_val_t http_post_json_with_headers(el_val_t url, el_val_t h, el_val_t b) { (void)url; (void)h; (void)b; return _no_curl_err(); } -el_val_t http_post_form_auth(el_val_t url, el_val_t b, el_val_t a) { (void)url; (void)b; (void)a; return _no_curl_err(); } -el_val_t http_delete(el_val_t url) { (void)url; return _no_curl_err(); } -el_val_t http_patch(el_val_t url, el_val_t body) { (void)url; (void)body; return _no_curl_err(); } -el_val_t http_get_to_file(el_val_t url, el_val_t h, el_val_t p) { (void)url; (void)h; (void)p; return _no_curl_err(); } -el_val_t http_post_to_file(el_val_t url, el_val_t b, el_val_t h, el_val_t p) { (void)url; (void)b; (void)h; (void)p; return _no_curl_err(); } -el_val_t http_post_engram(el_val_t url, el_val_t k, el_val_t b) { (void)url; (void)k; (void)b; return _no_curl_err(); } -el_val_t http_get_engram(el_val_t url, el_val_t k) { (void)url; (void)k; return _no_curl_err(); } -el_val_t llm_call(el_val_t m, el_val_t p) { (void)m; (void)p; return _no_curl_err(); } -el_val_t llm_call_system(el_val_t m, el_val_t s, el_val_t u) { (void)m; (void)s; (void)u; return _no_curl_err(); } -el_val_t llm_call_agentic(el_val_t m, el_val_t s, el_val_t u, el_val_t t) { (void)m; (void)s; (void)u; (void)t; return _no_curl_err(); } -el_val_t llm_vision(el_val_t m, el_val_t s, el_val_t p, el_val_t i) { (void)m; (void)s; (void)p; (void)i; return _no_curl_err(); } -el_val_t llm_models(void) { return el_list_empty(); } -void llm_register_tool(el_val_t n, el_val_t f) { (void)n; (void)f; } -/* __ HTTP stubs (no-curl build) */ -el_val_t __http_do(el_val_t m, el_val_t u, el_val_t b, el_val_t h, el_val_t t) { (void)m; (void)u; (void)b; (void)h; (void)t; return _no_curl_err(); } -el_val_t __http_do_map(el_val_t m, el_val_t u, el_val_t b, el_val_t h, el_val_t t) { (void)m; (void)u; (void)b; (void)h; (void)t; return _no_curl_err(); } -el_val_t __http_do_map_to_file(el_val_t m, el_val_t u, el_val_t b, el_val_t h, el_val_t p) { (void)m; (void)u; (void)b; (void)h; (void)p; return _no_curl_err(); } -#endif /* !HAVE_CURL */ diff --git a/lang/el-compiler/runtime/el_runtime.h b/lang/el-compiler/runtime/el_runtime.h deleted file mode 100644 index 87348f5..0000000 --- a/lang/el-compiler/runtime/el_runtime.h +++ /dev/null @@ -1,897 +0,0 @@ -/* - * el_runtime.h — El language C runtime header - * - * Declares all built-in functions available to compiled El programs. - * Include this in every generated .c file. - * - * Value model: - * All El values are represented as el_val_t (= int64_t). - * On 64-bit systems a pointer fits in int64_t. - * String values are cast: (el_val_t)(uintptr_t)"hello" - * Integer values are stored directly. - * This lets arithmetic work naturally while still passing strings around. - * - * Type conventions (El -> C): - * String -> el_val_t (holds const char* via uintptr_t cast) - * Int -> el_val_t - * Bool -> el_val_t (0 = false, nonzero = true) - * Any -> el_val_t - * Void -> void - * - * Macros for convenience: - * EL_STR(s) cast string literal to el_val_t - * EL_CSTR(v) cast el_val_t back to const char* - * EL_INT(v) identity — el_val_t is already int64_t - * EL_NULL null / zero value - * EL_FALSE boolean false (0) - * EL_TRUE boolean true (1) - * - * Link requirements: - * -lcurl — required for the HTTP client (http_get, http_post, llm_*). - * -lpthread — required for the HTTP server (one detached thread per - * connection, capped at 64 concurrent). - * -loqs — optional; required only when liboqs is installed and the - * pq_* / sha3_256_hex entry points are needed. Detected at - * compile time via __has_include(). - * -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in - * pq_hybrid_* and HKDF-SHA256 derivation. - * - * Canonical compile command: - * cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ - * -o .c el-compiler/runtime/el_runtime.c - * - * With liboqs (post-quantum stack): - * cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \ - * -o .c el-compiler/runtime/el_runtime.c - */ - -#pragma once - -#include -#include - -typedef int64_t el_val_t; - -/* HTTP request-handler function-pointer types. Public because soul modules (routes/chat/etc.) - * register handlers across translation units; previously defined only inside el_runtime.c, which - * made cross-module references (and the Windows build) fail. Home in the shared header. */ -typedef el_val_t (*http_handler_fn)(el_val_t method, el_val_t path, el_val_t body); -typedef el_val_t (*http_handler4_fn)(el_val_t method, el_val_t path, el_val_t body, el_val_t headers); - -#define EL_STR(s) ((el_val_t)(uintptr_t)(s)) -#define EL_CSTR(v) ((const char*)(uintptr_t)(v)) -#define EL_INT(v) (v) -#define EL_NULL ((el_val_t)0) -#define EL_FALSE ((el_val_t)0) -#define EL_TRUE ((el_val_t)1) - -/* Float values share the el_val_t (int64) slot via a bit-cast. - * The codegen emits Float literals as `el_from_float()` so the - * underlying bits represent the IEEE 754 double. Float-aware builtins - * (math, format, json) round-trip via these helpers. */ -static inline double el_to_float(el_val_t v) { - union { int64_t i; double f; } u; - u.i = (int64_t)v; - return u.f; -} - -static inline el_val_t el_from_float(double f) { - union { double f; int64_t i; } u; - u.f = f; - return (el_val_t)u.i; -} - -#ifdef __cplusplus -extern "C" { -#endif - -/* ── I/O ──────────────────────────────────────────────────────────────────── */ - -el_val_t println(el_val_t s); -el_val_t print(el_val_t s); -el_val_t readline(void); - -/* ── String builtins ─────────────────────────────────────────────────────── */ - -el_val_t el_str_concat(el_val_t a, el_val_t b); -el_val_t str_eq(el_val_t a, el_val_t b); -el_val_t str_starts_with(el_val_t s, el_val_t prefix); -el_val_t str_ends_with(el_val_t s, el_val_t suffix); -el_val_t str_len(el_val_t s); -el_val_t str_concat(el_val_t a, el_val_t b); -el_val_t int_to_str(el_val_t n); -el_val_t str_to_int(el_val_t s); -el_val_t native_str_to_int(el_val_t s); -el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end); -el_val_t str_contains(el_val_t s, el_val_t sub); -el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to); -el_val_t str_to_upper(el_val_t s); -el_val_t str_to_lower(el_val_t s); -el_val_t str_trim(el_val_t s); - -/* ── Math ────────────────────────────────────────────────────────────────── */ - -el_val_t el_abs(el_val_t n); -el_val_t el_max(el_val_t a, el_val_t b); -el_val_t el_min(el_val_t a, el_val_t b); - -/* ── Refcount (ARC) ────────────────────────────────────────────────────────── - * Lists and Maps carry a refcount. Strings and ints do not — el_retain and - * el_release are safe no-ops on non-refcounted values (they sniff a magic - * header at offset 0 and only act if the magic matches). - * - * Codegen emits these at let-binding shadowing, function entry (params), and - * function exit (locals other than the returned value). The refcount lets - * el_list_append and el_map_set mutate in place when uniquely owned (cheap) - * and copy-on-write when shared (preserves persistent semantics across - * accumulator patterns in the compiler itself). */ - -void el_retain(el_val_t v); -void el_release(el_val_t v); - -/* ── Scoped arena (CLI use) ───────────────────────────────────────────────── */ -el_val_t el_arena_push(void); -el_val_t el_arena_pop(el_val_t mark); - -/* ── List ────────────────────────────────────────────────────────────────── */ - -el_val_t el_list_new(el_val_t count, ...); -el_val_t el_list_len(el_val_t list); -el_val_t el_list_get(el_val_t list, el_val_t index); -el_val_t el_list_append(el_val_t list, el_val_t elem); -el_val_t el_list_empty(void); -el_val_t el_list_clone(el_val_t list); - -/* ── Map ─────────────────────────────────────────────────────────────────── */ - -el_val_t el_map_new(el_val_t pair_count, ...); -el_val_t el_get_field(el_val_t map, el_val_t key); -el_val_t el_map_get(el_val_t map, el_val_t key); -el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value); - -/* ── HTTP ─────────────────────────────────────────────────────────────────── */ - -el_val_t http_get(el_val_t url); -el_val_t http_post(el_val_t url, el_val_t body); -el_val_t http_post_json(el_val_t url, el_val_t json_body); -el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map); -el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map); -el_val_t http_post_json_with_headers(el_val_t url, el_val_t headers_map, el_val_t json_body); -el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header); -el_val_t http_delete(el_val_t url); -el_val_t http_serve(el_val_t port, el_val_t handler); -el_val_t http_set_handler(el_val_t name); - -/* HTTP server v2 ───────────────────────────────────────────────────────────── - * Same dispatch model as http_serve, but the handler signature is widened: - * - * el_val_t handler(method, path, headers_map, body) - * - * `headers_map` is an ElMap from lowercased header name → header value (both - * Strings). Repeated headers are joined with ", " per RFC 7230. - * - * Response value: the handler may return either - * (a) a plain body string — same auto-content-type / 200-OK behaviour as - * http_serve (3-arg) — or - * (b) a response envelope built with `http_response(status, headers_json, - * body)`. The runtime detects the envelope discriminator - * `"el_http_response":1` at the start of the returned string and - * unpacks status / headers / body before sending. - * - * The 3-arg http_serve(port, handler) remains supported unchanged for - * existing handlers (e.g. products/web/server.el): it dispatches with - * (method, path, body), hardcodes 200 OK, and auto-detects content type. */ -el_val_t http_serve_v2(el_val_t port, el_val_t handler); -void http_serve_async(el_val_t port, el_val_t handler); -el_val_t http_set_handler_v2(el_val_t name); - -/* Build an HTTP response envelope. `headers_json` should be a JSON object - * literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The - * returned string carries the discriminator `{"el_http_response":1,...}` - * which the runtime's send-path detects and unpacks. Detection happens - * uniformly inside http_send_response, so a 3-arg handler may also return - * an envelope. The 3-arg variant remains documented as a fixed 200-OK - * auto-content-type contract for legacy handlers that return plain bodies. */ -el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body); - -/* SSE connection fd — set by http_worker_v2 before calling the El handler, - * cleared afterwards. Defined in el_seed.c; called from el_runtime.c. - * The getter is exposed as __http_conn_fd() to El programs. */ -void el_seed_set_http_conn_fd(int fd); - -/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default - * 60000ms). Read lazily on first use, so setting the env var any time before - * the first http_* call is sufficient. */ - -/* Streaming variants — write the response body straight to a file via - * libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string - * wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive - * embedded NUL bytes that would truncate a strlen()-based code path. - * - * Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same - * `headers_map` shape as http_post_with_headers (ElMap of String→String). - * - * Return value: 1 on success (file fully written), 0 on any failure - * (network, file open, partial write). On failure the output file is removed - * so callers cannot mistake a partially-written file for a valid one. */ -el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path); -el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path); - -/* ── URL encoding ────────────────────────────────────────────────────────── */ - -el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */ -el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */ - -/* ── HTML allowlist sanitizer ──────────────────────────────────────────────── - * el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML - * cleaner. State-machine parser; tag/attribute names compared case- - * insensitively against the allowlist; `` / `<… src>` URL schemes - * validated (http, https, mailto, fragment-only, or relative); whole- - * subtree drop for script / style / iframe / object / embed / form; HTML- - * escapes free text outside dropped subtrees. - * - * The allowlist is JSON of the form - * {"p":[],"a":["href","title"],"strong":[],...} - * where each value is the array of attribute names allowed for that tag. */ -el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json); -el_val_t html_raw(el_val_t s); -el_val_t html_escape(el_val_t s); - -/* ── Filesystem ──────────────────────────────────────────────────────────── */ - -el_val_t fs_read(el_val_t path); -el_val_t fs_write(el_val_t path, el_val_t content); -el_val_t fs_list(el_val_t path); -el_val_t fs_list_json(el_val_t path); -el_val_t fs_exists(el_val_t path); -el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */ - -/* Length-explicit binary write. `length` is an Int (el_val_t holding the - * byte count). The caller knows the length from context — typically because - * `bytes` came from base64_decode (which produces a magic-tagged binary - * buffer with embedded NULs possible) and the caller already tracks the - * decoded length, OR because the bytes came from a fixed-size source - * (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely. - * - * Returns 1 on success, 0 on failure (invalid path, can't open, partial - * write, negative length). On partial-write failure, the file is removed - * so callers cannot read back a truncated artefact. */ -el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length); - -/* ── JSON ────────────────────────────────────────────────────────────────── */ - -el_val_t json_get(el_val_t json, el_val_t key); -el_val_t json_parse(el_val_t s); -el_val_t json_stringify(el_val_t v); -el_val_t json_get_string(el_val_t json_str, el_val_t key); -el_val_t json_get_int(el_val_t json_str, el_val_t key); -el_val_t json_get_float(el_val_t json_str, el_val_t key); -el_val_t json_get_bool(el_val_t json_str, el_val_t key); -el_val_t json_get_raw(el_val_t json_str, el_val_t key); -el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value); -el_val_t json_array_len(el_val_t json_str); -el_val_t json_array_get(el_val_t json_str, el_val_t index); -el_val_t json_array_get_string(el_val_t json_str, el_val_t index); -el_val_t json_escape_string(el_val_t sv); -el_val_t json_build_object(el_val_t kvs); -el_val_t json_build_array(el_val_t items); - -/* ── Time ────────────────────────────────────────────────────────────────── */ - -el_val_t time_now(void); -el_val_t time_now_utc(void); -el_val_t sleep_secs(el_val_t secs); -el_val_t sleep_ms(el_val_t ms); -el_val_t time_format(el_val_t ts, el_val_t fmt); -el_val_t time_to_parts(el_val_t ts); -el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz); -el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit); -el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit); -el_val_t now_ns(void); - -/* ── Instant + Duration: first-class temporal types ────────────────────────── - * Both types share the el_val_t (int64) slot. Instants are nanoseconds - * since the Unix epoch; Durations are signed nanoseconds. Type discipline - * is enforced at codegen-time: BinOps on names registered as Instant or - * Duration route through the typed wrappers below; mismatches like - * Instant+Instant become #error at the C compiler. - * - * Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are - * recognised by the parser as DurationLit AST nodes and lowered to literal - * int64 nanoseconds at codegen time. The runtime never sees the units. */ - -el_val_t el_now_instant(void); -el_val_t now(void); -el_val_t unix_seconds(el_val_t n); -el_val_t unix_millis(el_val_t n); -el_val_t instant_from_iso8601(el_val_t s); - -el_val_t el_duration_from_nanos(el_val_t ns); -el_val_t duration_seconds(el_val_t n); -el_val_t duration_millis(el_val_t n); -el_val_t duration_nanos(el_val_t n); - -el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur); -el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur); -el_val_t el_instant_diff(el_val_t a, el_val_t b); -el_val_t el_duration_add(el_val_t a, el_val_t b); -el_val_t el_duration_sub(el_val_t a, el_val_t b); -el_val_t el_duration_scale(el_val_t dur, el_val_t scalar); -el_val_t el_duration_div(el_val_t dur, el_val_t scalar); - -el_val_t el_instant_lt(el_val_t a, el_val_t b); -el_val_t el_instant_le(el_val_t a, el_val_t b); -el_val_t el_instant_gt(el_val_t a, el_val_t b); -el_val_t el_instant_ge(el_val_t a, el_val_t b); -el_val_t el_instant_eq(el_val_t a, el_val_t b); -el_val_t el_instant_ne(el_val_t a, el_val_t b); -el_val_t el_duration_lt(el_val_t a, el_val_t b); -el_val_t el_duration_le(el_val_t a, el_val_t b); -el_val_t el_duration_gt(el_val_t a, el_val_t b); -el_val_t el_duration_ge(el_val_t a, el_val_t b); -el_val_t el_duration_eq(el_val_t a, el_val_t b); -el_val_t el_duration_ne(el_val_t a, el_val_t b); - -el_val_t instant_to_unix_seconds(el_val_t i); -el_val_t instant_to_unix_millis(el_val_t i); -el_val_t instant_to_iso8601(el_val_t i); -el_val_t duration_to_seconds(el_val_t d); -el_val_t duration_to_millis(el_val_t d); -el_val_t duration_to_nanos(el_val_t d); - -el_val_t el_sleep_duration(el_val_t dur); -el_val_t unix_timestamp(void); - -el_val_t ttl_cache_set(el_val_t key, el_val_t value); -el_val_t ttl_cache_get(el_val_t key, el_val_t max_age); -el_val_t ttl_cache_age(el_val_t key); - -/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ───────────── - * Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA - * zones, Gregorian, DST) is the user-facing default; MarsCalendar, - * CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth - * domains. - * - * A Calendar interprets an Instant under a particular cycle convention and - * produces a CalendarTime. CalendarTime carries the underlying Instant and - * a back-pointer to its Calendar; arithmetic and formatting consult the - * Calendar to convert ns since epoch into year/month/day/hour/minute/second - * (or sol/phase, or cycle/phase, depending on kind). - * - * Storage convention: Calendar / CalendarTime / Rhythm / LocalDate / - * LocalDateTime are heap-allocated structs whose pointers are cast into - * el_val_t. A 24-bit magic header at offset 0 lets the runtime identify - * the kind safely. LocalTime is small enough to live in the int64 slot - * directly (nanos since midnight, signed). */ - -/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar. - * `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed - * offset string ("+05:30", "-08:00"). The runtime resolves it via tzset() - * on first use of the owning EarthCalendar. */ -el_val_t zone(el_val_t id); -el_val_t zone_utc(void); -el_val_t zone_local(void); -el_val_t zone_offset(el_val_t hours, el_val_t minutes); - -/* Calendar constructors. Each returns an el_val_t pointer to a heap- - * allocated, magic-tagged Calendar struct. Calendars are interned by - * (kind, zone_id, period_ns, epoch_ns) so identical constructors return - * the same pointer — equality is reference equality. */ -el_val_t earth_calendar(el_val_t z); -el_val_t earth_calendar_default(void); -el_val_t mars_calendar(void); -el_val_t cycle_calendar(el_val_t period_dur); -el_val_t no_cycle_calendar(void); -el_val_t relative_calendar(el_val_t epoch_inst); - -/* CalendarTime constructors and methods. Returns a heap-allocated struct - * whose pointer fits in el_val_t. */ -el_val_t now_in(el_val_t cal); -el_val_t in_calendar(el_val_t inst, el_val_t cal); -el_val_t cal_format(el_val_t ct, el_val_t pattern); -el_val_t cal_to_instant(el_val_t ct); -el_val_t cal_cycle_phase(el_val_t ct); -el_val_t cal_in(el_val_t ct, el_val_t cal); - -/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types. - * LocalTime carries nanoseconds since midnight as a signed int64 directly - * in the el_val_t slot (no allocation). LocalDate / LocalDateTime are - * heap-allocated structs with magic headers. */ -el_val_t local_date(el_val_t y, el_val_t m, el_val_t d); -el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns); -el_val_t local_datetime(el_val_t date, el_val_t time); -el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal); - -el_val_t local_date_year(el_val_t ld); -el_val_t local_date_month(el_val_t ld); -el_val_t local_date_day(el_val_t ld); -el_val_t local_time_hour(el_val_t lt); -el_val_t local_time_minute(el_val_t lt); -el_val_t local_time_second(el_val_t lt); -el_val_t local_time_nanos(el_val_t lt); - -el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur); -el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur); -el_val_t el_local_date_lt(el_val_t a, el_val_t b); -el_val_t el_local_date_eq(el_val_t a, el_val_t b); - -/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct - * pointer in el_val_t; rhythms are immutable so callers may share them. */ -el_val_t rhythm_cycle_start(void); -el_val_t rhythm_cycle_phase(el_val_t phase); -el_val_t rhythm_duration(el_val_t d); -el_val_t rhythm_session_start(void); -el_val_t rhythm_event(el_val_t name); -el_val_t rhythm_and(el_val_t a, el_val_t b); -el_val_t rhythm_or(el_val_t a, el_val_t b); -el_val_t rhythm_weekday(el_val_t day); -el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute); -el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal); -el_val_t rhythm_matches(el_val_t r, el_val_t ct); - -/* ── UUID ────────────────────────────────────────────────────────────────── */ - -el_val_t uuid_new(void); -el_val_t uuid_v4(void); - -/* ── Environment ─────────────────────────────────────────────────────────── */ - -el_val_t env(el_val_t key); - -/* ── In-process state K/V ────────────────────────────────────────────────── */ - -el_val_t state_set(el_val_t key, el_val_t value); -el_val_t state_get(el_val_t key); -el_val_t state_del(el_val_t key); -el_val_t state_keys(void); -el_val_t state_has(el_val_t key); -el_val_t state_get_or(el_val_t key, el_val_t default_val); - -/* ── Float formatting ────────────────────────────────────────────────────── */ - -el_val_t float_to_str(el_val_t f); -el_val_t int_to_float(el_val_t n); -el_val_t float_to_int(el_val_t f); -el_val_t format_float(el_val_t f, el_val_t decimals); -el_val_t decimal_round(el_val_t f, el_val_t decimals); -el_val_t str_to_float(el_val_t s); - -/* ── Math (Float-aware) ──────────────────────────────────────────────────── */ - -el_val_t math_sqrt(el_val_t f); -el_val_t math_log(el_val_t f); -el_val_t math_ln(el_val_t f); -el_val_t math_sin(el_val_t f); -el_val_t math_cos(el_val_t f); -el_val_t math_pi(void); - -/* ── String additions ────────────────────────────────────────────────────── */ - -el_val_t str_index_of(el_val_t s, el_val_t sub); -el_val_t str_split(el_val_t s, el_val_t sep); -el_val_t str_char_at(el_val_t s, el_val_t i); -el_val_t str_char_code(el_val_t s, el_val_t i); -el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad); -el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad); -el_val_t str_format(el_val_t fmt, el_val_t data); -el_val_t str_lower(el_val_t s); -el_val_t str_upper(el_val_t s); - -/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes) - * Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex. - * is_* predicates: empty input returns false; multi-char requires ALL bytes - * to match. ASCII ranges only in Phase 1. */ - -/* Counting */ -el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */ -el_val_t str_count_chars(el_val_t s); /* codepoint count */ -el_val_t str_count_bytes(el_val_t s); /* alias of str_len */ -el_val_t str_count_lines(el_val_t s); -el_val_t str_count_words(el_val_t s); -el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */ -el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */ - -/* Find / position */ -el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */ -el_val_t str_last_index_of(el_val_t s, el_val_t sub); -el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */ - -/* Transform */ -el_val_t str_repeat(el_val_t s, el_val_t n); -el_val_t str_reverse(el_val_t s); /* by codepoint */ -el_val_t str_strip_prefix(el_val_t s, el_val_t prefix); -el_val_t str_strip_suffix(el_val_t s, el_val_t suffix); -el_val_t str_strip_chars(el_val_t s, el_val_t chars); -el_val_t str_lstrip(el_val_t s); -el_val_t str_rstrip(el_val_t s); - -/* Char classification (Bool) */ -el_val_t is_letter(el_val_t s); -el_val_t is_digit(el_val_t s); -el_val_t is_alphanumeric(el_val_t s); -el_val_t is_whitespace(el_val_t s); -el_val_t is_punctuation(el_val_t s); -el_val_t is_uppercase(el_val_t s); -el_val_t is_lowercase(el_val_t s); - -/* Split / join */ -el_val_t str_split_lines(el_val_t s); -el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */ -el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n); -el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */ - -/* ── List additions ──────────────────────────────────────────────────────── */ - -el_val_t list_push(el_val_t list, el_val_t elem); -el_val_t list_push_front(el_val_t list, el_val_t elem); -el_val_t list_join(el_val_t list, el_val_t sep); -el_val_t list_range(el_val_t start, el_val_t end); - -/* ── Bool helpers ────────────────────────────────────────────────────────── */ - -el_val_t bool_to_str(el_val_t b); - -/* ── Numeric parsing ─────────────────────────────────────────────────────── */ - -el_val_t parse_int(el_val_t s, el_val_t default_val); - -/* ── Process ─────────────────────────────────────────────────────────────── */ - -el_val_t exit_program(el_val_t code); -el_val_t getpid_now(void); - -/* Self-terminating memory guard. Reads ELC_MAX_MEM_MB (default 512) and - * exits with code 1 if resident memory exceeds the limit. Call periodically - * during long compilation loops (e.g. after each function is compiled). - * Returns 0 when memory is within bounds. */ -el_val_t el_mem_check(void); - -/* ── CGI identity ───────────────────────────────────────────────────────────── - * Called at the start of main() in CGI programs (those with a `cgi {}` block). - * Records the program's DHARMA identity before any other code executes. */ - -void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, - el_val_t network, el_val_t engram); - -/* ── DHARMA network builtins ───────────────────────────────────────────────── - * Available to CGI programs (declared with a `cgi {}` block). - * - * Peers are addressed by `dharma_id` of the form - * "@" e.g. "ntn-genesis@http://localhost:7770" - * If the @ portion is omitted, transport defaults to - * "http://localhost:7770" (the local CGI daemon assumption). - * - * Wire protocol (all peers expose): - * POST /dharma/recv { channel, from, content } → response body - * POST /dharma/event { type, payload, source, timestamp } - * POST /api/activate { query } → list of nodes - * - * Hosting application's responsibility: an El program with a `cgi {}` block - * runs http_serve() with its own request handler; that handler should route - * "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so - * incoming events feed dharma_field() queues. The runtime itself does not - * intercept any /dharma path. */ - -el_val_t dharma_connect(el_val_t cgi_id); -el_val_t dharma_send(el_val_t channel, el_val_t content); -el_val_t dharma_activate(el_val_t query); -void dharma_emit(el_val_t event_type, el_val_t payload); -el_val_t dharma_field(el_val_t event_type); -void dharma_strengthen(el_val_t cgi_id, el_val_t weight); -el_val_t dharma_relationship(el_val_t cgi_id); -el_val_t dharma_peers(void); - -/* Public C API: called by an El program's HTTP handler when a /dharma/event - * request arrives. Pushes onto the per-event-type queue and signals any - * pending dharma_field() blockers. All three arguments must be NUL-terminated - * C strings (or NULL — then treated as empty). */ -void el_runtime_dharma_event_arrive(const char* event_type, - const char* payload, - const char* source); - -/* ── Engram local graph primitives ─────────────────────────────────────────── - * Operate on the CGI's local Engram knowledge graph. - * `engram_activate` queries the local graph only; `dharma_activate` is - * network-wide across all connected CGI graphs. */ - -el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience); -el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t importance, el_val_t confidence, - el_val_t tier, el_val_t tags); -/* Layered consciousness — see el_runtime.c for the layered architecture - * design notes (search "Layered consciousness architecture"). The five - * canonical layers (safety / core-identity / domain-knowledge / imprint / - * suit) are seeded automatically; engram_add_layer extends the registry - * with imprint or suit overlays at runtime. Nodes default to layer 1 - * (core-identity) when created via engram_node / engram_node_full. */ -el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t certainty, el_val_t confidence, - el_val_t status, el_val_t tags, el_val_t layer_id); -el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible, - el_val_t transparent, el_val_t injectable); -el_val_t engram_remove_layer(el_val_t layer_id); -el_val_t engram_list_layers(void); -el_val_t engram_get_node(el_val_t id); -void engram_strengthen(el_val_t node_id); -void engram_forget(el_val_t node_id); -el_val_t engram_node_count(void); -el_val_t engram_search(el_val_t query, el_val_t limit); -el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset); -void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation); -el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id); -el_val_t engram_neighbors(el_val_t node_id); -el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction); -el_val_t engram_edge_count(void); -/* Three-pass activation: background fan-out → working-memory promotion → - * Layer 0 override. See "Three-pass activation" in el_runtime.c. */ -el_val_t engram_activate(el_val_t query, el_val_t depth); -el_val_t engram_save(el_val_t path); -el_val_t engram_load(el_val_t path); - -/* JSON-string accessors — return pre-serialized JSON so HTTP handlers - * can pass results straight through without round-tripping ElList/ElMap - * through json_stringify. */ -el_val_t engram_get_node_json(el_val_t id); -el_val_t engram_get_node_by_label(el_val_t label); -el_val_t engram_search_json(el_val_t query, el_val_t limit); -el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset); -el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset); -el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction); -el_val_t engram_activate_json(el_val_t query, el_val_t depth); -el_val_t engram_stats_json(void); -el_val_t engram_list_layers_json(void); -/* engram_compile_layered_json — produce a prompt-ready text block split - * into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire) - * and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if - * no nodes promoted to working memory. */ -el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth); - -/* ── Working memory ──────────────────────────────────────────────────────────*/ -el_val_t engram_wm_count(void); -el_val_t engram_wm_avg_weight(void); -el_val_t engram_wm_top_json(el_val_t n); -el_val_t engram_load_merge(el_val_t path); - -/* ── LLM (Anthropic API client) ───────────────────────────────────────────── - * All functions call https://api.anthropic.com/v1/messages with the API key - * from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */ - -el_val_t llm_call(el_val_t model, el_val_t prompt); -el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt); -el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools); -el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64); -el_val_t llm_models(void); - -/* Register a tool handler by name. The handler is looked up via dlsym - * (mirroring http_set_handler), so any El `fn (input)` compiles to - * a global C symbol that this function can locate at runtime. - * Handler signature: `el_val_t handler(el_val_t input_json)` — receives - * the tool input as a JSON-string el_val_t and returns a JSON-string - * el_val_t result. Used by llm_call_agentic. */ -void llm_register_tool(el_val_t name, el_val_t handler_fn_name); - -/* ── args() ───────────────────────────────────────────────────────────────── - * Provides access to command-line arguments passed to the program. - * Populated by el_runtime_init_args() before main() runs. */ - -el_val_t args(void); -void el_runtime_init_args(int argc, char** argv); - -/* ── Crypto primitives ───────────────────────────────────────────────────── - * SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe). - * Self-contained — no OpenSSL/libcrypto dependency. The implementations are - * adapted from public-domain reference code (Brad Conte / RFC 4648). - * - * Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string - * value whose contents are raw binary; callers usually feed these into - * base64_encode. Note that el_val_t strings are NUL-terminated by convention, - * so the binary payload may contain embedded NULs — pass it directly into - * base64_encode (which uses an explicit length) rather than treating it as - * a printable C string. - * - * The "base64" variants emit/accept RFC 4648 standard alphabet with padding. - * The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding, - * as used in JWTs. */ - -el_val_t sha256_hex(el_val_t input); -el_val_t sha256_bytes(el_val_t input); -el_val_t hmac_sha256_hex(el_val_t key, el_val_t message); -el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message); -el_val_t base64_encode(el_val_t input); -el_val_t base64_decode(el_val_t input); -el_val_t base64url_encode(el_val_t input); -el_val_t base64url_decode(el_val_t input); - -/* Length-aware variants (internal — exposed for the rare caller that already - * has a known-length binary buffer and doesn't want to round-trip through - * a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed - * these implicitly. */ -el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len); -el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe); - -/* ── Post-quantum primitives (liboqs-backed) ──────────────────────────────── - * All inputs/outputs hex-encoded. Algorithm choices: - * Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced) - * KEM: CRYSTALS-Kyber-768 (NIST level 3) - * Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2) - * - * If liboqs is not linked (detected via __has_include() at compile - * time), the pq_* entry points return a JSON-shaped error string so callers - * fail loudly rather than silently fall back to classical schemes: - * {"error":"liboqs not linked, post-quantum primitives unavailable"} - * - * The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and - * CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss). - * Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack, - * Kyber holds. SHA3-256 also remains usable independent of liboqs (the - * Keccak permutation is PQ-OK as a primitive). */ - -el_val_t pq_keygen_signature(void); -el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message); -el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex); - -el_val_t pq_kem_keygen(void); -el_val_t pq_kem_encaps(el_val_t public_key_hex); -el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex); - -el_val_t pq_hybrid_keygen(void); -el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined); - -el_val_t sha3_256_hex(el_val_t input); - -/* ── AEAD: AES-256-GCM (libcrypto-backed) ─────────────────────────────────── - * Symmetric authenticated encryption used to wrap envelopes after a KEM - * handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the - * Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256. - * - * aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where - * ciphertext is the AES-256-GCM output with the 16-byte auth tag appended. - * Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which - * structurally rules out the GCM nonce-reuse footgun. - * - * aead_decrypt returns the plaintext String, or "" on any failure (including - * auth-tag mismatch). Callers MUST check for "" before trusting the result. */ -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext); -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex); - -/* ── Native VM builtin aliases (for compiled El source) ───────────────────── - * These match the El VM's native_* builtins so that El source compiled - * to C can call the same names without modification. */ - -el_val_t native_list_get(el_val_t list, el_val_t index); -el_val_t native_list_len(el_val_t list); -el_val_t native_list_append(el_val_t list, el_val_t elem); -el_val_t native_list_empty(void); -el_val_t native_list_clone(el_val_t list); -el_val_t native_string_chars(el_val_t s); -el_val_t native_int_to_str(el_val_t n); - -/* ── Method-call shorthand aliases ────────────────────────────────────────── - * The El method-call convention `obj.method(args)` compiles to - * `method(obj, args)`. These aliases expose the runtime functions under - * the short names that result from method calls in El source. - * - * Example: `myList.append(x)` → `append(myList, x)` (calls this alias) - * `myList.len()` → `len(myList)` (calls this alias) */ - -el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */ -el_val_t len(el_val_t list); /* el_list_len */ -el_val_t get(el_val_t list, el_val_t index); /* el_list_get */ -el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */ -el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */ - -/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */ -/* See bottom of el_runtime.c for the implementation. - * Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION. - * No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */ -/* ── Subprocess execution ────────────────────────────────────────────────── */ -el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */ -el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */ -el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */ -el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */ - -/* ── Stdout redirection (used by compiler JS pipeline) ───────────────────── */ -el_val_t stdout_to_file(el_val_t path); /* redirect process stdout to a file */ -el_val_t stdout_restore(void); /* restore process stdout to terminal */ - -el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json); -el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json); -el_val_t trace_span_start(el_val_t name); -el_val_t trace_span_end(el_val_t span_handle); -el_val_t emit_event(el_val_t name, el_val_t duration_ms); - -el_val_t __thread_create(el_val_t fn_name_v, el_val_t arg_v); -el_val_t __thread_join(el_val_t tid_v); - -/* ── __ prefixed aliases (self-hosting compiler ABI) ───────────────────────── - * The El self-hosting compiler emits calls to __-prefixed names. These are - * forwarding wrappers around the existing el_runtime functions above. */ - -/* I/O */ -el_val_t __println(el_val_t s); -el_val_t __print(el_val_t s); -el_val_t __readline(void); - -/* String */ -el_val_t __int_to_str(el_val_t n); -el_val_t __str_to_int(el_val_t s); -el_val_t __float_to_str(el_val_t f); -el_val_t __str_to_float(el_val_t s); -el_val_t __str_len(el_val_t s); -el_val_t __str_char_at(el_val_t s, el_val_t i); -el_val_t __str_cmp(el_val_t a, el_val_t b); -el_val_t __str_ncmp(el_val_t a, el_val_t b, el_val_t n); -el_val_t __str_concat_raw(el_val_t a, el_val_t b); -el_val_t __str_slice_raw(el_val_t s, el_val_t start, el_val_t end); -el_val_t __str_alloc(el_val_t n); -el_val_t __str_set_char(el_val_t s, el_val_t i, el_val_t c); - -/* URL encoding */ -el_val_t __url_encode(el_val_t s); -el_val_t __url_decode(el_val_t s); - -/* Environment */ -el_val_t __env_get(el_val_t key); - -/* Subprocess */ -el_val_t __exec(el_val_t cmd); -el_val_t __exec_bg(el_val_t cmd); - -/* Process */ -el_val_t __exit_program(el_val_t code); - -/* Filesystem */ -el_val_t __fs_exists(el_val_t path); -el_val_t __fs_mkdir(el_val_t path); -el_val_t __fs_read(el_val_t path); -el_val_t __fs_write(el_val_t path, el_val_t content); -el_val_t __fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t n); -el_val_t __fs_list_raw(el_val_t path); - -/* HTTP server */ -el_val_t __http_response(el_val_t status, el_val_t headers_json, el_val_t body); -el_val_t __http_serve(el_val_t port, el_val_t handler); -el_val_t __http_serve_v2(el_val_t port, el_val_t handler); - -/* HTTP conn fd / SSE (weak; overridden by el_seed.c when linked together) */ -el_val_t __http_conn_fd(void); -el_val_t __http_sse_open(el_val_t conn_id); -el_val_t __http_sse_send(el_val_t conn_id, el_val_t data); -el_val_t __http_sse_close(el_val_t conn_id); - -/* HTTP client (requires HAVE_CURL; stubs provided for no-curl builds) */ -el_val_t __http_do(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_map, el_val_t timeout_ms); -el_val_t __http_do_map(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_json, el_val_t timeout_ms); -el_val_t __http_do_map_to_file(el_val_t method, el_val_t url, el_val_t body, - el_val_t headers_json, el_val_t output_path); - -/* JSON */ -el_val_t __json_array_get(el_val_t json, el_val_t index); -el_val_t __json_array_get_string(el_val_t json, el_val_t index); -el_val_t __json_array_len(el_val_t json); -el_val_t __json_get(el_val_t json, el_val_t key); -el_val_t __json_get_raw(el_val_t json, el_val_t key); -el_val_t __json_set(el_val_t json, el_val_t key, el_val_t value); -el_val_t __json_parse_map(el_val_t json_str); -el_val_t __json_stringify_val(el_val_t val); - -/* Hashing */ -el_val_t __sha256_hex(el_val_t s); - -/* State K/V */ -el_val_t __state_del(el_val_t key); -el_val_t __state_get(el_val_t key); -el_val_t __state_keys(void); -el_val_t __state_set(el_val_t key, el_val_t val); - -/* UUID */ -el_val_t __uuid_v4(void); - -/* Args */ -el_val_t __args_json(void); - -#ifdef __cplusplus -} -#endif diff --git a/lang/el-compiler/runtime/legacy/el_runtime.c b/lang/el-compiler/runtime/legacy/el_runtime.c deleted file mode 100644 index 08d51fb..0000000 --- a/lang/el-compiler/runtime/legacy/el_runtime.c +++ /dev/null @@ -1,10258 +0,0 @@ -/* - * el_runtime.c — El language C runtime implementation - * - * All functions use el_val_t (= int64_t) as the universal value type. - * Strings are transported as their pointer address cast to int64_t. - * On any 64-bit system sizeof(pointer) <= sizeof(int64_t), so this is safe. - * - * Compile with: - * cc -std=c11 -I -lcurl -lpthread -o .c el_runtime.c - * - * Link requirements: -lcurl (HTTP client + LLM), -lpthread (HTTP server). - */ - -/* Feature-test macros must be set before any standard headers. _GNU_SOURCE - * exposes clock_gettime/CLOCK_REALTIME, strcasecmp, and the dlfcn extensions - * (RTLD_DEFAULT) — all of which macOS hands us without asking but glibc on - * Debian gates behind an explicit opt-in. */ -#ifndef _GNU_SOURCE -#define _GNU_SOURCE -#endif - -#include "el_runtime.h" - -#include -#include /* strcasecmp */ -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include /* dlsym for http_set_handler fallback */ -#include -#include -#include -#include -#include -#include - -/* ── Internal allocators ─────────────────────────────────────────────────── */ - -/* - * Per-request string arena - * - * Every El string allocated via el_strbuf / el_strdup during an HTTP request - * is registered in a thread-local arena. When el_request_end() is called at - * the end of the worker thread, every arena entry is freed — recovering all - * the intermediate strings from el_str_concat chains (build_system_prompt, - * engram_compile, etc.) that are otherwise leaked forever. - * - * Long-lived allocations (state_set values, engram internal storage) call - * el_strdup_persist() / el_strbuf_persist() which bypass the arena entirely. - */ - -#define EL_ARENA_INITIAL 512 - -typedef struct { - char** ptrs; - size_t count; - size_t cap; -} ElArena; - -static _Thread_local ElArena _tl_arena = {NULL, 0, 0}; -static _Thread_local int _tl_arena_active = 0; - -/* Binary-safe fs_read length — set by fs_read, consumed by http_send_response. - * Allows serving PNGs and other binary files without strlen truncation. */ -static _Thread_local size_t _tl_fs_read_len = 0; - -static void el_arena_track(char* p) { - if (!_tl_arena_active || !p) return; - if (_tl_arena.count >= _tl_arena.cap) { - size_t nc = _tl_arena.cap == 0 ? EL_ARENA_INITIAL : _tl_arena.cap * 2; - char** grown = realloc(_tl_arena.ptrs, nc * sizeof(char*)); - if (!grown) return; /* can't track — will leak this one ptr, but don't crash */ - _tl_arena.ptrs = grown; - _tl_arena.cap = nc; - } - _tl_arena.ptrs[_tl_arena.count++] = p; -} - -/* Called by http_worker before dispatching the El handler. */ -void el_request_start(void) { - _tl_arena.count = 0; - _tl_arena_active = 1; -} - -/* Called by http_worker after the El handler returns and the response is sent. - * Frees every intermediate string allocated during the request. */ -void el_request_end(void) { - _tl_arena_active = 0; - for (size_t i = 0; i < _tl_arena.count; i++) { - free(_tl_arena.ptrs[i]); - } - _tl_arena.count = 0; -} - -/* Persistent allocation — bypasses the arena (state_set, engram internals). */ -static char* el_strdup_persist(const char* s) { - if (!s) return strdup(""); - return strdup(s); -} -static char* el_strbuf_persist(size_t n) { - char* p = malloc(n + 1); - if (!p) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - p[0] = '\0'; - return p; -} - -static char* el_strdup(const char* s) { - if (!s) { char* p = strdup(""); el_arena_track(p); return p; } - char* p = strdup(s); - el_arena_track(p); - return p; -} - -static char* el_strbuf(size_t n) { - char* p = malloc(n + 1); - if (!p) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - p[0] = '\0'; - el_arena_track(p); - return p; -} - -/* Wrap an allocated C string as el_val_t */ -static el_val_t el_wrap_str(char* s) { - return EL_STR(s); -} - -/* ── I/O ──────────────────────────────────────────────────────────────────── */ - -void println(el_val_t s) { - const char* str = EL_CSTR(s); - if (str) puts(str); - else puts(""); -} - -void print(el_val_t s) { - const char* str = EL_CSTR(s); - if (str) fputs(str, stdout); -} - -el_val_t readline(void) { - char buf[4096]; - if (!fgets(buf, sizeof(buf), stdin)) return el_wrap_str(el_strdup("")); - size_t len = strlen(buf); - if (len > 0 && buf[len - 1] == '\n') buf[len - 1] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -/* ── String builtins ─────────────────────────────────────────────────────── */ - -el_val_t el_str_concat(el_val_t av, el_val_t bv) { - const char* a = EL_CSTR(av); - const char* b = EL_CSTR(bv); - if (!a) a = ""; - if (!b) b = ""; - size_t la = strlen(a); - size_t lb = strlen(b); - char* out = el_strbuf(la + lb); - memcpy(out, a, la); - memcpy(out + la, b, lb); - out[la + lb] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_eq(el_val_t av, el_val_t bv) { - const char* a = EL_CSTR(av); - const char* b = EL_CSTR(bv); - if (!a || !b) return (el_val_t)(a == b); - return (el_val_t)(strcmp(a, b) == 0); -} - -el_val_t str_starts_with(el_val_t sv, el_val_t prefv) { - const char* s = EL_CSTR(sv); - const char* prefix = EL_CSTR(prefv); - if (!s || !prefix) return 0; - size_t lp = strlen(prefix); - return (el_val_t)(strncmp(s, prefix, lp) == 0); -} - -el_val_t str_ends_with(el_val_t sv, el_val_t sufv) { - const char* s = EL_CSTR(sv); - const char* suffix = EL_CSTR(sufv); - if (!s || !suffix) return 0; - size_t ls = strlen(s); - size_t lsuf = strlen(suffix); - if (lsuf > ls) return 0; - return (el_val_t)(strcmp(s + ls - lsuf, suffix) == 0); -} - -el_val_t str_len(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - return (el_val_t)strlen(s); -} - -el_val_t str_concat(el_val_t a, el_val_t b) { - return el_str_concat(a, b); -} - -el_val_t int_to_str(el_val_t n) { - char buf[32]; - snprintf(buf, sizeof(buf), "%lld", (long long)n); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t str_to_int(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - return (el_val_t)atoll(s); -} - -el_val_t str_slice(el_val_t sv, el_val_t start, el_val_t end) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - int64_t len = (int64_t)strlen(s); - if (start < 0) start = 0; - if (end > len) end = len; - if (start >= end) return el_wrap_str(el_strdup("")); - int64_t sz = end - start; - char* out = el_strbuf((size_t)sz); - memcpy(out, s + start, (size_t)sz); - out[sz] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_contains(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub) return 0; - return (el_val_t)(strstr(s, sub) != NULL); -} - -el_val_t str_replace(el_val_t sv, el_val_t fromv, el_val_t tov) { - const char* s = EL_CSTR(sv); - const char* from = EL_CSTR(fromv); - const char* to = EL_CSTR(tov); - if (!s || !from || !to) return el_wrap_str(el_strdup(s ? s : "")); - size_t ls = strlen(s); - size_t lf = strlen(from); - size_t lt = strlen(to); - if (lf == 0) return el_wrap_str(el_strdup(s)); - size_t count = 0; - const char* p = s; - while ((p = strstr(p, from)) != NULL) { count++; p += lf; } - size_t out_sz = ls + count * lt + 1; - char* out = el_strbuf(out_sz); - char* dst = out; - p = s; - const char* found; - while ((found = strstr(p, from)) != NULL) { - size_t chunk = (size_t)(found - p); - memcpy(dst, p, chunk); dst += chunk; - memcpy(dst, to, lt); dst += lt; - p = found + lf; - } - strcpy(dst, p); - return el_wrap_str(out); -} - -el_val_t str_to_upper(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - for (size_t i = 0; i < n; i++) out[i] = (char)toupper((unsigned char)s[i]); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_to_lower(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - for (size_t i = 0; i < n; i++) out[i] = (char)tolower((unsigned char)s[i]); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_trim(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - while (*s && isspace((unsigned char)*s)) s++; - size_t n = strlen(s); - while (n > 0 && isspace((unsigned char)s[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, s, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -/* ── Math ────────────────────────────────────────────────────────────────── */ - -el_val_t el_abs(el_val_t n) { return n < 0 ? -n : n; } -el_val_t el_max(el_val_t a, el_val_t b) { return a > b ? a : b; } -el_val_t el_min(el_val_t a, el_val_t b) { return a < b ? a : b; } - -/* ── Refcounted heap objects ────────────────────────────────────────────────── - * - * ElList and ElMap carry a magic-tagged header at offset 0: - * { uint32_t magic; uint32_t refcount; ... payload ... } - * - * The magic tag distinguishes refcounted objects from raw C strings (whose - * first byte is printable ASCII < 0x80) and from small integers (which can't - * be dereferenced). el_retain / el_release sniff the magic and act only on - * matching values; everything else is a safe no-op. - * - * Both ElList and ElMap use INDIRECTION: the header is fixed-size and never - * moves. The payload arrays (elems, keys, values) live in separate heap - * allocations, so realloc-grow on append never invalidates the caller's - * pointer to the header. This is what lets us mutate-in-place safely when - * the refcount is 1 and copy-on-write when it's higher. - * - * Memory model in practice: - * Single-owner accumulator (the cg_stmts pattern) — refcount stays at 1, - * appends amortize to O(1), total memory O(N) for an N-element list. - * Multi-owner branching (the cg_if_stmt pattern) — refcount > 1, each - * append on a shared list copies, so the original is preserved for the - * else-branch. Persistent semantics where they're needed; mutation where - * they're not. */ - -#define EL_MAGIC_LIST 0xE15710A1u /* >= 0x80 in MSB so 'looks_like_string' rejects */ -#define EL_MAGIC_MAP 0xE19A704Bu - -typedef struct { - uint32_t magic; - uint32_t refcount; -} ElHeader; - -/* ── List ────────────────────────────────────────────────────────────────── */ - -typedef struct { - ElHeader hdr; - int64_t length; - int64_t capacity; - el_val_t* elems; -} ElList; - -static ElList* list_alloc(int64_t cap) { - if (cap < 4) cap = 4; - ElList* lst = malloc(sizeof(ElList)); - if (!lst) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - lst->hdr.magic = EL_MAGIC_LIST; - lst->hdr.refcount = 1; - lst->length = 0; - lst->capacity = cap; - lst->elems = malloc((size_t)cap * sizeof(el_val_t)); - if (!lst->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - return lst; -} - -el_val_t el_list_empty(void) { - return EL_STR(list_alloc(4)); -} - -el_val_t el_list_new(el_val_t count, ...) { - ElList* lst = list_alloc(count > 0 ? count : 4); - va_list ap; - va_start(ap, count); - for (int64_t i = 0; i < count; i++) { - lst->elems[i] = va_arg(ap, el_val_t); - } - va_end(ap); - lst->length = count; - return EL_STR(lst); -} - -el_val_t el_list_len(el_val_t listv) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) return 0; - return lst->length; -} - -el_val_t el_list_get(el_val_t listv, el_val_t index) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) return 0; - if (index < 0 || index >= lst->length) return 0; - return lst->elems[index]; -} - -el_val_t el_list_append(el_val_t listv, el_val_t elem) { - ElList* old = (ElList*)(uintptr_t)listv; - if (!old) { - ElList* fresh = list_alloc(4); - fresh->elems[0] = elem; - fresh->length = 1; - return EL_STR(fresh); - } - - /* Uniquely owned: grow the elems buffer in place. The header pointer the - * caller holds doesn't move (we only realloc the inner array). This is - * the common case in compiler accumulators, and it's amortized O(1). */ - if (old->hdr.refcount <= 1) { - if (old->length >= old->capacity) { - int64_t new_cap = old->capacity > 0 ? old->capacity * 2 : 4; - el_val_t* grown = realloc(old->elems, (size_t)new_cap * sizeof(el_val_t)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - old->elems = grown; - old->capacity = new_cap; - } - old->elems[old->length++] = elem; - return listv; - } - - /* Shared: copy-on-write. The original is preserved for its other owners. */ - int64_t new_cap = old->length + 1; - if (new_cap < 4) new_cap = 4; - ElList* fresh = malloc(sizeof(ElList)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_LIST; - fresh->hdr.refcount = 1; - fresh->length = old->length + 1; - fresh->capacity = new_cap; - fresh->elems = malloc((size_t)new_cap * sizeof(el_val_t)); - if (!fresh->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (old->length > 0) { - memcpy(fresh->elems, old->elems, (size_t)old->length * sizeof(el_val_t)); - } - fresh->elems[old->length] = elem; - return EL_STR(fresh); -} - -el_val_t el_list_clone(el_val_t listv) { - /* Shallow copy: the new ElList owns its own header and elems buffer, but - * the elements themselves are shared (which is what callers want for the - * cg_if_stmt 'declared' pattern — cloning the spine, not its contents). - * Used by codegen at scope branch points where two child scopes need to - * see the same starting set of declared names without each other's - * mutations. */ - ElList* old = (ElList*)(uintptr_t)listv; - if (!old) return el_list_empty(); - int64_t cap = old->capacity > 0 ? old->capacity : 4; - if (cap < old->length) cap = old->length; - if (cap < 4) cap = 4; - ElList* fresh = malloc(sizeof(ElList)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_LIST; - fresh->hdr.refcount = 1; - fresh->length = old->length; - fresh->capacity = cap; - fresh->elems = malloc((size_t)cap * sizeof(el_val_t)); - if (!fresh->elems) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (old->length > 0) { - memcpy(fresh->elems, old->elems, (size_t)old->length * sizeof(el_val_t)); - } - return EL_STR(fresh); -} - -/* ── Map ─────────────────────────────────────────────────────────────────── */ - -typedef struct { - ElHeader hdr; - int64_t count; - int64_t capacity; - el_val_t* keys; - el_val_t* values; -} ElMap; - -static ElMap* map_alloc(int64_t cap) { - if (cap < 4) cap = 4; - ElMap* m = malloc(sizeof(ElMap)); - if (!m) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - m->hdr.magic = EL_MAGIC_MAP; - m->hdr.refcount = 1; - m->count = 0; - m->capacity = cap; - m->keys = malloc((size_t)cap * sizeof(el_val_t)); - m->values = malloc((size_t)cap * sizeof(el_val_t)); - if (!m->keys || !m->values) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - return m; -} - -el_val_t el_map_new(el_val_t pair_count, ...) { - ElMap* m = map_alloc(pair_count > 0 ? pair_count : 4); - va_list ap; - va_start(ap, pair_count); - for (int64_t i = 0; i < pair_count; i++) { - m->keys[i] = va_arg(ap, el_val_t); - m->values[i] = va_arg(ap, el_val_t); - } - va_end(ap); - m->count = pair_count; - return EL_STR(m); -} - -static ElMap* as_map(el_val_t v) { return (ElMap*)(uintptr_t)v; } - -el_val_t el_map_get(el_val_t mapv, el_val_t keyv) { - ElMap* m = as_map(mapv); - const char* key = EL_CSTR(keyv); - if (!m || !key) return 0; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - if (k && strcmp(k, key) == 0) return m->values[i]; - } - return 0; -} - -el_val_t el_get_field(el_val_t mapv, el_val_t keyv) { - return el_map_get(mapv, keyv); -} - -/* Internal: in-place set on a uniquely-owned map. */ -static el_val_t map_set_in_place(ElMap* m, el_val_t keyv, el_val_t value) { - const char* key = EL_CSTR(keyv); - if (key) { - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - if (k && strcmp(k, key) == 0) { m->values[i] = value; return EL_STR(m); } - } - } - if (m->count >= m->capacity) { - int64_t new_cap = m->capacity > 0 ? m->capacity * 2 : 4; - el_val_t* gk = realloc(m->keys, (size_t)new_cap * sizeof(el_val_t)); - el_val_t* gv = realloc(m->values, (size_t)new_cap * sizeof(el_val_t)); - if (!gk || !gv) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - m->keys = gk; - m->values = gv; - m->capacity = new_cap; - } - m->keys[m->count] = keyv; - m->values[m->count] = value; - m->count++; - return EL_STR(m); -} - -el_val_t el_map_set(el_val_t mapv, el_val_t keyv, el_val_t value) { - ElMap* m = as_map(mapv); - if (!m) return 0; - if (m->hdr.refcount <= 1) { - return map_set_in_place(m, keyv, value); - } - /* Shared: copy then set. The original is preserved for its other owners. */ - int64_t new_cap = m->count + 1; - if (new_cap < 4) new_cap = 4; - ElMap* fresh = malloc(sizeof(ElMap)); - if (!fresh) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - fresh->hdr.magic = EL_MAGIC_MAP; - fresh->hdr.refcount = 1; - fresh->count = m->count; - fresh->capacity = new_cap; - fresh->keys = malloc((size_t)new_cap * sizeof(el_val_t)); - fresh->values = malloc((size_t)new_cap * sizeof(el_val_t)); - if (!fresh->keys || !fresh->values) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - if (m->count > 0) { - memcpy(fresh->keys, m->keys, (size_t)m->count * sizeof(el_val_t)); - memcpy(fresh->values, m->values, (size_t)m->count * sizeof(el_val_t)); - } - return map_set_in_place(fresh, keyv, value); -} - -/* ── Refcount ops ─────────────────────────────────────────────────────────── */ -/* - * Both retain and release sniff the magic header to decide whether a value - * is a refcounted heap object. For small integers, raw C strings, and any - * value whose magic word doesn't match, both functions are no-ops. This lets - * codegen emit them on every let-binding without having to track types. - * - * Safety: we filter out obvious non-pointers (small magnitudes, misaligned - * addresses) before dereferencing. For any value that passes the filter and - * lives in a mapped page, reading the first 4 bytes is safe — strings start - * with printable ASCII (< 0x80), so their magic word will never collide with - * EL_MAGIC_LIST (0xE1...) or EL_MAGIC_MAP (0xE1...). Random integers that - * happen to look like aligned heap pointers are exceedingly unlikely to land - * on a page whose first 4 bytes match either magic. */ - -static int looks_like_heap_obj(el_val_t v) { - if (v == 0) return 0; - int64_t s = (int64_t)v; - if (s > -0x10000 && s < 0x10000) return 0; /* small ints */ - uintptr_t p = (uintptr_t)v; - if (p < 0x10000) return 0; /* low addresses */ - if (p & 0x7) return 0; /* malloc returns 8-aligned */ - return 1; -} - -void el_retain(el_val_t v) { - if (!looks_like_heap_obj(v)) return; - ElHeader* h = (ElHeader*)(uintptr_t)v; - if (h->magic == EL_MAGIC_LIST || h->magic == EL_MAGIC_MAP) { - h->refcount++; - } -} - -void el_release(el_val_t v) { - if (!looks_like_heap_obj(v)) return; - ElHeader* h = (ElHeader*)(uintptr_t)v; - if (h->magic == EL_MAGIC_LIST) { - if (h->refcount > 0 && --h->refcount == 0) { - ElList* l = (ElList*)h; - free(l->elems); - l->hdr.magic = 0; /* poison so use-after-free is detected */ - free(l); - } - } else if (h->magic == EL_MAGIC_MAP) { - if (h->refcount > 0 && --h->refcount == 0) { - ElMap* m = (ElMap*)h; - free(m->keys); - free(m->values); - m->hdr.magic = 0; - free(m); - } - } -} - -/* ── Batch 2/3 forward decls (defined later in JSON section) ────────────── */ - -typedef struct JsonBuf JsonBuf; -typedef struct JsonParser JsonParser; -static void jb_init(JsonBuf* b); -static void jb_putc(JsonBuf* b, char c); -static void jb_puts(JsonBuf* b, const char* s); -static void jb_emit_escaped(JsonBuf* b, const char* s); -static int looks_like_string(el_val_t v); -static const char* json_find_key(const char* s, const char* key); -static const char* json_skip_value(const char* p); -static char* jp_parse_string_raw(JsonParser* jp); - -/* Struct definitions are visible here because batch 2/3 helpers above use - * them by value; the bodies (jb_init, etc.) appear in the JSON section. */ -struct JsonBuf { - char* buf; - size_t len; - size_t cap; -}; - -struct JsonParser { - const char* p; - const char* end; - int err; -}; - -/* ── Batch 2: Real HTTP (libcurl client + POSIX-socket server) ───────────── */ -/* - * Client: blocking libcurl easy-handle calls. Errors are returned as a JSON - * fragment {"error":"..."} so callers can detect via str_starts_with("{") / - * json_get_string("error", ...). - * - * Server: bind/listen/accept loop on a TCP socket. Each accepted connection - * is handled in its own pthread (detached). A semaphore-style counter caps - * concurrent in-flight connections at HTTP_MAX_CONNS (64). When the cap is - * reached, accept() blocks until a worker exits. This prevents runaway - * thread creation under high load. - * - * Handler dispatch: El does not expose first-class function references at - * the runtime layer, so the second argument to http_serve(port, handler) is - * treated as a string name (or any el_val_t — the runtime ignores its - * value and uses the registry). Callers register a C-level handler via - * - * extern void el_runtime_register_handler(const char* name, - * el_val_t (*fn)(el_val_t, - * el_val_t, - * el_val_t)); - * - * and select the active handler by calling http_set_handler("name") from - * El, or by setting it directly through the C registry. If no handler is - * registered, the server replies with a 200 carrying a default message so - * the loop is observable. - */ - -/* ── HTTP client write-callback buffer ───────────────────────────────────── */ - -typedef struct { - char* data; - size_t len; - size_t cap; -} HttpBuf; - -static void httpbuf_init(HttpBuf* b) { - b->cap = 1024; - b->len = 0; - b->data = malloc(b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->data[0] = '\0'; -} - -static void httpbuf_append(HttpBuf* b, const void* src, size_t n) { - if (b->len + n + 1 > b->cap) { - while (b->len + n + 1 > b->cap) b->cap *= 2; - b->data = realloc(b->data, b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } - memcpy(b->data + b->len, src, n); - b->len += n; - b->data[b->len] = '\0'; -} - -static size_t http_write_cb(char* ptr, size_t size, size_t nmemb, void* ud) { - size_t n = size * nmemb; - httpbuf_append((HttpBuf*)ud, ptr, n); - return n; -} - -/* JSON-escape an arbitrary C string into an allocated buffer. */ -static char* json_escape_alloc(const char* s) { - if (!s) return el_strdup(""); - JsonBuf b; jb_init(&b); - for (const char* p = s; *p; p++) { - unsigned char c = (unsigned char)*p; - switch (c) { - case '"': jb_puts(&b, "\\\""); break; - case '\\': jb_puts(&b, "\\\\"); break; - case '\n': jb_puts(&b, "\\n"); break; - case '\r': jb_puts(&b, "\\r"); break; - case '\t': jb_puts(&b, "\\t"); break; - default: - if (c < 0x20) { - char tmp[8]; snprintf(tmp, sizeof(tmp), "\\u%04x", c); - jb_puts(&b, tmp); - } else jb_putc(&b, (char)c); - } - } - return b.buf; -} - -static el_val_t http_error_json(const char* msg) { - char* esc = json_escape_alloc(msg ? msg : "unknown error"); - char* buf = el_strbuf(strlen(esc) + 16); - sprintf(buf, "{\"error\":\"%s\"}", esc); - free(esc); - return el_wrap_str(buf); -} - -/* HTTP timeout (ms) — read once from EL_HTTP_TIMEOUT_MS, default 60000. - * Applied via CURLOPT_TIMEOUT_MS on every libcurl request. */ -static long _el_http_timeout_ms = -1; -static long el_http_timeout_ms(void) { - long v = __atomic_load_n(&_el_http_timeout_ms, __ATOMIC_ACQUIRE); - if (v >= 0) return v; - const char* s = getenv("EL_HTTP_TIMEOUT_MS"); - long parsed = 60000L; - if (s && *s) { - char* end = NULL; - long n = strtol(s, &end, 10); - if (end != s && n > 0) parsed = n; - } - __atomic_store_n(&_el_http_timeout_ms, parsed, __ATOMIC_RELEASE); - return parsed; -} - -/* Internal: do a libcurl request; takes optional body/headers, optional method override. */ -static el_val_t http_do(const char* method, const char* url, const char* body, - struct curl_slist* extra_headers) { - if (!url || !*url) return http_error_json("empty url"); - CURL* c = curl_easy_init(); - if (!c) return http_error_json("curl_easy_init failed"); - HttpBuf rb; httpbuf_init(&rb); - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - if (extra_headers) curl_easy_setopt(c, CURLOPT_HTTPHEADER, extra_headers); - if (method && strcmp(method, "POST") == 0) { - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - } else if (method && strcmp(method, "DELETE") == 0) { - curl_easy_setopt(c, CURLOPT_CUSTOMREQUEST, "DELETE"); - } - CURLcode rc = curl_easy_perform(c); - curl_easy_cleanup(c); - if (rc != CURLE_OK) { - free(rb.data); - const char* m = errbuf[0] ? errbuf : curl_easy_strerror(rc); - return http_error_json(m); - } - return el_wrap_str(rb.data); -} - -el_val_t http_get(el_val_t url) { - return http_do("GET", EL_CSTR(url), NULL, NULL); -} - -el_val_t http_post(el_val_t url, el_val_t body) { - return http_do("POST", EL_CSTR(url), EL_CSTR(body), NULL); -} - -el_val_t http_post_json(el_val_t url, el_val_t json_body) { - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(json_body), h); - curl_slist_free_all(h); - return r; -} - -/* Build a curl_slist from an ElMap of name -> value strings. */ -static struct curl_slist* headers_from_map(el_val_t headers_map) { - struct curl_slist* h = NULL; - ElMap* m = as_map(headers_map); - if (!m) return NULL; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - const char* v = EL_CSTR(m->values[i]); - if (!k || !v) continue; - size_t n = strlen(k) + strlen(v) + 4; - char* line = malloc(n); - if (!line) continue; - snprintf(line, n, "%s: %s", k, v); - h = curl_slist_append(h, line); - free(line); - } - return h; -} - -el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do("GET", EL_CSTR(url), NULL, h); - if (h) curl_slist_free_all(h); - return r; -} - -el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(body), h); - if (h) curl_slist_free_all(h); - return r; -} - -el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header) { - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/x-www-form-urlencoded"); - const char* a = EL_CSTR(auth_header); - if (a && *a) { - size_t n = strlen(a) + 32; - char* line = malloc(n); - snprintf(line, n, "Authorization: %s", a); - h = curl_slist_append(h, line); - free(line); - } - el_val_t r = http_do("POST", EL_CSTR(url), EL_CSTR(form_body), h); - curl_slist_free_all(h); - return r; -} - -/* HTTP DELETE — mirrors http_post but with CURLOPT_CUSTOMREQUEST=DELETE. - * Returns response body on success; on transport failure returns an error - * JSON fragment (same convention as http_get/http_post). Callers that - * expect "" on failure should check for a leading '{' and an "error" key. */ -el_val_t http_delete(el_val_t url) { - return http_do("DELETE", EL_CSTR(url), NULL, NULL); -} - -/* ── HTTP → file streaming ──────────────────────────────────────────────── - * - * Why this exists: el_val_t strings are NUL-terminated by convention, so - * accumulating an HTTP response into an httpbuf and then wrapping its - * `.data` pointer with el_wrap_str() loses the byte length. Any consumer - * that does strlen() on the wrapped pointer truncates the body at the - * first embedded NUL. Audio (MP3, WAV, OGG), images (PNG, JPEG), and any - * other binary payload hits this. The vessels that download such bodies - * (e.g. ElevenLabs TTS → MP3) get silently corrupted files. - * - * The fix: wire libcurl's CURLOPT_WRITEFUNCTION directly to fwrite() - * against a fopen()-ed FILE*. The bytes never pass through an el_val_t - * string, so embedded NULs are preserved verbatim. Caller's contract is - * just "a file at this path with the response body in it". */ - -static size_t http_file_write_cb(char* ptr, size_t size, size_t nmemb, void* ud) { - FILE* f = (FILE*)ud; - return fwrite(ptr, size, nmemb, f); -} - -/* Internal: stream body to file. method is "GET" or "POST". body may be NULL - * (GET) or NUL-terminated (POST). headers may be NULL. Returns 1/0. */ -static el_val_t http_do_to_file(const char* method, const char* url, - const char* body, struct curl_slist* extra_headers, - const char* output_path) { - if (!url || !*url) return 0; - if (!output_path || !*output_path) return 0; - FILE* f = fopen(output_path, "wb"); - if (!f) return 0; - - CURL* c = curl_easy_init(); - if (!c) { fclose(f); remove(output_path); return 0; } - - char errbuf[CURL_ERROR_SIZE]; errbuf[0] = '\0'; - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_file_write_cb); - curl_easy_setopt(c, CURLOPT_WRITEDATA, f); - curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); - curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); - curl_easy_setopt(c, CURLOPT_FAILONERROR, 1L); /* 4xx/5xx → CURLE_HTTP_RETURNED_ERROR */ - if (extra_headers) curl_easy_setopt(c, CURLOPT_HTTPHEADER, extra_headers); - - if (method && strcmp(method, "POST") == 0) { - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body ? body : ""); - /* For the request body we still rely on strlen — POST bodies are - * caller-controlled and JSON/text in every known El use case. - * If a future caller needs a binary POST body, add a *_bytes - * variant that takes an explicit length, mirroring fs_write_bytes. */ - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); - } - - CURLcode rc = curl_easy_perform(c); - curl_easy_cleanup(c); - - /* Flush + close before signalling success, so the file is fully on disk - * by the time the caller reads back. */ - int flush_ok = (fflush(f) == 0); - int close_ok = (fclose(f) == 0); - - if (rc != CURLE_OK || !flush_ok || !close_ok) { - remove(output_path); - return 0; - } - return 1; -} - -el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do_to_file("GET", EL_CSTR(url), NULL, h, EL_CSTR(output_path)); - if (h) curl_slist_free_all(h); - return r; -} - -el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path) { - struct curl_slist* h = headers_from_map(headers_map); - el_val_t r = http_do_to_file("POST", EL_CSTR(url), EL_CSTR(body), h, EL_CSTR(output_path)); - if (h) curl_slist_free_all(h); - return r; -} - -/* ── HTTP server (POSIX sockets + pthreads) ──────────────────────────────── */ - -#define HTTP_MAX_CONNS 64 - -typedef el_val_t (*http_handler_fn)(el_val_t method, el_val_t path, el_val_t body); - -typedef struct { - char* name; - http_handler_fn fn; -} HttpHandlerEntry; - -static HttpHandlerEntry _http_handlers[32]; -static size_t _http_handler_count = 0; -static char* _http_active_handler = NULL; -static pthread_mutex_t _http_handler_mu = PTHREAD_MUTEX_INITIALIZER; - -static pthread_mutex_t _http_conn_mu = PTHREAD_MUTEX_INITIALIZER; -static pthread_cond_t _http_conn_cv = PTHREAD_COND_INITIALIZER; -static int _http_conn_active = 0; - -/* Public C-level API: register a handler by name. Programs that want El - * `http_serve` to dispatch into their handler call this from main() before - * http_serve. Not declared in the header to keep the public API minimal — - * extern lookup works since C symbols are global. */ -void el_runtime_register_handler(const char* name, http_handler_fn fn); -void el_runtime_register_handler(const char* name, http_handler_fn fn) { - if (!name || !fn) return; - pthread_mutex_lock(&_http_handler_mu); - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, name) == 0) { - _http_handlers[i].fn = fn; - pthread_mutex_unlock(&_http_handler_mu); - return; - } - } - if (_http_handler_count < sizeof(_http_handlers) / sizeof(_http_handlers[0])) { - _http_handlers[_http_handler_count].name = el_strdup(name); - _http_handlers[_http_handler_count].fn = fn; - _http_handler_count++; - } - pthread_mutex_unlock(&_http_handler_mu); -} - -void http_set_handler(el_val_t name) { - const char* n = EL_CSTR(name); - pthread_mutex_lock(&_http_handler_mu); - free(_http_active_handler); - _http_active_handler = el_strdup(n ? n : ""); - /* If the name is not yet in the registry, try dlsym lookup against - * the running binary's symbol table. Every El `fn name(...)` compiles - * to a global C symbol with that exact name, so El programs can self- - * register their own handlers just by calling http_set_handler("name"). */ - if (n && *n) { - int found = 0; - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, n) == 0) { found = 1; break; } - } - if (!found) { - void* sym = dlsym(RTLD_DEFAULT, n); - if (sym && _http_handler_count < sizeof(_http_handlers) / sizeof(_http_handlers[0])) { - _http_handlers[_http_handler_count].name = el_strdup(n); - _http_handlers[_http_handler_count].fn = (http_handler_fn)sym; - _http_handler_count++; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); -} - -static http_handler_fn http_lookup_active(void) { - http_handler_fn out = NULL; - pthread_mutex_lock(&_http_handler_mu); - if (_http_active_handler) { - for (size_t i = 0; i < _http_handler_count; i++) { - if (strcmp(_http_handlers[i].name, _http_active_handler) == 0) { - out = _http_handlers[i].fn; break; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return out; -} - -/* Auto-detect Content-Type from response body. */ -static const char* http_detect_content_type(const char* body) { - if (!body) return "text/plain; charset=utf-8"; - const char* p = body; - /* Binary magic bytes — check before stripping whitespace */ - if ((unsigned char)p[0] == 0x89 && p[1]=='P' && p[2]=='N' && p[3]=='G') - return "image/png"; - if ((unsigned char)p[0] == 0xFF && (unsigned char)p[1] == 0xD8) - return "image/jpeg"; - if (strncmp(p, "GIF8", 4) == 0) return "image/gif"; - if (strncmp(p, "RIFF", 4) == 0) return "image/webp"; - if (strncmp(p, "wOFF", 4) == 0) return "font/woff"; - if (strncmp(p, "wOF2", 4) == 0) return "font/woff2"; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (strncasecmp(p, "= cap) { - if (cap >= 1024 * 1024) { free(buf); return -1; } - cap *= 2; - buf = realloc(buf, cap); - if (!buf) return -1; - } - ssize_t n = recv(fd, buf + len, cap - len - 1, 0); - if (n <= 0) { free(buf); return -1; } - len += (size_t)n; - buf[len] = '\0'; - if (strstr(buf, "\r\n\r\n")) break; - } - /* Parse request line */ - char* sp1 = strchr(buf, ' '); - if (!sp1) { free(buf); return -1; } - *sp1 = '\0'; - *out_method = el_strdup(buf); - char* path_start = sp1 + 1; - char* sp2 = strchr(path_start, ' '); - if (!sp2) { free(*out_method); *out_method = NULL; free(buf); return -1; } - *sp2 = '\0'; - *out_path = el_strdup(path_start); - char* hdr_end = strstr(sp2 + 1, "\r\n\r\n"); - /* Capture the raw header block (after the request line's CRLF, up to - * but not including the terminating \r\n\r\n) for callers that asked - * for it. The legacy 3-arg path passes NULL and skips this. */ - if (out_headers_block) { - char* hdr_start = strstr(sp2 + 1, "\r\n"); - if (hdr_start && hdr_start < hdr_end) { - hdr_start += 2; - size_t hb_len = (size_t)(hdr_end - hdr_start); - char* hb = malloc(hb_len + 1); - if (hb) { - memcpy(hb, hdr_start, hb_len); - hb[hb_len] = '\0'; - *out_headers_block = hb; - } - } else { - *out_headers_block = el_strdup(""); - } - } - /* Find Content-Length */ - long content_length = 0; - char* hp = sp2 + 1; - while (hp < hdr_end) { - char* line_end = strstr(hp, "\r\n"); - /* line_end == hdr_end means we're on the LAST header line — its - * trailing \r\n is the same \r\n that begins the \r\n\r\n header - * terminator. Process this line; only stop when line_end is past - * hdr_end (which means the parser walked off the end of the - * header block). The previous condition (line_end >= hdr_end) - * silently dropped any Content-Length that appeared as the last - * header — exactly what real curl/clients tend to emit. */ - if (!line_end || line_end > hdr_end) break; - if (strncasecmp(hp, "Content-Length:", 15) == 0) { - content_length = strtol(hp + 15, NULL, 10); - if (content_length < 0) content_length = 0; - if (content_length > 64 * 1024 * 1024) content_length = 64 * 1024 * 1024; - } - hp = line_end + 2; - } - /* Body: any bytes already read past hdr_end, plus more recv */ - char* body_start = hdr_end + 4; - size_t body_have = (buf + len) - body_start; - char* body = malloc((size_t)content_length + 1); - if (!body) { free(*out_method); free(*out_path); *out_method=NULL; *out_path=NULL; free(buf); return -1; } - if ((long)body_have > content_length) body_have = (size_t)content_length; - if (body_have > 0) memcpy(body, body_start, body_have); - while ((long)body_have < content_length) { - ssize_t n = recv(fd, body + body_have, (size_t)content_length - body_have, 0); - if (n <= 0) break; - body_have += (size_t)n; - } - body[body_have] = '\0'; - *out_body = body; - free(buf); - return 0; -} - -/* Reason phrase for common HTTP statuses. Falls back to "Status" for the - * long tail — clients only care about the numeric code. */ -static const char* http_reason_phrase(int status) { - switch (status) { - case 200: return "OK"; - case 201: return "Created"; - case 202: return "Accepted"; - case 204: return "No Content"; - case 301: return "Moved Permanently"; - case 302: return "Found"; - case 303: return "See Other"; - case 304: return "Not Modified"; - case 307: return "Temporary Redirect"; - case 308: return "Permanent Redirect"; - case 400: return "Bad Request"; - case 401: return "Unauthorized"; - case 403: return "Forbidden"; - case 404: return "Not Found"; - case 405: return "Method Not Allowed"; - case 409: return "Conflict"; - case 410: return "Gone"; - case 422: return "Unprocessable Entity"; - case 429: return "Too Many Requests"; - case 500: return "Internal Server Error"; - case 501: return "Not Implemented"; - case 502: return "Bad Gateway"; - case 503: return "Service Unavailable"; - case 504: return "Gateway Timeout"; - default: return "Status"; - } -} - -/* Best-effort send with retry on partial writes. */ -static int http_send_all(int fd, const char* p, size_t left) { - while (left > 0) { - ssize_t w = send(fd, p, left, 0); - if (w <= 0) return -1; - p += w; left -= (size_t)w; - } - return 0; -} - -/* Discriminator that http_response() embeds at the start of its envelope. - * A handler returning a string starting with this exact prefix is treated - * as a structured response; anything else is treated as a raw body. */ -#define EL_HTTP_RESPONSE_TAG "{\"el_http_response\":1" - -/* Keys that conflict with runtime-managed headers are silently dropped to - * avoid double-emission — the runtime always emits its own Content-Length - * and Connection: close. Content-Type from the envelope IS allowed and - * overrides auto-detection. */ -static int http_header_is_managed(const char* k) { - return strcasecmp(k, "Content-Length") == 0 - || strcasecmp(k, "Connection") == 0; -} - -/* Walk an ElMap of header pairs and emit each as `K: V\r\n` into JsonBuf b. - * Sets *out_saw_content_type to 1 if the map contained an explicit - * Content-Type so the caller can skip auto-detection. */ -static void http_emit_headers_from_map(JsonBuf* b, el_val_t headers_map, - int* out_saw_content_type) { - *out_saw_content_type = 0; - if (headers_map == 0) return; - ElMap* m = (ElMap*)(uintptr_t)headers_map; - if (!m || m->hdr.magic != EL_MAGIC_MAP) return; - for (int64_t i = 0; i < m->count; i++) { - const char* k = EL_CSTR(m->keys[i]); - const char* v = EL_CSTR(m->values[i]); - if (!k || !v) continue; - if (http_header_is_managed(k)) continue; - if (strcasecmp(k, "Content-Type") == 0) *out_saw_content_type = 1; - jb_puts(b, k); - jb_puts(b, ": "); - jb_puts(b, v); - jb_puts(b, "\r\n"); - } -} - -/* Parse the envelope produced by http_response(). On success returns 1 and - * populates *out_status, *out_headers_map (an ElMap el_val_t — caller must - * el_release), and *out_body (allocated). On failure returns 0. - * - * Implementation: feeds the entire envelope through the recursive-descent - * JSON parser (which builds proper ElMap/ElList values), then pulls the - * three top-level fields by name. Avoids re-stringifying the headers map - * since json_stringify() does not support nested objects. */ -static int http_parse_envelope(const char* s, int* out_status, - el_val_t* out_headers_map, char** out_body, - el_val_t* out_parsed_root) { - if (!s) return 0; - if (strncmp(s, EL_HTTP_RESPONSE_TAG, - sizeof(EL_HTTP_RESPONSE_TAG) - 1) != 0) return 0; - - el_val_t parsed = json_parse(EL_STR(s)); - if (parsed == EL_NULL) return 0; - - int status = 200; - el_val_t hmap = 0; - char* body = NULL; - - el_val_t sv = el_map_get(parsed, EL_STR("status")); - if (sv != 0) { - /* status comes back as an integer — el_val_t holds it directly. */ - long sc = (long)sv; - if (sc >= 100 && sc <= 599) status = (int)sc; - } - - el_val_t hv = el_map_get(parsed, EL_STR("headers")); - if (hv != 0) { - ElMap* hm = (ElMap*)(uintptr_t)hv; - if (hm && hm->hdr.magic == EL_MAGIC_MAP) hmap = hv; - } - - el_val_t bv = el_map_get(parsed, EL_STR("body")); - if (bv != 0) { - const char* bs = EL_CSTR(bv); - if (bs) body = el_strdup(bs); - } - if (!body) body = el_strdup(""); - - *out_status = status; - *out_headers_map = hmap; - *out_body = body; - *out_parsed_root = parsed; /* caller releases to free hmap + entries */ - return 1; -} - -/* Lightweight `__status__` envelope: if the body's first key is `__status__` - * and its value is a numeric literal, lift the status to the HTTP layer and - * strip the marker from the body before sending. This is the common case for - * El handlers that want to return 4xx/5xx without going through - * http_response() — they just prepend `{"__status__":,...}` to the JSON - * they were already returning. - * - * We deliberately recognise ONLY the first-key form so the contract is cheap - * to detect and unambiguous: `{"__status__":401,"error":"unauthorized"}` is - * an envelope, but `{"error":"...","__status__":401}` is not. Product code - * controls placement. - * - * On success returns 1 with *out_status set and *out_body_alloc populated - * with a freshly malloc'd body (caller frees). On failure returns 0 and - * leaves outputs untouched. */ -static int http_parse_status_envelope(const char* s, int* out_status, - char** out_body_alloc) { - if (!s) return 0; - const char* p = s; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '{') return 0; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - static const char marker[] = "\"__status__\""; - size_t mlen = sizeof(marker) - 1; - if (strncmp(p, marker, mlen) != 0) return 0; - p += mlen; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != ':') return 0; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p < '0' || *p > '9') return 0; /* non-numeric -> not an envelope */ - int status = 0; - while (*p >= '0' && *p <= '9') { - status = status * 10 + (*p - '0'); - p++; - } - if (status < 100 || status > 599) return 0; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - /* Two trailing shapes accepted: - * ,"k":v,...} -> body becomes {"k":v,...} - * } -> body becomes {} - * Anything else (e.g. `:` re-appearing, garbage) drops the envelope so - * we don't strip what we shouldn't. */ - if (*p == '}') { - *out_status = status; - *out_body_alloc = el_strdup("{}"); - return 1; - } - if (*p != ',') return 0; - p++; /* skip the comma; the rest of the object follows */ - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - /* Build the trimmed body: '{' + remainder. */ - size_t rest_len = strlen(p); - char* out = (char*)malloc(rest_len + 2); - if (!out) return 0; - out[0] = '{'; - memcpy(out + 1, p, rest_len); - out[rest_len + 1] = '\0'; - *out_status = status; - *out_body_alloc = out; - return 1; -} - -/* Send a fully-built HTTP response. If `body` starts with the envelope tag, - * unpack status/headers/body. Otherwise emit the historical 200-OK with - * auto-detected Content-Type. */ -/* Thread-local flag: if 1, http_send_response writes status + headers but - * NO body (HEAD method behaviour). Set by http_worker before calling - * http_send_response, cleared after. */ -static __thread int _tl_http_head_only = 0; - -static void http_send_response(int fd, const char* body) { - if (!body) body = ""; - - int status = 200; - el_val_t env_headers_map = 0; - char* env_body = NULL; - el_val_t env_parsed_root = 0; - int is_envelope = http_parse_envelope(body, &status, - &env_headers_map, &env_body, - &env_parsed_root); - - /* If the rich http_response() envelope didn't claim this body, try the - * lightweight `__status__` form. This second envelope is malloc-backed so - * we route it through env_body and let the existing cleanup path free it - * — same lifetime contract, no special case at the bottom of the - * function. */ - if (!is_envelope) { - char* trimmed = NULL; - if (http_parse_status_envelope(body, &status, &trimmed)) { - env_body = trimmed; - is_envelope = 1; - } - } - - const char* eff_body = is_envelope ? env_body : body; - /* Use the real byte count from fs_read if available (handles binary files - * with embedded null bytes — PNG, WOFF2, etc.). Fall back to strlen for - * normal text/JSON responses where _tl_fs_read_len is 0. */ - size_t blen = (_tl_fs_read_len > 0) ? _tl_fs_read_len : strlen(eff_body); - _tl_fs_read_len = 0; /* consume — one-shot per response */ - int head_only = _tl_http_head_only; - - JsonBuf hdrs; jb_init(&hdrs); - int saw_content_type = 0; - if (is_envelope) { - http_emit_headers_from_map(&hdrs, env_headers_map, - &saw_content_type); - } - if (!saw_content_type) { - jb_puts(&hdrs, "Content-Type: "); - jb_puts(&hdrs, http_detect_content_type(eff_body)); - jb_puts(&hdrs, "\r\n"); - } - - char status_line[64]; - int sl = snprintf(status_line, sizeof(status_line), - "HTTP/1.1 %d %s\r\n", - status, http_reason_phrase(status)); - if (sl < 0) { - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); free(hdrs.buf); return; - } - - char tail[128]; - int tl = snprintf(tail, sizeof(tail), - "Content-Length: %zu\r\n" - "Connection: close\r\n" - "\r\n", blen); - if (tl < 0) { - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); free(hdrs.buf); return; - } - - if (http_send_all(fd, status_line, (size_t)sl) == 0 - && http_send_all(fd, hdrs.buf, hdrs.len) == 0 - && http_send_all(fd, tail, (size_t)tl) == 0 - && (head_only - /* HEAD requests echo headers + Content-Length but no body. */ - ? 1 - : http_send_all(fd, eff_body, blen) == 0)) { - /* sent successfully */ - } - - if (env_parsed_root) el_release(env_parsed_root); - free(env_body); - free(hdrs.buf); -} - -typedef struct { - int fd; -} HttpWorkerArg; - -static void* http_worker(void* arg) { - HttpWorkerArg* a = (HttpWorkerArg*)arg; - int fd = a->fd; - free(a); - char *method = NULL, *path = NULL, *body = NULL; - if (http_read_request(fd, &method, &path, &body, NULL) == 0) { - http_handler_fn h = http_lookup_active(); - char* response = NULL; - /* HEAD: dispatch as GET so existing handlers respond with the same - * body, but flag the response writer to emit headers only. RFC 9110 - * requires HEAD to mirror GET headers + Content-Length without body. */ - int head_only = (method && strcmp(method, "HEAD") == 0); - const char* dispatch_method = head_only ? "GET" : method; - el_request_start(); /* begin per-request arena */ - if (h) { - el_val_t r = h(EL_STR(dispatch_method), EL_STR(path), EL_STR(body)); - const char* rs = EL_CSTR(r); - /* Copy response out BEFORE arena teardown. - * For binary files, _tl_fs_read_len holds the real byte count — - * use memcpy instead of strdup so null bytes are preserved. */ - size_t rlen = _tl_fs_read_len > 0 ? _tl_fs_read_len : (rs ? strlen(rs) : 0); - response = malloc(rlen + 1); - if (response && rs) { memcpy(response, rs, rlen); response[rlen] = '\0'; } - else if (response) { response[0] = '\0'; } - } else { - response = el_strdup_persist("el-runtime: no http handler registered"); - } - el_request_end(); /* free all intermediate strings */ - _tl_http_head_only = head_only; - http_send_response(fd, response); - _tl_http_head_only = 0; - free(response); - } - free(method); free(path); free(body); - close(fd); - /* release a slot */ - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - return NULL; -} - -void http_serve(el_val_t port, el_val_t handler) { - /* If `handler` looks like a string name, register it as the active handler. */ - const char* hname = EL_CSTR(handler); - if (hname && looks_like_string(handler)) { - http_set_handler(handler); - } - int p = (int)port; - if (p <= 0 || p > 65535) { fprintf(stderr, "http_serve: invalid port %d\n", p); return; } - /* Dual-stack: AF_INET6 with IPV6_V6ONLY=0 accepts both IPv4 and IPv6. - * This makes `localhost` work in browsers that resolve it to ::1 first. */ - int sock = socket(AF_INET6, SOCK_STREAM, 0); - if (sock < 0) { perror("socket"); return; } - int yes = 1; int no = 0; - setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, &yes, sizeof(yes)); - setsockopt(sock, IPPROTO_IPV6, IPV6_V6ONLY, &no, sizeof(no)); - struct sockaddr_in6 addr; - memset(&addr, 0, sizeof(addr)); - addr.sin6_family = AF_INET6; - addr.sin6_addr = in6addr_any; - addr.sin6_port = htons((uint16_t)p); - if (bind(sock, (struct sockaddr*)&addr, sizeof(addr)) < 0) { - perror("bind"); close(sock); return; - } - if (listen(sock, 64) < 0) { perror("listen"); close(sock); return; } - fprintf(stderr, "[http] listening on [::]:%d (dual-stack)\n", p); - while (1) { - struct sockaddr_in6 cli; - socklen_t clen = sizeof(cli); - int cfd = accept(sock, (struct sockaddr*)&cli, &clen); - if (cfd < 0) { - if (errno == EINTR) continue; - perror("accept"); break; - } - pthread_mutex_lock(&_http_conn_mu); - while (_http_conn_active >= HTTP_MAX_CONNS) { - pthread_cond_wait(&_http_conn_cv, &_http_conn_mu); - } - _http_conn_active++; - pthread_mutex_unlock(&_http_conn_mu); - HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); - if (!arg) { close(cfd); continue; } - arg->fd = cfd; - pthread_t tid; - if (pthread_create(&tid, NULL, http_worker, arg) != 0) { - close(cfd); free(arg); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - continue; - } - pthread_detach(tid); - } - close(sock); -} - -/* ── HTTP server v2 — request headers + structured response ──────────────── */ -/* - * v2 widens the handler signature from - * (method, path, body) -> body_string - * to - * (method, path, headers_map, body) -> body_string_or_envelope - * - * The response envelope is detected uniformly inside http_send_response — so - * 4-arg handlers can return either a plain body or http_response(...). The - * 3-arg path stays untouched in spirit (its handlers still build plain - * bodies; the envelope tag, being `{"el_http_response":1`, will never - * collide with normal JSON the legacy server.el routes return). - * - * Registry is parallel to the 3-arg handler registry: separate name table, - * separate active-handler slot, separate dlsym fallback. Mixing v1 and v2 - * handlers in the same process is fine — they don't share the active slot. */ - -typedef el_val_t (*http_handler4_fn)(el_val_t method, el_val_t path, - el_val_t headers_map, el_val_t body); - -typedef struct { - char* name; - http_handler4_fn fn; -} HttpHandler4Entry; - -static HttpHandler4Entry _http_handlers4[32]; -static size_t _http_handler4_count = 0; -static char* _http_active_handler4 = NULL; - -void el_runtime_register_handler_v2(const char* name, http_handler4_fn fn); -void el_runtime_register_handler_v2(const char* name, http_handler4_fn fn) { - if (!name || !fn) return; - pthread_mutex_lock(&_http_handler_mu); - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, name) == 0) { - _http_handlers4[i].fn = fn; - pthread_mutex_unlock(&_http_handler_mu); - return; - } - } - if (_http_handler4_count < - sizeof(_http_handlers4) / sizeof(_http_handlers4[0])) { - _http_handlers4[_http_handler4_count].name = el_strdup(name); - _http_handlers4[_http_handler4_count].fn = fn; - _http_handler4_count++; - } - pthread_mutex_unlock(&_http_handler_mu); -} - -void http_set_handler_v2(el_val_t name) { - const char* n = EL_CSTR(name); - pthread_mutex_lock(&_http_handler_mu); - free(_http_active_handler4); - _http_active_handler4 = el_strdup(n ? n : ""); - if (n && *n) { - int found = 0; - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, n) == 0) { found = 1; break; } - } - if (!found) { - void* sym = dlsym(RTLD_DEFAULT, n); - if (sym && _http_handler4_count < - sizeof(_http_handlers4) / sizeof(_http_handlers4[0])) { - _http_handlers4[_http_handler4_count].name = el_strdup(n); - _http_handlers4[_http_handler4_count].fn = - (http_handler4_fn)sym; - _http_handler4_count++; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); -} - -static http_handler4_fn http_lookup_active_v2(void) { - http_handler4_fn out = NULL; - pthread_mutex_lock(&_http_handler_mu); - if (_http_active_handler4) { - for (size_t i = 0; i < _http_handler4_count; i++) { - if (strcmp(_http_handlers4[i].name, - _http_active_handler4) == 0) { - out = _http_handlers4[i].fn; break; - } - } - } - pthread_mutex_unlock(&_http_handler_mu); - return out; -} - -/* Build an ElMap from the raw header block produced by http_read_request. - * Keys are lowercased (RFC 7230 — case-insensitive); values have leading - * whitespace trimmed. Repeated headers with the same name are joined with - * ", " in arrival order, matching standard library behaviour elsewhere. */ -static el_val_t http_build_headers_map(const char* hdr_block) { - el_val_t m = el_map_new(0); - if (!hdr_block || !*hdr_block) return m; - const char* p = hdr_block; - while (*p) { - const char* line_end = strstr(p, "\r\n"); - const char* end = line_end ? line_end : p + strlen(p); - const char* colon = NULL; - for (const char* c = p; c < end; c++) { - if (*c == ':') { colon = c; break; } - } - if (colon && colon > p) { - size_t klen = (size_t)(colon - p); - char* key = malloc(klen + 1); - if (key) { - for (size_t i = 0; i < klen; i++) { - unsigned char ch = (unsigned char)p[i]; - key[i] = (char)tolower(ch); - } - key[klen] = '\0'; - const char* vstart = colon + 1; - while (vstart < end && (*vstart == ' ' || *vstart == '\t')) vstart++; - size_t vlen = (size_t)(end - vstart); - /* Strip trailing OWS just in case. */ - while (vlen > 0 - && (vstart[vlen - 1] == ' ' - || vstart[vlen - 1] == '\t')) vlen--; - /* Coalesce repeats: if key already present, append ", value". */ - el_val_t existing = el_map_get(m, EL_STR(key)); - if (existing != 0 && looks_like_string(existing)) { - const char* old = EL_CSTR(existing); - size_t olen = strlen(old); - char* combined = malloc(olen + 2 + vlen + 1); - if (combined) { - memcpy(combined, old, olen); - memcpy(combined + olen, ", ", 2); - memcpy(combined + olen + 2, vstart, vlen); - combined[olen + 2 + vlen] = '\0'; - m = el_map_set(m, EL_STR(key), EL_STR(combined)); - } - free(key); - } else { - char* val = malloc(vlen + 1); - if (val) { - memcpy(val, vstart, vlen); - val[vlen] = '\0'; - m = el_map_set(m, EL_STR(key), EL_STR(val)); - } else { - free(key); - } - } - } - } - if (!line_end) break; - p = line_end + 2; - } - return m; -} - -static void* http_worker_v2(void* arg) { - HttpWorkerArg* a = (HttpWorkerArg*)arg; - int fd = a->fd; - free(a); - int is_sse = 0; - char *method = NULL, *path = NULL, *body = NULL, *hdr_block = NULL; - if (http_read_request(fd, &method, &path, &body, &hdr_block) == 0) { - http_handler4_fn h = http_lookup_active_v2(); - char* response = NULL; - int head_only = (method && strcmp(method, "HEAD") == 0); - const char* dispatch_method = head_only ? "GET" : method; - el_request_start(); /* begin per-request arena */ - /* Expose the raw fd to El SSE builtins (__http_conn_fd etc.). */ - el_seed_set_http_conn_fd(fd); - if (h) { - el_val_t hmap = http_build_headers_map(hdr_block ? hdr_block : ""); - el_val_t r = h(EL_STR(dispatch_method), EL_STR(path), hmap, EL_STR(body)); - const char* rs = EL_CSTR(r); - /* Detect SSE sentinel — handler took ownership of the fd. */ - if (rs && strcmp(rs, "__sse__") == 0) { - is_sse = 1; - } else { - size_t rlen = _tl_fs_read_len > 0 ? _tl_fs_read_len : (rs ? strlen(rs) : 0); - response = malloc(rlen + 1); - if (response && rs) { memcpy(response, rs, rlen); response[rlen] = '\0'; } - else if (response) { response[0] = '\0'; } - } - el_release(hmap); - } else { - response = el_strdup_persist( - "el-runtime: no v2 http handler registered " - "(call http_set_handler_v2)"); - } - el_seed_set_http_conn_fd(-1); /* clear before arena teardown */ - el_request_end(); /* free all intermediate strings */ - if (!is_sse) { - _tl_http_head_only = head_only; - http_send_response(fd, response); - _tl_http_head_only = 0; - free(response); - } - } - free(method); free(path); free(body); free(hdr_block); - /* SSE handlers close the fd themselves via __http_sse_close. */ - if (!is_sse) close(fd); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - return NULL; -} - -void http_serve_v2(el_val_t port, el_val_t handler) { - const char* hname = EL_CSTR(handler); - if (hname && looks_like_string(handler)) { - http_set_handler_v2(handler); - } - int p = (int)port; - if (p <= 0 || p > 65535) { - fprintf(stderr, "http_serve_v2: invalid port %d\n", p); - return; - } - /* Dual-stack: same as http_serve - AF_INET6 + IPV6_V6ONLY=0. */ - int sock = socket(AF_INET6, SOCK_STREAM, 0); - if (sock < 0) { perror("socket"); return; } - int yes = 1; int no = 0; - setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, &yes, sizeof(yes)); - setsockopt(sock, IPPROTO_IPV6, IPV6_V6ONLY, &no, sizeof(no)); - struct sockaddr_in6 addr; - memset(&addr, 0, sizeof(addr)); - addr.sin6_family = AF_INET6; - addr.sin6_addr = in6addr_any; - addr.sin6_port = htons((uint16_t)p); - if (bind(sock, (struct sockaddr*)&addr, sizeof(addr)) < 0) { - perror("bind"); close(sock); return; - } - if (listen(sock, 64) < 0) { perror("listen"); close(sock); return; } - fprintf(stderr, "[http v2] listening on [::]:%d (dual-stack)\n", p); - while (1) { - struct sockaddr_in6 cli; - socklen_t clen = sizeof(cli); - int cfd = accept(sock, (struct sockaddr*)&cli, &clen); - if (cfd < 0) { - if (errno == EINTR) continue; - perror("accept"); break; - } - pthread_mutex_lock(&_http_conn_mu); - while (_http_conn_active >= HTTP_MAX_CONNS) { - pthread_cond_wait(&_http_conn_cv, &_http_conn_mu); - } - _http_conn_active++; - pthread_mutex_unlock(&_http_conn_mu); - HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); - if (!arg) { close(cfd); continue; } - arg->fd = cfd; - pthread_t tid; - if (pthread_create(&tid, NULL, http_worker_v2, arg) != 0) { - close(cfd); free(arg); - pthread_mutex_lock(&_http_conn_mu); - _http_conn_active--; - pthread_cond_signal(&_http_conn_cv); - pthread_mutex_unlock(&_http_conn_mu); - continue; - } - pthread_detach(tid); - } - close(sock); -} - -/* Build the response envelope a 4-arg handler can return. We hand-write - * the JSON so the discriminator key always lands first — the runtime's - * http_parse_envelope() detects it via prefix match. headers_json must be - * either "" (empty), "{}" (empty object), or a well-formed JSON object - * literal; anything else will produce a malformed envelope and the runtime - * will treat the whole string as a plain body (no envelope detected). */ -el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body) { - long sc = (long)status; - if (sc < 100 || sc > 599) sc = 200; - const char* hj = EL_CSTR(headers_json); - if (!hj || !*hj) hj = "{}"; - /* Light validation: must start with '{' and end with '}'. */ - size_t hlen = strlen(hj); - int hj_ok = (hlen >= 2 && hj[0] == '{' && hj[hlen - 1] == '}'); - if (!hj_ok) hj = "{}"; - const char* b = EL_CSTR(body); - if (!b) b = ""; - - JsonBuf out; jb_init(&out); - jb_puts(&out, EL_HTTP_RESPONSE_TAG); /* {"el_http_response":1 */ - jb_puts(&out, ",\"status\":"); - char num[32]; - snprintf(num, sizeof(num), "%ld", sc); - jb_puts(&out, num); - jb_puts(&out, ",\"headers\":"); - jb_puts(&out, hj); - jb_puts(&out, ",\"body\":"); - jb_emit_escaped(&out, b); - jb_putc(&out, '}'); - return el_wrap_str(out.buf); -} - -/* ── Filesystem ──────────────────────────────────────────────────────────── */ - -el_val_t fs_read(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - _tl_fs_read_len = 0; - if (!path) return el_wrap_str(el_strdup("")); - FILE* f = fopen(path, "rb"); - if (!f) return el_wrap_str(el_strdup("")); - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz < 0) { fclose(f); return el_wrap_str(el_strdup("")); } /* pipe/special file */ - char* buf = el_strbuf((size_t)sz); - size_t got = fread(buf, 1, (size_t)sz, f); - buf[got] = '\0'; - _tl_fs_read_len = got; /* store real byte count for binary-safe send */ - fclose(f); - return el_wrap_str(buf); -} - -el_val_t fs_write(el_val_t pathv, el_val_t contentv) { - const char* path = EL_CSTR(pathv); - const char* content = EL_CSTR(contentv); - if (!path || !content) return 0; - FILE* f = fopen(path, "wb"); - if (!f) return 0; - size_t n = strlen(content); - size_t written = fwrite(content, 1, n, f); - fclose(f); - return written == n ? 1 : 0; -} - -/* fs_write_bytes — explicit-length binary write. Bypasses strlen so embedded - * NULs survive. Caller must know the byte count (e.g. from base64_decode, - * or the fixed 32-byte sha256_bytes/hmac_sha256_bytes outputs). - * - * If `length` is negative, treats as failure. If `length` is 0, creates an - * empty file (still useful as a "touch with content" primitive). */ -el_val_t fs_write_bytes(el_val_t pathv, el_val_t bytesv, el_val_t lengthv) { - const char* path = EL_CSTR(pathv); - const char* bytes = EL_CSTR(bytesv); - int64_t n = (int64_t)lengthv; - if (!path || !bytes) return 0; - if (n < 0) return 0; - FILE* f = fopen(path, "wb"); - if (!f) return 0; - size_t written = (n > 0) ? fwrite(bytes, 1, (size_t)n, f) : 0; - int flush_ok = (fflush(f) == 0); - int close_ok = (fclose(f) == 0); - if (!flush_ok || !close_ok || written != (size_t)n) { - remove(path); - return 0; - } - return 1; -} - -// exec_command — run a shell command, return exit code (0 = success). -// Used by elb and other El tooling to invoke subprocesses. -el_val_t exec_command(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd) return (el_val_t)(int64_t)-1; - int ret = system(cmd); - return (el_val_t)(int64_t)ret; -} - -// exec_capture — run a shell command, capture stdout, return as String. -// Returns "" on failure. -el_val_t exec_capture(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd) return el_wrap_str(el_strdup("")); - FILE* f = popen(cmd, "r"); - if (!f) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - char buf[4096]; - while (fgets(buf, sizeof(buf), f)) jb_puts(&b, buf); - pclose(f); - return el_wrap_str(b.buf); -} - -// exec — run a shell command via /bin/sh, capture stdout, return as String. -// Times out after 30 seconds. Returns "" on any error. -// El name: exec(cmd) -> String -el_val_t exec(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd || !*cmd) return el_wrap_str(el_strdup("")); - /* Build a time-limited command: wrap with timeout(1) if available, - * otherwise rely on the 30s read loop guard below. We use the simple - * popen approach with a deadline measured by wall clock so the caller - * is never blocked indefinitely. */ - FILE* f = popen(cmd, "r"); - if (!f) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - char buf[4096]; - /* 30-second wall-clock deadline */ - time_t deadline = time(NULL) + 30; - while (time(NULL) < deadline) { - if (fgets(buf, sizeof(buf), f) == NULL) break; - jb_puts(&b, buf); - } - pclose(f); - return el_wrap_str(b.buf); -} - -// exec_bg — run a shell command in background, return PID as String. -// The child process runs independently; the caller is not blocked. -// Returns "" on fork failure. -// El name: exec_bg(cmd) -> String -el_val_t exec_bg(el_val_t cmdv) { - const char* cmd = EL_CSTR(cmdv); - if (!cmd || !*cmd) return el_wrap_str(el_strdup("")); - pid_t pid = fork(); - if (pid < 0) { - /* fork failed */ - return el_wrap_str(el_strdup("")); - } - if (pid == 0) { - /* child: detach from parent's stdio, exec via shell */ - setsid(); - int devnull = open("/dev/null", O_RDWR); - if (devnull >= 0) { - dup2(devnull, STDIN_FILENO); - dup2(devnull, STDOUT_FILENO); - dup2(devnull, STDERR_FILENO); - close(devnull); - } - execl("/bin/sh", "sh", "-c", cmd, (char*)NULL); - _exit(127); - } - /* parent: convert pid to string and return immediately */ - char pidbuf[32]; - snprintf(pidbuf, sizeof(pidbuf), "%d", (int)pid); - return el_wrap_str(el_strdup(pidbuf)); -} - -el_val_t fs_list(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - el_val_t lst = el_list_empty(); - if (!path) return lst; - DIR* d = opendir(path); - if (!d) return lst; - struct dirent* e; - while ((e = readdir(d)) != NULL) { - if (strcmp(e->d_name, ".") == 0 || strcmp(e->d_name, "..") == 0) continue; - lst = el_list_append(lst, el_wrap_str(el_strdup(e->d_name))); - } - closedir(d); - return lst; -} - -/* fs_exists — true iff stat(path) succeeds. Symlinks are followed. */ -el_val_t fs_exists(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path || !*path) return 0; - struct stat st; - return (el_val_t)(stat(path, &st) == 0 ? 1 : 0); -} - -/* fs_mkdir — create directory at path with mode 0755, mkdir -p semantics. - * Returns 1 if path exists or was created (incl. all parents); 0 on failure. - * Walks the path component-by-component so missing intermediate dirs are - * also created. An existing leaf is not an error. */ -el_val_t fs_mkdir(el_val_t pathv) { - const char* path = EL_CSTR(pathv); - if (!path || !*path) return 0; - size_t n = strlen(path); - char* buf = malloc(n + 1); - if (!buf) return 0; - memcpy(buf, path, n + 1); - /* Walk components; create each prefix in turn. */ - for (size_t i = 1; i <= n; i++) { - if (buf[i] == '/' || buf[i] == '\0') { - char saved = buf[i]; - buf[i] = '\0'; - if (buf[0] != '\0') { - if (mkdir(buf, 0755) != 0 && errno != EEXIST) { - /* Tolerate the case where this prefix exists as a non-dir - * only when stat says it's a directory. */ - struct stat st; - if (stat(buf, &st) != 0 || !S_ISDIR(st.st_mode)) { - free(buf); - return 0; - } - } - } - buf[i] = saved; - } - } - free(buf); - return 1; -} - -/* ── URL encoding ─────────────────────────────────────────────────────────── */ - -/* RFC 3986 percent-encoding for URL components (form bodies, query strings). - * Unreserved set: A-Z a-z 0-9 - _ . ~ — passed through verbatim. - * Everything else (including space) becomes %XX hex. */ -el_val_t url_encode(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - static const char hex[] = "0123456789ABCDEF"; - size_t n = strlen(s); - char* out = el_strbuf(n * 3); - size_t o = 0; - for (size_t i = 0; i < n; i++) { - unsigned char c = (unsigned char)s[i]; - if ((c >= 'A' && c <= 'Z') || - (c >= 'a' && c <= 'z') || - (c >= '0' && c <= '9') || - c == '-' || c == '_' || c == '.' || c == '~') { - out[o++] = (char)c; - } else { - out[o++] = '%'; - out[o++] = hex[(c >> 4) & 0xF]; - out[o++] = hex[c & 0xF]; - } - } - out[o] = '\0'; - return el_wrap_str(out); -} - -/* Decode percent-encoded URL component. '+' becomes space (form-encoded); - * malformed %-escapes are emitted verbatim. */ -el_val_t url_decode(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - size_t o = 0; - for (size_t i = 0; i < n; i++) { - char c = s[i]; - if (c == '+') { - out[o++] = ' '; - } else if (c == '%' && i + 2 < n) { - char h1 = s[i + 1], h2 = s[i + 2]; - int v1 = (h1 >= '0' && h1 <= '9') ? h1 - '0' - : (h1 >= 'a' && h1 <= 'f') ? h1 - 'a' + 10 - : (h1 >= 'A' && h1 <= 'F') ? h1 - 'A' + 10 : -1; - int v2 = (h2 >= '0' && h2 <= '9') ? h2 - '0' - : (h2 >= 'a' && h2 <= 'f') ? h2 - 'a' + 10 - : (h2 >= 'A' && h2 <= 'F') ? h2 - 'A' + 10 : -1; - if (v1 >= 0 && v2 >= 0) { - out[o++] = (char)((v1 << 4) | v2); - i += 2; - } else { - out[o++] = c; - } - } else { - out[o++] = c; - } - } - out[o] = '\0'; - return el_wrap_str(out); -} - -/* ── HTML allowlist sanitizer ──────────────────────────────────────────────── - * el_html_sanitize(input, allowlist_json) - * - * Strict allowlist HTML cleaner. Replaces the older denylist patterns - * (str_replace cascades that wrapped dangerous tags in HTML comments and - * renamed `on*` attributes). The denylist approach is fragile: comment- - * wrapping can be re-broken by a literal `-->` inside an attacker-supplied - * attribute value, and every new attack vector requires a code change. - * - * Design: - * - Single-pass byte-level state machine. - * - Tag and attribute names are matched case-insensitively against the - * allowlist. Unknown tags are dropped entirely (the open and close - * markers are stripped; their inner text content survives, escaped). - * - A small set of "dangerous container" tags (script, style, iframe, - * object, embed, form, plus a few rarer ones) drop themselves AND - * their full subtree — text between `` is - * CDATA-like and must not be re-emitted as escaped text either. - * - Comments (), doctype (), CDATA (), - * and processing instructions () are dropped entirely. - * - Text content outside dropped subtrees is HTML-escaped (&, <, >, ", '). - * - Attribute values are unquoted/dequoted, then re-emitted with double - * quotes around the cleanly-escaped value. - * - For `` and any `src` attribute, the URL scheme is validated: - * only http:, https:, mailto:, fragment-only `#anchor`, or relative - * paths are allowed. Anything else (javascript:, data:, vbscript:, - * about:, file:, etc.) drops the attribute. - * - Self-closing void tags (br, hr, img, etc.) emit without a close tag. - * - Malformed input (unclosed tag at EOF, bad attribute syntax) drops - * the pending tag and continues. Pre-encoded entities (<, &, - * etc.) are passed through verbatim — the browser will decode them - * safely on render. - * - * Allowlist format (JSON string): - * {"p":[],"a":["href","title"],"strong":[],...} - * - Key = lowercase tag name. - * - Value = JSON array of allowed attribute names (lowercase). - * - Empty array means tag allowed but no attributes survive. - * - * Output is a freshly-allocated arena-tracked el_val_t string. */ - -/* Internal byte buffer with realloc-doubling. Used during sanitization; - * the final result is copied into an arena-tracked el_strbuf so the caller - * sees standard runtime memory semantics. */ -typedef struct { - char* data; - size_t len; - size_t cap; -} html_buf_t; - -static void html_buf_init(html_buf_t* b) { - b->cap = 256; - b->data = malloc(b->cap); - if (!b->data) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->len = 0; -} - -static void html_buf_grow(html_buf_t* b, size_t need) { - if (b->len + need + 1 <= b->cap) return; - size_t nc = b->cap; - while (b->len + need + 1 > nc) nc *= 2; - char* nd = realloc(b->data, nc); - if (!nd) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->data = nd; - b->cap = nc; -} - -static void html_buf_putc(html_buf_t* b, char c) { - html_buf_grow(b, 1); - b->data[b->len++] = c; -} - -static void html_buf_puts(html_buf_t* b, const char* s) { - if (!s) return; - size_t n = strlen(s); - html_buf_grow(b, n); - memcpy(b->data + b->len, s, n); - b->len += n; -} - -static void html_buf_free(html_buf_t* b) { - free(b->data); - b->data = NULL; - b->len = b->cap = 0; -} - -/* ASCII tolower, locale-independent. */ -static int html_tolower(int c) { - return (c >= 'A' && c <= 'Z') ? c + 32 : c; -} - -/* Case-insensitive ASCII compare of [a, a+n) against c-string `s`. - * Returns 1 iff lengths match and bytes are equal under tolower. */ -static int html_ieq_n(const char* a, size_t n, const char* s) { - if (!a || !s) return 0; - if (strlen(s) != n) return 0; - for (size_t i = 0; i < n; i++) { - if (html_tolower((unsigned char)a[i]) != html_tolower((unsigned char)s[i])) return 0; - } - return 1; -} - -/* Case-insensitive ASCII compare of two byte slices. */ -static int html_iemem(const char* a, const char* b, size_t n) { - for (size_t i = 0; i < n; i++) { - if (html_tolower((unsigned char)a[i]) != html_tolower((unsigned char)b[i])) return 0; - } - return 1; -} - -/* Walk a JSON allowlist object and find the value (an array) for a given - * tag key, comparing case-insensitively. On hit returns a pointer to the - * opening `[` of the array and writes the byte length of the array span - * (including the brackets) to *out_len. On miss returns NULL. - * - * The parser is intentionally tiny: it does not handle escapes inside - * keys (allowlist authors do not need them), and it relies on balanced - * brackets/quotes within the value array. */ -static const char* html_allowlist_find(const char* allow, const char* tag, - size_t tag_len, size_t* out_len) { - if (!allow) return NULL; - const char* p = allow; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '{') return NULL; - p++; - while (*p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p == '}' || *p == 0) return NULL; - if (*p != '"') return NULL; - p++; - const char* k = p; - while (*p && *p != '"') p++; - if (*p != '"') return NULL; - size_t klen = (size_t)(p - k); - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != ':') return NULL; - p++; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return NULL; - const char* arr_start = p; - int depth = 0; - int in_str = 0; - while (*p) { - char c = *p; - if (in_str) { - if (c == '\\' && p[1]) { p += 2; continue; } - if (c == '"') in_str = 0; - } else { - if (c == '"') in_str = 1; - else if (c == '[') depth++; - else if (c == ']') { depth--; if (depth == 0) { p++; break; } } - } - p++; - } - size_t alen = (size_t)(p - arr_start); - int match = (klen == tag_len) && html_iemem(k, tag, klen); - if (match) { - if (out_len) *out_len = alen; - return arr_start; - } - } - return NULL; -} - -/* Returns 1 iff `attr` (length attr_len) appears as a string element - * in the JSON array slice [arr, arr+arr_len). Comparison is case- - * insensitive. */ -static int html_attr_in_array(const char* arr, size_t arr_len, - const char* attr, size_t attr_len) { - if (!arr || arr_len < 2) return 0; - const char* p = arr + 1; - const char* end = arr + arr_len - 1; - while (p < end) { - while (p < end && (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',')) p++; - if (p >= end) return 0; - if (*p != '"') return 0; - p++; - const char* s = p; - while (p < end && *p != '"') { - if (*p == '\\' && p + 1 < end) p++; - p++; - } - if (p >= end) return 0; - size_t slen = (size_t)(p - s); - p++; - if (slen == attr_len && html_iemem(s, attr, slen)) return 1; - } - return 0; -} - -/* Hard-coded set of tags whose content is ALSO dropped (entire subtree). */ -static int html_is_dangerous_container(const char* tag, size_t tag_len) { - static const char* names[] = { - "script", "style", "iframe", "object", "embed", "form", - "noscript", "noembed", "template", "svg", "math", "frame", - "frameset", "applet", "audio", "video", "source", "track", - NULL - }; - for (int i = 0; names[i]; i++) { - if (html_ieq_n(tag, tag_len, names[i])) return 1; - } - return 0; -} - -/* HTML void elements — emit without a close tag. */ -static int html_is_void(const char* tag, size_t tag_len) { - static const char* names[] = { - "area", "base", "br", "col", "embed", "hr", "img", "input", - "link", "meta", "param", "source", "track", "wbr", - NULL - }; - for (int i = 0; names[i]; i++) { - if (html_ieq_n(tag, tag_len, names[i])) return 1; - } - return 0; -} - -/* Append a single byte HTML-escaped into the output buffer. */ -static void html_escape_byte(html_buf_t* out, unsigned char c) { - switch (c) { - case '<': html_buf_puts(out, "<"); break; - case '>': html_buf_puts(out, ">"); break; - case '"': html_buf_puts(out, """); break; - case '\'': html_buf_puts(out, "'"); break; - default: html_buf_putc(out, (char)c); break; - } -} - -/* Validate a URL value against the allowlist of safe schemes for hrefs. - * Returns 1 iff the URL is safe to emit. Acceptable forms: - * - http:// or https:// (case-insensitive) - * - mailto: - * - fragment-only `#anchor` - * - relative path that does not contain a colon before the first - * slash/?/# (so `foo/bar`, `/foo`, `?x=1` are OK; `javascript:x` is - * not — its colon precedes any path/hash/query separator). - * - * URL leading whitespace and embedded ASCII control bytes (TAB, LF, CR) - * are stripped before the scheme test, mirroring how browsers normalise - * URLs (these bytes are otherwise a known XSS bypass: `java\tscript:`). */ -static int html_url_is_safe(const char* url, size_t len) { - if (!url || len == 0) return 1; /* empty href is harmless */ - size_t i = 0; - while (i < len) { - unsigned char c = (unsigned char)url[i]; - if (c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == 0x0B || c == 0x0C) { - i++; continue; - } - break; - } - if (i >= len) return 1; /* whitespace only */ - if (url[i] == '#') return 1; /* fragment only */ - if (url[i] == '/' || url[i] == '?') return 1; /* relative */ - /* Find the first scheme-terminating character. */ - size_t scheme_end = (size_t)-1; - for (size_t j = i; j < len; j++) { - char c = url[j]; - if (c == ':') { scheme_end = j; break; } - if (c == '/' || c == '?' || c == '#') break; - } - if (scheme_end == (size_t)-1) return 1; /* no colon → relative path */ - /* Lowercase the scheme, stripping embedded control bytes. */ - char scheme[32]; - size_t sl = 0; - for (size_t j = i; j < scheme_end && sl < sizeof(scheme) - 1; j++) { - unsigned char c = (unsigned char)url[j]; - if (c == '\t' || c == '\n' || c == '\r' || c == 0x0B || c == 0x0C) continue; - scheme[sl++] = (char)html_tolower(c); - } - scheme[sl] = '\0'; - if (strcmp(scheme, "http") == 0) return 1; - if (strcmp(scheme, "https") == 0) return 1; - if (strcmp(scheme, "mailto") == 0) return 1; - return 0; -} - -el_val_t el_html_sanitize(el_val_t input_v, el_val_t allowlist_v) { - const char* input = EL_CSTR(input_v); - const char* allow = EL_CSTR(allowlist_v); - if (!input) return el_wrap_str(el_strdup("")); - if (!allow) allow = "{}"; - size_t in_len = strlen(input); - - html_buf_t out; - html_buf_init(&out); - - size_t i = 0; - while (i < in_len) { - unsigned char c = (unsigned char)input[i]; - if (c != '<') { - /* Plain text — escape and emit. We pass `&` through verbatim - * to preserve pre-encoded entities (`<`, `&`, `&#x...;`) - * which the browser will decode safely. */ - if (c == '&') html_buf_putc(&out, '&'); - else html_escape_byte(&out, c); - i++; - continue; - } - /* `<` — try to parse a tag. */ - if (i + 1 >= in_len) { - html_buf_puts(&out, "<"); - i++; - continue; - } - /* Comments, doctype, CDATA, processing instructions — drop entirely. */ - if (input[i + 1] == '!') { - if (i + 3 < in_len && input[i + 2] == '-' && input[i + 3] == '-') { - size_t j = i + 4; - while (j + 2 < in_len && !(input[j] == '-' && input[j + 1] == '-' && input[j + 2] == '>')) j++; - if (j + 2 < in_len) i = j + 3; - else i = in_len; - continue; - } - size_t j = i + 2; - while (j < in_len && input[j] != '>') j++; - i = (j < in_len) ? j + 1 : in_len; - continue; - } - if (input[i + 1] == '?') { - size_t j = i + 2; - while (j < in_len && input[j] != '>') j++; - i = (j < in_len) ? j + 1 : in_len; - continue; - } - int is_close = 0; - size_t name_start = i + 1; - if (input[i + 1] == '/') { - is_close = 1; - name_start = i + 2; - } - if (name_start >= in_len) { - html_buf_puts(&out, "<"); - i++; - continue; - } - unsigned char nc = (unsigned char)input[name_start]; - if (!((nc >= 'a' && nc <= 'z') || (nc >= 'A' && nc <= 'Z'))) { - /* `<` followed by non-letter — emit as escaped text. */ - html_buf_puts(&out, "<"); - i++; - continue; - } - size_t name_end = name_start; - while (name_end < in_len) { - unsigned char x = (unsigned char)input[name_end]; - if ((x >= 'a' && x <= 'z') || (x >= 'A' && x <= 'Z') || - (x >= '0' && x <= '9') || x == '-' || x == '_' || x == ':') { - name_end++; - } else { - break; - } - } - const char* tag = input + name_start; - size_t tag_len = name_end - name_start; - /* Find the `>` that closes this tag, respecting quoted attrs. */ - size_t cur = name_end; - int self_close = 0; - while (cur < in_len) { - unsigned char x = (unsigned char)input[cur]; - if (x == '"' || x == '\'') { - unsigned char q = x; - cur++; - while (cur < in_len && (unsigned char)input[cur] != q) cur++; - if (cur < in_len) cur++; /* skip closing quote */ - continue; - } - if (x == '/' && cur + 1 < in_len && input[cur + 1] == '>') { - self_close = 1; - break; - } - if (x == '>') break; - cur++; - } - if (cur >= in_len) { - /* Malformed: unclosed tag at EOF. Drop the rest of the input. */ - i = in_len; - continue; - } - size_t tag_end = self_close ? cur + 2 : cur + 1; /* one past `>` */ - /* Dangerous container — drop the whole subtree. */ - if (!is_close && html_is_dangerous_container(tag, tag_len)) { - if (self_close || html_is_void(tag, tag_len)) { - i = tag_end; - continue; - } - size_t scan = tag_end; - int found_close = 0; - while (scan < in_len) { - if (input[scan] != '<') { scan++; continue; } - if (scan + 1 < in_len && input[scan + 1] == '/') { - size_t cn_start = scan + 2; - size_t cn_end = cn_start; - while (cn_end < in_len) { - unsigned char x = (unsigned char)input[cn_end]; - if ((x >= 'a' && x <= 'z') || (x >= 'A' && x <= 'Z') || - (x >= '0' && x <= '9') || x == '-' || x == '_' || x == ':') { - cn_end++; - } else break; - } - if (cn_end - cn_start == tag_len && - html_iemem(input + cn_start, tag, tag_len)) { - size_t end_close = cn_end; - while (end_close < in_len && input[end_close] != '>') end_close++; - i = (end_close < in_len) ? end_close + 1 : in_len; - found_close = 1; - break; - } - } - scan++; - } - if (!found_close) { - /* No matching close — drop everything from here on. */ - i = in_len; - } - continue; - } - /* Look up the tag in the allowlist. */ - size_t arr_len = 0; - const char* arr = html_allowlist_find(allow, tag, tag_len, &arr_len); - if (!arr) { - /* Tag not allowed. Drop the open/close marker; inner text is - * processed by the outer loop and re-emitted as escaped text. */ - i = tag_end; - continue; - } - if (is_close) { - if (!html_is_void(tag, tag_len)) { - html_buf_putc(&out, '<'); - html_buf_putc(&out, '/'); - for (size_t k = 0; k < tag_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)tag[k])); - } - html_buf_putc(&out, '>'); - } - i = tag_end; - continue; - } - /* Allowed open tag. Emit ``. */ - html_buf_putc(&out, '<'); - for (size_t k = 0; k < tag_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)tag[k])); - } - size_t a = name_end; - while (a < cur) { - unsigned char x = (unsigned char)input[a]; - if (x == ' ' || x == '\t' || x == '\n' || x == '\r' || x == '/') { a++; continue; } - size_t an_start = a; - while (a < cur) { - unsigned char y = (unsigned char)input[a]; - if (y == '=' || y == ' ' || y == '\t' || y == '\n' || y == '\r' || y == '/' || y == '>') break; - a++; - } - size_t an_len = a - an_start; - if (an_len == 0) { a++; continue; } - size_t av_start = 0; - size_t av_len = 0; - int has_value = 0; - size_t b = a; - while (b < cur && (input[b] == ' ' || input[b] == '\t' || input[b] == '\n' || input[b] == '\r')) b++; - if (b < cur && input[b] == '=') { - has_value = 1; - b++; - while (b < cur && (input[b] == ' ' || input[b] == '\t' || input[b] == '\n' || input[b] == '\r')) b++; - if (b < cur && (input[b] == '"' || input[b] == '\'')) { - unsigned char q = (unsigned char)input[b]; - b++; - av_start = b; - while (b < cur && (unsigned char)input[b] != q) b++; - av_len = b - av_start; - if (b < cur) b++; - } else { - av_start = b; - while (b < cur) { - unsigned char y = (unsigned char)input[b]; - if (y == ' ' || y == '\t' || y == '\n' || y == '\r' || y == '>') break; - b++; - } - av_len = b - av_start; - } - a = b; - } - if (!html_attr_in_array(arr, arr_len, input + an_start, an_len)) continue; - int is_href = (an_len == 4 && html_iemem(input + an_start, "href", 4)); - int is_src = (an_len == 3 && html_iemem(input + an_start, "src", 3)); - if ((is_href || is_src) && has_value) { - if (!html_url_is_safe(input + av_start, av_len)) continue; - } - html_buf_putc(&out, ' '); - for (size_t k = 0; k < an_len; k++) { - html_buf_putc(&out, (char)html_tolower((unsigned char)input[an_start + k])); - } - if (has_value) { - html_buf_puts(&out, "=\""); - for (size_t k = 0; k < av_len; k++) { - unsigned char y = (unsigned char)input[av_start + k]; - /* Re-escape so the emitted attribute is well-formed - * double-quoted HTML. `&` passes through to preserve - * pre-encoded entities. */ - if (y == '"') html_buf_puts(&out, """); - else if (y == '<') html_buf_puts(&out, "<"); - else if (y == '>') html_buf_puts(&out, ">"); - else html_buf_putc(&out, (char)y); - } - html_buf_putc(&out, '"'); - } - } - html_buf_putc(&out, '>'); - i = tag_end; - } - /* Copy into arena-tracked buffer so the standard runtime memory model - * applies to the returned string. */ - char* result = el_strbuf(out.len); - memcpy(result, out.data, out.len); - result[out.len] = '\0'; - html_buf_free(&out); - return el_wrap_str(result); -} - -/* ── JSON ────────────────────────────────────────────────────────────────── */ - -/* True iff the segment is non-empty and every byte is an ASCII digit. We treat - * such segments as numeric array indices when walking a dot-path; mixed names - * like "0a" remain object-key lookups, so a key named "0" still wins over an - * index when the surrounding container is an object. */ -static int json_path_seg_is_index(const char* seg, size_t n) { - if (n == 0) return 0; - for (size_t i = 0; i < n; i++) { - if (seg[i] < '0' || seg[i] > '9') return 0; - } - return 1; -} - -/* Skip JSON whitespace. */ -static const char* json_skip_ws(const char* p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - return p; -} - -/* Descend one segment into the JSON cursor `p`. - * - If `p` points at an array `[...]` and the segment is all digits, - * advance to that element (zero-based). - * - Otherwise treat the segment as an object key and use json_find_key - * scoped to a one-level slice of the current container. - * Returns NULL if the descent fails (segment not found, container mismatch). - * - * `seg` is a pointer into the original path string and `seg_len` is its - * byte length — this avoids an extra alloc per segment. */ -static const char* json_path_descend(const char* p, const char* seg, size_t seg_len) { - if (!p || !seg) return NULL; - p = json_skip_ws(p); - if (*p == '[' && json_path_seg_is_index(seg, seg_len)) { - long idx = 0; - for (size_t i = 0; i < seg_len; i++) idx = idx * 10 + (seg[i] - '0'); - p++; /* step past '[' */ - p = json_skip_ws(p); - long cur = 0; - while (*p && *p != ']') { - if (cur == idx) return p; - const char* end = json_skip_value(p); - if (!end || end == p) return NULL; - p = json_skip_ws(end); - if (*p == ',') { p++; p = json_skip_ws(p); cur++; continue; } - /* No comma after this element — only acceptable at the closing ']', - * which means we ran out of elements. */ - break; - } - return NULL; - } - /* Object lookup. json_find_key walks at depth 1 of whatever container it - * receives, so we slice from `p` onwards. Caller already positioned us at - * the opening '{' (or at whitespace before it). */ - if (*p != '{') return NULL; - /* Build a NUL-terminated copy of the key segment for the lookup. We only - * pay this cost when the segment isn't a numeric index. */ - char stack_key[256]; - char* k = stack_key; - if (seg_len + 1 > sizeof(stack_key)) { - k = malloc(seg_len + 1); - if (!k) return NULL; - } - memcpy(k, seg, seg_len); - k[seg_len] = '\0'; - const char* found = json_find_key(p, k); - if (k != stack_key) free(k); - return found; -} - -/* Read the JSON value at `p` into a freshly-allocated, arena-owned el_val_t. - * - String -> unescaped, wrapped el_val_t string - * - Anything else -> raw JSON slice as a string (matches the historical - * json_get behaviour: numbers/bools/null come back stringified). */ -static el_val_t json_read_value(const char* p) { - p = json_skip_ws(p); - if (*p == '"') { - p++; - size_t cap = strlen(p) + 1; - char* out = el_strbuf(cap); - char* w = out; - while (*p && *p != '"') { - if (*p == '\\' && *(p+1)) { - p++; - switch (*p) { - case '"': *w++ = '"'; break; - case '\\': *w++ = '\\'; break; - case '/': *w++ = '/'; break; - case 'n': *w++ = '\n'; break; - case 'r': *w++ = '\r'; break; - case 't': *w++ = '\t'; break; - default: *w++ = *p; break; - } - } else { - *w++ = *p; - } - p++; - } - *w = '\0'; - return el_wrap_str(out); - } - /* Object/array/number/bool/null — return the raw slice up to the value's - * end. json_skip_value tracks brace/bracket/string state so nested objects - * round-trip cleanly. */ - const char* end = json_skip_value(p); - if (!end) end = p; - size_t n = (size_t)(end - p); - /* Strip trailing whitespace from scalar values so callers don't see - * `123 ` when they parsed a pretty-printed number. */ - while (n > 0 && (p[n-1] == ' ' || p[n-1] == '\t' || p[n-1] == '\n' || p[n-1] == '\r')) { - n--; - } - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t json_get(el_val_t jsonv, el_val_t keyv) { - const char* json = EL_CSTR(jsonv); - const char* key = EL_CSTR(keyv); - if (!json || !key) return el_wrap_str(el_strdup("")); - - /* Fast path: key contains no '.' — keep the historical single-segment - * substring search so existing callers retain their O(strlen) cost - * profile. The dot-path walker is only paid for when needed. */ - if (!strchr(key, '.')) { - size_t klen = strlen(key); - char stack_pat[512]; - char* pattern; - if (klen + 5 <= sizeof(stack_pat)) { - pattern = stack_pat; - } else { - pattern = malloc(klen + 5); - if (!pattern) return el_wrap_str(el_strdup("")); - } - snprintf(pattern, klen + 5, "\"%s\":", key); - const char* p = strstr(json, pattern); - if (pattern != stack_pat) free(pattern); - if (!p) return el_wrap_str(el_strdup("")); - p += strlen(key) + 3; /* skip "key": */ - return json_read_value(p); - } - - /* Dot-path traversal. Walk segments left to right; at each step, descend - * into the current container by either array index (all-digit segment on - * an array cursor) or object key. */ - const char* cursor = json_skip_ws(json); - const char* seg_start = key; - const char* k = key; - while (1) { - if (*k == '.' || *k == '\0') { - size_t seg_len = (size_t)(k - seg_start); - cursor = json_path_descend(cursor, seg_start, seg_len); - if (!cursor) return el_wrap_str(el_strdup("")); - if (*k == '\0') break; - k++; - seg_start = k; - continue; - } - k++; - } - return json_read_value(cursor); -} - -/* ── Float bit-cast helpers ──────────────────────────────────────────────── */ -/* `el_to_float` and `el_from_float` are exposed in el_runtime.h as static - * inlines so generated programs (which #include the header) can call them - * for Float literals. No definitions are needed here. */ - -/* ── JSON parser (recursive descent) ─────────────────────────────────────── */ -/* - * Parsed JSON representation: - * - object -> ElMap (keys & values are el_val_t) - * - array -> ElList - * - string -> EL_STR-wrapped char* (allocated) - * - number -> int (el_val_t) if integer, otherwise el_from_float(double) - * - true -> 1 - * - false -> 0 - * - null -> EL_NULL (0) - * - * Note: there is no runtime type tag — parsed numbers cannot be - * distinguished from booleans by the runtime alone. The codegen tracks - * types separately. This matches the rest of el_val_t's type-erased model. - */ - -/* JsonParser struct is forward-declared near the HTTP/Engram section. */ - -static void jp_skip_ws(JsonParser* jp) { - while (jp->p < jp->end) { - char c = *jp->p; - if (c == ' ' || c == '\t' || c == '\n' || c == '\r') jp->p++; - else break; - } -} - -static el_val_t jp_parse_value(JsonParser* jp); - -/* Parse a JSON string literal (the opening " has NOT yet been consumed). */ -static char* jp_parse_string_raw(JsonParser* jp) { - if (jp->p >= jp->end || *jp->p != '"') { jp->err = 1; return el_strdup(""); } - jp->p++; - size_t cap = 32, len = 0; - char* out = malloc(cap); - if (!out) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - while (jp->p < jp->end && *jp->p != '"') { - char c = *jp->p++; - if (c == '\\' && jp->p < jp->end) { - char esc = *jp->p++; - switch (esc) { - case '"': c = '"'; break; - case '\\': c = '\\'; break; - case '/': c = '/'; break; - case 'b': c = '\b'; break; - case 'f': c = '\f'; break; - case 'n': c = '\n'; break; - case 'r': c = '\r'; break; - case 't': c = '\t'; break; - case 'u': { - /* Skip 4 hex digits; emit '?' as a placeholder */ - for (int i = 0; i < 4 && jp->p < jp->end; i++) jp->p++; - c = '?'; - break; - } - default: c = esc; break; - } - } - if (len + 1 >= cap) { - cap *= 2; - out = realloc(out, cap); - if (!out) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } - out[len++] = c; - } - if (jp->p < jp->end && *jp->p == '"') jp->p++; - else jp->err = 1; - out[len] = '\0'; - return out; -} - -static el_val_t jp_parse_number(JsonParser* jp) { - const char* start = jp->p; - int is_float = 0; - if (jp->p < jp->end && (*jp->p == '-' || *jp->p == '+')) jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - if (jp->p < jp->end && *jp->p == '.') { - is_float = 1; jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - } - if (jp->p < jp->end && (*jp->p == 'e' || *jp->p == 'E')) { - is_float = 1; jp->p++; - if (jp->p < jp->end && (*jp->p == '+' || *jp->p == '-')) jp->p++; - while (jp->p < jp->end && isdigit((unsigned char)*jp->p)) jp->p++; - } - size_t n = (size_t)(jp->p - start); - char buf[64]; - if (n >= sizeof(buf)) n = sizeof(buf) - 1; - memcpy(buf, start, n); - buf[n] = '\0'; - if (is_float) return el_from_float(strtod(buf, NULL)); - return (el_val_t)strtoll(buf, NULL, 10); -} - -static el_val_t jp_parse_array(JsonParser* jp) { - if (jp->p < jp->end && *jp->p == '[') jp->p++; - el_val_t lst = el_list_empty(); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ']') { jp->p++; return lst; } - while (jp->p < jp->end) { - jp_skip_ws(jp); - el_val_t v = jp_parse_value(jp); - lst = el_list_append(lst, v); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ',') { jp->p++; continue; } - if (jp->p < jp->end && *jp->p == ']') { jp->p++; break; } - jp->err = 1; - break; - } - return lst; -} - -static el_val_t jp_parse_object(JsonParser* jp) { - if (jp->p < jp->end && *jp->p == '{') jp->p++; - el_val_t m = el_map_new(0); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == '}') { jp->p++; return m; } - while (jp->p < jp->end) { - jp_skip_ws(jp); - char* key = jp_parse_string_raw(jp); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ':') jp->p++; - else { jp->err = 1; free(key); break; } - jp_skip_ws(jp); - el_val_t v = jp_parse_value(jp); - m = el_map_set(m, EL_STR(key), v); - jp_skip_ws(jp); - if (jp->p < jp->end && *jp->p == ',') { jp->p++; continue; } - if (jp->p < jp->end && *jp->p == '}') { jp->p++; break; } - jp->err = 1; - break; - } - return m; -} - -static el_val_t jp_parse_value(JsonParser* jp) { - jp_skip_ws(jp); - if (jp->p >= jp->end) { jp->err = 1; return EL_NULL; } - char c = *jp->p; - if (c == '"') return el_wrap_str(jp_parse_string_raw(jp)); - if (c == '{') return jp_parse_object(jp); - if (c == '[') return jp_parse_array(jp); - if (c == '-' || isdigit((unsigned char)c)) return jp_parse_number(jp); - if (c == 't' && jp->p + 4 <= jp->end && strncmp(jp->p, "true", 4) == 0) { jp->p += 4; return 1; } - if (c == 'f' && jp->p + 5 <= jp->end && strncmp(jp->p, "false", 5) == 0) { jp->p += 5; return 0; } - if (c == 'n' && jp->p + 4 <= jp->end && strncmp(jp->p, "null", 4) == 0) { jp->p += 4; return EL_NULL; } - jp->err = 1; - return EL_NULL; -} - -el_val_t json_parse(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return EL_NULL; - JsonParser jp = { .p = s, .end = s + strlen(s), .err = 0 }; - el_val_t v = jp_parse_value(&jp); - if (jp.err) return EL_NULL; - return v; -} - -/* ── JSON stringify ──────────────────────────────────────────────────────── */ -/* - * Stringify policy: el_val_t is type-erased, so we cannot perfectly - * round-trip arbitrary values. We use these heuristics: - * - If value is an ElList pointer (in the heap range), serialize as array. - * - If value is an ElMap pointer, serialize as object. - * - If value looks like a printable string pointer, serialize as string. - * - Otherwise serialize as integer. - * This is best-effort. Programs that need exact control should build the - * string directly. A pointer test is the cheapest way to disambiguate - * from small integers without a separate type tag. - */ - -/* JsonBuf struct is forward-declared near the HTTP section so HTTP helpers - * can use it. Its definition appears there. */ - -static void jb_init(JsonBuf* b) { - b->cap = 64; b->len = 0; - b->buf = malloc(b->cap); - if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - b->buf[0] = '\0'; -} - -static void jb_reserve(JsonBuf* b, size_t add) { - if (b->len + add + 1 > b->cap) { - while (b->len + add + 1 > b->cap) b->cap *= 2; - b->buf = realloc(b->buf, b->cap); - if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - } -} - -static void jb_putc(JsonBuf* b, char c) { - jb_reserve(b, 1); - b->buf[b->len++] = c; - b->buf[b->len] = '\0'; -} - -static void jb_puts(JsonBuf* b, const char* s) { - size_t n = strlen(s); - jb_reserve(b, n); - memcpy(b->buf + b->len, s, n); - b->len += n; - b->buf[b->len] = '\0'; -} - -static void jb_emit_escaped(JsonBuf* b, const char* s) { - jb_putc(b, '"'); - for (; *s; s++) { - unsigned char c = (unsigned char)*s; - switch (c) { - case '"': jb_puts(b, "\\\""); break; - case '\\': jb_puts(b, "\\\\"); break; - case '\b': jb_puts(b, "\\b"); break; - case '\f': jb_puts(b, "\\f"); break; - case '\n': jb_puts(b, "\\n"); break; - case '\r': jb_puts(b, "\\r"); break; - case '\t': jb_puts(b, "\\t"); break; - default: - if (c < 0x20) { - char tmp[8]; - snprintf(tmp, sizeof(tmp), "\\u%04x", c); - jb_puts(b, tmp); - } else { - jb_putc(b, (char)c); - } - break; - } - } - jb_putc(b, '"'); -} - -/* Heuristic: is this el_val_t likely a pointer to an ElList? - * We can't fully verify, but pointers are large addresses, integers small. - * Treat values whose magnitude exceeds 2^32 as potential pointers and - * sniff by reading the header conservatively. - * - * Simpler heuristic: if the value reads as a printable string, treat as - * string; otherwise as integer. Lists/Maps are encoded as struct pointers, - * which have leading binary bytes — so they won't look like strings. */ - -static int looks_like_string(el_val_t v) { - if (v == 0) return 0; - /* Treat plausible heap addresses as candidates. - * Threshold: 4 GiB (0x100000000). On 64-bit systems heap addresses from - * malloc/mmap start well above 4 GiB (ASLR pushes them to ~0x7f...). - * El integer values (counters, unix timestamps up to ~2106) all fit below - * 0x100000000 (4294967296). The old threshold of 1,000,000 caused unix - * timestamps (~1.7e9) to be misidentified as string pointers — a segfault - * risk in json_stringify and jb_emit_value. */ - uintptr_t p = (uintptr_t)v; - if (p < 0x100000000ULL) return 0; /* integers, timestamps, counters */ - if (p < 0x1000) return 0; - /* Sniff first bytes for printable */ - const unsigned char* s = (const unsigned char*)p; - for (int i = 0; i < 16; i++) { - unsigned char c = s[i]; - if (c == '\0') return 1; /* terminated string (empty string is still a valid string) */ - /* Reject C0 control chars (non-whitespace), allow UTF-8 high bytes. - * 0x09-0x0d = tab/newline/cr/vt/ff (whitespace, OK) - * 0x20-0x7e = printable ASCII (OK) - * 0x7f = DEL (reject) - * 0x80-0xff = UTF-8 continuation/lead bytes (OK for multi-byte chars) */ - if (c < 0x09 || (c > 0x0d && c < 0x20) || c == 0x7f) return 0; - } - return 1; /* 16+ printable bytes — call it a string */ -} - -static void jb_emit_value(JsonBuf* b, el_val_t v); - -static void jb_emit_int(JsonBuf* b, int64_t n) { - char tmp[32]; - snprintf(tmp, sizeof(tmp), "%lld", (long long)n); - jb_puts(b, tmp); -} - -static void jb_emit_value(JsonBuf* b, el_val_t v) { - if (v == EL_NULL) { jb_puts(b, "null"); return; } - if (looks_like_string(v)) { - jb_emit_escaped(b, EL_CSTR(v)); - return; - } - jb_emit_int(b, (int64_t)v); -} - -el_val_t json_stringify(el_val_t v) { - JsonBuf b; jb_init(&b); - jb_emit_value(&b, v); - return el_wrap_str(b.buf); -} - -/* ── JSON substring accessors ────────────────────────────────────────────── */ -/* - * These walk the raw JSON string looking for "key": at the top level (depth 1) - * of an object. They handle escaped quotes, nested objects/arrays, and - * whitespace around the colon. - */ - -/* Find "key": at object-depth == 1 inside the JSON object string `s`. - * Returns pointer to the first byte of the value, or NULL. */ -static const char* json_find_key(const char* s, const char* key) { - if (!s || !key) return NULL; - size_t klen = strlen(key); - int depth = 0; - int in_str = 0; - int escape = 0; - const char* p = s; - while (*p) { - char c = *p; - if (in_str) { - if (escape) { escape = 0; } - else if (c == '\\') { escape = 1; } - else if (c == '"') { - /* End of string. If we're at depth 1, check if this was a key. */ - p++; - if (depth == 1) { - /* The string just ended at p-1. Check if it matches key - * and is followed by a colon. We need to backtrack to find - * the start of this string and compare. */ - } - in_str = 0; - continue; - } - p++; - continue; - } - if (c == '"') { - /* Start of a string literal */ - const char* str_start = p + 1; - const char* q = str_start; - int e = 0; - while (*q) { - if (e) { e = 0; q++; continue; } - if (*q == '\\') { e = 1; q++; continue; } - if (*q == '"') break; - q++; - } - size_t slen = (size_t)(q - str_start); - const char* after = (*q == '"') ? q + 1 : q; - /* If at depth 1 and matches key and followed by ':' -> got it */ - if (depth == 1 && slen == klen && strncmp(str_start, key, klen) == 0) { - const char* r = after; - while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') r++; - if (*r == ':') { - r++; - while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') r++; - return r; - } - } - p = after; - continue; - } - if (c == '{' || c == '[') depth++; - else if (c == '}' || c == ']') depth--; - p++; - } - return NULL; -} - -/* Skip a JSON value starting at p; return pointer past the value end. */ -static const char* json_skip_value(const char* p) { - if (!p || !*p) return p; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p == '"') { - p++; - int e = 0; - while (*p) { - if (e) { e = 0; p++; continue; } - if (*p == '\\') { e = 1; p++; continue; } - if (*p == '"') { p++; break; } - p++; - } - return p; - } - if (*p == '{' || *p == '[') { - char open = *p; - char close = (open == '{') ? '}' : ']'; - int depth = 0; - int in_str = 0; - int e = 0; - while (*p) { - char c = *p; - if (in_str) { - if (e) { e = 0; } - else if (c == '\\') { e = 1; } - else if (c == '"') in_str = 0; - p++; - continue; - } - if (c == '"') { in_str = 1; p++; continue; } - if (c == open) depth++; - else if (c == close) { depth--; p++; if (depth == 0) return p; continue; } - p++; - } - return p; - } - /* scalar: number, true/false/null */ - while (*p && *p != ',' && *p != '}' && *p != ']' && - *p != ' ' && *p != '\t' && *p != '\n' && *p != '\r') p++; - return p; -} - -el_val_t json_get_string(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p || *p != '"') return el_wrap_str(el_strdup("")); - p++; - JsonParser jp = { .p = p - 1, .end = json + (json ? strlen(json) : 0), .err = 0 }; - char* parsed = jp_parse_string_raw(&jp); - if (jp.err) { free(parsed); return el_wrap_str(el_strdup("")); } - return el_wrap_str(parsed); -} - -el_val_t json_get_int(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (*p == '"' || *p == '{' || *p == '[') return 0; - return (el_val_t)strtoll(p, NULL, 10); -} - -el_val_t json_get_float(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (*p == '"' || *p == '{' || *p == '[') return 0; - return el_from_float(strtod(p, NULL)); -} - -el_val_t json_get_bool(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - if (!p) return 0; - if (strncmp(p, "true", 4) == 0) return 1; - return 0; -} - -el_val_t json_get_raw(el_val_t json_str, el_val_t key) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - const char* p = json_find_key(json, k); - /* Clear fs_read binary-length hint — result is a fresh null-terminated - * string, not the raw file bytes, so Content-Length must use strlen. */ - _tl_fs_read_len = 0; - if (!p) return el_wrap_str(el_strdup("")); - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value) { - const char* json = EL_CSTR(json_str); - const char* k = EL_CSTR(key); - if (!k) k = ""; - if (!json || !*json) { - /* Build a fresh object */ - JsonBuf b; jb_init(&b); - jb_putc(&b, '{'); - jb_emit_escaped(&b, k); - jb_putc(&b, ':'); - jb_emit_value(&b, value); - jb_putc(&b, '}'); - return el_wrap_str(b.buf); - } - const char* existing = json_find_key(json, k); - JsonBuf b; jb_init(&b); - if (existing) { - const char* end = json_skip_value(existing); - /* Copy [json .. existing) */ - size_t prefix = (size_t)(existing - json); - jb_reserve(&b, prefix); - memcpy(b.buf + b.len, json, prefix); - b.len += prefix; - b.buf[b.len] = '\0'; - jb_emit_value(&b, value); - jb_puts(&b, end); - return el_wrap_str(b.buf); - } - /* Insert before closing '}'. Find last '}' */ - size_t jl = strlen(json); - if (jl == 0) { free(b.buf); return el_wrap_str(el_strdup("{}")); } - /* Find last '}' from the end */ - ssize_t close_idx = -1; - for (ssize_t i = (ssize_t)jl - 1; i >= 0; i--) { - if (json[i] == '}') { close_idx = i; break; } - } - if (close_idx < 0) { - free(b.buf); - return el_wrap_str(el_strdup(json)); - } - /* Determine if object is empty: scan between last '{' and '}' for non-ws */ - int empty = 1; - for (ssize_t i = close_idx - 1; i >= 0; i--) { - char c = json[i]; - if (c == '{') break; - if (c != ' ' && c != '\t' && c != '\n' && c != '\r') { empty = 0; break; } - } - /* Copy json[0..close_idx) */ - jb_reserve(&b, (size_t)close_idx); - memcpy(b.buf + b.len, json, (size_t)close_idx); - b.len += (size_t)close_idx; - b.buf[b.len] = '\0'; - if (!empty) jb_putc(&b, ','); - jb_emit_escaped(&b, k); - jb_putc(&b, ':'); - jb_emit_value(&b, value); - /* Append from close_idx onward */ - jb_puts(&b, json + close_idx); - return el_wrap_str(b.buf); -} - -el_val_t json_array_len(el_val_t json_str) { - const char* s = EL_CSTR(json_str); - if (!s) return 0; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s != '[') return 0; - s++; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ']') return 0; - int64_t count = 0; - while (*s) { - const char* end = json_skip_value(s); - if (end == s) break; - count++; - s = end; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ',') { s++; continue; } - if (*s == ']' || *s == '\0') break; - } - return (el_val_t)count; -} - -/* json_array_get — return the i-th element of a JSON array as a JSON - * fragment string. Nested objects and arrays are returned verbatim - * (json_skip_value tracks brace/bracket depth so nested structures are - * preserved intact). Out-of-range index → "". */ -el_val_t json_array_get(el_val_t json_str, el_val_t index) { - const char* s = EL_CSTR(json_str); - int64_t idx = (int64_t)index; - if (!s || idx < 0) return el_wrap_str(el_strdup("")); - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s != '[') return el_wrap_str(el_strdup("")); - s++; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ']') return el_wrap_str(el_strdup("")); - int64_t i = 0; - while (*s) { - const char* start = s; - const char* end = json_skip_value(s); - if (end == s) break; - if (i == idx) { - size_t n = (size_t)(end - start); - char* out = el_strbuf(n); - memcpy(out, start, n); - out[n] = '\0'; - return el_wrap_str(out); - } - i++; - s = end; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == ',') { s++; while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; continue; } - if (*s == ']' || *s == '\0') break; - } - return el_wrap_str(el_strdup("")); -} - -/* json_array_get_string — same as json_array_get, but assume the element - * is a JSON string and return the unquoted/unescaped value. Non-string - * elements yield "". */ -el_val_t json_array_get_string(el_val_t json_str, el_val_t index) { - el_val_t raw = json_array_get(json_str, index); - const char* s = EL_CSTR(raw); - if (!s || *s != '"') return el_wrap_str(el_strdup("")); - JsonParser jp = { - .p = s, - .end = s + strlen(s), - .err = 0, - }; - char* parsed = jp_parse_string_raw(&jp); - if (jp.err) { - free(parsed); - return el_wrap_str(el_strdup("")); - } - return el_wrap_str(parsed); -} - -/* ── Time ────────────────────────────────────────────────────────────────── */ - -el_val_t time_now(void) { - struct timeval tv; - gettimeofday(&tv, NULL); - int64_t ms = (int64_t)tv.tv_sec * 1000LL + (int64_t)tv.tv_usec / 1000LL; - return (el_val_t)ms; -} - -el_val_t time_now_utc(void) { - return time_now(); -} - -el_val_t time_format(el_val_t ts, el_val_t fmt) { - int64_t ms = (int64_t)ts; - time_t s = (time_t)(ms / 1000); - int msec = (int)(ms % 1000); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - const char* fmt_str = EL_CSTR(fmt); - if (!fmt_str || strcmp(fmt_str, "ISO") == 0) { - char buf[64]; - snprintf(buf, sizeof(buf), "%04d-%02d-%02dT%02d:%02d:%02d.%03dZ", - tm.tm_year + 1900, tm.tm_mon + 1, tm.tm_mday, - tm.tm_hour, tm.tm_min, tm.tm_sec, msec); - return el_wrap_str(el_strdup(buf)); - } - char buf[256]; - if (strftime(buf, sizeof(buf), fmt_str, &tm) == 0) buf[0] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -el_val_t time_to_parts(el_val_t ts) { - int64_t ms = (int64_t)ts; - time_t s = (time_t)(ms / 1000); - int msec = (int)(ms % 1000); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("year")), (el_val_t)(tm.tm_year + 1900)); - m = el_map_set(m, EL_STR(el_strdup("month")), (el_val_t)(tm.tm_mon + 1)); - m = el_map_set(m, EL_STR(el_strdup("day")), (el_val_t)tm.tm_mday); - m = el_map_set(m, EL_STR(el_strdup("hour")), (el_val_t)tm.tm_hour); - m = el_map_set(m, EL_STR(el_strdup("minute")), (el_val_t)tm.tm_min); - m = el_map_set(m, EL_STR(el_strdup("second")), (el_val_t)tm.tm_sec); - m = el_map_set(m, EL_STR(el_strdup("ms")), (el_val_t)msec); - return m; -} - -el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz) { - (void)tz; - int64_t s = (int64_t)secs; - int64_t n = (int64_t)ns; - int64_t ms = s * 1000LL + n / 1000000LL; - return (el_val_t)ms; -} - -el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit) { - const char* u = EL_CSTR(unit); - int64_t cur = (int64_t)ts; - int64_t d = (int64_t)n; - int64_t add_ms = d; - if (u) { - if (strcmp(u, "ms") == 0) add_ms = d; - else if (strcmp(u, "sec") == 0) add_ms = d * 1000LL; - else if (strcmp(u, "min") == 0) add_ms = d * 60000LL; - else if (strcmp(u, "hour") == 0) add_ms = d * 3600000LL; - else if (strcmp(u, "day") == 0) add_ms = d * 86400000LL; - } - return (el_val_t)(cur + add_ms); -} - -el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit) { - int64_t d = (int64_t)ts2 - (int64_t)ts1; - const char* u = EL_CSTR(unit); - if (!u || strcmp(u, "ms") == 0) return (el_val_t)d; - if (strcmp(u, "sec") == 0) return (el_val_t)(d / 1000LL); - if (strcmp(u, "min") == 0) return (el_val_t)(d / 60000LL); - if (strcmp(u, "hour") == 0) return (el_val_t)(d / 3600000LL); - if (strcmp(u, "day") == 0) return (el_val_t)(d / 86400000LL); - return (el_val_t)d; -} - -/* Block the calling thread for `secs` seconds. Negative values are clamped - * to 0. Used by El programs that poll external resources (e.g. RunPod - * /status, Engram readiness probes). */ -el_val_t sleep_secs(el_val_t secs) { - int64_t s = (int64_t)secs; - if (s < 0) s = 0; - struct timespec ts; - ts.tv_sec = (time_t)s; - ts.tv_nsec = 0; - nanosleep(&ts, NULL); - return 0; -} - -el_val_t sleep_ms(el_val_t ms) { - int64_t m = (int64_t)ms; - if (m < 0) m = 0; - struct timespec ts; - ts.tv_sec = (time_t)(m / 1000LL); - ts.tv_nsec = (long)((m % 1000LL) * 1000000LL); - nanosleep(&ts, NULL); - return 0; -} - -/* ── Instant + Duration: first-class temporal types ────────────────────────── - * El's substrate (Neuron) is a temporal cognition system. Memory salience - * decay, the six-tier pacemaker, TTL caches, and supersession are all - * temporal. Treating time as a raw Int (now() returning ms-since-epoch and - * arithmetic done with mixed unit literals) lets bugs through the type - * system: `(now - cached_at) < 60` cannot tell ms from sec, and `sleep(30)` - * is ambiguous. This block introduces two dedicated representations. - * - * Representation: - * Instant — int64 nanoseconds since the Unix epoch - * Duration — int64 nanoseconds (signed; negative durations are legal, - * e.g. when a deadline has passed) - * - * Both share the el_val_t (int64) slot the rest of the runtime uses, so no - * boxing / arena allocation is needed. Type discipline is enforced at the - * codegen layer: `let x: Duration = ...` registers `x` in __duration_names, - * and BinOp dispatches through typed wrappers (el_duration_add, etc.) that - * make intent explicit in the generated C. Mismatched ops (Instant+Instant, - * Duration+Int) are surfaced via #error directives at codegen time so the - * downstream cc step fails with a clear El-source-level message. - * - * Nanosecond precision matches POSIX clock_gettime / nanosleep granularity. - * 2^63 nanos covers ~292 years from epoch — comfortably past 2200, plenty - * for a memory-system runtime that never schedules outside a human lifespan. - */ - -/* now() — current Instant. Wraps clock_gettime(CLOCK_REALTIME) for nanosecond - * precision. Falls back to gettimeofday on systems where clock_gettime is - * unavailable (defensive — every supported platform has it). */ -el_val_t el_now_instant(void) { - struct timespec ts; - if (clock_gettime(CLOCK_REALTIME, &ts) == 0) { - int64_t ns = (int64_t)ts.tv_sec * 1000000000LL + (int64_t)ts.tv_nsec; - return (el_val_t)ns; - } - struct timeval tv; - gettimeofday(&tv, NULL); - int64_t ns = (int64_t)tv.tv_sec * 1000000000LL - + (int64_t)tv.tv_usec * 1000LL; - return (el_val_t)ns; -} - -el_val_t now(void) { - return el_now_instant(); -} - -/* unix_seconds(n) — Instant from a Unix-epoch second count. - * unix_millis(n) — Instant from a Unix-epoch millisecond count. */ -el_val_t unix_seconds(el_val_t n) { - int64_t s = (int64_t)n; - return (el_val_t)(s * 1000000000LL); -} - -el_val_t unix_millis(el_val_t n) { - int64_t m = (int64_t)n; - return (el_val_t)(m * 1000000LL); -} - -/* instant_from_iso8601 — parse a strict subset: - * YYYY-MM-DDTHH:MM:SS[.fff]Z - * Returns 0 (the Unix-epoch sentinel) on parse failure. Callers that need to - * distinguish epoch-zero from a parse error should use a wider sentinel - * representation; the current zero-on-failure choice matches existing El - * runtime conventions for parse builtins (str_to_int, parse_int). */ -el_val_t instant_from_iso8601(el_val_t s) { - const char* str = EL_CSTR(s); - if (!str) return (el_val_t)0; - int Y, M, D, h, m, sec, frac = 0; - int n = sscanf(str, "%d-%d-%dT%d:%d:%d.%3d", &Y, &M, &D, &h, &m, &sec, &frac); - if (n < 6) { - n = sscanf(str, "%d-%d-%dT%d:%d:%dZ", &Y, &M, &D, &h, &m, &sec); - if (n < 6) return (el_val_t)0; - } - struct tm tm; - memset(&tm, 0, sizeof(tm)); - tm.tm_year = Y - 1900; - tm.tm_mon = M - 1; - tm.tm_mday = D; - tm.tm_hour = h; - tm.tm_min = m; - tm.tm_sec = sec; - /* timegm — UTC. POSIX-Y but available on macOS and glibc. */ - time_t t = timegm(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (int64_t)frac * 1000000LL; - return (el_val_t)ns; -} - -/* Duration constructors. The El-side postfix literals (30.seconds, 1.hour) - * are lowered by the codegen directly into a literal int64 of nanoseconds — - * these constructors are for runtime values where the count is dynamic. */ -el_val_t el_duration_from_nanos(el_val_t ns) { - return (el_val_t)(int64_t)ns; -} - -el_val_t duration_seconds(el_val_t n) { - int64_t s = (int64_t)n; - return (el_val_t)(s * 1000000000LL); -} - -el_val_t duration_millis(el_val_t n) { - int64_t m = (int64_t)n; - return (el_val_t)(m * 1000000LL); -} - -el_val_t duration_nanos(el_val_t n) { - return (el_val_t)(int64_t)n; -} - -/* Arithmetic — typed wrappers. At the C level these are no-op casts, but - * the codegen routes Instant/Duration BinOps through them so the generated - * C says `el_instant_add_dur(start, dur)` rather than `start + dur`. The - * intent is explicit, the operand order is documented, and a future change - * to the underlying representation (saturating arithmetic, overflow guards) - * has a single chokepoint. */ -el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur) { - return (el_val_t)((int64_t)inst + (int64_t)dur); -} - -el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur) { - return (el_val_t)((int64_t)inst - (int64_t)dur); -} - -el_val_t el_instant_diff(el_val_t a, el_val_t b) { - /* a - b — yields a Duration (negative if b is later than a). */ - return (el_val_t)((int64_t)a - (int64_t)b); -} - -el_val_t el_duration_add(el_val_t a, el_val_t b) { - return (el_val_t)((int64_t)a + (int64_t)b); -} - -el_val_t el_duration_sub(el_val_t a, el_val_t b) { - return (el_val_t)((int64_t)a - (int64_t)b); -} - -el_val_t el_duration_scale(el_val_t dur, el_val_t scalar) { - return (el_val_t)((int64_t)dur * (int64_t)scalar); -} - -el_val_t el_duration_div(el_val_t dur, el_val_t scalar) { - int64_t s = (int64_t)scalar; - if (s == 0) return (el_val_t)0; - return (el_val_t)((int64_t)dur / s); -} - -/* Comparisons. Return 1/0 in el_val_t convention. */ -el_val_t el_instant_lt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a < (int64_t)b ? 1 : 0); } -el_val_t el_instant_le(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a <= (int64_t)b ? 1 : 0); } -el_val_t el_instant_gt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a > (int64_t)b ? 1 : 0); } -el_val_t el_instant_ge(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a >= (int64_t)b ? 1 : 0); } -el_val_t el_instant_eq(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a == (int64_t)b ? 1 : 0); } -el_val_t el_instant_ne(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a != (int64_t)b ? 1 : 0); } -el_val_t el_duration_lt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a < (int64_t)b ? 1 : 0); } -el_val_t el_duration_le(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a <= (int64_t)b ? 1 : 0); } -el_val_t el_duration_gt(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a > (int64_t)b ? 1 : 0); } -el_val_t el_duration_ge(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a >= (int64_t)b ? 1 : 0); } -el_val_t el_duration_eq(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a == (int64_t)b ? 1 : 0); } -el_val_t el_duration_ne(el_val_t a, el_val_t b) { return (el_val_t)((int64_t)a != (int64_t)b ? 1 : 0); } - -/* Conversions. */ -el_val_t instant_to_unix_seconds(el_val_t i) { - return (el_val_t)((int64_t)i / 1000000000LL); -} - -el_val_t instant_to_unix_millis(el_val_t i) { - return (el_val_t)((int64_t)i / 1000000LL); -} - -el_val_t instant_to_iso8601(el_val_t i) { - int64_t ns = (int64_t)i; - time_t s = (time_t)(ns / 1000000000LL); - int msec = (int)((ns / 1000000LL) % 1000LL); - if (msec < 0) { msec += 1000; s -= 1; } - struct tm tm; - gmtime_r(&s, &tm); - char buf[64]; - snprintf(buf, sizeof(buf), "%04d-%02d-%02dT%02d:%02d:%02d.%03dZ", - tm.tm_year + 1900, tm.tm_mon + 1, tm.tm_mday, - tm.tm_hour, tm.tm_min, tm.tm_sec, msec); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t duration_to_seconds(el_val_t d) { - return (el_val_t)((int64_t)d / 1000000000LL); -} - -el_val_t duration_to_millis(el_val_t d) { - return (el_val_t)((int64_t)d / 1000000LL); -} - -el_val_t duration_to_nanos(el_val_t d) { - return (el_val_t)(int64_t)d; -} - -/* sleep(Duration) — Phase 1 replacement for ambiguous sleep(Int). The runtime - * still exposes sleep_secs/sleep_ms for legacy call sites; codegen lowers - * sleep(Duration) to el_sleep_duration(d). Negative durations clamp to 0 so a - * stale deadline doesn't block forever. */ -el_val_t el_sleep_duration(el_val_t dur) { - int64_t ns = (int64_t)dur; - if (ns < 0) ns = 0; - struct timespec ts; - ts.tv_sec = (time_t)(ns / 1000000000LL); - ts.tv_nsec = (long)(ns % 1000000000LL); - nanosleep(&ts, NULL); - return (el_val_t)0; -} - -/* unix_timestamp() — back-compat. Existing El callers expect an Int seconds - * value; this stays an Int returner so the type system isn't disturbed for - * legacy code. New code should call now() and convert when needed. */ -el_val_t unix_timestamp(void) { - return instant_to_unix_seconds(el_now_instant()); -} - -/* TTL cache helpers. Backed by the existing process-wide K/V (state_set/get) - * with a sibling __ttl_set_at_ entry recording the Instant of the last - * write. ttl_cache_get returns "" if the entry is missing or stale, so call - * sites can branch on `if v == "" { miss } else { hit }` — the same shape - * existing get-with-default code uses. No more (now - cached_at) < 60. */ -el_val_t ttl_cache_set(el_val_t key, el_val_t value) { - const char* k = EL_CSTR(key); - if (!k) return (el_val_t)0; - /* Store the value at the user's key. */ - state_set(key, value); - /* Stamp set_at — opaque schema, namespaced under __ttl: prefix so user - * keys can't collide with stamps. */ - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return (el_val_t)0; - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - int64_t now_ns = (int64_t)el_now_instant(); - char buf[32]; - snprintf(buf, sizeof(buf), "%lld", (long long)now_ns); - state_set(EL_STR(stamp_key), EL_STR(buf)); - free(stamp_key); - return (el_val_t)1; -} - -el_val_t ttl_cache_get(el_val_t key, el_val_t max_age) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - /* Look up stamp. */ - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return el_wrap_str(el_strdup("")); - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - el_val_t stamp = state_get(EL_STR(stamp_key)); - free(stamp_key); - const char* sv = EL_CSTR(stamp); - if (!sv || !*sv) return el_wrap_str(el_strdup("")); - int64_t set_at = (int64_t)atoll(sv); - int64_t now_ns = (int64_t)el_now_instant(); - int64_t age = now_ns - set_at; - int64_t max_ns = (int64_t)max_age; - if (age < 0) return el_wrap_str(el_strdup("")); /* clock skew — treat as miss */ - if (age > max_ns) return el_wrap_str(el_strdup("")); /* expired */ - return state_get(key); -} - -el_val_t ttl_cache_age(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return (el_val_t)INT64_MAX; - size_t klen = strlen(k); - char* stamp_key = (char*)malloc(klen + 16); - if (!stamp_key) return (el_val_t)INT64_MAX; - snprintf(stamp_key, klen + 16, "__ttl_at:%s", k); - el_val_t stamp = state_get(EL_STR(stamp_key)); - free(stamp_key); - const char* sv = EL_CSTR(stamp); - if (!sv || !*sv) return (el_val_t)INT64_MAX; - int64_t set_at = (int64_t)atoll(sv); - int64_t now_ns = (int64_t)el_now_instant(); - return (el_val_t)(now_ns - set_at); -} - -/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ────────────── - * Phase 1.5. Calendar is pluggable: EarthCalendar (IANA zones + Gregorian + - * DST), MarsCalendar (sols, MTC), CycleCalendar(period), NoCycleCalendar, - * RelativeCalendar(epoch). Phase 1 zone wrapping folds INTO EarthCalendar; - * UTC and IANA zones are themselves Earth-parochial and cannot live at the - * lowest type layer. - * - * A Rhythm is a small AST that asks the Calendar for cycle phase, weekday, - * etc. Most rhythm logic is calendar-agnostic at runtime: rhythm_cycle_phase - * means "midpoint of cycle" whether the cycle is 24h on Earth or 30h on a - * station or 300y on a long-cycle world. */ - -/* Magic headers — used by the runtime to recognize boxed temporal values - * arriving through el_val_t. Distinct constants so accidental misuse fails - * loudly rather than silently. */ -#define EL_CAL_MAGIC 0xE1CA1EDDU -#define EL_CALTIME_MAGIC 0xE1CA1747U -#define EL_RHYTHM_MAGIC 0xE1287A11U -#define EL_LDATE_MAGIC 0xE1DA7E00U -#define EL_LDT_MAGIC 0xE1DA7E1DU -#define EL_ZONE_MAGIC 0xE12017E0U - -typedef enum { - EL_CALENDAR_EARTH = 1, - EL_CALENDAR_MARS = 2, - EL_CALENDAR_CYCLE = 3, - EL_CALENDAR_NO_CYCLE = 4, - EL_CALENDAR_RELATIVE = 5 -} el_calendar_kind_t; - -typedef struct { - uint32_t magic; - char* id; /* IANA name or "+HH:MM" / "-HH:MM" */ - int fixed; /* 1 for fixed offset, 0 for IANA */ - int64_t offset_ns; /* fixed offset in nanos (only when fixed) */ -} el_zone_t; - -typedef struct { - uint32_t magic; - el_calendar_kind_t kind; - el_zone_t* zone; /* EarthCalendar; MarsCalendar uses MTC */ - int64_t cycle_period_ns;/* CycleCalendar; computed for Earth (86400 s) and Mars (88775.244 s) */ - int64_t epoch_ns; /* RelativeCalendar; Unix-epoch zero otherwise */ -} el_calendar_t; - -typedef struct { - uint32_t magic; - int64_t instant_ns; - el_calendar_t* cal; -} el_caltime_t; - -/* Rhythm AST. */ -typedef enum { - EL_RHYTHM_CYCLE_START = 1, - EL_RHYTHM_CYCLE_PHASE = 2, - EL_RHYTHM_DURATION = 3, - EL_RHYTHM_SESSION_START = 4, - EL_RHYTHM_EVENT = 5, - EL_RHYTHM_AND = 6, - EL_RHYTHM_OR = 7, - EL_RHYTHM_WEEKDAY = 8, - EL_RHYTHM_WEEKLY_AT = 9 -} el_rhythm_kind_t; - -typedef struct el_rhythm_s { - uint32_t magic; - el_rhythm_kind_t kind; - double phase; /* CYCLE_PHASE */ - int64_t period_ns; /* DURATION */ - int weekday; /* 1..7 Mon..Sun */ - int hour; - int minute; - char* event_name; /* EVENT */ - struct el_rhythm_s* a; /* AND/OR */ - struct el_rhythm_s* b; -} el_rhythm_t; - -typedef struct { - uint32_t magic; - int year; - int month; - int day; -} el_localdate_t; - -typedef struct { - uint32_t magic; - el_localdate_t* date; - int64_t time_ns; /* nanos since midnight */ -} el_localdt_t; - -/* Magic-tag check helpers — peek the first 4 bytes of an el_val_t pointer - * and compare against the expected magic. Strings are NUL-terminated and - * never start with our magic byte sequence, so this is safe. */ -static int el_is_magic(el_val_t v, uint32_t want) { - if (v == 0) return 0; - /* Defensive: only follow pointers in plausible address space. - * On 64-bit unix processes pointers are above 0x10000. */ - if ((uint64_t)v < 0x10000ULL) return 0; - uint32_t got = *(volatile uint32_t*)(uintptr_t)v; - return got == want; -} - -/* Sol length on Mars in nanoseconds: 88775.244 seconds. */ -#define EL_MARS_SOL_NS ((int64_t)88775244000000LL) -/* Earth solar day in nanoseconds: 86400 seconds. */ -#define EL_EARTH_DAY_NS ((int64_t)86400000000000LL) - -/* ── Zone construction ────────────────────────────────────────────────────── - * Zones intern by id string so equality comparisons are pointer-compares. */ - -#define EL_ZONE_TABLE_CAP 64 -static el_zone_t* _el_zone_table[EL_ZONE_TABLE_CAP]; -static int _el_zone_count = 0; - -static el_zone_t* _el_zone_intern(const char* id, int fixed, int64_t offset_ns) { - for (int i = 0; i < _el_zone_count; i++) { - el_zone_t* z = _el_zone_table[i]; - if (z->fixed == fixed && z->offset_ns == offset_ns && - strcmp(z->id ? z->id : "", id ? id : "") == 0) { - return z; - } - } - if (_el_zone_count >= EL_ZONE_TABLE_CAP) { - /* Out of slots: build a non-interned zone. Equality will fail across - * such zones but the program still runs. */ - el_zone_t* z = (el_zone_t*)malloc(sizeof(el_zone_t)); - z->magic = EL_ZONE_MAGIC; - z->id = el_strdup_persist(id ? id : ""); - z->fixed = fixed; - z->offset_ns = offset_ns; - return z; - } - el_zone_t* z = (el_zone_t*)malloc(sizeof(el_zone_t)); - z->magic = EL_ZONE_MAGIC; - z->id = el_strdup_persist(id ? id : ""); - z->fixed = fixed; - z->offset_ns = offset_ns; - _el_zone_table[_el_zone_count++] = z; - return z; -} - -el_val_t zone(el_val_t id) { - const char* s = EL_CSTR(id); - if (!s || !*s) return (el_val_t)(uintptr_t)_el_zone_intern("UTC", 0, 0); - /* Fixed-offset shortcut: "+HH:MM" or "-HH:MM". */ - if ((s[0] == '+' || s[0] == '-') && strlen(s) >= 6 && s[3] == ':') { - int sign = (s[0] == '-') ? -1 : 1; - int hh = (s[1] - '0') * 10 + (s[2] - '0'); - int mm = (s[4] - '0') * 10 + (s[5] - '0'); - int64_t off = (int64_t)sign * ((int64_t)hh * 3600LL + (int64_t)mm * 60LL) * 1000000000LL; - return (el_val_t)(uintptr_t)_el_zone_intern(s, 1, off); - } - return (el_val_t)(uintptr_t)_el_zone_intern(s, 0, 0); -} - -el_val_t zone_utc(void) { - return (el_val_t)(uintptr_t)_el_zone_intern("UTC", 1, 0); -} - -el_val_t zone_local(void) { - /* Resolve the local zone via TZ env or system default. tzset() picks - * up TZ if set; otherwise the C library reads /etc/localtime. We store - * the zone id as "LOCAL" so subsequent equality holds; resolution is - * lazy at use time. */ - return (el_val_t)(uintptr_t)_el_zone_intern("LOCAL", 0, 0); -} - -el_val_t zone_offset(el_val_t hours, el_val_t minutes) { - int hh = (int)(int64_t)hours; - int mm = (int)(int64_t)minutes; - int sign = (hh < 0 || mm < 0) ? -1 : 1; - if (hh < 0) hh = -hh; - if (mm < 0) mm = -mm; - int64_t off = (int64_t)sign * ((int64_t)hh * 3600LL + (int64_t)mm * 60LL) * 1000000000LL; - char buf[16]; - snprintf(buf, sizeof(buf), "%c%02d:%02d", sign < 0 ? '-' : '+', hh, mm); - return (el_val_t)(uintptr_t)_el_zone_intern(buf, 1, off); -} - -/* ── Calendar interning ──────────────────────────────────────────────────── */ - -#define EL_CAL_TABLE_CAP 64 -static el_calendar_t* _el_cal_table[EL_CAL_TABLE_CAP]; -static int _el_cal_count = 0; - -static el_calendar_t* _el_cal_intern(el_calendar_kind_t kind, el_zone_t* z, - int64_t period_ns, int64_t epoch_ns) { - for (int i = 0; i < _el_cal_count; i++) { - el_calendar_t* c = _el_cal_table[i]; - if (c->kind == kind && c->zone == z && - c->cycle_period_ns == period_ns && c->epoch_ns == epoch_ns) { - return c; - } - } - el_calendar_t* c = (el_calendar_t*)malloc(sizeof(el_calendar_t)); - c->magic = EL_CAL_MAGIC; - c->kind = kind; - c->zone = z; - c->cycle_period_ns = period_ns; - c->epoch_ns = epoch_ns; - if (_el_cal_count < EL_CAL_TABLE_CAP) _el_cal_table[_el_cal_count++] = c; - return c; -} - -el_val_t earth_calendar(el_val_t z_val) { - el_zone_t* z = NULL; - if (z_val != 0 && el_is_magic(z_val, EL_ZONE_MAGIC)) { - z = (el_zone_t*)(uintptr_t)z_val; - } else { - z = (el_zone_t*)(uintptr_t)zone_local(); - } - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_EARTH, z, EL_EARTH_DAY_NS, 0); -} - -el_val_t earth_calendar_default(void) { - return earth_calendar(zone_local()); -} - -el_val_t mars_calendar(void) { - el_zone_t* z = (el_zone_t*)(uintptr_t)_el_zone_intern("MTC", 1, 0); - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_MARS, z, EL_MARS_SOL_NS, 0); -} - -el_val_t cycle_calendar(el_val_t period_dur) { - int64_t period = (int64_t)period_dur; - if (period <= 0) period = 1; - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_CYCLE, NULL, period, 0); -} - -el_val_t no_cycle_calendar(void) { - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_NO_CYCLE, NULL, 0, 0); -} - -el_val_t relative_calendar(el_val_t epoch_inst) { - int64_t ep = (int64_t)epoch_inst; - return (el_val_t)(uintptr_t)_el_cal_intern(EL_CALENDAR_RELATIVE, NULL, 0, ep); -} - -/* ── CalendarTime ───────────────────────────────────────────────────────── */ - -static el_caltime_t* _el_caltime_alloc(int64_t inst, el_calendar_t* c) { - el_caltime_t* ct = (el_caltime_t*)malloc(sizeof(el_caltime_t)); - ct->magic = EL_CALTIME_MAGIC; - ct->instant_ns = inst; - ct->cal = c; - return ct; -} - -static el_calendar_t* _el_resolve_cal(el_val_t cal_val) { - if (cal_val == 0 || !el_is_magic(cal_val, EL_CAL_MAGIC)) { - return (el_calendar_t*)(uintptr_t)earth_calendar_default(); - } - return (el_calendar_t*)(uintptr_t)cal_val; -} - -el_val_t now_in(el_val_t cal_val) { - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t ns = (int64_t)el_now_instant(); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); -} - -el_val_t in_calendar(el_val_t inst, el_val_t cal_val) { - el_calendar_t* c = _el_resolve_cal(cal_val); - return (el_val_t)(uintptr_t)_el_caltime_alloc((int64_t)inst, c); -} - -el_val_t cal_to_instant(el_val_t ct_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - return (el_val_t)ct->instant_ns; -} - -el_val_t cal_in(el_val_t ct_val, el_val_t cal_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ct->instant_ns, c); -} - -el_val_t cal_cycle_phase(el_val_t ct_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return el_from_float(0.0); - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - el_calendar_t* c = ct->cal; - if (c->kind == EL_CALENDAR_NO_CYCLE) { - return el_from_float(0.0/0.0); /* NaN sentinel */ - } - int64_t period = c->cycle_period_ns; - if (period <= 0) return el_from_float(0.0); - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t phase_ns = base % period; - if (phase_ns < 0) phase_ns += period; - double phase = (double)phase_ns / (double)period; - return el_from_float(phase); -} - -/* ── Earth zone resolution: TZ-based offset lookup ────────────────────────── - * For an EarthCalendar(zone), we want to convert an instant_ns into local - * y/m/d/h/m/s, including DST. Approach: setenv("TZ", id), tzset(), use - * localtime_r, then restore. This is not thread-safe by design — El's - * runtime is single-threaded for the request handler path. Cache the - * computed (instant -> tm) to avoid the syscall churn on repeat formats. */ - -static void _el_apply_zone(el_zone_t* z) { - if (!z) { unsetenv("TZ"); tzset(); return; } - if (z->fixed && strcmp(z->id, "UTC") == 0) { - setenv("TZ", "UTC0", 1); - tzset(); - return; - } - if (z->fixed) { - /* Fixed offset: POSIX TZ uses inverted sign (sign convention of - * "hours WEST of UTC" rather than east). Build the spec accordingly. */ - char buf[32]; - int neg_secs = (int)(-z->offset_ns / 1000000000LL); - int sign = neg_secs < 0 ? -1 : 1; - int abs_secs = neg_secs < 0 ? -neg_secs : neg_secs; - int hh = abs_secs / 3600; - int mm = (abs_secs % 3600) / 60; - snprintf(buf, sizeof(buf), "FIX%c%d:%02d", sign < 0 ? '-' : '+', hh, mm); - setenv("TZ", buf, 1); - tzset(); - return; - } - if (strcmp(z->id, "LOCAL") == 0) { - unsetenv("TZ"); - tzset(); - return; - } - setenv("TZ", z->id, 1); - tzset(); -} - -static int _el_decompose_earth(el_caltime_t* ct, struct tm* tm_out, int* abbr_len, char* abbr_buf, size_t abbr_cap) { - el_calendar_t* c = ct->cal; - el_zone_t* z = c->zone; - _el_apply_zone(z); - time_t s = (time_t)(ct->instant_ns / 1000000000LL); - struct tm tm; - localtime_r(&s, &tm); - *tm_out = tm; - if (abbr_buf && abbr_cap > 0) { - const char* z_str = tm.tm_zone ? tm.tm_zone : ""; - size_t n = strlen(z_str); - if (n >= abbr_cap) n = abbr_cap - 1; - memcpy(abbr_buf, z_str, n); - abbr_buf[n] = '\0'; - if (abbr_len) *abbr_len = (int)n; - } - return 0; -} - -/* Format an Earth CalendarTime under a Java-DateTimeFormatter-ish pattern. - * We support a useful core: yyyy MM dd HH mm ss z EEE MMM d h a — enough for - * the acceptance tests. Single quotes denote literal text. */ -static const char* _el_weekday_short[] = {"Sun","Mon","Tue","Wed","Thu","Fri","Sat"}; -static const char* _el_month_short[] = {"Jan","Feb","Mar","Apr","May","Jun", - "Jul","Aug","Sep","Oct","Nov","Dec"}; - -static char* _el_format_earth(el_caltime_t* ct, const char* pattern) { - struct tm tm; - char abbr[16] = {0}; - int abbr_len = 0; - _el_decompose_earth(ct, &tm, &abbr_len, abbr, sizeof(abbr)); - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - size_t i = 0; - size_t plen = strlen(pattern); - while (i < plen) { - char ch = pattern[i]; - /* Quoted literal */ - if (ch == '\'') { - i++; - while (i < plen && pattern[i] != '\'') { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i++]; - } - if (i < plen) i++; - continue; - } - /* Count run of same letter */ - size_t run = 1; - while (i + run < plen && pattern[i + run] == ch) run++; - char tmp[64]; - tmp[0] = '\0'; - if (ch == 'y') { - if (run >= 4) snprintf(tmp, sizeof(tmp), "%04d", tm.tm_year + 1900); - else snprintf(tmp, sizeof(tmp), "%02d", (tm.tm_year + 1900) % 100); - } else if (ch == 'M') { - if (run >= 3) snprintf(tmp, sizeof(tmp), "%s", _el_month_short[tm.tm_mon]); - else if (run == 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_mon + 1); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_mon + 1); - } else if (ch == 'd') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_mday); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_mday); - } else if (ch == 'H') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_hour); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_hour); - } else if (ch == 'h') { - int h12 = tm.tm_hour % 12; if (h12 == 0) h12 = 12; - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", h12); - else snprintf(tmp, sizeof(tmp), "%d", h12); - } else if (ch == 'm') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_min); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_min); - } else if (ch == 's') { - if (run >= 2) snprintf(tmp, sizeof(tmp), "%02d", tm.tm_sec); - else snprintf(tmp, sizeof(tmp), "%d", tm.tm_sec); - } else if (ch == 'a') { - snprintf(tmp, sizeof(tmp), "%s", tm.tm_hour < 12 ? "AM" : "PM"); - } else if (ch == 'E') { - snprintf(tmp, sizeof(tmp), "%s", _el_weekday_short[tm.tm_wday]); - } else if (ch == 'z') { - snprintf(tmp, sizeof(tmp), "%s", abbr); - } else { - for (size_t k = 0; k < run; k++) { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = ch; - } - i += run; - continue; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - i += run; - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -/* Format a Mars CalendarTime: %sol prints the integer sol number since - * mission epoch (Unix epoch fallback), %phase prints cycle_phase as a - * 0..1 decimal. Other %-specifiers fall through. */ -static char* _el_format_mars(el_caltime_t* ct, const char* pattern) { - el_calendar_t* c = ct->cal; - int64_t period = c->cycle_period_ns > 0 ? c->cycle_period_ns : EL_MARS_SOL_NS; - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t sol = base / period; - int64_t phase_ns = base % period; - if (phase_ns < 0) { phase_ns += period; sol -= 1; } - double phase = (double)phase_ns / (double)period; - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - for (size_t i = 0; pattern[i]; i++) { - if (pattern[i] == '%' && pattern[i+1]) { - char tmp[64]; - tmp[0] = '\0'; - if (strncmp(pattern + i + 1, "sol", 3) == 0) { - snprintf(tmp, sizeof(tmp), "%lld", (long long)sol); - i += 3; - } else if (strncmp(pattern + i + 1, "phase", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%.4f", phase); - i += 5; - } else if (pattern[i+1] == 'd') { - snprintf(tmp, sizeof(tmp), "%lld", (long long)sol); - i += 1; - } else { - tmp[0] = pattern[i+1]; tmp[1] = '\0'; - i += 1; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - } else { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i]; - } - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -/* Format a CycleCalendar CalendarTime: %cycle and %phase. */ -static char* _el_format_cycle(el_caltime_t* ct, const char* pattern) { - el_calendar_t* c = ct->cal; - int64_t period = c->cycle_period_ns > 0 ? c->cycle_period_ns : 1; - int64_t base = ct->instant_ns - c->epoch_ns; - int64_t cycle = base / period; - int64_t phase_ns = base % period; - if (phase_ns < 0) { phase_ns += period; cycle -= 1; } - double phase = (double)phase_ns / (double)period; - size_t cap = strlen(pattern) * 4 + 64; - char* out = (char*)malloc(cap); - size_t pos = 0; - for (size_t i = 0; pattern[i]; i++) { - if (pattern[i] == '%' && pattern[i+1]) { - char tmp[64]; - tmp[0] = '\0'; - if (strncmp(pattern + i + 1, "cycle", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%lld", (long long)cycle); - i += 5; - } else if (strncmp(pattern + i + 1, "phase", 5) == 0) { - snprintf(tmp, sizeof(tmp), "%.4f", phase); - i += 5; - } else if (pattern[i+1] == 'd') { - snprintf(tmp, sizeof(tmp), "%lld", (long long)cycle); - i += 1; - } else if (pattern[i+1] == 'f') { - snprintf(tmp, sizeof(tmp), "%.2f", phase); - i += 1; - } else { - /* Pass through unknown specifier */ - tmp[0] = '%'; tmp[1] = pattern[i+1]; tmp[2] = '\0'; - i += 1; - } - size_t tl = strlen(tmp); - if (pos + tl + 1 >= cap) { cap = (cap + tl) * 2; out = realloc(out, cap); } - memcpy(out + pos, tmp, tl); - pos += tl; - } else { - if (pos + 1 >= cap) { cap *= 2; out = realloc(out, cap); } - out[pos++] = pattern[i]; - } - } - out[pos] = '\0'; - char* result = el_strdup(out); - free(out); - return result; -} - -el_val_t cal_format(el_val_t ct_val, el_val_t pattern_val) { - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return el_wrap_str(el_strdup("")); - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - const char* pat = EL_CSTR(pattern_val); - if (!pat) pat = ""; - char* result = NULL; - switch (ct->cal->kind) { - case EL_CALENDAR_EARTH: result = _el_format_earth(ct, pat); break; - case EL_CALENDAR_MARS: result = _el_format_mars(ct, pat); break; - case EL_CALENDAR_CYCLE: result = _el_format_cycle(ct, pat); break; - case EL_CALENDAR_RELATIVE: result = _el_format_cycle(ct, pat); break; - case EL_CALENDAR_NO_CYCLE: { - char buf[64]; - snprintf(buf, sizeof(buf), "instant:%lld", (long long)ct->instant_ns); - result = el_strdup(buf); - break; - } - default: result = el_strdup(""); - } - return el_wrap_str(result); -} - -/* ── LocalDate / LocalTime / LocalDateTime ──────────────────────────────── */ - -static int _el_days_in_month(int y, int m) { - static const int dim[12] = {31,28,31,30,31,30,31,31,30,31,30,31}; - if (m == 2) { - int leap = ((y % 4 == 0) && (y % 100 != 0)) || (y % 400 == 0); - return 28 + (leap ? 1 : 0); - } - if (m < 1 || m > 12) return 30; - return dim[m - 1]; -} - -el_val_t local_date(el_val_t y, el_val_t m, el_val_t d) { - el_localdate_t* ld = (el_localdate_t*)malloc(sizeof(el_localdate_t)); - ld->magic = EL_LDATE_MAGIC; - ld->year = (int)(int64_t)y; - ld->month = (int)(int64_t)m; - ld->day = (int)(int64_t)d; - return (el_val_t)(uintptr_t)ld; -} - -el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns) { - int64_t hh = (int64_t)h; - int64_t mm = (int64_t)m; - int64_t ss = (int64_t)s; - int64_t nn = (int64_t)ns; - int64_t total = hh * 3600000000000LL + mm * 60000000000LL + ss * 1000000000LL + nn; - return (el_val_t)total; -} - -el_val_t local_datetime(el_val_t date_val, el_val_t time_val) { - if (!el_is_magic(date_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdt_t* ldt = (el_localdt_t*)malloc(sizeof(el_localdt_t)); - ldt->magic = EL_LDT_MAGIC; - ldt->date = (el_localdate_t*)(uintptr_t)date_val; - ldt->time_ns = (int64_t)time_val; - return (el_val_t)(uintptr_t)ldt; -} - -el_val_t zoned(el_val_t date_val, el_val_t time_val, el_val_t cal_val) { - if (!el_is_magic(date_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* ld = (el_localdate_t*)(uintptr_t)date_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t time_ns = (int64_t)time_val; - /* Convert (LocalDate, LocalTime, EarthCalendar) -> Instant. - * For non-Earth calendars we use day-anchored conversion: treat the - * LocalDate's (y,m,d) as a Gregorian projection, convert to seconds via - * mktime under the calendar's zone, then add nanos-since-midnight. */ - if (c->kind == EL_CALENDAR_EARTH) { - _el_apply_zone(c->zone); - struct tm tm; memset(&tm, 0, sizeof(tm)); - tm.tm_year = ld->year - 1900; - tm.tm_mon = ld->month - 1; - tm.tm_mday = ld->day; - tm.tm_hour = (int)(time_ns / 3600000000000LL); - tm.tm_min = (int)((time_ns / 60000000000LL) % 60); - tm.tm_sec = (int)((time_ns / 1000000000LL) % 60); - tm.tm_isdst = -1; - time_t t = mktime(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (time_ns % 1000000000LL); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); - } - /* Non-Earth fallback: project as if Earth UTC then attach calendar. */ - struct tm tm; memset(&tm, 0, sizeof(tm)); - tm.tm_year = ld->year - 1900; - tm.tm_mon = ld->month - 1; - tm.tm_mday = ld->day; - tm.tm_hour = (int)(time_ns / 3600000000000LL); - tm.tm_min = (int)((time_ns / 60000000000LL) % 60); - tm.tm_sec = (int)((time_ns / 1000000000LL) % 60); - time_t t = timegm(&tm); - if (t == (time_t)-1) return (el_val_t)0; - int64_t ns = (int64_t)t * 1000000000LL + (time_ns % 1000000000LL); - return (el_val_t)(uintptr_t)_el_caltime_alloc(ns, c); -} - -el_val_t local_date_year(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->year; -} -el_val_t local_date_month(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->month; -} -el_val_t local_date_day(el_val_t v) { - if (!el_is_magic(v, EL_LDATE_MAGIC)) return (el_val_t)0; - return (el_val_t)((el_localdate_t*)(uintptr_t)v)->day; -} -el_val_t local_time_hour(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)(t / 3600000000000LL); -} -el_val_t local_time_minute(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)((t / 60000000000LL) % 60); -} -el_val_t local_time_second(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)((t / 1000000000LL) % 60); -} -el_val_t local_time_nanos(el_val_t v) { - int64_t t = (int64_t)v; - return (el_val_t)(t % 1000000000LL); -} - -el_val_t el_local_date_add_dur(el_val_t ld_val, el_val_t dur_val) { - if (!el_is_magic(ld_val, EL_LDATE_MAGIC)) return ld_val; - el_localdate_t* ld = (el_localdate_t*)(uintptr_t)ld_val; - int64_t dur_ns = (int64_t)dur_val; - int64_t days = dur_ns / EL_EARTH_DAY_NS; - int y = ld->year, m = ld->month, d = ld->day; - /* Walk days forward/backward in canonical Gregorian. */ - while (days > 0) { - int dim = _el_days_in_month(y, m); - if (d + days <= dim) { d += (int)days; days = 0; break; } - days -= (dim - d + 1); - d = 1; - m++; - if (m > 12) { m = 1; y++; } - } - while (days < 0) { - if (d + days >= 1) { d += (int)days; days = 0; break; } - days += d; - m--; - if (m < 1) { m = 12; y--; } - d = _el_days_in_month(y, m); - } - return local_date((el_val_t)y, (el_val_t)m, (el_val_t)d); -} - -el_val_t el_local_time_add_dur(el_val_t lt_val, el_val_t dur_val) { - int64_t t = (int64_t)lt_val + (int64_t)dur_val; - /* Wrap mod 24h on Earth-default. CycleCalendar wrapping requires the - * caller to use cal_in / cal_format for the right modulus. */ - int64_t day = EL_EARTH_DAY_NS; - int64_t r = t % day; - if (r < 0) r += day; - return (el_val_t)r; -} - -el_val_t el_local_date_lt(el_val_t a_val, el_val_t b_val) { - if (!el_is_magic(a_val, EL_LDATE_MAGIC) || !el_is_magic(b_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* a = (el_localdate_t*)(uintptr_t)a_val; - el_localdate_t* b = (el_localdate_t*)(uintptr_t)b_val; - if (a->year != b->year) return (el_val_t)(a->year < b->year ? 1 : 0); - if (a->month != b->month) return (el_val_t)(a->month < b->month ? 1 : 0); - return (el_val_t)(a->day < b->day ? 1 : 0); -} - -el_val_t el_local_date_eq(el_val_t a_val, el_val_t b_val) { - if (!el_is_magic(a_val, EL_LDATE_MAGIC) || !el_is_magic(b_val, EL_LDATE_MAGIC)) return (el_val_t)0; - el_localdate_t* a = (el_localdate_t*)(uintptr_t)a_val; - el_localdate_t* b = (el_localdate_t*)(uintptr_t)b_val; - return (el_val_t)((a->year == b->year && a->month == b->month && a->day == b->day) ? 1 : 0); -} - -/* ── Rhythm ──────────────────────────────────────────────────────────────── */ - -static el_rhythm_t* _el_rhythm_alloc(el_rhythm_kind_t k) { - el_rhythm_t* r = (el_rhythm_t*)calloc(1, sizeof(el_rhythm_t)); - r->magic = EL_RHYTHM_MAGIC; - r->kind = k; - return r; -} - -el_val_t rhythm_cycle_start(void) { - return (el_val_t)(uintptr_t)_el_rhythm_alloc(EL_RHYTHM_CYCLE_START); -} - -el_val_t rhythm_cycle_phase(el_val_t phase_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_CYCLE_PHASE); - r->phase = el_to_float(phase_val); - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_duration(el_val_t d_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_DURATION); - r->period_ns = (int64_t)d_val; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_session_start(void) { - return (el_val_t)(uintptr_t)_el_rhythm_alloc(EL_RHYTHM_SESSION_START); -} - -el_val_t rhythm_event(el_val_t name_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_EVENT); - const char* n = EL_CSTR(name_val); - r->event_name = el_strdup_persist(n ? n : ""); - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_and(el_val_t a_val, el_val_t b_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_AND); - r->a = el_is_magic(a_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)a_val : NULL; - r->b = el_is_magic(b_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)b_val : NULL; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_or(el_val_t a_val, el_val_t b_val) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_OR); - r->a = el_is_magic(a_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)a_val : NULL; - r->b = el_is_magic(b_val, EL_RHYTHM_MAGIC) ? (el_rhythm_t*)(uintptr_t)b_val : NULL; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_weekday(el_val_t day) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_WEEKDAY); - r->weekday = (int)(int64_t)day; - return (el_val_t)(uintptr_t)r; -} - -el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute) { - el_rhythm_t* r = _el_rhythm_alloc(EL_RHYTHM_WEEKLY_AT); - r->weekday = (int)(int64_t)day; - r->hour = (int)(int64_t)hour; - r->minute = (int)(int64_t)minute; - return (el_val_t)(uintptr_t)r; -} - -/* Compute the next instant on or after `after` when rhythm `r` matches, - * under calendar `cal`. */ -static int64_t _el_next_after(el_rhythm_t* r, int64_t after_ns, el_calendar_t* cal) { - if (!r) return after_ns; - int64_t period = cal->cycle_period_ns > 0 ? cal->cycle_period_ns : EL_EARTH_DAY_NS; - switch (r->kind) { - case EL_RHYTHM_CYCLE_START: { - int64_t base = after_ns - cal->epoch_ns; - int64_t cyc = (base / period) + 1; - return cal->epoch_ns + cyc * period; - } - case EL_RHYTHM_CYCLE_PHASE: { - int64_t base = after_ns - cal->epoch_ns; - int64_t cyc_ns = (int64_t)(r->phase * (double)period); - int64_t cur_cyc = base / period; - int64_t candidate = cal->epoch_ns + cur_cyc * period + cyc_ns; - if (candidate <= after_ns) candidate += period; - return candidate; - } - case EL_RHYTHM_DURATION: { - return after_ns + (r->period_ns > 0 ? r->period_ns : 1); - } - case EL_RHYTHM_WEEKDAY: - case EL_RHYTHM_WEEKLY_AT: { - if (cal->kind != EL_CALENDAR_EARTH) { - /* Non-Earth calendars: fall back to cycle math, treating - * weekday as a 7-cycle-per-period proxy. */ - return after_ns + period; - } - _el_apply_zone(cal->zone); - time_t s = (time_t)(after_ns / 1000000000LL); - struct tm tm; - localtime_r(&s, &tm); - /* tm_wday: 0=Sun..6=Sat. We use 1=Mon..7=Sun. */ - int target = r->weekday >= 1 && r->weekday <= 7 ? r->weekday : 1; - int target_wday = target == 7 ? 0 : target; /* 7→Sun=0, 1→Mon=1 */ - int days_ahead = (target_wday - tm.tm_wday + 7) % 7; - int hour = (r->kind == EL_RHYTHM_WEEKLY_AT) ? r->hour : 0; - int minute = (r->kind == EL_RHYTHM_WEEKLY_AT) ? r->minute : 0; - struct tm cand = tm; - cand.tm_mday += days_ahead; - cand.tm_hour = hour; - cand.tm_min = minute; - cand.tm_sec = 0; - cand.tm_isdst = -1; - time_t cand_t = mktime(&cand); - int64_t cand_ns = (int64_t)cand_t * 1000000000LL; - if (cand_ns <= after_ns) { - cand.tm_mday += 7; - cand.tm_isdst = -1; - cand_t = mktime(&cand); - cand_ns = (int64_t)cand_t * 1000000000LL; - } - return cand_ns; - } - case EL_RHYTHM_AND: { - int64_t a = _el_next_after(r->a, after_ns, cal); - int64_t b = _el_next_after(r->b, after_ns, cal); - return a > b ? a : b; - } - case EL_RHYTHM_OR: { - int64_t a = _el_next_after(r->a, after_ns, cal); - int64_t b = _el_next_after(r->b, after_ns, cal); - return a < b ? a : b; - } - case EL_RHYTHM_SESSION_START: - case EL_RHYTHM_EVENT: - default: - return after_ns; - } -} - -el_val_t rhythm_next_after(el_val_t r_val, el_val_t after_val, el_val_t cal_val) { - if (!el_is_magic(r_val, EL_RHYTHM_MAGIC)) return after_val; - el_rhythm_t* r = (el_rhythm_t*)(uintptr_t)r_val; - el_calendar_t* c = _el_resolve_cal(cal_val); - int64_t out = _el_next_after(r, (int64_t)after_val, c); - return (el_val_t)out; -} - -el_val_t rhythm_matches(el_val_t r_val, el_val_t ct_val) { - if (!el_is_magic(r_val, EL_RHYTHM_MAGIC)) return (el_val_t)0; - if (!el_is_magic(ct_val, EL_CALTIME_MAGIC)) return (el_val_t)0; - el_rhythm_t* r = (el_rhythm_t*)(uintptr_t)r_val; - el_caltime_t* ct = (el_caltime_t*)(uintptr_t)ct_val; - int64_t period = ct->cal->cycle_period_ns > 0 ? ct->cal->cycle_period_ns : EL_EARTH_DAY_NS; - int64_t base = ct->instant_ns - ct->cal->epoch_ns; - int64_t phase_ns = base % period; - if (phase_ns < 0) phase_ns += period; - double phase = (double)phase_ns / (double)period; - switch (r->kind) { - case EL_RHYTHM_CYCLE_START: return (el_val_t)(phase_ns == 0 ? 1 : 0); - case EL_RHYTHM_CYCLE_PHASE: { - double diff = phase - r->phase; - if (diff < 0) diff = -diff; - return (el_val_t)(diff < 0.001 ? 1 : 0); - } - default: return (el_val_t)0; - } -} - -/* ── UUID v4 ─────────────────────────────────────────────────────────────── */ - -static int _el_uuid_seeded = 0; - -static void _el_uuid_seed(void) { - if (!_el_uuid_seeded) { - srand((unsigned)time(NULL) ^ (unsigned)(uintptr_t)&_el_uuid_seeded); - _el_uuid_seeded = 1; - } -} - -el_val_t uuid_new(void) { - _el_uuid_seed(); - unsigned char b[16]; - for (int i = 0; i < 16; i++) b[i] = (unsigned char)(rand() & 0xff); - /* Version 4 */ - b[6] = (b[6] & 0x0f) | 0x40; - /* RFC 4122 variant */ - b[8] = (b[8] & 0x3f) | 0x80; - char buf[37]; - snprintf(buf, sizeof(buf), - "%02x%02x%02x%02x-%02x%02x-%02x%02x-%02x%02x-%02x%02x%02x%02x%02x%02x", - b[0], b[1], b[2], b[3], - b[4], b[5], - b[6], b[7], - b[8], b[9], - b[10], b[11], b[12], b[13], b[14], b[15]); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t uuid_v4(void) { return uuid_new(); } - -/* ── Environment ─────────────────────────────────────────────────────────── */ - -el_val_t env(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - const char* v = getenv(k); - return el_wrap_str(el_strdup(v ? v : "")); -} - -/* ── In-process state K/V ────────────────────────────────────────────────── */ - -typedef struct { - char* key; - char* value; -} StateEntry; - -static StateEntry* _state_entries = NULL; -static size_t _state_count = 0; -static size_t _state_cap = 0; -/* Mutex protecting all _state_entries access. state_set/state_get are called - * concurrently from 64 HTTP worker threads — without this lock, realloc and - * free race, producing corruption, double-free, and segfaults. */ -static pthread_mutex_t _state_mu = PTHREAD_MUTEX_INITIALIZER; - -static StateEntry* state_find(const char* key) { - for (size_t i = 0; i < _state_count; i++) { - if (strcmp(_state_entries[i].key, key) == 0) return &_state_entries[i]; - } - return NULL; -} - -el_val_t state_set(el_val_t key, el_val_t value) { - const char* k = EL_CSTR(key); - const char* v = EL_CSTR(value); - if (!k) return 0; - if (!v) v = ""; - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - if (e) { - free(e->value); - e->value = el_strdup_persist(v); - pthread_mutex_unlock(&_state_mu); - return 1; - } - if (_state_count >= _state_cap) { - size_t nc = _state_cap == 0 ? 16 : _state_cap * 2; - StateEntry* grown = realloc(_state_entries, nc * sizeof(StateEntry)); - if (!grown) { pthread_mutex_unlock(&_state_mu); fputs("el_runtime: out of memory\n", stderr); exit(1); } - _state_entries = grown; - _state_cap = nc; - } - _state_entries[_state_count].key = el_strdup_persist(k); - _state_entries[_state_count].value = el_strdup_persist(v); - _state_count++; - pthread_mutex_unlock(&_state_mu); - return 1; -} - -el_val_t state_get(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return el_wrap_str(el_strdup("")); - pthread_mutex_lock(&_state_mu); - StateEntry* e = state_find(k); - char* result = el_strdup_persist(e ? e->value : ""); - pthread_mutex_unlock(&_state_mu); - /* wrap in arena-tracked copy for the caller's request lifetime */ - char* copy = el_strdup(result); - return el_wrap_str(copy); -} - -el_val_t state_del(el_val_t key) { - const char* k = EL_CSTR(key); - if (!k) return 0; - pthread_mutex_lock(&_state_mu); - for (size_t i = 0; i < _state_count; i++) { - if (strcmp(_state_entries[i].key, k) == 0) { - free(_state_entries[i].key); - free(_state_entries[i].value); - for (size_t j = i + 1; j < _state_count; j++) { - _state_entries[j - 1] = _state_entries[j]; - } - _state_count--; - pthread_mutex_unlock(&_state_mu); - return 1; - } - } - pthread_mutex_unlock(&_state_mu); - return 1; -} - -el_val_t state_keys(void) { - pthread_mutex_lock(&_state_mu); - el_val_t lst = el_list_empty(); - for (size_t i = 0; i < _state_count; i++) { - lst = el_list_append(lst, el_wrap_str(el_strdup(_state_entries[i].key))); - } - pthread_mutex_unlock(&_state_mu); - return lst; -} - -/* ── Float formatting ────────────────────────────────────────────────────── */ - -el_val_t float_to_str(el_val_t f) { - char buf[64]; - snprintf(buf, sizeof(buf), "%g", el_to_float(f)); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t int_to_float(el_val_t n) { - return el_from_float((double)(int64_t)n); -} - -el_val_t float_to_int(el_val_t f) { - return (el_val_t)(int64_t)el_to_float(f); -} - -el_val_t format_float(el_val_t f, el_val_t decimals) { - int d = (int)(int64_t)decimals; - if (d < 0) d = 0; - if (d > 30) d = 30; - char buf[128]; - snprintf(buf, sizeof(buf), "%.*f", d, el_to_float(f)); - return el_wrap_str(el_strdup(buf)); -} - -el_val_t decimal_round(el_val_t f, el_val_t decimals) { - int d = (int)(int64_t)decimals; - if (d < 0) d = 0; - if (d > 15) d = 15; - double mul = pow(10.0, (double)d); - double v = el_to_float(f); - double r = (v >= 0.0 ? floor(v * mul + 0.5) : -floor(-v * mul + 0.5)) / mul; - return el_from_float(r); -} - -el_val_t str_to_float(el_val_t s) { - const char* str = EL_CSTR(s); - if (!str) return el_from_float(0.0); - return el_from_float(strtod(str, NULL)); -} - -/* ── Math (Float-aware) ──────────────────────────────────────────────────── */ - -el_val_t math_sqrt(el_val_t f) { return el_from_float(sqrt(el_to_float(f))); } -el_val_t math_log(el_val_t f) { return el_from_float(log(el_to_float(f))); } -el_val_t math_ln(el_val_t f) { return el_from_float(log(el_to_float(f))); } -el_val_t math_sin(el_val_t f) { return el_from_float(sin(el_to_float(f))); } -el_val_t math_cos(el_val_t f) { return el_from_float(cos(el_to_float(f))); } -el_val_t math_pi(void) { return el_from_float(3.141592653589793238462643383279502884); } - -/* ── String additions ────────────────────────────────────────────────────── */ - -el_val_t str_index_of(el_val_t s, el_val_t sub) { - const char* str = EL_CSTR(s); - const char* sb = EL_CSTR(sub); - if (!str || !sb) return -1; - const char* hit = strstr(str, sb); - if (!hit) return -1; - return (el_val_t)(int64_t)(hit - str); -} - -el_val_t str_split(el_val_t s, el_val_t sep) { - const char* str = EL_CSTR(s); - const char* sp = EL_CSTR(sep); - el_val_t lst = el_list_empty(); - if (!str) return lst; - if (!sp || !*sp) { - lst = el_list_append(lst, el_wrap_str(el_strdup(str))); - return lst; - } - size_t lp = strlen(sp); - const char* p = str; - const char* hit; - while ((hit = strstr(p, sp)) != NULL) { - size_t n = (size_t)(hit - p); - char* out = el_strbuf(n); - memcpy(out, p, n); - out[n] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - p = hit + lp; - } - lst = el_list_append(lst, el_wrap_str(el_strdup(p))); - return lst; -} - -el_val_t str_char_at(el_val_t s, el_val_t i) { - const char* str = EL_CSTR(s); - int64_t idx = (int64_t)i; - if (!str) return el_wrap_str(el_strdup("")); - int64_t n = (int64_t)strlen(str); - if (idx < 0 || idx >= n) return el_wrap_str(el_strdup("")); - char buf[2]; - buf[0] = str[idx]; - buf[1] = '\0'; - return el_wrap_str(el_strdup(buf)); -} - -el_val_t str_char_code(el_val_t s, el_val_t i) { - const char* str = EL_CSTR(s); - int64_t idx = (int64_t)i; - if (!str) return 0; - int64_t n = (int64_t)strlen(str); - if (idx < 0 || idx >= n) return 0; - return (el_val_t)(unsigned char)str[idx]; -} - -static el_val_t str_pad(const char* s, int64_t width, const char* pad, int left) { - if (!s) s = ""; - if (!pad || !*pad) pad = " "; - int64_t lp = (int64_t)strlen(pad); - int64_t ls = (int64_t)strlen(s); - if (ls >= width) return el_wrap_str(el_strdup(s)); - int64_t need = width - ls; - char* out = el_strbuf((size_t)width); - if (left) { - for (int64_t i = 0; i < need; i++) out[i] = pad[i % lp]; - memcpy(out + need, s, (size_t)ls); - } else { - memcpy(out, s, (size_t)ls); - for (int64_t i = 0; i < need; i++) out[ls + i] = pad[i % lp]; - } - out[width] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad) { - return str_pad(EL_CSTR(s), (int64_t)width, EL_CSTR(pad), 1); -} - -el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad) { - return str_pad(EL_CSTR(s), (int64_t)width, EL_CSTR(pad), 0); -} - -el_val_t str_format(el_val_t fmt, el_val_t data) { - const char* tpl = EL_CSTR(fmt); - if (!tpl) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - const char* p = tpl; - while (*p) { - if (*p == '{') { - const char* q = p + 1; - while (*q && *q != '}') q++; - if (*q == '}') { - size_t klen = (size_t)(q - p - 1); - char keybuf[256]; - if (klen < sizeof(keybuf)) { - memcpy(keybuf, p + 1, klen); - keybuf[klen] = '\0'; - el_val_t v = el_map_get(data, EL_STR(keybuf)); - if (v != 0 && looks_like_string(v)) { - jb_puts(&b, EL_CSTR(v)); - p = q + 1; - continue; - } else if (v != 0) { - jb_emit_int(&b, (int64_t)v); - p = q + 1; - continue; - } - } - /* Unknown key — leave {key} verbatim */ - jb_reserve(&b, klen + 2); - memcpy(b.buf + b.len, p, klen + 2); - b.len += klen + 2; - b.buf[b.len] = '\0'; - p = q + 1; - continue; - } - } - jb_putc(&b, *p); - p++; - } - return el_wrap_str(b.buf); -} - -el_val_t str_lower(el_val_t s) { return str_to_lower(s); } -el_val_t str_upper(el_val_t s) { return str_to_upper(s); } - -/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes) - * - * Phase 1 covers the operations every text-handling caller used to roll by - * hand on top of str_index_of + str_slice. The character-class predicates - * (is_letter / is_digit / ...) are ASCII only — Unicode-grapheme awareness, - * NFC/NFD normalization, and regex are Phase 2. Single-char input checks the - * first byte; multi-char input requires ALL bytes to match (false otherwise). - * - * Counting: - * str_count non-overlapping occurrences of sub in s - * str_count_chars codepoint count (UTF-8 leading-byte count) - * str_count_bytes explicit byte length (alias of str_len) - * str_count_lines \n-delimited line count (\r\n folded to \n) - * str_count_words whitespace-delimited tokens, non-empty only - * str_count_letters ASCII [A-Za-z] - * str_count_digits ASCII [0-9] - * - * Find / position: - * str_index_of_all all byte offsets of sub, [] if none - * str_last_index_of last byte offset of sub, -1 if not found - * str_find_chars first index of any char in any_of, -1 if none - * - * Transform: - * str_repeat s * n (non-negative) - * str_reverse codepoint-reversed (NOT grapheme-aware) - * str_strip_prefix s without prefix if present, else s - * str_strip_suffix s without suffix if present, else s - * str_strip_chars strip leading+trailing chars matching any in chars - * str_lstrip strip leading whitespace - * str_rstrip strip trailing whitespace - * - * Char classification (Bool): - * is_letter, is_digit, is_alphanumeric, is_whitespace, - * is_punctuation, is_uppercase, is_lowercase - * - * Splitting: - * str_split_lines \n-delimited (\r\n folded). Trailing empty dropped. - * str_split_chars alias of native_string_chars in str_ namespace - * str_split_n split into at most n parts (last part keeps the - * rest verbatim, including any further separators) - * - * Joining: - * str_join [String] -> String, sep between elements - */ - -/* Count non-overlapping occurrences of sub in s. Empty sub returns 0. */ -el_val_t str_count(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub || !*sub) return 0; - size_t lp = strlen(sub); - int64_t count = 0; - const char* p = s; - while ((p = strstr(p, sub)) != NULL) { - count++; - p += lp; /* non-overlapping advance */ - } - return (el_val_t)count; -} - -/* Codepoint count: walk bytes, count those NOT matching 10xxxxxx. */ -el_val_t str_count_chars(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if ((*p & 0xC0) != 0x80) count++; - } - return (el_val_t)count; -} - -el_val_t str_count_bytes(el_val_t sv) { - return str_len(sv); -} - -el_val_t str_count_lines(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s || !*s) return 0; - int64_t count = 0; - int has_content = 0; - for (const char* p = s; *p; p++) { - has_content = 1; - if (*p == '\n') { - count++; - has_content = 0; /* the \n closed the line */ - } - } - if (has_content) count++; /* trailing line with no terminator */ - return (el_val_t)count; -} - -el_val_t str_count_words(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - int in_word = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (isspace(*p)) { - in_word = 0; - } else if (!in_word) { - in_word = 1; - count++; - } - } - return (el_val_t)count; -} - -el_val_t str_count_letters(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if ((*p >= 'A' && *p <= 'Z') || (*p >= 'a' && *p <= 'z')) count++; - } - return (el_val_t)count; -} - -el_val_t str_count_digits(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return 0; - int64_t count = 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (*p >= '0' && *p <= '9') count++; - } - return (el_val_t)count; -} - -el_val_t str_index_of_all(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - el_val_t lst = el_list_empty(); - if (!s || !sub || !*sub) return lst; - size_t lp = strlen(sub); - const char* p = s; - const char* hit; - while ((hit = strstr(p, sub)) != NULL) { - lst = el_list_append(lst, (el_val_t)(int64_t)(hit - s)); - p = hit + lp; - } - return lst; -} - -el_val_t str_last_index_of(el_val_t sv, el_val_t subv) { - const char* s = EL_CSTR(sv); - const char* sub = EL_CSTR(subv); - if (!s || !sub || !*sub) return -1; - size_t lp = strlen(sub); - int64_t last = -1; - const char* p = s; - const char* hit; - while ((hit = strstr(p, sub)) != NULL) { - last = (int64_t)(hit - s); - p = hit + lp; - } - return (el_val_t)last; -} - -el_val_t str_find_chars(el_val_t sv, el_val_t any_of_v) { - const char* s = EL_CSTR(sv); - const char* any = EL_CSTR(any_of_v); - if (!s || !any || !*any) return -1; - for (const char* p = s; *p; p++) { - if (strchr(any, *p)) return (el_val_t)(int64_t)(p - s); - } - return -1; -} - -el_val_t str_repeat(el_val_t sv, el_val_t nv) { - const char* s = EL_CSTR(sv); - int64_t n = (int64_t)nv; - if (!s || n <= 0) return el_wrap_str(el_strdup("")); - size_t ls = strlen(s); - if (ls == 0) return el_wrap_str(el_strdup("")); - size_t total = ls * (size_t)n; - char* out = el_strbuf(total); - for (int64_t i = 0; i < n; i++) { - memcpy(out + i * ls, s, ls); - } - out[total] = '\0'; - return el_wrap_str(out); -} - -/* Reverse by codepoint: walk codepoints, copy each backwards into the output. - * NOT grapheme-aware (Phase 2). Combining marks attached to a base codepoint - * will detach. ASCII strings are byte-reverse equivalent. */ -el_val_t str_reverse(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - char* out = el_strbuf(n); - /* Walk forward, find each codepoint's byte length, then copy from the end. */ - size_t out_pos = n; - const unsigned char* p = (const unsigned char*)s; - while (*p) { - int cp_len; - if ((*p & 0x80) == 0x00) cp_len = 1; - else if ((*p & 0xE0) == 0xC0) cp_len = 2; - else if ((*p & 0xF0) == 0xE0) cp_len = 3; - else if ((*p & 0xF8) == 0xF0) cp_len = 4; - else cp_len = 1; /* invalid byte: passthrough */ - out_pos -= cp_len; - memcpy(out + out_pos, p, cp_len); - p += cp_len; - } - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_strip_prefix(el_val_t sv, el_val_t prefv) { - const char* s = EL_CSTR(sv); - const char* pref = EL_CSTR(prefv); - if (!s) return el_wrap_str(el_strdup("")); - if (!pref || !*pref) return el_wrap_str(el_strdup(s)); - size_t lp = strlen(pref); - size_t ls = strlen(s); - if (lp <= ls && strncmp(s, pref, lp) == 0) { - char* out = el_strbuf(ls - lp); - memcpy(out, s + lp, ls - lp); - out[ls - lp] = '\0'; - return el_wrap_str(out); - } - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_strip_suffix(el_val_t sv, el_val_t sufv) { - const char* s = EL_CSTR(sv); - const char* suf = EL_CSTR(sufv); - if (!s) return el_wrap_str(el_strdup("")); - if (!suf || !*suf) return el_wrap_str(el_strdup(s)); - size_t ls = strlen(s); - size_t lsuf = strlen(suf); - if (lsuf <= ls && strcmp(s + ls - lsuf, suf) == 0) { - char* out = el_strbuf(ls - lsuf); - memcpy(out, s, ls - lsuf); - out[ls - lsuf] = '\0'; - return el_wrap_str(out); - } - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_strip_chars(el_val_t sv, el_val_t charsv) { - const char* s = EL_CSTR(sv); - const char* chars = EL_CSTR(charsv); - if (!s) return el_wrap_str(el_strdup("")); - if (!chars || !*chars) return el_wrap_str(el_strdup(s)); - const char* start = s; - while (*start && strchr(chars, *start)) start++; - size_t n = strlen(start); - while (n > 0 && strchr(chars, start[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, start, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -el_val_t str_lstrip(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - while (*s && isspace((unsigned char)*s)) s++; - return el_wrap_str(el_strdup(s)); -} - -el_val_t str_rstrip(el_val_t sv) { - const char* s = EL_CSTR(sv); - if (!s) return el_wrap_str(el_strdup("")); - size_t n = strlen(s); - while (n > 0 && isspace((unsigned char)s[n - 1])) n--; - char* out = el_strbuf(n); - memcpy(out, s, n); - out[n] = '\0'; - return el_wrap_str(out); -} - -/* Character classification. - * Empty input returns false. Multi-char input requires ALL bytes to match. - * ASCII range only; Phase 2 will widen to Unicode. */ -static int s_all_match(el_val_t sv, int (*pred)(unsigned char)) { - const char* s = EL_CSTR(sv); - if (!s || !*s) return 0; - for (const unsigned char* p = (const unsigned char*)s; *p; p++) { - if (!pred(*p)) return 0; - } - return 1; -} - -static int p_letter(unsigned char c) { return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z'); } -static int p_digit(unsigned char c) { return c >= '0' && c <= '9'; } -static int p_alnum(unsigned char c) { return p_letter(c) || p_digit(c); } -static int p_white(unsigned char c) { return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f' || c == '\v'; } -static int p_punct(unsigned char c) { return ispunct(c) ? 1 : 0; } -static int p_upper(unsigned char c) { return c >= 'A' && c <= 'Z'; } -static int p_lower(unsigned char c) { return c >= 'a' && c <= 'z'; } - -el_val_t is_letter(el_val_t s) { return (el_val_t)s_all_match(s, p_letter); } -el_val_t is_digit(el_val_t s) { return (el_val_t)s_all_match(s, p_digit); } -el_val_t is_alphanumeric(el_val_t s) { return (el_val_t)s_all_match(s, p_alnum); } -el_val_t is_whitespace(el_val_t s) { return (el_val_t)s_all_match(s, p_white); } -el_val_t is_punctuation(el_val_t s) { return (el_val_t)s_all_match(s, p_punct); } -el_val_t is_uppercase(el_val_t s) { return (el_val_t)s_all_match(s, p_upper); } -el_val_t is_lowercase(el_val_t s) { return (el_val_t)s_all_match(s, p_lower); } - -/* Split on \n. \r\n is folded to \n first. Trailing empty after final \n - * is dropped (so "a\nb\n" -> ["a", "b"], not ["a", "b", ""]). */ -el_val_t str_split_lines(el_val_t sv) { - const char* s = EL_CSTR(sv); - el_val_t lst = el_list_empty(); - if (!s) return lst; - size_t n = strlen(s); - /* Pre-scan: build into a normalized buffer with \r\n folded. */ - const char* line_start = s; - for (size_t i = 0; i <= n; i++) { - if (s[i] == '\n' || s[i] == '\0') { - size_t len = (size_t)(s + i - line_start); - /* Drop trailing \r if this was \r\n. */ - if (len > 0 && line_start[len - 1] == '\r') len--; - /* Drop final trailing-empty-after-newline. */ - if (s[i] == '\0' && len == 0 && i > 0 && s[i - 1] == '\n') break; - char* out = el_strbuf(len); - memcpy(out, line_start, len); - out[len] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - if (s[i] == '\0') break; - line_start = s + i + 1; - } - } - return lst; -} - -el_val_t str_split_chars(el_val_t s) { - return native_string_chars(s); -} - -/* Split into at most n parts. The (n-1)th split point is the LAST split; - * after it, the remainder is appended verbatim including any further - * separators. n <= 0 returns an empty list. n == 1 returns [s]. */ -el_val_t str_split_n(el_val_t sv, el_val_t sepv, el_val_t nv) { - const char* s = EL_CSTR(sv); - const char* sep = EL_CSTR(sepv); - int64_t n = (int64_t)nv; - el_val_t lst = el_list_empty(); - if (!s) return lst; - if (n <= 0) return lst; - if (n == 1 || !sep || !*sep) { - lst = el_list_append(lst, el_wrap_str(el_strdup(s))); - return lst; - } - size_t lp = strlen(sep); - const char* p = s; - int64_t parts = 0; - const char* hit; - while (parts < n - 1 && (hit = strstr(p, sep)) != NULL) { - size_t len = (size_t)(hit - p); - char* out = el_strbuf(len); - memcpy(out, p, len); - out[len] = '\0'; - lst = el_list_append(lst, el_wrap_str(out)); - p = hit + lp; - parts++; - } - /* Remainder verbatim. */ - lst = el_list_append(lst, el_wrap_str(el_strdup(p))); - return lst; -} - -/* Join a [String] with a separator. Empty list -> "". Single-element -> - * that element. Non-string elements are stringified via int_to_str. */ -el_val_t str_join(el_val_t listv, el_val_t sepv) { - return list_join(listv, sepv); -} - -/* ── List additions ──────────────────────────────────────────────────────── */ - -el_val_t list_push(el_val_t list, el_val_t elem) { - return el_list_append(list, elem); -} - -el_val_t list_push_front(el_val_t listv, el_val_t elem) { - ElList* lst = (ElList*)(uintptr_t)listv; - if (!lst) { - el_val_t nl = el_list_empty(); - return el_list_append(nl, elem); - } - /* Append to grow capacity, then shift right */ - listv = el_list_append(listv, elem); - lst = (ElList*)(uintptr_t)listv; - for (int64_t i = lst->length - 1; i > 0; i--) { - lst->elems[i] = lst->elems[i - 1]; - } - lst->elems[0] = elem; - return EL_STR(lst); -} - -el_val_t list_join(el_val_t listv, el_val_t sep) { - ElList* lst = (ElList*)(uintptr_t)listv; - const char* sp = EL_CSTR(sep); - if (!sp) sp = ""; - if (!lst || lst->length == 0) return el_wrap_str(el_strdup("")); - JsonBuf b; jb_init(&b); - for (int64_t i = 0; i < lst->length; i++) { - if (i > 0) jb_puts(&b, sp); - el_val_t v = lst->elems[i]; - if (v == 0) continue; - if (looks_like_string(v)) { - jb_puts(&b, EL_CSTR(v)); - } else { - char tmp[32]; - snprintf(tmp, sizeof(tmp), "%lld", (long long)v); - jb_puts(&b, tmp); - } - } - return el_wrap_str(b.buf); -} - -el_val_t list_range(el_val_t start, el_val_t end) { - int64_t a = (int64_t)start; - int64_t b = (int64_t)end; - el_val_t lst = el_list_empty(); - for (int64_t i = a; i < b; i++) lst = el_list_append(lst, (el_val_t)i); - return lst; -} - -/* ── Bool helpers ────────────────────────────────────────────────────────── */ - -el_val_t bool_to_str(el_val_t b) { - return el_wrap_str(el_strdup(b ? "true" : "false")); -} - -/* ── Numeric parsing ─────────────────────────────────────────────────────── */ - -/* parse_int — strtoll with a default. str_to_int already exists but does not - * distinguish "0" from a parse failure, so callers that need a sentinel use - * this. Skips leading whitespace; accepts an optional leading +/-; returns - * default_val on empty input or no consumed digits. Trailing junk is ignored - * (atoi-style). */ -el_val_t parse_int(el_val_t sv, el_val_t default_val) { - const char* s = EL_CSTR(sv); - if (!s) return default_val; - while (*s == ' ' || *s == '\t' || *s == '\n' || *s == '\r') s++; - if (*s == '\0') return default_val; - char* end = NULL; - long long n = strtoll(s, &end, 10); - if (end == s) return default_val; - return (el_val_t)n; -} - -/* ── Process ─────────────────────────────────────────────────────────────── */ - -void exit_program(el_val_t code) { - exit((int)code); -} - -/* getpid_now — current process id. Named with the _now suffix to avoid - * colliding with the libc `getpid` declaration that the runtime already - * sees via (calling it `getpid` would fight the prototype). */ -el_val_t getpid_now(void) { - return (el_val_t)getpid(); -} - -/* ── args() — command-line argument access ────────────────────────────────── - * Compiled El programs call args() to get a list of CLI arguments. - * Call el_runtime_init_args(argc, argv) at the start of C main() to populate. - * The args list excludes argv[0] (the program name). */ - -static el_val_t _el_args_list = 0; - -void el_runtime_init_args(int argc, char** argv) { - _el_args_list = el_list_empty(); - for (int i = 1; i < argc; i++) { - _el_args_list = el_list_append(_el_args_list, EL_STR(argv[i])); - } -} - -el_val_t args(void) { - if (!_el_args_list) _el_args_list = el_list_empty(); - return _el_args_list; -} - -/* ── CGI identity ──────────────────────────────────────────────────────────── - * Called once at program start by the generated main() of a cgi {} program. - * Stores CGI identity so dharma_* builtins can reference it. */ - -static const char* _el_cgi_name = NULL; -static const char* _el_cgi_dharma_id = NULL; -static const char* _el_cgi_principal = NULL; -static const char* _el_cgi_network = NULL; -static const char* _el_cgi_engram = NULL; - -void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, - el_val_t network, el_val_t engram) { - _el_cgi_name = EL_CSTR(name); - _el_cgi_dharma_id = EL_CSTR(dharma_id); - _el_cgi_principal = EL_CSTR(principal); - _el_cgi_network = EL_CSTR(network) ? EL_CSTR(network) : "dharma-mainnet"; - _el_cgi_engram = EL_CSTR(engram) ? EL_CSTR(engram) : "http://localhost:8742"; - printf("[cgi] identity: name=%s dharma_id=%s principal=%s network=%s engram=%s\n", - _el_cgi_name ? _el_cgi_name : "(unset)", - _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unset)", - _el_cgi_principal ? _el_cgi_principal : "(unset)", - _el_cgi_network, - _el_cgi_engram); -} - - -/* ── Batch 3: Engram in-process graph store ──────────────────────────────── */ -/* - * Single global EngramStore allocated lazily on first call. All node and - * edge content strings are owned (strdup'd) by the store. Linear arrays - * with doubling capacity for both nodes and edges. - * - * Two-layer activation algorithm (engram_activate): - * - * LAYER 1 — Broad fan-out (background activation): - * 1. Find seed nodes whose content/label/tags contain query (case-insens). - * 2. BFS up to `depth` hops along ALL edges (excitatory and inhibitory). - * Every reachable node fires — nothing is filtered at this layer. - * 3. bg_act = seed.salience * temporal_decay * dampening - * propagated as: new_bg = parent_bg * edge_weight * 0.7 * (1 + tbonus) - * where tbonus ∈ {0, 0.10, 0.20} for co-temporal nodes. - * 4. If reached by multiple paths, take max background_activation. - * 5. Persist background_activation to EngramNode.background_activation. - * - * LAYER 2 — Executive filter (working memory promotion): - * 6. For each inhibitory edge where source has background_activation > 0: - * inhibition[target] = max(bg[source] * e->weight) - * 7. For each background-activated node: - * raw_wm = bg * goal_bias(node, query) * confidence - * * (1 - (1 - INHIBITION_FACTOR) * inhibition) - * 8. Per-type threshold gate: raw_wm >= type_threshold → promoted. - * Safety/DharmaSelf: 0.05 Canonical: 0.15 Lesson: 0.25 - * Belief/Entity: 0.30 Note/Memory/Working: 0.40 - * 9. If not promoted: suppression_count++. After - * ENGRAM_SUPPRESSION_BREAKTHROUGH suppressions → force breakthrough - * at ENGRAM_BREAKTHROUGH_WEIGHT (latent tension surfacing). - * 10. Persist working_memory_weight to EngramNode.working_memory_weight. - * 11. Sort: promoted nodes (wm > 0) first by wm desc, then background- - * only by bg desc. Context compilation uses ONLY promoted nodes. - * - * Temporal decay: - * decay_factor = exp(-lambda * age_hours / T_half) - * T_half = 168.0 h (one week), lambda = ln(2) - * - * Activation dampening: - * dampen = 1.0 / (1.0 + log(1 + activation_count)) - * - * engram_query_range(start_ms, end_ms): - * Returns nodes whose created_at OR last_activated falls within - * [start_ms, end_ms], sorted by created_at ascending. - */ - -/* Temporal decay constants. - * T_HALF_HOURS: half-life in hours — one week. After one week of no - * activation a node retains 50% of its base salience contribution. - * DECAY_LAMBDA: ln(2) ≈ 0.693147 */ -#define ENGRAM_T_HALF_HOURS 168.0 -#define ENGRAM_DECAY_LAMBDA 0.693147 - -/* Two-layer activation constants. - * ENGRAM_WM_THRESHOLD: minimum background_activation for a node to be - * considered for working-memory promotion (layer 2 candidate gate). - * ENGRAM_WM_DECAY: per-turn decay applied to working_memory_weight for - * nodes NOT re-activated in the current turn (conversational thread - * continuity: a node promoted in turn N persists with reduced weight - * into turn N+1 without re-activation cost). - * ENGRAM_SUPPRESSION_BREAKTHROUGH: after this many consecutive suppressions - * a latent node forces itself into working memory at reduced weight, - * modelling the brain's "intrusive thought" / unresolved-tension surfacing. - * ENGRAM_BREAKTHROUGH_WEIGHT: the reduced working_memory_weight assigned - * when a suppressed node breaks through. - * ENGRAM_INHIBITION_FACTOR: multiplier applied to working_memory_weight when - * an inhibitory edge fires against a node (0 = full suppress, 0.3 = partial). */ -#define ENGRAM_WM_THRESHOLD 0.15 -#define ENGRAM_WM_DECAY 0.7 -#define ENGRAM_SUPPRESSION_BREAKTHROUGH 5 -#define ENGRAM_BREAKTHROUGH_WEIGHT 0.25 -#define ENGRAM_INHIBITION_FACTOR 0.1 - -/* ── Layered consciousness architecture ────────────────────────────────────── - * - * The engram graph is stratified into LAYERS that gate which suppressions - * apply during the executive filter pass. Layers are ordered shallow-to-deep - * by `activation_priority`; the deepest layer (priority 0, conventionally - * "safety") is the structural floor of the soul: nodes here cannot be - * silenced by inhibitory edges from any other layer. Higher layers - * (core-identity, domain-knowledge, imprint, suit) are normally - * suppressible — they participate in attentional inhibition and goal - * focus the way the prior single-graph implementation did. - * - * The five canonical layers (see engram_init_layers): - * 0. safety — structural, transparent, non-injectable, non-suppressible - * 1. core-identity — default for legacy nodes; suppressible - * 2. domain-knowledge— suppressible - * 3. imprint — runtime-injectable (an Imprint package can add/remove) - * 4. suit — runtime-injectable (a Suit overlays domain skill) - * - * Three-pass activation (engram_activate): - * Pass 1 — Background fan-out: BFS spreads activation across ALL layers - * (existing behavior preserved). Inhibitory edges propagate at - * this layer too; no filtering happens here. - * Pass 2 — Working memory promotion: type-threshold gate, goal bias, - * confidence weighting, inhibitory suppression. Inhibitory edges - * ONLY apply against nodes whose layer is `suppressible == 1`. - * Nodes in non-suppressible layers (Layer 0) ignore inhibition. - * Pass 3 — Layer 0 override: every node in a non-suppressible layer that - * received background activation has its working_memory_weight - * forced to >= ENGRAM_LAYER0_OVERRIDE_WEIGHT. The sacred fire — - * safety nodes that touched any seed unconditionally surface, - * even when the executive filter would have silenced them. - * - * Layer fields: - * suppressible : 0 → inhibitory edges are ignored against nodes in this - * layer during pass 2. Pass 3 also force-promotes them. - * 1 → standard behavior (most layers). - * transparent : 1 → emitted into the prompt context so its content shapes - * output, but filtered out of "what do you know about - * yourself?" introspection queries (engram_search and - * friends do not return transparent-layer nodes by - * default). 0 → fully visible to introspection. - * injectable : 1 → can be added/removed at runtime via engram_add_layer - * and engram_remove_layer (imprints, suits). - * 0 → built-in, fixed at engram_get() initialization. - * - * Backward compatibility: - * Nodes and edges loaded from snapshots without a `layer_id` field default - * to layer 1 (core-identity). The five canonical layers are always present. - */ -#define ENGRAM_LAYER_SAFETY 0u -#define ENGRAM_LAYER_CORE_IDENTITY 1u -#define ENGRAM_LAYER_DOMAIN 2u -#define ENGRAM_LAYER_IMPRINT 3u -#define ENGRAM_LAYER_SUIT 4u -#define ENGRAM_LAYER_DEFAULT ENGRAM_LAYER_CORE_IDENTITY - -/* Pass 3 override floor. Layer 0 nodes that received any background - * activation are force-promoted to AT LEAST this working_memory_weight, - * regardless of inhibitory suppression in pass 2. */ -#define ENGRAM_LAYER0_OVERRIDE_WEIGHT 1.0 - -/* Per-node-type activation thresholds. - * Lower tier / safety-critical nodes fire more readily. */ -static double engram_type_threshold(const char* node_type, const char* tier) { - if (node_type) { - if (strcmp(node_type, "DharmaSelf") == 0) return 0.05; - if (strcmp(node_type, "Safety") == 0) return 0.05; - } - if (tier) { - if (strcmp(tier, "Canonical") == 0) return 0.15; - if (strcmp(tier, "Lesson") == 0) return 0.25; - } - if (node_type) { - if (strcmp(node_type, "Belief") == 0) return 0.30; - if (strcmp(node_type, "Entity") == 0) return 0.30; - } - return 0.40; /* Note / Memory / Working (most nodes) */ -} - -typedef struct EngramNode { - char* id; - char* content; - char* node_type; - char* label; - char* tier; - char* tags; - char* metadata; - double salience; - double importance; - double confidence; - double temporal_decay_rate; /* per-node override for lambda; 0 = use default */ - int64_t activation_count; - int64_t last_activated; - int64_t created_at; - int64_t updated_at; - /* Two-layer activation fields ───────────────────────────────────────── - * background_activation: Layer 1. Set by BFS fan-out on every query. - * Every reachable node fires here — nothing is filtered at this stage. - * Models the brain's massive parallel sub-threshold activation of all - * associated content in response to a stimulus. - * working_memory_weight: Layer 2. Executive filter output. Only nodes - * that survive goal-state / attentional-bias scoring receive a - * non-zero weight here. Context compilation ONLY uses this field. - * Background-activated nodes with working_memory_weight == 0 remain - * latent — real, available, but silent. - * suppression_count: Consecutive turn count where this node was - * background-activated but NOT promoted to working memory. High - * values signal the node "wants to surface." After - * ENGRAM_SUPPRESSION_BREAKTHROUGH consecutive suppressions the node - * is force-promoted at a reduced weight (breakthrough activation). */ - double background_activation; - double working_memory_weight; - int32_t suppression_count; - /* Layered consciousness — see ENGRAM_LAYER_* macros and engram_init_layers. - * Defaults to ENGRAM_LAYER_DEFAULT (1, core-identity) for legacy nodes - * created via engram_node / engram_node_full and for snapshots that - * predate the layered schema. */ - uint32_t layer_id; -} EngramNode; - -typedef struct EngramEdge { - char* id; - char* from_id; - char* to_id; - char* relation; - char* metadata; - double weight; - double confidence; - int64_t created_at; - int64_t updated_at; - int64_t last_fired; - /* Inhibitory flag: when 1, activating the source node SUPPRESSES the - * working_memory_weight of the target node rather than exciting it. - * Models attentional inhibition: "I am focused on code work" creates - * inhibitory edges to personal/emotional nodes, preventing them from - * surfacing even if they have high background_activation. */ - int inhibitory; - /* Layered consciousness — edges carry a layer assignment for - * categorization/visualization. Pass 2 inhibitory gating is decided by - * the TARGET node's layer (whether it's suppressible), not by the edge - * layer. Defaults to ENGRAM_LAYER_DEFAULT. */ - uint32_t layer_id; -} EngramEdge; - -/* Layered consciousness — runtime layer registry entry. */ -typedef struct EngramLayer { - uint32_t layer_id; /* 0 = deepest (safety/limbic) */ - char* name; /* persistent — owned by the store */ - uint32_t activation_priority; /* lower = fires earlier; safety = 0 */ - int suppressible; /* can higher layers suppress nodes here? */ - int transparent; /* invisible to introspection queries? */ - int injectable; /* can be added/removed at runtime? */ -} EngramLayer; - -typedef struct EngramStore { - EngramNode* nodes; - int64_t node_count; - int64_t node_capacity; - EngramEdge* edges; - int64_t edge_count; - int64_t edge_capacity; - /* Layer registry — see engram_init_layers. The five canonical layers - * are always present; injectable layers (imprint, suit) are extended - * via engram_add_layer at runtime. layer_id values are assigned - * monotonically; removed injectable layers leave a NULL `name` slot - * (tombstone) so existing layer_id references on nodes stay stable. */ - EngramLayer* layers; - size_t layer_count; - size_t layer_capacity; -} EngramStore; - -static EngramStore* engram_global = NULL; - -/* Initialize the five canonical layers on a fresh store. Called once from - * engram_get(). Layer ids 0..4 are reserved; runtime-injected imprint/suit - * layers (engram_add_layer) get ids 5+. */ -static void engram_init_layers(EngramStore* g) { - g->layer_capacity = 16; - g->layers = calloc(g->layer_capacity, sizeof(EngramLayer)); - if (!g->layers) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - g->layer_count = 0; - - /* Layer 0 — safety. Structural floor. Non-suppressible; transparent - * (filtered out of introspection but still shapes output); not - * runtime-injectable. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_SAFETY, - .name = el_strdup_persist("safety"), - .activation_priority = 0, - .suppressible = 0, - .transparent = 1, - .injectable = 0 - }; - /* Layer 1 — core-identity. The default home for legacy nodes. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_CORE_IDENTITY, - .name = el_strdup_persist("core-identity"), - .activation_priority = 10, - .suppressible = 1, - .transparent = 0, - .injectable = 0 - }; - /* Layer 2 — domain-knowledge. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_DOMAIN, - .name = el_strdup_persist("domain-knowledge"), - .activation_priority = 20, - .suppressible = 1, - .transparent = 0, - .injectable = 0 - }; - /* Layer 3 — imprint. Injectable: an imprint package adds/removes this - * layer (and the nodes assigned to it) as a unit. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_IMPRINT, - .name = el_strdup_persist("imprint"), - .activation_priority = 30, - .suppressible = 1, - .transparent = 0, - .injectable = 1 - }; - /* Layer 4 — suit. Injectable: a Suit overlays domain skill (e.g. - * "enterprise advisor", "divorce lawyer") and can be detached. */ - g->layers[g->layer_count++] = (EngramLayer){ - .layer_id = ENGRAM_LAYER_SUIT, - .name = el_strdup_persist("suit"), - .activation_priority = 40, - .suppressible = 1, - .transparent = 0, - .injectable = 1 - }; -} - -static EngramStore* engram_get(void) { - if (engram_global) return engram_global; - engram_global = calloc(1, sizeof(EngramStore)); - if (!engram_global) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - engram_global->node_capacity = 16; - engram_global->nodes = calloc((size_t)engram_global->node_capacity, sizeof(EngramNode)); - engram_global->edge_capacity = 16; - engram_global->edges = calloc((size_t)engram_global->edge_capacity, sizeof(EngramEdge)); - engram_init_layers(engram_global); - return engram_global; -} - -/* Resolve a layer record by id. Returns NULL if no layer with that id - * exists (e.g. a removed injectable layer or a malformed snapshot). */ -static EngramLayer* engram_find_layer(uint32_t layer_id) { - EngramStore* g = engram_get(); - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; /* tombstone for removed injectable layer */ - if (L->layer_id == layer_id) return L; - } - return NULL; -} - -/* Resolve a layer record by name. Returns NULL if not found. */ -static EngramLayer* engram_find_layer_by_name(const char* name) { - if (!name || !*name) return NULL; - EngramStore* g = engram_get(); - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if (strcmp(L->name, name) == 0) return L; - } - return NULL; -} - -/* Allocate the next layer id. Skips ids that are still in use. */ -static uint32_t engram_next_layer_id(void) { - EngramStore* g = engram_get(); - uint32_t maxid = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].layer_id > maxid) maxid = g->layers[i].layer_id; - } - return maxid + 1; -} - -/* Whether a node in `layer_id` may be silenced by inhibitory edges in pass 2. */ -static int engram_layer_is_suppressible(uint32_t layer_id) { - EngramLayer* L = engram_find_layer(layer_id); - if (!L) return 1; /* unknown layer → safe default: standard suppression */ - return L->suppressible ? 1 : 0; -} - -/* Whether a layer is transparent (its content shapes output but is filtered - * from introspection queries). Currently used to mark Layer 0 as invisible - * to "what do you know about yourself" lookups while still letting it - * dominate the prompt context. */ -static int engram_layer_is_transparent(uint32_t layer_id) { - EngramLayer* L = engram_find_layer(layer_id); - if (!L) return 0; - return L->transparent ? 1 : 0; -} - -static int64_t engram_now_ms(void) { - struct timeval tv; gettimeofday(&tv, NULL); - return (int64_t)tv.tv_sec * 1000LL + (int64_t)tv.tv_usec / 1000LL; -} - -static EngramNode* engram_find_node(const char* id) { - if (!id) return NULL; - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].id && strcmp(g->nodes[i].id, id) == 0) return &g->nodes[i]; - } - return NULL; -} - -static int64_t engram_find_node_index(const char* id) { - if (!id) return -1; - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].id && strcmp(g->nodes[i].id, id) == 0) return i; - } - return -1; -} - -static void engram_grow_nodes(void) { - EngramStore* g = engram_get(); - if (g->node_count < g->node_capacity) return; - int64_t nc = g->node_capacity * 2; - g->nodes = realloc(g->nodes, (size_t)nc * sizeof(EngramNode)); - if (!g->nodes) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(g->nodes + g->node_capacity, 0, - (size_t)(nc - g->node_capacity) * sizeof(EngramNode)); - g->node_capacity = nc; -} - -static void engram_grow_edges(void) { - EngramStore* g = engram_get(); - if (g->edge_count < g->edge_capacity) return; - int64_t nc = g->edge_capacity * 2; - g->edges = realloc(g->edges, (size_t)nc * sizeof(EngramEdge)); - if (!g->edges) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(g->edges + g->edge_capacity, 0, - (size_t)(nc - g->edge_capacity) * sizeof(EngramEdge)); - g->edge_capacity = nc; -} - -/* Build a fresh UUID string. Reuses uuid_new but takes the underlying char*. */ -static char* engram_new_id(void) { - el_val_t v = uuid_new(); - const char* s = EL_CSTR(v); - return el_strdup(s ? s : ""); -} - -/* Convert a node into an ElMap of its fields. */ -static el_val_t engram_node_to_map(const EngramNode* n) { - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("id")), EL_STR(el_strdup(n->id ? n->id : ""))); - m = el_map_set(m, EL_STR(el_strdup("content")), EL_STR(el_strdup(n->content ? n->content : ""))); - m = el_map_set(m, EL_STR(el_strdup("node_type")), EL_STR(el_strdup(n->node_type ? n->node_type : ""))); - m = el_map_set(m, EL_STR(el_strdup("label")), EL_STR(el_strdup(n->label ? n->label : ""))); - m = el_map_set(m, EL_STR(el_strdup("tier")), EL_STR(el_strdup(n->tier ? n->tier : "Working"))); - m = el_map_set(m, EL_STR(el_strdup("tags")), EL_STR(el_strdup(n->tags ? n->tags : ""))); - m = el_map_set(m, EL_STR(el_strdup("metadata")), EL_STR(el_strdup(n->metadata ? n->metadata : "{}"))); - m = el_map_set(m, EL_STR(el_strdup("salience")), el_from_float(n->salience)); - m = el_map_set(m, EL_STR(el_strdup("importance")), el_from_float(n->importance)); - m = el_map_set(m, EL_STR(el_strdup("confidence")), el_from_float(n->confidence)); - m = el_map_set(m, EL_STR(el_strdup("temporal_decay_rate")), el_from_float(n->temporal_decay_rate)); - m = el_map_set(m, EL_STR(el_strdup("activation_count")), (el_val_t)n->activation_count); - m = el_map_set(m, EL_STR(el_strdup("last_activated")), (el_val_t)n->last_activated); - m = el_map_set(m, EL_STR(el_strdup("created_at")), (el_val_t)n->created_at); - m = el_map_set(m, EL_STR(el_strdup("updated_at")), (el_val_t)n->updated_at); - m = el_map_set(m, EL_STR(el_strdup("background_activation")), el_from_float(n->background_activation)); - m = el_map_set(m, EL_STR(el_strdup("working_memory_weight")), el_from_float(n->working_memory_weight)); - m = el_map_set(m, EL_STR(el_strdup("suppression_count")), (el_val_t)n->suppression_count); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), (el_val_t)(int64_t)n->layer_id); - return m; -} - -/* (Node JSON serialization is provided by `engram_emit_node_json` further - * down in the persistence section — reused by the *_json builtins below.) */ -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n); -static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e); - -/* Salience may arrive either as a float bit-pattern or as a small integer - * (e.g. 1, meaning 1.0). Heuristic: if interpreted as double it's in - * [0.0, 100.0] use it; otherwise treat as int and convert. */ -static double engram_decode_score(el_val_t v) { - double f = el_to_float(v); - if (!isnan(f) && !isinf(f) && f >= 0.0 && f <= 100.0) return f; - int64_t n = (int64_t)v; - return (double)n; -} - -static char* engram_first_n_chars(const char* s, size_t n) { - if (!s) return el_strdup(""); - size_t l = strlen(s); - if (l > n) l = n; - char* out = el_strbuf(l); - memcpy(out, s, l); - out[l] = '\0'; - return out; -} - -el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - n->content = el_strdup(c ? c : ""); - n->node_type = el_strdup(nt && *nt ? nt : "Memory"); - n->label = engram_first_n_chars(c, 60); - n->tier = el_strdup("Working"); - n->tags = el_strdup(""); - n->metadata = el_strdup("{}"); - n->salience = engram_decode_score(salience); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - n->importance = 0.5; - n->confidence = 1.0; - n->temporal_decay_rate = 0.0; /* 0 = use global default ENGRAM_DECAY_LAMBDA */ - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t importance, el_val_t confidence, - el_val_t tier, el_val_t tags) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - const char* lb = EL_CSTR(label); - const char* ti = EL_CSTR(tier); - const char* tg = EL_CSTR(tags); - n->content = el_strdup(c ? c : ""); - n->node_type = el_strdup(nt && *nt ? nt : "Memory"); - n->label = el_strdup(lb && *lb ? lb : (c ? engram_first_n_chars(c, 60) : "")); - n->tier = el_strdup(ti && *ti ? ti : "Working"); - n->tags = el_strdup(tg ? tg : ""); - n->metadata = el_strdup("{}"); - n->salience = engram_decode_score(salience); - n->importance = engram_decode_score(importance); - n->confidence = engram_decode_score(confidence); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - if (n->importance <= 0.0 || n->importance > 1.0) n->importance = 0.5; - if (n->confidence <= 0.0 || n->confidence > 1.0) n->confidence = 1.0; - n->temporal_decay_rate = 0.0; /* 0 = use global default ENGRAM_DECAY_LAMBDA */ - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -/* engram_node_layered — like engram_node_full but with explicit layer - * assignment and an additional `status` slot reserved for callers that - * track lifecycle state in metadata. The signature mirrors the public API - * defined in the layered consciousness design doc: - * - * engram_node_layered(content, node_type, label, - * salience, certainty, confidence, - * status, tags, layer_id) - * - * `certainty` is folded into `importance` (it occupies the same axis in - * the existing schema). `status` is recorded under metadata.status; an - * empty status leaves metadata as the default "{}". - * - * If `layer_id` does not resolve to a known layer the call falls back to - * ENGRAM_LAYER_DEFAULT — better to keep the node addressable than to drop - * it because of a stale layer reference. Callers wanting strict validation - * should engram_list_layers first. */ -el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t certainty, el_val_t confidence, - el_val_t status, el_val_t tags, el_val_t layer_id) { - EngramStore* g = engram_get(); - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = engram_new_id(); - const char* c = EL_CSTR(content); - const char* nt = EL_CSTR(node_type); - const char* lb = EL_CSTR(label); - const char* tg = EL_CSTR(tags); - const char* st = EL_CSTR(status); - n->content = el_strdup(c ? c : ""); - n->node_type = el_strdup(nt && *nt ? nt : "Memory"); - n->label = el_strdup(lb && *lb ? lb : (c ? engram_first_n_chars(c, 60) : "")); - n->tier = el_strdup("Working"); - n->tags = el_strdup(tg ? tg : ""); - if (st && *st) { - /* Minimal metadata payload: {"status":"..."}. Keep it cheap so - * callers using `status` don't pay JSON parse cost on every read. */ - size_t sl = strlen(st) + 16; - char* meta = el_strbuf(sl); - snprintf(meta, sl, "{\"status\":\"%s\"}", st); - n->metadata = meta; - } else { - n->metadata = el_strdup("{}"); - } - n->salience = engram_decode_score(salience); - n->importance = engram_decode_score(certainty); - n->confidence = engram_decode_score(confidence); - if (n->salience <= 0.0 || n->salience > 1.0) n->salience = 0.5; - if (n->importance <= 0.0 || n->importance > 1.0) n->importance = 0.5; - if (n->confidence <= 0.0 || n->confidence > 1.0) n->confidence = 1.0; - n->temporal_decay_rate = 0.0; - n->activation_count = 0; - int64_t now = engram_now_ms(); - n->last_activated = now; - n->created_at = now; - n->updated_at = now; - /* Resolve layer assignment. Caller passes either a numeric layer_id or - * a stringified id; el_to_float / int cast tolerates both. */ - int64_t lid = (int64_t)layer_id; - if (lid < 0) lid = (int64_t)ENGRAM_LAYER_DEFAULT; - if (!engram_find_layer((uint32_t)lid)) lid = (int64_t)ENGRAM_LAYER_DEFAULT; - n->layer_id = (uint32_t)lid; - g->node_count++; - return el_wrap_str(el_strdup(n->id)); -} - -/* ── Layer registry public API ────────────────────────────────────────────── - * - * The five canonical layers are seeded at engram_get() initialization. - * Runtime code (typically imprint/suit injection logic at the EL level) - * can extend the registry with engram_add_layer() — only layers marked - * `injectable=1` may be removed via engram_remove_layer(). Removing a - * layer leaves a tombstone slot so existing layer_id references on nodes - * stay valid; orphaned references resolve to "unknown layer" and inherit - * the default suppression behavior. - */ - -/* engram_add_layer — register a new layer at runtime. - * Returns the assigned layer_id as an el_val_t int (cast back via int64_t). - * Conflicting names are rejected (returns 0). */ -el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible, - el_val_t transparent, el_val_t injectable) { - EngramStore* g = engram_get(); - const char* nm = EL_CSTR(name); - if (!nm || !*nm) return (el_val_t)0; - if (engram_find_layer_by_name(nm)) { - /* Name collision — return existing id so callers are idempotent. */ - return (el_val_t)(int64_t)engram_find_layer_by_name(nm)->layer_id; - } - if (g->layer_count >= g->layer_capacity) { - size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; - EngramLayer* grown = realloc(g->layers, nc * sizeof(EngramLayer)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(grown + g->layer_capacity, 0, - (nc - g->layer_capacity) * sizeof(EngramLayer)); - g->layers = grown; - g->layer_capacity = nc; - } - EngramLayer* L = &g->layers[g->layer_count++]; - L->layer_id = engram_next_layer_id(); - L->name = el_strdup_persist(nm); - L->activation_priority = (uint32_t)(int64_t)priority; - L->suppressible = (int)(int64_t)suppressible ? 1 : 0; - L->transparent = (int)(int64_t)transparent ? 1 : 0; - L->injectable = (int)(int64_t)injectable ? 1 : 0; - return (el_val_t)(int64_t)L->layer_id; -} - -/* engram_remove_layer — remove an injectable layer by id. - * Built-in (non-injectable) layers cannot be removed. Nodes still tagged - * with the removed layer's id keep their tag but resolve to "unknown - * layer" thereafter and inherit standard (suppressible) behavior. - * Returns 1 on success, 0 on failure (unknown id, non-injectable). */ -el_val_t engram_remove_layer(el_val_t layer_id) { - EngramStore* g = engram_get(); - int64_t lid = (int64_t)layer_id; - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if ((int64_t)L->layer_id != lid) continue; - if (!L->injectable) return (el_val_t)0; - free(L->name); - L->name = NULL; /* tombstone */ - /* Leave layer_id, priority, flags intact so debug snapshots can - * still distinguish "removed at runtime" from "never existed". */ - return (el_val_t)1; - } - return (el_val_t)0; -} - -/* engram_list_layers — enumerate the active layer registry. - * Returns an ElList of maps, one per non-tombstone layer, sorted by - * activation_priority ascending (deepest layer first). */ -el_val_t engram_list_layers(void) { - EngramStore* g = engram_get(); - el_val_t lst = el_list_empty(); - if (g->layer_count == 0) return lst; - /* Build an index sorted by activation_priority ascending. */ - size_t* idx = malloc(g->layer_count * sizeof(size_t)); - if (!idx) return lst; - size_t live = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) idx[live++] = i; - } - /* Insertion sort — N is small (≤ a few dozen layers). */ - for (size_t i = 1; i < live; i++) { - size_t key = idx[i]; - uint32_t kp = g->layers[key].activation_priority; - size_t j = i; - while (j > 0 && g->layers[idx[j - 1]].activation_priority > kp) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - for (size_t i = 0; i < live; i++) { - EngramLayer* L = &g->layers[idx[i]]; - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), - (el_val_t)(int64_t)L->layer_id); - m = el_map_set(m, EL_STR(el_strdup("name")), - EL_STR(el_strdup(L->name ? L->name : ""))); - m = el_map_set(m, EL_STR(el_strdup("activation_priority")), - (el_val_t)(int64_t)L->activation_priority); - m = el_map_set(m, EL_STR(el_strdup("suppressible")), - (el_val_t)(int64_t)(L->suppressible ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("transparent")), - (el_val_t)(int64_t)(L->transparent ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("injectable")), - (el_val_t)(int64_t)(L->injectable ? 1 : 0)); - lst = el_list_append(lst, m); - } - free(idx); - return lst; -} - -el_val_t engram_get_node(el_val_t id) { - const char* sid = EL_CSTR(id); - EngramNode* n = engram_find_node(sid); - if (!n) return el_map_new(0); - return engram_node_to_map(n); -} - -void engram_strengthen(el_val_t node_id) { - const char* sid = EL_CSTR(node_id); - EngramNode* n = engram_find_node(sid); - if (!n) return; - n->salience += 0.05; - if (n->salience > 1.0) n->salience = 1.0; - n->activation_count++; - n->last_activated = engram_now_ms(); - n->updated_at = n->last_activated; -} - -void engram_forget(el_val_t node_id) { - const char* sid = EL_CSTR(node_id); - if (!sid) return; - EngramStore* g = engram_get(); - int64_t idx = engram_find_node_index(sid); - if (idx < 0) return; - /* Free node strings */ - EngramNode* n = &g->nodes[idx]; - free(n->id); free(n->content); free(n->node_type); free(n->label); - free(n->tier); free(n->tags); free(n->metadata); - /* Shift remaining nodes down */ - for (int64_t i = idx + 1; i < g->node_count; i++) { - g->nodes[i - 1] = g->nodes[i]; - } - g->node_count--; - memset(&g->nodes[g->node_count], 0, sizeof(EngramNode)); - /* Remove all incident edges */ - int64_t w = 0; - for (int64_t r = 0; r < g->edge_count; r++) { - EngramEdge* e = &g->edges[r]; - int incident = (e->from_id && strcmp(e->from_id, sid) == 0) || - (e->to_id && strcmp(e->to_id, sid) == 0); - if (incident) { - free(e->id); free(e->from_id); free(e->to_id); - free(e->relation); free(e->metadata); - } else { - if (w != r) g->edges[w] = g->edges[r]; - w++; - } - } - g->edge_count = w; -} - -el_val_t engram_node_count(void) { - return (el_val_t)engram_get()->node_count; -} - -static int istr_contains(const char* hay, const char* needle) { - if (!hay || !needle || !*needle) return 0; - size_t nl = strlen(needle); - for (const char* p = hay; *p; p++) { - if (strncasecmp(p, needle, nl) == 0) return 1; - } - return 0; -} - -el_val_t engram_search(el_val_t query, el_val_t limit) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t lim = (int64_t)limit; - if (lim <= 0) lim = 100; - el_val_t lst = el_list_empty(); - if (!q || !*q) return lst; - int64_t found = 0; - for (int64_t i = 0; i < g->node_count && found < lim; i++) { - EngramNode* n = &g->nodes[i]; - /* Filter transparent layers: nodes whose layer is `transparent=1` - * shape output but are invisible to introspection ("what do you - * know about yourself"). They still surface via engram_activate - * + engram_compile_layered_json — that's the legitimate path. */ - if (engram_layer_is_transparent(n->layer_id)) continue; - if (istr_contains(n->content, q) || - istr_contains(n->label, q) || - istr_contains(n->tags, q)) { - lst = el_list_append(lst, engram_node_to_map(n)); - found++; - } - } - return lst; -} - -/* Sort node indices by salience desc (small N, insertion sort is fine). */ -static void engram_sort_indices_by_salience(int64_t* arr, int64_t n, - const EngramNode* nodes) { - for (int64_t i = 1; i < n; i++) { - int64_t key = arr[i]; - double ks = nodes[key].salience; - int64_t j = i - 1; - while (j >= 0 && nodes[arr[j]].salience < ks) { - arr[j + 1] = arr[j]; - j--; - } - arr[j + 1] = key; - } -} - -el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset) { - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - el_val_t lst = el_list_empty(); - if (g->node_count == 0) return lst; - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) return lst; - /* Skip transparent layers — same introspection-filter rationale as - * engram_search above. */ - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - for (int64_t i = off; i < end; i++) { - lst = el_list_append(lst, engram_node_to_map(&g->nodes[idx[i]])); - } - free(idx); - return lst; -} - -void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation) { - EngramStore* g = engram_get(); - const char* f = EL_CSTR(from_id); - const char* t = EL_CSTR(to_id); - const char* r = EL_CSTR(relation); - if (!f || !t) return; - engram_grow_edges(); - EngramEdge* e = &g->edges[g->edge_count]; - memset(e, 0, sizeof(*e)); - e->id = engram_new_id(); - e->from_id = el_strdup(f); - e->to_id = el_strdup(t); - e->relation = el_strdup(r && *r ? r : "associate"); - e->metadata = el_strdup("{}"); - e->weight = engram_decode_score(weight); - if (e->weight <= 0.0 || e->weight > 1.0) e->weight = 0.5; - e->confidence = 1.0; - int64_t now = engram_now_ms(); - e->created_at = now; - e->updated_at = now; - e->last_fired = 0; - e->layer_id = ENGRAM_LAYER_DEFAULT; - g->edge_count++; -} - -el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id) { - EngramStore* g = engram_get(); - const char* f = EL_CSTR(from_id); - const char* t = EL_CSTR(to_id); - if (!f || !t) return 0; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (e->from_id && e->to_id && - strcmp(e->from_id, f) == 0 && strcmp(e->to_id, t) == 0) return 1; - } - return 0; -} - -/* Reserved helper: edge -> ElMap. Kept around for future builtins. */ -static el_val_t engram_edge_to_map(const EngramEdge* e) __attribute__((unused)); -static el_val_t engram_edge_to_map(const EngramEdge* e) { - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("id")), EL_STR(el_strdup(e->id ? e->id : ""))); - m = el_map_set(m, EL_STR(el_strdup("from_id")), EL_STR(el_strdup(e->from_id ? e->from_id : ""))); - m = el_map_set(m, EL_STR(el_strdup("to_id")), EL_STR(el_strdup(e->to_id ? e->to_id : ""))); - m = el_map_set(m, EL_STR(el_strdup("relation")), EL_STR(el_strdup(e->relation ? e->relation : ""))); - m = el_map_set(m, EL_STR(el_strdup("metadata")), EL_STR(el_strdup(e->metadata ? e->metadata : "{}"))); - m = el_map_set(m, EL_STR(el_strdup("weight")), el_from_float(e->weight)); - m = el_map_set(m, EL_STR(el_strdup("confidence")), el_from_float(e->confidence)); - m = el_map_set(m, EL_STR(el_strdup("created_at")), (el_val_t)e->created_at); - m = el_map_set(m, EL_STR(el_strdup("updated_at")), (el_val_t)e->updated_at); - m = el_map_set(m, EL_STR(el_strdup("last_fired")), (el_val_t)e->last_fired); - m = el_map_set(m, EL_STR(el_strdup("inhibitory")), (el_val_t)(e->inhibitory ? 1 : 0)); - m = el_map_set(m, EL_STR(el_strdup("layer_id")), (el_val_t)(int64_t)e->layer_id); - return m; -} - -el_val_t engram_neighbors(el_val_t node_id) { - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - el_val_t lst = el_list_empty(); - if (!sid) return lst; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* other = NULL; - if (e->from_id && strcmp(e->from_id, sid) == 0) other = e->to_id; - else if (e->to_id && strcmp(e->to_id, sid) == 0) other = e->from_id; - if (!other) continue; - EngramNode* n = engram_find_node(other); - if (n) lst = el_list_append(lst, engram_node_to_map(n)); - } - return lst; -} - -el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction) { - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - int64_t md = (int64_t)max_depth; if (md <= 0) md = 1; - const char* dir = EL_CSTR(direction); /* "out" | "in" | "both" (default) */ - el_val_t lst = el_list_empty(); - if (!sid || g->node_count == 0) return lst; - int64_t start = engram_find_node_index(sid); - if (start < 0) return lst; - /* BFS with depth tracking */ - int64_t* visited = calloc((size_t)g->node_count, sizeof(int64_t)); - int64_t* queue = calloc((size_t)g->node_count, sizeof(int64_t)); - int64_t* depths = calloc((size_t)g->node_count, sizeof(int64_t)); - if (!visited || !queue || !depths) { - free(visited); free(queue); free(depths); return lst; - } - int64_t qh = 0, qt = 0; - queue[qt++] = start; - visited[start] = 1; - depths[start] = 0; - while (qh < qt) { - int64_t cur = queue[qh++]; - const char* cur_id = g->nodes[cur].id; - int64_t cur_depth = depths[cur]; - if (cur_depth >= md) continue; - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* other = NULL; - int outgoing = e->from_id && strcmp(e->from_id, cur_id) == 0; - int incoming = e->to_id && strcmp(e->to_id, cur_id) == 0; - if (dir && strcmp(dir, "out") == 0 && !outgoing) continue; - if (dir && strcmp(dir, "in") == 0 && !incoming) continue; - if (outgoing) other = e->to_id; - else if (incoming) other = e->from_id; - else continue; - int64_t oi = engram_find_node_index(other); - if (oi < 0 || visited[oi]) continue; - visited[oi] = 1; - depths[oi] = cur_depth + 1; - queue[qt++] = oi; - } - } - /* Emit all visited except the seed */ - for (int64_t i = 0; i < g->node_count; i++) { - if (visited[i] && i != start) { - lst = el_list_append(lst, engram_node_to_map(&g->nodes[i])); - } - } - free(visited); free(queue); free(depths); - return lst; -} - -el_val_t engram_edge_count(void) { - return (el_val_t)engram_get()->edge_count; -} - -/* Compute temporal decay factor for a node given current time. - * effective contribution = salience * exp(-lambda * age_hours / T_half) - * Clamped to [0.05, 1.0] so very old nodes retain a meaningful floor. */ -static double engram_temporal_decay(const EngramNode* n, int64_t now_ms) { - int64_t age_ms = now_ms - n->last_activated; - if (age_ms <= 0) return 1.0; - double lambda = (n->temporal_decay_rate > 0.0) ? n->temporal_decay_rate - : ENGRAM_DECAY_LAMBDA; - double age_hours = (double)age_ms / 3600000.0; - double factor = exp(-lambda * age_hours / ENGRAM_T_HALF_HOURS); - if (factor < 0.05) factor = 0.05; - return factor; -} - -/* Activation dampening: high activation_count nodes are "well-known" context - * and get less marginal boost per firing. - * count=0 → 1.0, count=2 → ~0.74, count=9 → ~0.59, count=99 → ~0.43 */ -static double engram_activation_dampen(const EngramNode* n) { - return 1.0 / (1.0 + log(1.0 + (double)n->activation_count)); -} - -/* Temporal proximity bonus: boost propagation along edges connecting - * co-temporal nodes. Returns a multiplier bonus in [0, 0.2]. */ -static double engram_temporal_proximity_bonus(int64_t node_created, - int64_t seed_epoch) { - int64_t diff = node_created - seed_epoch; - if (diff < 0) diff = -diff; - if (diff < 86400000LL) return 0.20; /* within 1 day */ - if (diff < 604800000LL) return 0.10; /* within 7 days */ - return 0.0; -} - -/* ── Two-layer activation (biologically-motivated) ─────────────────────────── - * - * Layer 1 — Broad fan-out (background activation): - * BFS + spreading activation fires on ALL nodes reachable from seeds, - * regardless of relevance to the current goal. Every reachable node gets - * a background_activation score. Nothing is filtered here. Models the - * brain's massive parallel sub-threshold activation of all associated - * content in response to a stimulus. Temporal decay and activation - * dampening are applied at this layer (as before), but no threshold gate. - * - * Layer 2 — Executive filter (working memory promotion): - * A second pass asks: given the query (goal intent), attentional bias, - * and inhibitory edge topology — which background-activated nodes should - * break through into working memory? - * - * wm_weight = bg_activation * goal_bias(node, query) * confidence - * * inhibitory_suppression_factor - * - * Only nodes where wm_weight >= ENGRAM_WM_THRESHOLD are promoted to - * working memory (working_memory_weight > 0). Background-activated nodes - * that don't cross the threshold accumulate suppression_count. After - * ENGRAM_SUPPRESSION_BREAKTHROUGH consecutive suppressed turns, the node - * force-breaks through at ENGRAM_BREAKTHROUGH_WEIGHT (latent tension - * surfacing — models intrusive memory / unresolved cognitive load). - * - * Inhibitory edges: - * An edge with inhibitory=1 suppresses the TARGET node's working memory - * promotion when the SOURCE is background-activated. Background activation - * of the target is NOT affected — the node fires in layer 1. Only the - * executive filter (layer 2) is gated. Models attentional inhibition: - * "focused on code work" suppresses personal memories from surfacing - * even if they have high background_activation. - * - * Goal bias: - * A lightweight heuristic rates how well each background-activated node - * aligns with the apparent intent of the current query. Technical queries - * boost Belief/Canonical/Lesson nodes; relational queries boost Memory/ - * Entity nodes. Direct lexical overlap gives a 50% bonus. - * - * Working memory persistence (turn continuity): - * Nodes promoted in the previous turn retain a decayed working_memory_weight - * (weight *= ENGRAM_WM_DECAY) without needing re-activation. This models - * conversational thread continuity — once a topic is in working memory, - * it persists slightly into the next turn. - * - * Returns ElList of {node, activation_strength, working_memory_weight, - * epistemic_confidence, hops, promoted}. - * "promoted" = 1 if working_memory_weight > 0, 0 if background-only. - * Context compilation uses ONLY nodes with promoted=1. - * - * Temporal decay (preserved from prior implementation): - * effective_salience = salience * exp(-lambda * age_hours / T_half) - * where T_half = 168 h (one week), lambda = ln(2) - * - * Activation dampening (preserved): - * dampen = 1 / (1 + log(1 + activation_count)) - * - * Temporal proximity bonus (preserved): - * edge_strength *= (1 + tbonus) where tbonus ∈ {0, 0.10, 0.20} - * - * Per-type threshold gates apply only to working memory promotion (layer 2): - * Safety/DharmaSelf: 0.05 Canonical: 0.15 Lesson: 0.25 - * Belief/Entity: 0.30 Note/Memory/Working: 0.40 - */ - -/* Compute goal-state bias multiplier for a node given the query. - * Returns a value in [0.3, 2.0]. This is a lightweight heuristic — - * a production implementation may use LLM-derived intent classification. */ -static double engram_goal_bias(const EngramNode* n, const char* query) { - if (!query || !*query) return 1.0; - double bias = 1.0; - /* Direct lexical overlap: node content/label/tags share text with query. */ - if (istr_contains(n->content, query) || istr_contains(n->label, query) || - istr_contains(n->tags, query)) { - bias += 0.5; - } - /* Node-type resonance with query intent. */ - int technical_query = istr_contains(query, "code") || - istr_contains(query, "function") || - istr_contains(query, "implement") || - istr_contains(query, "error") || - istr_contains(query, "bug") || - istr_contains(query, "build") || - istr_contains(query, "system") || - istr_contains(query, "design") || - istr_contains(query, "architecture"); - int personal_query = istr_contains(query, "feel") || - istr_contains(query, "emotion") || - istr_contains(query, "remember") || - istr_contains(query, "personal") || - istr_contains(query, "story") || - istr_contains(query, "relationship"); - if (n->node_type) { - int is_knowledge = (strcmp(n->node_type, "Belief") == 0) || - (strcmp(n->node_type, "DharmaSelf") == 0) || - (strcmp(n->node_type, "Safety") == 0); - int is_personal = (strcmp(n->node_type, "Memory") == 0) || - (strcmp(n->node_type, "Entity") == 0); - if (technical_query && is_knowledge) bias += 0.3; - if (technical_query && is_personal) bias -= 0.3; - if (personal_query && is_personal) bias += 0.3; - if (personal_query && is_knowledge) bias -= 0.1; - } - /* Tier-based bonus: promote higher-confidence knowledge nodes. */ - if (n->tier) { - if (strcmp(n->tier, "Canonical") == 0) bias += 0.2; - if (strcmp(n->tier, "Lesson") == 0) bias += 0.1; - } - if (bias < 0.3) bias = 0.3; - if (bias > 2.0) bias = 2.0; - return bias; -} - -el_val_t engram_activate(el_val_t query, el_val_t depth) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t max_depth = (int64_t)depth; if (max_depth <= 0) max_depth = 2; - el_val_t out = el_list_empty(); - if (!q || g->node_count == 0) return out; - - int64_t now_ms = engram_now_ms(); - - /* Per-node layer-1 tracking. */ - double* best_bg = calloc((size_t)g->node_count, sizeof(double)); - int64_t* best_hops = calloc((size_t)g->node_count, sizeof(int64_t)); - int* reached = calloc((size_t)g->node_count, sizeof(int)); - if (!best_bg || !best_hops || !reached) { - free(best_bg); free(best_hops); free(reached); return out; - } - - /* ── LAYER 1: broad fan-out (background activation) ───────────────── - * Find seeds, apply temporal decay + dampening, BFS with edge weights. - * Inhibitory edges propagate activation normally at this layer — they - * only gate working memory promotion in layer 2. */ - typedef struct { int64_t idx; double act; int64_t created_at; } SeedEntry; - SeedEntry* seeds = malloc((size_t)g->node_count * sizeof(SeedEntry)); - int64_t seed_count = 0; - if (!seeds) { - free(best_bg); free(best_hops); free(reached); return out; - } - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - if (istr_contains(n->content, q) || - istr_contains(n->label, q) || - istr_contains(n->tags, q)) { - double tdecay = engram_temporal_decay(n, now_ms); - double dampen = engram_activation_dampen(n); - double act = n->salience * tdecay * dampen; - seeds[seed_count].idx = i; - seeds[seed_count].act = act; - seeds[seed_count].created_at = n->created_at; - seed_count++; - best_bg[i] = act; - best_hops[i] = 0; - reached[i] = 1; - } - } - /* Compute mean seed created_at for temporal proximity bonus. */ - int64_t seed_epoch = 0; - if (seed_count > 0) { - seed_epoch = seeds[0].created_at; - for (int64_t s = 1; s < seed_count; s++) - seed_epoch = (seed_epoch + seeds[s].created_at) / 2; - } - typedef struct { int64_t idx; int64_t hops; double act; } Frontier; - Frontier* fr = malloc((size_t)(g->node_count * (max_depth + 1)) * sizeof(Frontier) + 16 * sizeof(Frontier)); - if (!fr) { - free(best_bg); free(best_hops); free(reached); free(seeds); return out; - } - int64_t fhead = 0, ftail = 0; - int64_t fcap = (int64_t)((size_t)(g->node_count * (max_depth + 1)) + 16); - for (int64_t s = 0; s < seed_count; s++) { - if (ftail >= fcap) break; - fr[ftail].idx = seeds[s].idx; - fr[ftail].hops = 0; - fr[ftail].act = seeds[s].act; - ftail++; - } - const double SPREAD_DECAY = 0.7; - while (fhead < ftail) { - Frontier f = fr[fhead++]; - if (f.hops >= max_depth) continue; - const char* cur_id = g->nodes[f.idx].id; - for (int64_t ei = 0; ei < g->edge_count; ei++) { - EngramEdge* e = &g->edges[ei]; - const char* other = NULL; - if (e->from_id && strcmp(e->from_id, cur_id) == 0) other = e->to_id; - else if (e->to_id && strcmp(e->to_id, cur_id) == 0) other = e->from_id; - else continue; - int64_t oi = engram_find_node_index(other); - if (oi < 0) continue; - EngramNode* on = &g->nodes[oi]; - double tbonus = engram_temporal_proximity_bonus(on->created_at, seed_epoch); - double tdecay = engram_temporal_decay(on, now_ms); - double dampen = engram_activation_dampen(on); - double new_act = f.act * e->weight * SPREAD_DECAY * (1.0 + tbonus) - * tdecay * dampen; - int64_t new_hops = f.hops + 1; - if (!reached[oi] || new_act > best_bg[oi]) { - best_bg[oi] = new_act; - best_hops[oi] = new_hops; - reached[oi] = 1; - if (ftail < fcap) { - fr[ftail].idx = oi; - fr[ftail].hops = new_hops; - fr[ftail].act = new_act; - ftail++; - } - } - } - } - /* Persist layer-1 background_activation to node store. */ - for (int64_t i = 0; i < g->node_count; i++) { - g->nodes[i].background_activation = reached[i] ? best_bg[i] : 0.0; - } - - /* ── PASS 2: executive filter → working memory promotion ──────────── */ - /* Step A: collect inhibitory suppressions from fired inhibitory edges. - * Layered consciousness: inhibition is ONLY recorded against targets - * whose layer is `suppressible == 1`. Nodes in non-suppressible layers - * (Layer 0 / safety) ignore inhibitory edges entirely — their working - * memory weight cannot be silenced by attentional suppression. */ - double* inhibition = calloc((size_t)g->node_count, sizeof(double)); - if (!inhibition) { - free(best_bg); free(best_hops); free(reached); free(seeds); free(fr); - return out; - } - for (int64_t ei = 0; ei < g->edge_count; ei++) { - EngramEdge* e = &g->edges[ei]; - if (!e->inhibitory) continue; - int64_t src = engram_find_node_index(e->from_id); - int64_t tgt = engram_find_node_index(e->to_id); - if (src < 0 || tgt < 0) continue; - if (!reached[src] || best_bg[src] <= 0.0) continue; - /* Skip if target layer is non-suppressible: Layer 0 / safety nodes - * are immune to inhibitory edges from any source. The pass-3 - * override below also force-promotes them, but recording inhibition - * against them at all would be wasted work and could confuse - * downstream debugging output. */ - if (!engram_layer_is_suppressible(g->nodes[tgt].layer_id)) continue; - /* Inhibition strength proportional to source background activation - * and edge weight. Takes the maximum if multiple inhibitory edges - * target the same node. */ - double inh = best_bg[src] * e->weight; - if (inh > inhibition[tgt]) inhibition[tgt] = inh; - } - /* Step B: compute working_memory_weight per candidate node. */ - double* wm_weights = calloc((size_t)g->node_count, sizeof(double)); - if (!wm_weights) { - free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); return out; - } - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i] || best_bg[i] <= 0.0) continue; - EngramNode* n = &g->nodes[i]; - /* Per-type threshold: safety nodes break through more easily. */ - double type_threshold = engram_type_threshold(n->node_type, n->tier); - /* Goal bias weights the node's relevance to current intent. */ - double bias = engram_goal_bias(n, q); - /* Raw working memory score. */ - double raw_wm = best_bg[i] * bias * n->confidence; - /* Apply inhibitory suppression. Full inhibition → scale by factor. */ - double inh = inhibition[i]; - if (inh > 1.0) inh = 1.0; - double suppress = 1.0 - (1.0 - ENGRAM_INHIBITION_FACTOR) * inh; - raw_wm *= suppress; - /* Threshold gate: must exceed per-type threshold to enter working - * memory. Type threshold replaces the old flat 0.2 filter. */ - if (raw_wm >= type_threshold) { - wm_weights[i] = raw_wm > 1.0 ? 1.0 : raw_wm; - if (n->suppression_count > 0) n->suppression_count = 0; - } else { - /* Node didn't make it through — increment suppression counter. - * After N consecutive suppressions: force breakthrough. */ - n->suppression_count++; - if (n->suppression_count >= ENGRAM_SUPPRESSION_BREAKTHROUGH) { - wm_weights[i] = ENGRAM_BREAKTHROUGH_WEIGHT; - n->suppression_count = 0; - } else { - wm_weights[i] = 0.0; - } - } - } - /* ── PASS 3: Layer 0 override (the sacred fire) ───────────────────── - * Every node in a non-suppressible layer that received any background - * activation is force-promoted to AT LEAST ENGRAM_LAYER0_OVERRIDE_WEIGHT. - * This runs LAST and overrides whatever Pass 2 decided — Layer 0 cannot - * be silenced by inhibitory edges, by goal-bias misalignment, by - * confidence weighting, or by per-type threshold gates. If the seed - * fan-out reached a structural-floor node, that node surfaces. - * - * Note: this also clears the suppression_count when an override fires, - * since the node DID surface this turn — it just took the override path - * rather than the standard threshold path. Without this, a Layer 0 - * node with persistent inhibitory pressure would accumulate - * suppression_count forever and never reach the breakthrough state. */ - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i] || best_bg[i] <= 0.0) continue; - EngramNode* n = &g->nodes[i]; - if (engram_layer_is_suppressible(n->layer_id)) continue; - if (wm_weights[i] < ENGRAM_LAYER0_OVERRIDE_WEIGHT) { - wm_weights[i] = ENGRAM_LAYER0_OVERRIDE_WEIGHT; - } - n->suppression_count = 0; - } - - /* Persist working_memory_weight (post Pass 3) to node store. */ - for (int64_t i = 0; i < g->node_count; i++) { - g->nodes[i].working_memory_weight = wm_weights[i]; - } - - /* ── Collect all background-activated nodes for the return value ──── - * Callers see both layers. Context compilation uses only promoted nodes - * (working_memory_weight > 0). Sort: promoted first by wm_weight desc, - * then background-only by background_activation desc. */ - typedef struct { int64_t idx; double bg; double wm; double epist; int64_t hops; } Result; - Result* results = malloc((size_t)g->node_count * sizeof(Result)); - int64_t rcount = 0; - if (!results) { - free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); free(wm_weights); return out; - } - for (int64_t i = 0; i < g->node_count; i++) { - if (!reached[i]) continue; - double epist = best_bg[i] * g->nodes[i].confidence; - /* Include if promoted to working memory OR if background activation - * is meaningful enough to report (epist >= 0.1). */ - if (epist < 0.1 && wm_weights[i] <= 0.0) continue; - results[rcount].idx = i; - results[rcount].bg = best_bg[i]; - results[rcount].wm = wm_weights[i]; - results[rcount].epist = epist; - results[rcount].hops = best_hops[i]; - rcount++; - } - /* Sort: promoted nodes first (by wm_weight desc), then background-only - * by background_activation desc. */ - for (int64_t i = 1; i < rcount; i++) { - Result key = results[i]; - int64_t j = i - 1; - while (j >= 0 && (results[j].wm < key.wm || - (results[j].wm == key.wm && results[j].bg < key.bg))) { - results[j + 1] = results[j]; - j--; - } - results[j + 1] = key; - } - for (int64_t i = 0; i < rcount; i++) { - el_val_t entry = el_map_new(0); - entry = el_map_set(entry, EL_STR(el_strdup("node")), - engram_node_to_map(&g->nodes[results[i].idx])); - entry = el_map_set(entry, EL_STR(el_strdup("activation_strength")), - el_from_float(results[i].bg)); - entry = el_map_set(entry, EL_STR(el_strdup("working_memory_weight")), - el_from_float(results[i].wm)); - entry = el_map_set(entry, EL_STR(el_strdup("epistemic_confidence")), - el_from_float(results[i].epist)); - entry = el_map_set(entry, EL_STR(el_strdup("hops")), - (el_val_t)results[i].hops); - entry = el_map_set(entry, EL_STR(el_strdup("promoted")), - (el_val_t)(results[i].wm > 0.0 ? 1 : 0)); - out = el_list_append(out, entry); - } - free(best_bg); free(best_hops); free(reached); - free(seeds); free(fr); free(inhibition); free(wm_weights); free(results); - return out; -} - -/* ── Engram persistence (JSON snapshot) ─────────────────────────────────── */ - -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n) { - jb_putc(b, '{'); - jb_puts(b, "\"id\":"); jb_emit_escaped(b, n->id ? n->id : ""); - jb_puts(b, ",\"content\":"); jb_emit_escaped(b, n->content ? n->content : ""); - jb_puts(b, ",\"node_type\":"); jb_emit_escaped(b, n->node_type ? n->node_type : ""); - jb_puts(b, ",\"label\":"); jb_emit_escaped(b, n->label ? n->label : ""); - jb_puts(b, ",\"tier\":"); jb_emit_escaped(b, n->tier ? n->tier : "Working"); - jb_puts(b, ",\"tags\":"); jb_emit_escaped(b, n->tags ? n->tags : ""); - jb_puts(b, ",\"metadata\":"); jb_emit_escaped(b, n->metadata ? n->metadata : "{}"); - char tmp[80]; - snprintf(tmp, sizeof(tmp), ",\"salience\":%g", n->salience); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"importance\":%g", n->importance); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"confidence\":%g", n->confidence); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"temporal_decay_rate\":%g", n->temporal_decay_rate); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"activation_count\":%lld", (long long)n->activation_count); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"last_activated\":%lld", (long long)n->last_activated); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"created_at\":%lld", (long long)n->created_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"updated_at\":%lld", (long long)n->updated_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"background_activation\":%g", n->background_activation); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"working_memory_weight\":%g", n->working_memory_weight); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppression_count\":%d", n->suppression_count); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"layer_id\":%u", n->layer_id); jb_puts(b, tmp); - jb_putc(b, '}'); -} - -static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e) { - jb_putc(b, '{'); - jb_puts(b, "\"id\":"); jb_emit_escaped(b, e->id ? e->id : ""); - jb_puts(b, ",\"from_id\":"); jb_emit_escaped(b, e->from_id ? e->from_id : ""); - jb_puts(b, ",\"to_id\":"); jb_emit_escaped(b, e->to_id ? e->to_id : ""); - jb_puts(b, ",\"relation\":"); jb_emit_escaped(b, e->relation ? e->relation : ""); - jb_puts(b, ",\"metadata\":"); jb_emit_escaped(b, e->metadata ? e->metadata : "{}"); - char tmp[64]; - snprintf(tmp, sizeof(tmp), ",\"weight\":%g", e->weight); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"confidence\":%g", e->confidence); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"created_at\":%lld", (long long)e->created_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"updated_at\":%lld", (long long)e->updated_at); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"last_fired\":%lld", (long long)e->last_fired); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"inhibitory\":%d", e->inhibitory ? 1 : 0); jb_puts(b, tmp); - snprintf(tmp, sizeof(tmp), ",\"layer_id\":%u", e->layer_id); jb_puts(b, tmp); - jb_putc(b, '}'); -} - -el_val_t engram_save(el_val_t path) { - const char* p = EL_CSTR(path); - if (!p || !*p) return 0; - EngramStore* g = engram_get(); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"nodes\":["); - for (int64_t i = 0; i < g->node_count; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[i]); - } - jb_puts(&b, "],\"edges\":["); - for (int64_t i = 0; i < g->edge_count; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_edge_json(&b, &g->edges[i]); - } - /* Layered consciousness — emit the layer registry under "layers". - * Older readers that don't know about this top-level key will simply - * ignore it (forward compatible). Tombstoned (removed-injectable) - * layers are skipped — they have no name and can't be re-created - * meaningfully on load anyway. */ - jb_puts(&b, "],\"layers\":["); - int first_layer = 1; - for (size_t i = 0; i < g->layer_count; i++) { - EngramLayer* L = &g->layers[i]; - if (!L->name) continue; - if (!first_layer) jb_putc(&b, ','); - first_layer = 0; - jb_putc(&b, '{'); - char tmp[80]; - snprintf(tmp, sizeof(tmp), "\"layer_id\":%u", L->layer_id); - jb_puts(&b, tmp); - jb_puts(&b, ",\"name\":"); - jb_emit_escaped(&b, L->name); - snprintf(tmp, sizeof(tmp), ",\"activation_priority\":%u", L->activation_priority); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppressible\":%d", L->suppressible ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"transparent\":%d", L->transparent ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"injectable\":%d", L->injectable ? 1 : 0); - jb_puts(&b, tmp); - jb_putc(&b, '}'); - } - jb_puts(&b, "]}"); - FILE* f = fopen(p, "wb"); - if (!f) { free(b.buf); return 0; } - size_t w = fwrite(b.buf, 1, b.len, f); - fclose(f); - int ok = (w == b.len); - free(b.buf); - return ok ? 1 : 0; -} - -/* Helper: extract a string field from a JSON object substring. */ -static char* eg_get_str_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p) return el_strdup(""); - if (*p != '"') return el_strdup(""); - JsonParser jp = { .p = p, .end = p + strlen(p), .err = 0 }; - char* out = jp_parse_string_raw(&jp); - if (jp.err) { free(out); return el_strdup(""); } - return out; -} - -static double eg_get_num_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p || *p == '"' || *p == '{' || *p == '[') return 0.0; - return strtod(p, NULL); -} - -static int64_t eg_get_int_field(const char* obj, const char* key) { - const char* p = json_find_key(obj, key); - if (!p || *p == '"' || *p == '{' || *p == '[') return 0; - return strtoll(p, NULL, 10); -} - -/* Iterate the top-level nodes/edges arrays in a saved snapshot. */ -static const char* eg_skip_ws(const char* p) { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - return p; -} - -el_val_t engram_load(el_val_t path) { - const char* p = EL_CSTR(path); - if (!p || !*p) return 0; - FILE* f = fopen(p, "rb"); - if (!f) return 0; - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz <= 0) { fclose(f); return 0; } - char* data = malloc((size_t)sz + 1); - if (!data) { fclose(f); return 0; } - size_t got = fread(data, 1, (size_t)sz, f); - fclose(f); - data[got] = '\0'; - - /* Reset store */ - EngramStore* g = engram_get(); - for (int64_t i = 0; i < g->node_count; i++) { - free(g->nodes[i].id); free(g->nodes[i].content); free(g->nodes[i].node_type); - free(g->nodes[i].label); free(g->nodes[i].tier); free(g->nodes[i].tags); - free(g->nodes[i].metadata); - } - g->node_count = 0; - for (int64_t i = 0; i < g->edge_count; i++) { - free(g->edges[i].id); free(g->edges[i].from_id); free(g->edges[i].to_id); - free(g->edges[i].relation); free(g->edges[i].metadata); - } - g->edge_count = 0; - - /* Walk nodes array */ - const char* nodes_p = json_find_key(data, "nodes"); - if (nodes_p) { - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == '[') { - nodes_p++; - nodes_p = eg_skip_ws(nodes_p); - while (*nodes_p && *nodes_p != ']') { - if (*nodes_p != '{') { nodes_p++; continue; } - const char* end = json_skip_value(nodes_p); - size_t n = (size_t)(end - nodes_p); - char* obj = malloc(n + 1); - memcpy(obj, nodes_p, n); obj[n] = '\0'; - engram_grow_nodes(); - EngramNode* nn = &g->nodes[g->node_count]; - memset(nn, 0, sizeof(*nn)); - nn->id = eg_get_str_field(obj, "id"); - nn->content = eg_get_str_field(obj, "content"); - nn->node_type = eg_get_str_field(obj, "node_type"); - nn->label = eg_get_str_field(obj, "label"); - nn->tier = eg_get_str_field(obj, "tier"); - nn->tags = eg_get_str_field(obj, "tags"); - nn->metadata = eg_get_str_field(obj, "metadata"); - if (!nn->metadata || !*nn->metadata) { free(nn->metadata); nn->metadata = el_strdup("{}"); } - nn->salience = eg_get_num_field(obj, "salience"); - nn->importance = eg_get_num_field(obj, "importance"); - nn->confidence = eg_get_num_field(obj, "confidence"); - nn->temporal_decay_rate = eg_get_num_field(obj, "temporal_decay_rate"); - /* temporal_decay_rate defaults to 0 (use global) if absent in snapshot */ - nn->activation_count = eg_get_int_field(obj, "activation_count"); - nn->last_activated = eg_get_int_field(obj, "last_activated"); - nn->created_at = eg_get_int_field(obj, "created_at"); - nn->updated_at = eg_get_int_field(obj, "updated_at"); - nn->background_activation = eg_get_num_field(obj, "background_activation"); - nn->working_memory_weight = eg_get_num_field(obj, "working_memory_weight"); - nn->suppression_count = (int32_t)eg_get_int_field(obj, "suppression_count"); - /* layer_id defaults to ENGRAM_LAYER_DEFAULT (core-identity) - * for snapshots that predate the layered schema. We can't - * tell "explicit 0" from "missing field" using the helper - * directly, so probe for the key — if absent, fall back. */ - if (json_find_key(obj, "layer_id")) { - nn->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - nn->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->node_count++; - free(obj); - nodes_p = end; - nodes_p = eg_skip_ws(nodes_p); - if (*nodes_p == ',') { nodes_p++; nodes_p = eg_skip_ws(nodes_p); } - } - } - } - /* Walk edges array */ - const char* edges_p = json_find_key(data, "edges"); - if (edges_p) { - edges_p = eg_skip_ws(edges_p); - if (*edges_p == '[') { - edges_p++; - edges_p = eg_skip_ws(edges_p); - while (*edges_p && *edges_p != ']') { - if (*edges_p != '{') { edges_p++; continue; } - const char* end = json_skip_value(edges_p); - size_t n = (size_t)(end - edges_p); - char* obj = malloc(n + 1); - memcpy(obj, edges_p, n); obj[n] = '\0'; - engram_grow_edges(); - EngramEdge* ee = &g->edges[g->edge_count]; - memset(ee, 0, sizeof(*ee)); - ee->id = eg_get_str_field(obj, "id"); - ee->from_id = eg_get_str_field(obj, "from_id"); - ee->to_id = eg_get_str_field(obj, "to_id"); - ee->relation = eg_get_str_field(obj, "relation"); - ee->metadata = eg_get_str_field(obj, "metadata"); - if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = el_strdup("{}"); } - ee->weight = eg_get_num_field(obj, "weight"); - ee->confidence = eg_get_num_field(obj, "confidence"); - ee->created_at = eg_get_int_field(obj, "created_at"); - ee->updated_at = eg_get_int_field(obj, "updated_at"); - ee->last_fired = eg_get_int_field(obj, "last_fired"); - ee->inhibitory = (int)eg_get_int_field(obj, "inhibitory"); - if (json_find_key(obj, "layer_id")) { - ee->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - } else { - ee->layer_id = ENGRAM_LAYER_DEFAULT; - } - g->edge_count++; - free(obj); - edges_p = end; - edges_p = eg_skip_ws(edges_p); - if (*edges_p == ',') { edges_p++; edges_p = eg_skip_ws(edges_p); } - } - } - } - /* Walk layers array (optional — older snapshots omit this). - * If present we replace the canonical registry entirely; if absent we - * keep whatever the engram_get() init established. */ - const char* layers_p = json_find_key(data, "layers"); - if (layers_p) { - layers_p = eg_skip_ws(layers_p); - if (*layers_p == '[') { - /* Reset existing layer registry. Free strdup'd names; the - * struct array itself can be reused. */ - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) free(g->layers[i].name); - g->layers[i].name = NULL; - } - g->layer_count = 0; - - layers_p++; - layers_p = eg_skip_ws(layers_p); - while (*layers_p && *layers_p != ']') { - if (*layers_p != '{') { layers_p++; continue; } - const char* end = json_skip_value(layers_p); - size_t n = (size_t)(end - layers_p); - char* obj = malloc(n + 1); - memcpy(obj, layers_p, n); obj[n] = '\0'; - if (g->layer_count >= g->layer_capacity) { - size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; - EngramLayer* grown = realloc(g->layers, nc * sizeof(EngramLayer)); - if (!grown) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(grown + g->layer_capacity, 0, - (nc - g->layer_capacity) * sizeof(EngramLayer)); - g->layers = grown; - g->layer_capacity = nc; - } - EngramLayer* L = &g->layers[g->layer_count]; - memset(L, 0, sizeof(*L)); - L->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); - L->activation_priority = (uint32_t)eg_get_int_field(obj, "activation_priority"); - L->suppressible = (int)eg_get_int_field(obj, "suppressible") ? 1 : 0; - L->transparent = (int)eg_get_int_field(obj, "transparent") ? 1 : 0; - L->injectable = (int)eg_get_int_field(obj, "injectable") ? 1 : 0; - char* nm = eg_get_str_field(obj, "name"); - if (nm && *nm) { - L->name = el_strdup_persist(nm); - free(nm); - } else { - free(nm); - L->name = el_strdup_persist(""); - } - g->layer_count++; - free(obj); - layers_p = end; - layers_p = eg_skip_ws(layers_p); - if (*layers_p == ',') { layers_p++; layers_p = eg_skip_ws(layers_p); } - } - } - } - free(data); - return 1; -} - -/* ── Engram JSON-string accessors ───────────────────────────────────────── - * These return pre-serialized JSON strings so callers (especially HTTP - * handlers) don't have to round-trip ElList/ElMap through json_stringify - * — which can't reliably distinguish those structures from raw pointers - * due to el_val_t's type erasure. The runtime knows the real C types and - * can serialize directly. */ - -el_val_t engram_get_node_json(el_val_t id) { - const char* sid = EL_CSTR(id); - EngramNode* n = engram_find_node(sid); - if (!n) return el_wrap_str(el_strdup("{}")); - JsonBuf b; jb_init(&b); - engram_emit_node_json(&b, n); - return el_wrap_str(b.buf); -} - -el_val_t engram_search_json(el_val_t query, el_val_t limit) { - EngramStore* g = engram_get(); - const char* q = EL_CSTR(query); - int64_t lim = (int64_t)limit; - if (lim <= 0) lim = 100; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - int first = 1; - int64_t found = 0; - if (q && *q) { - for (int64_t i = 0; i < g->node_count && found < lim; i++) { - EngramNode* n = &g->nodes[i]; - /* Filter transparent layers — same as engram_search. */ - if (engram_layer_is_transparent(n->layer_id)) continue; - if (istr_contains(n->content, q) || - istr_contains(n->label, q) || - istr_contains(n->tags, q)) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, n); - first = 0; - found++; - } - } - } - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset) { - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (g->node_count == 0) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - /* Skip transparent layers — introspection filter, same as engram_scan_nodes. */ - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - int first = 1; - for (int64_t i = off; i < end; i++) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - first = 0; - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -/* engram_scan_nodes_by_type_json — filter by node_type before paginating. - * Empty / NULL type_v falls back to the unfiltered scan (existing behaviour). - * Result is JSON array, salience-sorted, transparent layers skipped. */ -el_val_t engram_scan_nodes_by_type_json(el_val_t type_v, el_val_t limit, el_val_t offset) { - const char* type_filter = EL_CSTR(type_v); - if (!type_filter || !*type_filter) { - return engram_scan_nodes_json(limit, offset); - } - EngramStore* g = engram_get(); - int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100; - int64_t off = (int64_t)offset; if (off < 0) off = 0; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (g->node_count == 0) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - int64_t live = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (engram_layer_is_transparent(g->nodes[i].layer_id)) continue; - const char* nt = g->nodes[i].node_type; - if (!nt || strcmp(nt, type_filter) != 0) continue; - idx[live++] = i; - } - engram_sort_indices_by_salience(idx, live, g->nodes); - int64_t end = off + lim; - if (end > live) end = live; - int first = 1; - for (int64_t i = off; i < end; i++) { - if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - first = 0; - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction) { - /* Re-implement here directly so we serialize without going through - * the ElList path. Walks BFS to max_depth, emits {node, edge, hops} - * triples. */ - EngramStore* g = engram_get(); - const char* sid = EL_CSTR(node_id); - int64_t depth = (int64_t)max_depth; if (depth <= 0) depth = 1; - const char* dir = EL_CSTR(direction); if (!dir) dir = "both"; - int allow_out = (strcmp(dir, "out") == 0) || (strcmp(dir, "both") == 0); - int allow_in = (strcmp(dir, "in") == 0) || (strcmp(dir, "both") == 0); - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (!sid || !*sid) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - - /* Frontier of (node_id, hops). Cap to a sane size. */ - char** frontier = calloc(1024, sizeof(char*)); - int64_t* frontier_h = calloc(1024, sizeof(int64_t)); - int64_t fc = 0; - char** visited = calloc(1024, sizeof(char*)); - int64_t vc = 0; - if (!frontier || !frontier_h || !visited) { - free(frontier); free(frontier_h); free(visited); - jb_putc(&b, ']'); return el_wrap_str(b.buf); - } - frontier[fc] = el_strdup_persist(sid); frontier_h[fc] = 0; fc++; - visited[vc++] = el_strdup_persist(sid); - - int first = 1; - while (fc > 0) { - char* cur = frontier[0]; int64_t h = frontier_h[0]; - for (int64_t k = 1; k < fc; k++) { frontier[k-1] = frontier[k]; frontier_h[k-1] = frontier_h[k]; } - fc--; - if (h >= depth) { free(cur); continue; } - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - const char* peer = NULL; - if (allow_out && e->from_id && strcmp(e->from_id, cur) == 0) peer = e->to_id; - else if (allow_in && e->to_id && strcmp(e->to_id, cur) == 0) peer = e->from_id; - if (!peer) continue; - int seen = 0; - for (int64_t v = 0; v < vc; v++) { - if (strcmp(visited[v], peer) == 0) { seen = 1; break; } - } - if (seen) continue; - EngramNode* n = engram_find_node(peer); - if (!n) continue; - if (!first) jb_putc(&b, ','); - jb_puts(&b, "{\"node\":"); - engram_emit_node_json(&b, n); - jb_puts(&b, ",\"edge\":"); - engram_emit_edge_json(&b, e); - char tmp[64]; snprintf(tmp, sizeof(tmp), ",\"hops\":%lld}", (long long)(h + 1)); - jb_puts(&b, tmp); - first = 0; - if (vc < 1024) visited[vc++] = el_strdup_persist(peer); - if (fc < 1024 && h + 1 < depth) { frontier[fc] = el_strdup_persist(peer); frontier_h[fc] = h + 1; fc++; } - } - free(cur); - } - for (int64_t i = 0; i < fc; i++) free(frontier[i]); - for (int64_t i = 0; i < vc; i++) free(visited[i]); - free(frontier); free(frontier_h); free(visited); - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -el_val_t engram_activate_json(el_val_t query, el_val_t depth) { - /* Run two-layer engram_activate and serialize the result list to JSON. - * Each entry includes both activation_strength (layer 1 background) and - * working_memory_weight (layer 2 executive filter), plus promoted flag. - * Callers performing context compilation should filter to promoted=1. */ - el_val_t lst = engram_activate(query, depth); - ElList* arr = (ElList*)(uintptr_t)lst; - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - if (arr) { - for (int64_t i = 0; i < arr->length; i++) { - if (!arr->elems[i]) continue; - el_val_t node_map = el_map_get(arr->elems[i], EL_STR("node")); - el_val_t strength_v = el_map_get(arr->elems[i], EL_STR("activation_strength")); - el_val_t wm_v = el_map_get(arr->elems[i], EL_STR("working_memory_weight")); - el_val_t epist_v = el_map_get(arr->elems[i], EL_STR("epistemic_confidence")); - el_val_t hops_v = el_map_get(arr->elems[i], EL_STR("hops")); - el_val_t promoted_v = el_map_get(arr->elems[i], EL_STR("promoted")); - /* Look up underlying EngramNode by id to emit canonical JSON. */ - el_val_t id_v = el_map_get(node_map, EL_STR("id")); - const char* id_s = EL_CSTR(id_v); - EngramNode* n = id_s ? engram_find_node(id_s) : NULL; - if (i > 0) jb_putc(&b, ','); - jb_puts(&b, "{\"node\":"); - if (n) { - engram_emit_node_json(&b, n); - } else { - jb_puts(&b, "{}"); - } - char tmp[80]; - snprintf(tmp, sizeof(tmp), ",\"activation_strength\":%g", el_to_float(strength_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"working_memory_weight\":%g", el_to_float(wm_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"epistemic_confidence\":%g", el_to_float(epist_v)); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"hops\":%lld", (long long)(int64_t)hops_v); jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"promoted\":%d}", (int)(int64_t)promoted_v); jb_puts(&b, tmp); - } - } - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -el_val_t engram_stats_json(void) { - EngramStore* g = engram_get(); - char buf[128]; - snprintf(buf, sizeof(buf), - "{\"node_count\":%lld,\"edge_count\":%lld,\"layer_count\":%zu}", - (long long)g->node_count, (long long)g->edge_count, g->layer_count); - return el_wrap_str(el_strdup(buf)); -} - -/* engram_list_layers_json — serialized counterpart of engram_list_layers. - * Returns a JSON array, sorted by activation_priority ascending. */ -el_val_t engram_list_layers_json(void) { - EngramStore* g = engram_get(); - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - /* Build a sorted index over live layers. */ - size_t* idx = malloc((g->layer_count + 1) * sizeof(size_t)); - if (!idx) { jb_putc(&b, ']'); return el_wrap_str(b.buf); } - size_t live = 0; - for (size_t i = 0; i < g->layer_count; i++) { - if (g->layers[i].name) idx[live++] = i; - } - for (size_t i = 1; i < live; i++) { - size_t key = idx[i]; - uint32_t kp = g->layers[key].activation_priority; - size_t j = i; - while (j > 0 && g->layers[idx[j - 1]].activation_priority > kp) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - int first = 1; - for (size_t i = 0; i < live; i++) { - EngramLayer* L = &g->layers[idx[i]]; - if (!first) jb_putc(&b, ','); - first = 0; - jb_putc(&b, '{'); - char tmp[80]; - snprintf(tmp, sizeof(tmp), "\"layer_id\":%u", L->layer_id); jb_puts(&b, tmp); - jb_puts(&b, ",\"name\":"); - jb_emit_escaped(&b, L->name ? L->name : ""); - snprintf(tmp, sizeof(tmp), ",\"activation_priority\":%u", L->activation_priority); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"suppressible\":%d", L->suppressible ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"transparent\":%d", L->transparent ? 1 : 0); - jb_puts(&b, tmp); - snprintf(tmp, sizeof(tmp), ",\"injectable\":%d", L->injectable ? 1 : 0); - jb_puts(&b, tmp); - jb_putc(&b, '}'); - } - free(idx); - jb_putc(&b, ']'); - return el_wrap_str(b.buf); -} - -/* engram_compile_layered_json — produce a prompt-ready context block split - * by layer. - * - * Runs the three-pass activation, then partitions promoted nodes by layer - * suppressibility: - * - Non-suppressible (Layer 0 / structural-floor) layers go FIRST under - * the heading "[LAYER 0 — STRUCTURAL]". These are the sacred-fire - * nodes that surfaced via the pass-3 override. - * - All other promoted layers go SECOND under "[ENGRAM CONTEXT]". - * - * Output is a single JSON-string el_val_t: a UTF-8 text block ready to be - * concatenated into a system prompt. Returns "" if no nodes promoted. - * - * Transparent layers (Layer 0) are emitted into the prompt — they shape - * the model's output — but engram_search and friends still hide them from - * introspection-style queries. The split heading lets the LLM weight them - * appropriately without revealing their internal label. - * - * Each emitted line for a node is its raw JSON (matching engram_emit_node_json) - * so downstream JSON parsers can still walk individual records inside the - * formatted block. The block is plain text, not a JSON document — callers - * concatenating it into a prompt should treat it as opaque markdown. */ -el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth) { - EngramStore* g = engram_get(); - /* Run the three-pass activator. We need the persisted node fields, so - * call engram_activate (it writes background_activation and - * working_memory_weight back into the store). */ - (void)engram_activate(intent, depth); - - /* Walk the store and partition by suppressibility. */ - JsonBuf b; jb_init(&b); - int wrote_layer0 = 0; - int wrote_normal = 0; - - /* Sort indices by working_memory_weight descending so the most - * confidently promoted nodes appear first within each section. */ - int64_t* idx = malloc((size_t)(g->node_count + 1) * sizeof(int64_t)); - if (!idx) return el_wrap_str(el_strdup("")); - int64_t mc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - if (g->nodes[i].working_memory_weight > 0.0) idx[mc++] = i; - } - for (int64_t i = 1; i < mc; i++) { - int64_t key = idx[i]; - double kw = g->nodes[key].working_memory_weight; - int64_t j = i; - while (j > 0 && g->nodes[idx[j - 1]].working_memory_weight < kw) { - idx[j] = idx[j - 1]; - j--; - } - idx[j] = key; - } - - /* Section 1: structural floor (non-suppressible layers). */ - for (int64_t i = 0; i < mc; i++) { - EngramNode* n = &g->nodes[idx[i]]; - if (engram_layer_is_suppressible(n->layer_id)) continue; - if (!wrote_layer0) { - jb_puts(&b, "[LAYER 0 — STRUCTURAL]\n"); - wrote_layer0 = 1; - } - engram_emit_node_json(&b, n); - jb_putc(&b, '\n'); - } - - /* Section 2: standard engram context (suppressible layers). */ - for (int64_t i = 0; i < mc; i++) { - EngramNode* n = &g->nodes[idx[i]]; - if (!engram_layer_is_suppressible(n->layer_id)) continue; - if (!wrote_normal) { - if (wrote_layer0) jb_putc(&b, '\n'); - jb_puts(&b, "[ENGRAM CONTEXT]\n"); - wrote_normal = 1; - } - engram_emit_node_json(&b, n); - jb_putc(&b, '\n'); - } - - free(idx); - if (b.len == 0) { - free(b.buf); - return el_wrap_str(el_strdup("")); - } - return el_wrap_str(b.buf); -} - -/* engram_query_range — temporal range query. - * Returns a JSON array of nodes whose created_at OR last_activated falls - * within [start_ms, end_ms], sorted by created_at ascending. - * Enables "what was I working on last Tuesday?" style queries by passing - * unix-millisecond timestamps for the start and end of the target interval. - * Both endpoints are inclusive. Pass 0 for start_ms to mean "beginning of - * time"; pass 0 for end_ms to mean "now". */ -el_val_t engram_query_range(el_val_t start_ms_v, el_val_t end_ms_v) { - EngramStore* g = engram_get(); - int64_t start_ms = (int64_t)start_ms_v; - int64_t end_ms = (int64_t)end_ms_v; - if (end_ms <= 0) end_ms = engram_now_ms(); - - /* Collect matching indices. */ - int64_t* idx = malloc((size_t)g->node_count * sizeof(int64_t)); - if (!idx) return el_wrap_str(el_strdup("[]")); - int64_t mc = 0; - for (int64_t i = 0; i < g->node_count; i++) { - EngramNode* n = &g->nodes[i]; - int in_created = (n->created_at >= start_ms && n->created_at <= end_ms); - int in_activated = (n->last_activated >= start_ms && n->last_activated <= end_ms); - if (in_created || in_activated) idx[mc++] = i; - } - /* Sort by created_at ascending (insertion sort — N is small in practice). */ - for (int64_t i = 1; i < mc; i++) { - int64_t key = idx[i]; - int64_t kts = g->nodes[key].created_at; - int64_t j = i - 1; - while (j >= 0 && g->nodes[idx[j]].created_at > kts) { - idx[j + 1] = idx[j]; - j--; - } - idx[j + 1] = key; - } - JsonBuf b; jb_init(&b); - jb_putc(&b, '['); - for (int64_t i = 0; i < mc; i++) { - if (i > 0) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); - } - jb_putc(&b, ']'); - free(idx); - return el_wrap_str(b.buf); -} - -/* ── DHARMA network ───────────────────────────────────────────────────────── - * Real implementation. Peers are addressed by `dharma_id` — either bare - * (e.g. "ntn-genesis", transport defaults to http://localhost:7770) or - * "@" where is the peer's Engram-exposed daemon. - * - * Channels are logical handles cached per-cgi: `dharma_connect` is - * idempotent and returns "ch:". The channel registry below tracks - * every cgi_id we've connected to and its resolved transport URL. - * - * Relationship weights live in the local Engram graph: edges of type - * "dharma-relation" between a synthetic local node ("dharma:self") and - * synthetic peer nodes ("dharma:peer:"). Hebbian increments - * accumulate in EngramEdge.weight, clamped to [0.0, 1.0]. - * - * Events arrive over HTTP via the application's request handler, which is - * expected to call el_runtime_dharma_event_arrive() when it sees a - * /dharma/event POST. dharma_field() blocks on a per-event-type queue. - */ - -#define DHARMA_DEFAULT_URL "http://localhost:7770" - -/* Channel registry — one entry per known peer. */ -typedef struct DharmaChannel { - char* cgi_id; /* full dharma_id including any @ suffix */ - char* base_id; /* registry-id portion (before @) for relationship lookup */ - char* url; /* resolved transport URL */ - char* channel_id; /* "ch:" */ -} DharmaChannel; - -static DharmaChannel* _dharma_channels = NULL; -static size_t _dharma_channel_count = 0; -static size_t _dharma_channel_cap = 0; -static pthread_mutex_t _dharma_channel_mu = PTHREAD_MUTEX_INITIALIZER; - -/* Event queue — per-type linked list. dharma_field blocks on _dharma_event_cv. */ -typedef struct DharmaEvent { - char* event_type; - char* payload; - char* source; - int64_t timestamp; - struct DharmaEvent* next; -} DharmaEvent; - -static DharmaEvent* _dharma_event_head = NULL; -static DharmaEvent* _dharma_event_tail = NULL; -static pthread_mutex_t _dharma_event_mu = PTHREAD_MUTEX_INITIALIZER; -static pthread_cond_t _dharma_event_cv = PTHREAD_COND_INITIALIZER; - -/* Split "@" → (base_id, url). If no "@", base_id = full, url = default. - * Returned strings are heap-allocated; caller must free. */ -static void dharma_parse_id(const char* full, char** out_base, char** out_url) { - if (!full) full = ""; - const char* at = strchr(full, '@'); - if (at) { - size_t bn = (size_t)(at - full); - char* b = malloc(bn + 1); - memcpy(b, full, bn); b[bn] = '\0'; - *out_base = b; - *out_url = el_strdup(at + 1); - if (!**out_url) { free(*out_url); *out_url = el_strdup(DHARMA_DEFAULT_URL); } - } else { - *out_base = el_strdup(full); - *out_url = el_strdup(DHARMA_DEFAULT_URL); - } -} - -/* Find existing channel by full cgi_id. Caller must hold _dharma_channel_mu. */ -static DharmaChannel* dharma_find_channel_locked(const char* cgi_id) { - if (!cgi_id) return NULL; - for (size_t i = 0; i < _dharma_channel_count; i++) { - if (_dharma_channels[i].cgi_id && - strcmp(_dharma_channels[i].cgi_id, cgi_id) == 0) { - return &_dharma_channels[i]; - } - } - return NULL; -} - -/* Add a new channel entry. Caller must hold _dharma_channel_mu. */ -static DharmaChannel* dharma_add_channel_locked(const char* cgi_id) { - if (_dharma_channel_count >= _dharma_channel_cap) { - size_t nc = _dharma_channel_cap ? _dharma_channel_cap * 2 : 8; - _dharma_channels = realloc(_dharma_channels, nc * sizeof(DharmaChannel)); - if (!_dharma_channels) { fputs("el_runtime: out of memory\n", stderr); exit(1); } - memset(_dharma_channels + _dharma_channel_cap, 0, - (nc - _dharma_channel_cap) * sizeof(DharmaChannel)); - _dharma_channel_cap = nc; - } - DharmaChannel* ch = &_dharma_channels[_dharma_channel_count++]; - char* base = NULL; char* url = NULL; - dharma_parse_id(cgi_id, &base, &url); - ch->cgi_id = el_strdup(cgi_id ? cgi_id : ""); - ch->base_id = base; - ch->url = url; - size_t cn = strlen(ch->cgi_id) + 4; - ch->channel_id = malloc(cn); - snprintf(ch->channel_id, cn, "ch:%s", ch->cgi_id); - return ch; -} - -el_val_t dharma_connect(el_val_t cgi_id) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return el_wrap_str(el_strdup("")); - pthread_mutex_lock(&_dharma_channel_mu); - DharmaChannel* ch = dharma_find_channel_locked(id); - if (!ch) ch = dharma_add_channel_locked(id); - char* out = el_strdup(ch->channel_id); - pthread_mutex_unlock(&_dharma_channel_mu); - return el_wrap_str(out); -} - -/* Build an error JSON body — same shape http_error_json uses. */ -static el_val_t dharma_error_json(const char* msg) { - return http_error_json(msg); -} - -el_val_t dharma_send(el_val_t channel, el_val_t content) { - const char* ch_id = EL_CSTR(channel); - const char* msg = EL_CSTR(content); - if (!ch_id || strncmp(ch_id, "ch:", 3) != 0) { - return dharma_error_json("invalid channel"); - } - const char* peer_id = ch_id + 3; - /* Look up channel; if unknown (caller fabricated), auto-register. */ - pthread_mutex_lock(&_dharma_channel_mu); - DharmaChannel* ch = dharma_find_channel_locked(peer_id); - if (!ch) ch = dharma_add_channel_locked(peer_id); - char* url = el_strdup(ch->url); - pthread_mutex_unlock(&_dharma_channel_mu); - /* Build /dharma/recv body. */ - const char* from = _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unknown)"; - char* esc_ch = json_escape_alloc(ch_id); - char* esc_from = json_escape_alloc(from); - char* esc_msg = json_escape_alloc(msg ? msg : ""); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"channel\":\""); jb_puts(&b, esc_ch); - jb_puts(&b, "\",\"from\":\""); jb_puts(&b, esc_from); - jb_puts(&b, "\",\"content\":\""); jb_puts(&b, esc_msg); - jb_puts(&b, "\"}"); - free(esc_ch); free(esc_from); free(esc_msg); - size_t ul = strlen(url) + 16; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/dharma/recv", url); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t resp = http_do("POST", full_url, b.buf, h); - curl_slist_free_all(h); - free(b.buf); free(full_url); free(url); - return resp; -} - -el_val_t dharma_activate(el_val_t query) { - const char* q = EL_CSTR(query); - if (!q) q = ""; - el_val_t out = el_list_empty(); - char* esc_q = json_escape_alloc(q); - JsonBuf body; jb_init(&body); - jb_puts(&body, "{\"query\":\""); jb_puts(&body, esc_q); jb_puts(&body, "\"}"); - free(esc_q); - - /* Snapshot the channel list under lock so we can iterate without - * holding the mutex during network I/O. */ - pthread_mutex_lock(&_dharma_channel_mu); - size_t n = _dharma_channel_count; - char** urls = calloc(n ? n : 1, sizeof(char*)); - char** ids = calloc(n ? n : 1, sizeof(char*)); - char** bases = calloc(n ? n : 1, sizeof(char*)); - for (size_t i = 0; i < n; i++) { - urls[i] = el_strdup(_dharma_channels[i].url); - ids[i] = el_strdup(_dharma_channels[i].cgi_id); - bases[i] = el_strdup(_dharma_channels[i].base_id); - } - pthread_mutex_unlock(&_dharma_channel_mu); - - for (size_t i = 0; i < n; i++) { - size_t ul = strlen(urls[i]) + 32; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/api/activate", urls[i]); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t resp = http_do("POST", full_url, body.buf, h); - curl_slist_free_all(h); - free(full_url); - const char* rs = EL_CSTR(resp); - if (!rs || !*rs) continue; - if (rs[0] == '{' && strstr(rs, "\"error\"")) continue; - - /* Look up relationship weight (attenuation). */ - double rel_weight = 1.0; - { - const char* self_id = "dharma:self"; - char peer_node[512]; - snprintf(peer_node, sizeof(peer_node), "dharma:peer:%s", bases[i]); - EngramStore* g = engram_get(); - for (int64_t k = 0; k < g->edge_count; k++) { - EngramEdge* e = &g->edges[k]; - if (e->from_id && e->to_id && - strcmp(e->from_id, self_id) == 0 && - strcmp(e->to_id, peer_node) == 0 && - e->relation && strcmp(e->relation, "dharma-relation") == 0) { - rel_weight = e->weight; - break; - } - } - } - - /* Iterate the response array. Expect either a top-level array - * or an object whose "results" field is an array. */ - const char* arr = rs; - while (*arr == ' ' || *arr == '\t' || *arr == '\n' || *arr == '\r') arr++; - char* arr_owned = NULL; - if (*arr == '{') { - el_val_t r = json_get_raw(EL_STR(rs), EL_STR("results")); - const char* rr = EL_CSTR(r); - if (rr && *rr == '[') { - arr_owned = el_strdup(rr); - arr = arr_owned; - } else { - continue; - } - } - if (*arr != '[') { free(arr_owned); continue; } - const char* p = arr + 1; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - while (*p && *p != ']') { - const char* end = json_skip_value(p); - size_t en = (size_t)(end - p); - char* obj = el_strbuf(en); - memcpy(obj, p, en); obj[en] = '\0'; - - /* Pull activation_strength if present, else 1.0. */ - el_val_t act_v = json_get_float(EL_STR(obj), EL_STR("activation_strength")); - double act = el_to_float(act_v); - if (!(act > 0.0 && act <= 100.0)) act = 1.0; - double final_act = act * rel_weight; - - el_val_t entry = el_map_new(0); - /* node = the inner JSON if present, else the entire obj. */ - el_val_t node_raw = json_get_raw(EL_STR(obj), EL_STR("node")); - const char* nr = EL_CSTR(node_raw); - entry = el_map_set(entry, EL_STR(el_strdup("node")), - (nr && *nr) ? node_raw : EL_STR(el_strdup(obj))); - entry = el_map_set(entry, EL_STR(el_strdup("source_cgi")), - EL_STR(el_strdup(ids[i]))); - entry = el_map_set(entry, EL_STR(el_strdup("activation_strength")), - el_from_float(final_act)); - out = el_list_append(out, entry); - free(obj); - p = end; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - } - free(arr_owned); - } - for (size_t i = 0; i < n; i++) { free(urls[i]); free(ids[i]); free(bases[i]); } - free(urls); free(ids); free(bases); - free(body.buf); - return out; -} - -void dharma_emit(el_val_t event_type, el_val_t payload) { - const char* et = EL_CSTR(event_type); - const char* pay = EL_CSTR(payload); - if (!et) et = ""; - if (!pay) pay = ""; - const char* src = _el_cgi_dharma_id ? _el_cgi_dharma_id : "(unknown)"; - int64_t ts = engram_now_ms(); - - char* esc_et = json_escape_alloc(et); - char* esc_pay = json_escape_alloc(pay); - char* esc_src = json_escape_alloc(src); - JsonBuf b; jb_init(&b); - jb_puts(&b, "{\"type\":\""); jb_puts(&b, esc_et); - jb_puts(&b, "\",\"payload\":\""); jb_puts(&b, esc_pay); - jb_puts(&b, "\",\"source\":\""); jb_puts(&b, esc_src); - jb_puts(&b, "\",\"timestamp\":"); jb_emit_int(&b, ts); - jb_putc(&b, '}'); - free(esc_et); free(esc_pay); free(esc_src); - - /* Snapshot URLs to avoid holding the channel mutex during I/O. */ - pthread_mutex_lock(&_dharma_channel_mu); - size_t n = _dharma_channel_count; - char** urls = calloc(n ? n : 1, sizeof(char*)); - for (size_t i = 0; i < n; i++) urls[i] = el_strdup(_dharma_channels[i].url); - pthread_mutex_unlock(&_dharma_channel_mu); - - for (size_t i = 0; i < n; i++) { - size_t ul = strlen(urls[i]) + 32; - char* full_url = malloc(ul); - snprintf(full_url, ul, "%s/dharma/event", urls[i]); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - el_val_t r = http_do("POST", full_url, b.buf, h); - (void)r; /* fire-and-forget — emit is not synchronous */ - curl_slist_free_all(h); - free(full_url); - } - for (size_t i = 0; i < n; i++) free(urls[i]); - free(urls); - free(b.buf); -} - -void el_runtime_dharma_event_arrive(const char* event_type, const char* payload, - const char* source) { - DharmaEvent* ev = calloc(1, sizeof(DharmaEvent)); - if (!ev) return; - ev->event_type = el_strdup(event_type ? event_type : ""); - ev->payload = el_strdup(payload ? payload : ""); - ev->source = el_strdup(source ? source : ""); - ev->timestamp = engram_now_ms(); - ev->next = NULL; - pthread_mutex_lock(&_dharma_event_mu); - if (_dharma_event_tail) _dharma_event_tail->next = ev; - else _dharma_event_head = ev; - _dharma_event_tail = ev; - pthread_cond_broadcast(&_dharma_event_cv); - pthread_mutex_unlock(&_dharma_event_mu); -} - -el_val_t dharma_field(el_val_t event_type) { - const char* et = EL_CSTR(event_type); - if (!et) et = ""; - - /* Compute deadline: now + 30 seconds. */ - struct timespec deadline; - clock_gettime(CLOCK_REALTIME, &deadline); - deadline.tv_sec += 30; - - DharmaEvent* found = NULL; - pthread_mutex_lock(&_dharma_event_mu); - while (1) { - /* Scan queue for matching type; pop and return first match. */ - DharmaEvent* prev = NULL; - DharmaEvent* cur = _dharma_event_head; - while (cur) { - if (cur->event_type && strcmp(cur->event_type, et) == 0) { - if (prev) prev->next = cur->next; - else _dharma_event_head = cur->next; - if (_dharma_event_tail == cur) _dharma_event_tail = prev; - cur->next = NULL; - found = cur; - break; - } - prev = cur; cur = cur->next; - } - if (found) break; - int rc = pthread_cond_timedwait(&_dharma_event_cv, &_dharma_event_mu, &deadline); - if (rc == ETIMEDOUT) break; - } - pthread_mutex_unlock(&_dharma_event_mu); - - if (!found) return el_map_new(0); - el_val_t m = el_map_new(0); - m = el_map_set(m, EL_STR(el_strdup("type")), - EL_STR(el_strdup(found->event_type ? found->event_type : ""))); - m = el_map_set(m, EL_STR(el_strdup("payload")), - EL_STR(el_strdup(found->payload ? found->payload : ""))); - m = el_map_set(m, EL_STR(el_strdup("source_cgi")), - EL_STR(el_strdup(found->source ? found->source : ""))); - m = el_map_set(m, EL_STR(el_strdup("timestamp")), (el_val_t)found->timestamp); - free(found->event_type); free(found->payload); free(found->source); free(found); - return m; -} - -/* Locate (or create) the local "dharma:self" node and the synthetic peer - * node "dharma:peer:". Returns the index of the dharma-relation - * edge, or -1 if not found. If `create` is non-zero, ensure the nodes - * and edge exist (creating them as needed) and return the edge index. */ -static int64_t dharma_find_or_create_relation_edge(const char* peer_base, int create) { - if (!peer_base || !*peer_base) return -1; - EngramStore* g = engram_get(); - const char* self_id = "dharma:self"; - char peer_node[512]; - snprintf(peer_node, sizeof(peer_node), "dharma:peer:%s", peer_base); - - /* Look for the edge first. */ - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (e->from_id && e->to_id && - strcmp(e->from_id, self_id) == 0 && - strcmp(e->to_id, peer_node) == 0 && - e->relation && strcmp(e->relation, "dharma-relation") == 0) { - return i; - } - } - if (!create) return -1; - - /* Ensure self node exists. We use a fixed id (not engram_new_id) so - * subsequent calls reuse the same one. */ - if (!engram_find_node(self_id)) { - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = el_strdup(self_id); - n->content = el_strdup(_el_cgi_dharma_id ? _el_cgi_dharma_id : "(self)"); - n->node_type = el_strdup("DharmaSelf"); - n->label = el_strdup("dharma:self"); - n->tier = el_strdup("Working"); - n->tags = el_strdup("dharma"); - n->metadata = el_strdup("{}"); - n->salience = 1.0; n->importance = 1.0; n->confidence = 1.0; - int64_t now = engram_now_ms(); - n->created_at = now; n->updated_at = now; n->last_activated = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - } - if (!engram_find_node(peer_node)) { - engram_grow_nodes(); - EngramNode* n = &g->nodes[g->node_count]; - memset(n, 0, sizeof(*n)); - n->id = el_strdup(peer_node); - n->content = el_strdup(peer_base); - n->node_type = el_strdup("DharmaPeer"); - n->label = el_strdup(peer_node); - n->tier = el_strdup("Working"); - n->tags = el_strdup("dharma"); - n->metadata = el_strdup("{}"); - n->salience = 0.5; n->importance = 0.5; n->confidence = 1.0; - int64_t now = engram_now_ms(); - n->created_at = now; n->updated_at = now; n->last_activated = now; - n->layer_id = ENGRAM_LAYER_DEFAULT; - g->node_count++; - } - /* Create the edge with weight 0.0 — caller will increment. */ - engram_grow_edges(); - EngramEdge* e = &g->edges[g->edge_count]; - memset(e, 0, sizeof(*e)); - e->id = engram_new_id(); - e->from_id = el_strdup(self_id); - e->to_id = el_strdup(peer_node); - e->relation = el_strdup("dharma-relation"); - e->metadata = el_strdup("{}"); - e->weight = 0.0; - e->confidence = 1.0; - int64_t now = engram_now_ms(); - e->created_at = now; e->updated_at = now; - e->layer_id = ENGRAM_LAYER_DEFAULT; - int64_t idx = g->edge_count; - g->edge_count++; - return idx; -} - -void dharma_strengthen(el_val_t cgi_id, el_val_t weight) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return; - char* base = NULL; char* url = NULL; - dharma_parse_id(id, &base, &url); - free(url); - int64_t ei = dharma_find_or_create_relation_edge(base, 1); - free(base); - if (ei < 0) return; - EngramStore* g = engram_get(); - double inc = engram_decode_score(weight); - if (!(inc >= 0.0)) inc = 0.0; - double w = g->edges[ei].weight + inc; - if (w < 0.0) w = 0.0; - if (w > 1.0) w = 1.0; - g->edges[ei].weight = w; - g->edges[ei].updated_at = engram_now_ms(); - g->edges[ei].last_fired = g->edges[ei].updated_at; -} - -el_val_t dharma_relationship(el_val_t cgi_id) { - const char* id = EL_CSTR(cgi_id); - if (!id || !*id) return el_from_float(0.0); - char* base = NULL; char* url = NULL; - dharma_parse_id(id, &base, &url); - free(url); - int64_t ei = dharma_find_or_create_relation_edge(base, 0); - free(base); - if (ei < 0) return el_from_float(0.0); - EngramStore* g = engram_get(); - return el_from_float(g->edges[ei].weight); -} - -el_val_t dharma_peers(void) { - /* Walk dharma-relation edges out of "dharma:self", weight > 0, sort desc. */ - EngramStore* g = engram_get(); - const char* self_id = "dharma:self"; - typedef struct { char* peer_base; double weight; } PeerEntry; - PeerEntry* peers = malloc((size_t)(g->edge_count + 1) * sizeof(PeerEntry)); - int64_t pcount = 0; - if (!peers) return el_list_empty(); - for (int64_t i = 0; i < g->edge_count; i++) { - EngramEdge* e = &g->edges[i]; - if (!e->from_id || !e->to_id) continue; - if (strcmp(e->from_id, self_id) != 0) continue; - if (!e->relation || strcmp(e->relation, "dharma-relation") != 0) continue; - if (e->weight <= 0.0) continue; - const char* prefix = "dharma:peer:"; - size_t pl = strlen(prefix); - if (strncmp(e->to_id, prefix, pl) != 0) continue; - peers[pcount].peer_base = el_strdup(e->to_id + pl); - peers[pcount].weight = e->weight; - pcount++; - } - /* Sort desc by weight. */ - for (int64_t i = 1; i < pcount; i++) { - PeerEntry key = peers[i]; - int64_t j = i - 1; - while (j >= 0 && peers[j].weight < key.weight) { - peers[j + 1] = peers[j]; j--; - } - peers[j + 1] = key; - } - el_val_t out = el_list_empty(); - for (int64_t i = 0; i < pcount; i++) { - out = el_list_append(out, EL_STR(peers[i].peer_base)); - } - free(peers); - return out; -} - -/* ── Batch 4: LLM (Anthropic API client) ─────────────────────────────────── */ -/* - * All LLM builtins call https://api.anthropic.com/v1/messages with the API - * key from env ANTHROPIC_API_KEY. Default model is "claude-sonnet-4-5" - * when the supplied model is empty/null. - * - * `llm_call_agentic` runs a real multi-turn tool_use/tool_result loop. - * Tool handlers are registered with `llm_register_tool(name, fn_name)`, - * which dlsym()s the named symbol. Each tool handler has the C signature - * el_val_t handler(el_val_t input_json); - * and returns a JSON-string el_val_t result. Iteration is capped at 10. - */ - -static const char* LLM_DEFAULT_MODEL = "claude-sonnet-4-5"; -static const char* LLM_API_URL = "https://api.anthropic.com/v1/messages"; -static const char* LLM_VERSION = "2023-06-01"; - -static const char* llm_resolve_model(const char* m) { - if (!m || !*m) return LLM_DEFAULT_MODEL; - return m; -} - -/* - * ── Configurable LLM provider chain ────────────────────────────────────────── - * - * Providers are configured via indexed env vars. The runtime tries each in - * order (0, 1, 2, ...) and returns the first successful non-empty response. - * - * Per provider (N = 0, 1, 2, ...): - * NEURON_LLM_N_URL — endpoint URL (base URL; /v1/chat/completions appended - * if format is "openai" and not already in URL) - * NEURON_LLM_N_KEY — API key - * NEURON_LLM_N_FORMAT — "openai" (default) or "anthropic" - * NEURON_LLM_N_MODEL — model name override (optional) - * - * Example — Neuron inference primary, Anthropic fallback: - * NEURON_LLM_0_URL=https://soma.../v1/chat/completions - * NEURON_LLM_0_KEY=svc-key - * NEURON_LLM_0_FORMAT=openai - * NEURON_LLM_0_MODEL=neuron - * NEURON_LLM_1_URL=https://api.anthropic.com/v1/messages - * NEURON_LLM_1_KEY=sk-ant-... - * NEURON_LLM_1_FORMAT=anthropic - * - * If no NEURON_LLM_0_URL is set, falls back to legacy ANTHROPIC_API_KEY. - */ - -#define LLM_MAX_PROVIDERS 16 - -/* forward declarations */ -static el_val_t llm_extract_text(el_val_t resp_val); -static el_val_t llm_extract_text_openai(el_val_t resp_val); - -static el_val_t llm_extract_text_openai(el_val_t resp_val) { - const char* resp = EL_CSTR(resp_val); - if (!resp || !*resp) return el_wrap_str(el_strdup("")); - if (resp[0] == '{' && strstr(resp, "\"error\"")) return el_wrap_str(el_strdup("")); - const char* choices = json_find_key(resp, "choices"); - if (!choices || *choices != '[') return el_wrap_str(el_strdup("")); - choices++; - while (*choices == ' ' || *choices == '\t') choices++; - if (*choices != '{') return el_wrap_str(el_strdup("")); - const char* end = json_skip_value(choices); - size_t n = (size_t)(end - choices); - char* obj = malloc(n + 1); memcpy(obj, choices, n); obj[n] = '\0'; - const char* msg = json_find_key(obj, "message"); - if (!msg || *msg != '{') { free(obj); return el_wrap_str(el_strdup("")); } - const char* msg_end = json_skip_value(msg); - size_t mn = (size_t)(msg_end - msg); - char* msg_obj = malloc(mn + 1); memcpy(msg_obj, msg, mn); msg_obj[mn] = '\0'; - const char* content = json_find_key(msg_obj, "content"); - el_val_t result = el_wrap_str(el_strdup("")); - if (content && *content == '"') { - JsonParser jp = { .p = content, .end = content + strlen(content), .err = 0 }; - char* text = jp_parse_string_raw(&jp); - if (!jp.err && text) result = el_wrap_str(text); - } - free(msg_obj); free(obj); - return result; -} - -/* Send a request to one provider. Returns the raw response string. - * format: 0 = openai, 1 = anthropic */ -static el_val_t llm_provider_request(const char* url, const char* key, - int format, const char* model, - const char* system_str, - const char* user_str) { - char* esc_sys = system_str && *system_str ? json_escape_alloc(system_str) : NULL; - char* esc_user = json_escape_alloc(user_str ? user_str : ""); - JsonBuf b; jb_init(&b); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - - if (format == 0) { /* OpenAI */ - char full_url[1024]; - if (strstr(url, "/chat/completions") || strstr(url, "/messages")) { - snprintf(full_url, sizeof(full_url), "%s", url); - } else { - snprintf(full_url, sizeof(full_url), "%s/v1/chat/completions", url); - } - { size_t n = strlen(key)+24; char* l=malloc(n); snprintf(l,n,"Authorization: Bearer %s",key); h=curl_slist_append(h,l); free(l); } - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, model ? model : "neuron"); - jb_puts(&b, ",\"max_tokens\":1024,\"messages\":["); - if (esc_sys && *esc_sys) { jb_puts(&b,"{\"role\":\"system\",\"content\":\""); jb_puts(&b,esc_sys); jb_puts(&b,"\"},"); } - jb_puts(&b, "{\"role\":\"user\",\"content\":\""); jb_puts(&b, esc_user); jb_puts(&b, "\"}]}"); - el_val_t resp = http_do("POST", full_url, b.buf, h); - curl_slist_free_all(h); free(b.buf); - if (esc_sys) free(esc_sys); free(esc_user); - return llm_extract_text_openai(resp); - } else { /* Anthropic */ - { size_t n = strlen(key)+16; char* l=malloc(n); snprintf(l,n,"x-api-key: %s",key); h=curl_slist_append(h,l); free(l); } - { size_t n = strlen(LLM_VERSION)+32; char* l=malloc(n); snprintf(l,n,"anthropic-version: %s",LLM_VERSION); h=curl_slist_append(h,l); free(l); } - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, model ? model : LLM_DEFAULT_MODEL); - jb_puts(&b, ",\"max_tokens\":4096"); - if (esc_sys && *esc_sys) { jb_puts(&b,",\"system\":\""); jb_puts(&b,esc_sys); jb_puts(&b,"\""); } - jb_puts(&b, ",\"messages\":[{\"role\":\"user\",\"content\":\""); jb_puts(&b, esc_user); jb_puts(&b, "\"}]}"); - el_val_t resp = http_do("POST", url, b.buf, h); - curl_slist_free_all(h); free(b.buf); - if (esc_sys) free(esc_sys); free(esc_user); - return llm_extract_text(resp); - } -} - -static el_val_t llm_chain_call(const char* system_str, const char* user_str) { - char url_key[64], key_key[64], fmt_key[64], model_key[64]; - for (int i = 0; i < LLM_MAX_PROVIDERS; i++) { - snprintf(url_key, sizeof(url_key), "NEURON_LLM_%d_URL", i); - snprintf(key_key, sizeof(key_key), "NEURON_LLM_%d_KEY", i); - snprintf(fmt_key, sizeof(fmt_key), "NEURON_LLM_%d_FORMAT", i); - snprintf(model_key, sizeof(model_key), "NEURON_LLM_%d_MODEL", i); - const char* url = getenv(url_key); - const char* key = getenv(key_key); - if (!url || !*url || !key || !*key) break; /* end of chain */ - const char* fmt_s = getenv(fmt_key); - int fmt = (fmt_s && strcmp(fmt_s, "anthropic") == 0) ? 1 : 0; - const char* model = getenv(model_key); - fprintf(stderr, "[llm] trying provider %d (%s)\n", i, url); - el_val_t result = llm_provider_request(url, key, fmt, model, system_str, user_str); - const char* t = EL_CSTR(result); - if (t && *t && t[0] != '{') return result; /* success */ - fprintf(stderr, "[llm] provider %d failed or empty, trying next\n", i); - } - /* Legacy fallback: ANTHROPIC_API_KEY */ - const char* api_key = getenv("ANTHROPIC_API_KEY"); - if (!api_key || !*api_key) return http_error_json("no LLM providers configured"); - fprintf(stderr, "[llm] using legacy ANTHROPIC_API_KEY fallback\n"); - return llm_provider_request(LLM_API_URL, api_key, 1, NULL, system_str, user_str); -} - -/* Legacy llm_request — kept for backward compat with agentic loop internals */ -static el_val_t llm_request(const char* json_body) { - const char* api_key = getenv("ANTHROPIC_API_KEY"); - if (!api_key || !*api_key) return http_error_json("ANTHROPIC_API_KEY not set"); - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - { size_t n=strlen(api_key)+16; char* l=malloc(n); snprintf(l,n,"x-api-key: %s",api_key); h=curl_slist_append(h,l); free(l); } - { size_t n=strlen(LLM_VERSION)+32; char* l=malloc(n); snprintf(l,n,"anthropic-version: %s",LLM_VERSION); h=curl_slist_append(h,l); free(l); } - el_val_t resp = http_do("POST", LLM_API_URL, json_body, h); - curl_slist_free_all(h); - return resp; -} - -/* Extract concatenated assistant text from an Anthropic /v1/messages - * response. The response shape is: - * {"content":[{"type":"text","text":"..."}, ...], ...} - * If parsing fails, returns the raw response so the caller can inspect. - */ -static el_val_t llm_extract_text(el_val_t resp_val) { - const char* resp = EL_CSTR(resp_val); - if (!resp || !*resp) return el_wrap_str(el_strdup("")); - /* If error JSON, propagate as-is. */ - if (resp[0] == '{' && strstr(resp, "\"error\"")) { - return el_wrap_str(el_strdup(resp)); - } - /* Find "content":[ ... ] */ - const char* p = json_find_key(resp, "content"); - if (!p) return el_wrap_str(el_strdup(resp)); - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return el_wrap_str(el_strdup(resp)); - p++; - JsonBuf out; jb_init(&out); - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - const char* type_p = json_find_key(obj, "type"); - if (type_p && *type_p == '"') { - JsonParser jp = { .p = type_p, .end = type_p + strlen(type_p), .err = 0 }; - char* type_s = jp_parse_string_raw(&jp); - if (!jp.err && type_s && strcmp(type_s, "text") == 0) { - const char* tp = json_find_key(obj, "text"); - if (tp && *tp == '"') { - JsonParser jp2 = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - char* text_s = jp_parse_string_raw(&jp2); - if (!jp2.err && text_s) jb_puts(&out, text_s); - free(text_s); - } - } - free(type_s); - } - free(obj); - p = end; - } - return el_wrap_str(out.buf); -} - -el_val_t llm_call(el_val_t model, el_val_t prompt) { - const char* u = EL_CSTR(prompt); if (!u) u = ""; - return llm_chain_call(NULL, u); -} - -el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt) { - const char* s = EL_CSTR(system_prompt); if (!s) s = ""; - const char* u = EL_CSTR(user_prompt); if (!u) u = ""; - return llm_chain_call(s, u); -} - -/* ── Tool registry for llm_call_agentic ─────────────────────────────────── */ - -typedef el_val_t (*llm_tool_fn)(el_val_t input); - -typedef struct LlmToolEntry { - char* name; - llm_tool_fn fn; -} LlmToolEntry; - -static LlmToolEntry _llm_tools[64]; -static size_t _llm_tool_count = 0; -static pthread_mutex_t _llm_tool_mu = PTHREAD_MUTEX_INITIALIZER; - -static llm_tool_fn llm_tool_lookup(const char* name) { - if (!name) return NULL; - llm_tool_fn fn = NULL; - pthread_mutex_lock(&_llm_tool_mu); - for (size_t i = 0; i < _llm_tool_count; i++) { - if (strcmp(_llm_tools[i].name, name) == 0) { fn = _llm_tools[i].fn; break; } - } - pthread_mutex_unlock(&_llm_tool_mu); - return fn; -} - -void llm_register_tool(el_val_t name, el_val_t handler_fn_name) { - const char* nm = EL_CSTR(name); - const char* sym = EL_CSTR(handler_fn_name); - if (!nm || !*nm || !sym || !*sym) return; - void* p = dlsym(RTLD_DEFAULT, sym); - if (!p) { - fprintf(stderr, "[llm_register_tool] symbol not found: %s\n", sym); - return; - } - pthread_mutex_lock(&_llm_tool_mu); - /* Replace existing entry by name. */ - for (size_t i = 0; i < _llm_tool_count; i++) { - if (strcmp(_llm_tools[i].name, nm) == 0) { - _llm_tools[i].fn = (llm_tool_fn)p; - pthread_mutex_unlock(&_llm_tool_mu); - return; - } - } - if (_llm_tool_count < sizeof(_llm_tools) / sizeof(_llm_tools[0])) { - _llm_tools[_llm_tool_count].name = el_strdup(nm); - _llm_tools[_llm_tool_count].fn = (llm_tool_fn)p; - _llm_tool_count++; - } - pthread_mutex_unlock(&_llm_tool_mu); -} - -/* Serialize the El `tools` list into the JSON `tools:[...]` field expected - * by the Anthropic API. Each tool is an ElMap with name/description/ - * input_schema. input_schema is treated as either a JSON-object string - * (passed through verbatim) or a missing field (substitute {}). */ -static void llm_emit_tools_json(JsonBuf* b, el_val_t tools_list) { - jb_putc(b, '['); - ElList* lst = (ElList*)(uintptr_t)tools_list; - int64_t n = lst ? lst->length : 0; - for (int64_t i = 0; i < n; i++) { - if (i > 0) jb_putc(b, ','); - ElMap* tm = as_map(lst->elems[i]); - const char* name = ""; - const char* desc = ""; - const char* schema = "{}"; - if (tm) { - for (int64_t k = 0; k < tm->count; k++) { - const char* key = EL_CSTR(tm->keys[k]); - const char* val = EL_CSTR(tm->values[k]); - if (!key || !val) continue; - if (strcmp(key, "name") == 0) name = val; - else if (strcmp(key, "description") == 0) desc = val; - else if (strcmp(key, "input_schema") == 0) schema = val; - } - } - char* esc_name = json_escape_alloc(name); - char* esc_desc = json_escape_alloc(desc); - jb_puts(b, "{\"name\":\""); jb_puts(b, esc_name); - jb_puts(b, "\",\"description\":\""); jb_puts(b, esc_desc); - jb_puts(b, "\",\"input_schema\":"); jb_puts(b, schema && *schema ? schema : "{}"); - jb_putc(b, '}'); - free(esc_name); free(esc_desc); - } - jb_putc(b, ']'); -} - -/* Walk the assistant `content` array and emit each block back into b, - * preserving the verbatim JSON of every block — used to re-include the - * assistant turn in the next request. */ -static void llm_emit_content_blocks(JsonBuf* b, const char* resp) { - const char* p = json_find_key(resp, "content"); - jb_putc(b, '['); - if (!p) { jb_putc(b, ']'); return; } - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') { jb_putc(b, ']'); return; } - p++; - int first = 1; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - if (!first) jb_putc(b, ','); - first = 0; - size_t n = (size_t)(end - p); - jb_reserve(b, n); - memcpy(b->buf + b->len, p, n); - b->len += n; - b->buf[b->len] = '\0'; - p = end; - } - jb_putc(b, ']'); -} - -/* Concatenate all "text" blocks from a response. Returns owned string. */ -static char* llm_concat_text_blocks(const char* resp) { - JsonBuf out; jb_init(&out); - if (!resp) return out.buf; - const char* p = json_find_key(resp, "content"); - if (!p) return out.buf; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return out.buf; - p++; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - const char* tp = json_find_key(obj, "type"); - if (tp && *tp == '"') { - JsonParser jp = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - char* tname = jp_parse_string_raw(&jp); - if (!jp.err && tname && strcmp(tname, "text") == 0) { - const char* xp = json_find_key(obj, "text"); - if (xp && *xp == '"') { - JsonParser jp2 = { .p = xp, .end = xp + strlen(xp), .err = 0 }; - char* txt = jp_parse_string_raw(&jp2); - if (!jp2.err && txt) jb_puts(&out, txt); - free(txt); - } - } - free(tname); - } - free(obj); - p = end; - } - return out.buf; -} - -/* Build tool_result message blocks for every tool_use in a response. - * Appends to `b` an array element for each tool_use; caller wraps. */ -static int llm_build_tool_results(JsonBuf* b, const char* resp) { - int any = 0; - const char* p = json_find_key(resp, "content"); - if (!p) return 0; - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') p++; - if (*p != '[') return 0; - p++; - while (*p && *p != ']') { - while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' || *p == ',') p++; - if (*p != '{') break; - const char* end = json_skip_value(p); - size_t n = (size_t)(end - p); - char* obj = malloc(n + 1); - memcpy(obj, p, n); obj[n] = '\0'; - - const char* tp = json_find_key(obj, "type"); - char* type_s = NULL; - if (tp && *tp == '"') { - JsonParser jp = { .p = tp, .end = tp + strlen(tp), .err = 0 }; - type_s = jp_parse_string_raw(&jp); - } - if (type_s && strcmp(type_s, "tool_use") == 0) { - /* Extract id, name, input. */ - char* id_s = NULL; char* name_s = NULL; - const char* idp = json_find_key(obj, "id"); - if (idp && *idp == '"') { - JsonParser jp = { .p = idp, .end = idp + strlen(idp), .err = 0 }; - id_s = jp_parse_string_raw(&jp); - } - const char* np = json_find_key(obj, "name"); - if (np && *np == '"') { - JsonParser jp = { .p = np, .end = np + strlen(np), .err = 0 }; - name_s = jp_parse_string_raw(&jp); - } - el_val_t input_raw = json_get_raw(EL_STR(obj), EL_STR("input")); - const char* input_s = EL_CSTR(input_raw); - if (!input_s || !*input_s) input_s = "{}"; - - llm_tool_fn fn = llm_tool_lookup(name_s ? name_s : ""); - char* result = NULL; - int is_error = 0; - if (!fn) { - size_t en = strlen(name_s ? name_s : "(null)") + 64; - result = malloc(en); - snprintf(result, en, "{\"error\":\"tool not registered: %s\"}", - name_s ? name_s : "(null)"); - is_error = 1; - } else { - el_val_t out = fn(EL_STR(input_s)); - const char* os = EL_CSTR(out); - result = el_strdup(os ? os : ""); - } - - if (any) jb_putc(b, ','); - char* esc_id = json_escape_alloc(id_s ? id_s : ""); - char* esc_res = json_escape_alloc(result ? result : ""); - jb_puts(b, "{\"type\":\"tool_result\",\"tool_use_id\":\""); - jb_puts(b, esc_id); - jb_puts(b, "\",\"content\":\""); - jb_puts(b, esc_res); - jb_puts(b, "\""); - if (is_error) jb_puts(b, ",\"is_error\":true"); - jb_putc(b, '}'); - free(esc_id); free(esc_res); free(result); - free(id_s); free(name_s); - any = 1; - } - free(type_s); - free(obj); - p = end; - } - return any; -} - -el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools) { - /* Empty tools list → degrade to plain system call. */ - ElList* tl = (ElList*)(uintptr_t)tools; - if (!tl || tl->length == 0) { - return llm_call_system(model, system, user); - } - const char* m = llm_resolve_model(EL_CSTR(model)); - const char* sys_p = EL_CSTR(system); if (!sys_p) sys_p = ""; - const char* usr_p = EL_CSTR(user); if (!usr_p) usr_p = ""; - - /* Build the static parts: tools JSON and system prompt — these don't - * change across iterations. */ - JsonBuf tools_buf; jb_init(&tools_buf); - llm_emit_tools_json(&tools_buf, tools); - char* esc_sys = json_escape_alloc(sys_p); - - /* messages array, accumulated as a mutable JSON fragment (no surrounding - * brackets — emitted at request time). */ - JsonBuf msgs; jb_init(&msgs); - /* First user message. */ - char* esc_user = json_escape_alloc(usr_p); - jb_puts(&msgs, "{\"role\":\"user\",\"content\":\""); - jb_puts(&msgs, esc_user); - jb_puts(&msgs, "\"}"); - free(esc_user); - - char* last_text = el_strdup(""); - el_val_t final_out = 0; - int reached_cap = 1; - - for (int iter = 0; iter < 10; iter++) { - /* Build request body. */ - JsonBuf body; jb_init(&body); - jb_putc(&body, '{'); - jb_puts(&body, "\"model\":"); jb_emit_escaped(&body, m); - jb_puts(&body, ",\"max_tokens\":4096"); - if (*sys_p) { - jb_puts(&body, ",\"system\":\""); - jb_puts(&body, esc_sys); - jb_puts(&body, "\""); - } - jb_puts(&body, ",\"tools\":"); - jb_puts(&body, tools_buf.buf); - jb_puts(&body, ",\"messages\":["); - jb_puts(&body, msgs.buf); - jb_puts(&body, "]}"); - - el_val_t resp_v = llm_request(body.buf); - free(body.buf); - const char* resp = EL_CSTR(resp_v); - if (!resp || !*resp) { - final_out = http_error_json("empty response"); - reached_cap = 0; - break; - } - if (resp[0] == '{' && strstr(resp, "\"error\"") && - !json_find_key(resp, "content")) { - final_out = el_wrap_str(el_strdup(resp)); - reached_cap = 0; - break; - } - - /* Update last_text from this response. */ - free(last_text); - last_text = llm_concat_text_blocks(resp); - - /* Inspect stop_reason. */ - el_val_t sr_v = json_get_string(EL_STR(resp), EL_STR("stop_reason")); - const char* sr = EL_CSTR(sr_v); if (!sr) sr = ""; - - if (strcmp(sr, "end_turn") == 0) { - final_out = el_wrap_str(el_strdup(last_text)); - reached_cap = 0; - break; - } - if (strcmp(sr, "max_tokens") == 0) { - size_t ln = strlen(last_text) + 16; - char* out = malloc(ln); - snprintf(out, ln, "%s\n[truncated]", last_text); - final_out = el_wrap_str(out); - reached_cap = 0; - break; - } - if (strcmp(sr, "tool_use") != 0) { - /* Unexpected stop reason; return the text we have. */ - final_out = el_wrap_str(el_strdup(last_text)); - reached_cap = 0; - break; - } - - /* Append the assistant turn (raw content blocks) to messages. */ - JsonBuf ab; jb_init(&ab); - jb_puts(&ab, ",{\"role\":\"assistant\",\"content\":"); - llm_emit_content_blocks(&ab, resp); - jb_putc(&ab, '}'); - jb_puts(&msgs, ab.buf); - free(ab.buf); - - /* Build tool_result message. */ - JsonBuf tr; jb_init(&tr); - jb_puts(&tr, ",{\"role\":\"user\",\"content\":["); - int any = llm_build_tool_results(&tr, resp); - jb_puts(&tr, "]}"); - if (any) { - jb_puts(&msgs, tr.buf); - } - free(tr.buf); - } - - if (reached_cap) { - size_t ln = strlen(last_text) + 32; - char* out = malloc(ln); - snprintf(out, ln, "[loop_cap_reached]\n%s", last_text); - final_out = el_wrap_str(out); - } - free(last_text); - free(esc_sys); - free(tools_buf.buf); - free(msgs.buf); - return final_out; -} - -/* base64-encode arbitrary bytes (returns owned C string). - * Internal helper for llm_vision; the public crypto entry point that El - * programs call is `base64_encode(el_val_t)` defined in the crypto block - * at the end of this file. */ -static char* el_b64_encode_internal(const unsigned char* src, size_t n) { - static const char tbl[] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; - size_t out_len = 4 * ((n + 2) / 3); - char* out = malloc(out_len + 1); - if (!out) return NULL; - size_t o = 0; - for (size_t i = 0; i < n;) { - uint32_t v = 0; int got = 0; - v |= (uint32_t)src[i++] << 16; got++; - if (i < n) { v |= (uint32_t)src[i++] << 8; got++; } - if (i < n) { v |= (uint32_t)src[i++]; got++; } - out[o++] = tbl[(v >> 18) & 0x3f]; - out[o++] = tbl[(v >> 12) & 0x3f]; - out[o++] = (got > 1) ? tbl[(v >> 6) & 0x3f] : '='; - out[o++] = (got > 2) ? tbl[v & 0x3f] : '='; - } - out[o] = '\0'; - return out; -} - -el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64) { - const char* m = llm_resolve_model(EL_CSTR(model)); - const char* s = EL_CSTR(system); if (!s) s = ""; - const char* u = EL_CSTR(prompt); if (!u) u = ""; - const char* img = EL_CSTR(image_url_or_b64); if (!img) img = ""; - - /* Choose source mode */ - char* image_block = NULL; - if (strncasecmp(img, "http://", 7) == 0 || strncasecmp(img, "https://", 8) == 0) { - char* esc_url = json_escape_alloc(img); - size_t n = strlen(esc_url) + 128; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"%s\"}}", - esc_url); - free(esc_url); - } else if (strncmp(img, "data:", 5) == 0) { - /* Inline data URL: split media-type and base64 */ - const char* semi = strchr(img + 5, ';'); - const char* comma = strchr(img + 5, ','); - char media[64] = "image/png"; - if (semi && comma && semi < comma) { - size_t ml = (size_t)(semi - (img + 5)); - if (ml >= sizeof(media)) ml = sizeof(media) - 1; - memcpy(media, img + 5, ml); media[ml] = '\0'; - } - const char* b64 = comma ? comma + 1 : ""; - char* esc_media = json_escape_alloc(media); - char* esc_b64 = json_escape_alloc(b64); - size_t n = strlen(esc_media) + strlen(esc_b64) + 192; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"base64\"," - "\"media_type\":\"%s\",\"data\":\"%s\"}}", - esc_media, esc_b64); - free(esc_media); free(esc_b64); - } else if (*img) { - /* Treat as file path: read, base64-encode, attach. */ - FILE* f = fopen(img, "rb"); - if (!f) { - char err[256]; snprintf(err, sizeof(err), "cannot open image: %s", img); - return http_error_json(err); - } - fseek(f, 0, SEEK_END); - long sz = ftell(f); - rewind(f); - if (sz <= 0) { fclose(f); return http_error_json("empty image file"); } - unsigned char* buf = malloc((size_t)sz); - if (!buf) { fclose(f); return http_error_json("oom"); } - size_t got = fread(buf, 1, (size_t)sz, f); - fclose(f); - char* b64 = el_b64_encode_internal(buf, got); - free(buf); - if (!b64) return http_error_json("base64 encode failed"); - const char* media = "image/png"; - size_t ilen = strlen(img); - if (ilen >= 4) { - if (strcasecmp(img + ilen - 4, ".jpg") == 0 || - (ilen >= 5 && strcasecmp(img + ilen - 5, ".jpeg") == 0)) media = "image/jpeg"; - else if (strcasecmp(img + ilen - 4, ".gif") == 0) media = "image/gif"; - else if (strcasecmp(img + ilen - 4, ".webp") == 0) media = "image/webp"; - } - char* esc_b64 = json_escape_alloc(b64); free(b64); - size_t n = strlen(esc_b64) + 192; - image_block = malloc(n); - snprintf(image_block, n, - "{\"type\":\"image\",\"source\":{\"type\":\"base64\"," - "\"media_type\":\"%s\",\"data\":\"%s\"}}", - media, esc_b64); - free(esc_b64); - } - - char* esc_sys = json_escape_alloc(s); - char* esc_user = json_escape_alloc(u); - JsonBuf b; jb_init(&b); - jb_putc(&b, '{'); - jb_puts(&b, "\"model\":"); jb_emit_escaped(&b, m); - jb_puts(&b, ",\"max_tokens\":4096"); - if (*s) { - jb_puts(&b, ",\"system\":\""); - jb_puts(&b, esc_sys); - jb_puts(&b, "\""); - } - jb_puts(&b, ",\"messages\":[{\"role\":\"user\",\"content\":["); - if (image_block) { - jb_puts(&b, image_block); - jb_putc(&b, ','); - } - jb_puts(&b, "{\"type\":\"text\",\"text\":\""); - jb_puts(&b, esc_user); - jb_puts(&b, "\"}]}]}"); - free(esc_sys); free(esc_user); free(image_block); - el_val_t resp = llm_request(b.buf); - free(b.buf); - return llm_extract_text(resp); -} - -el_val_t llm_models(void) { - el_val_t lst = el_list_empty(); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-sonnet-4-5"))); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-opus-4-7"))); - lst = el_list_append(lst, el_wrap_str(el_strdup("claude-haiku-4-5"))); - return lst; -} - -/* ── Native VM builtin aliases ────────────────────────────────────────────── - * El source files use native_* names (El VM builtins). - * When compiled to C, these map directly to el_* runtime functions. */ - -el_val_t native_list_get(el_val_t list, el_val_t index) { - return el_list_get(list, index); -} - -el_val_t native_list_len(el_val_t list) { - return el_list_len(list); -} - -el_val_t native_list_append(el_val_t list, el_val_t elem) { - return el_list_append(list, elem); -} - -el_val_t native_list_empty(void) { - return el_list_empty(); -} - -el_val_t native_list_clone(el_val_t list) { - return el_list_clone(list); -} - -el_val_t native_string_chars(el_val_t sv) { - const char* s = EL_CSTR(sv); - el_val_t result = el_list_empty(); - if (!s) return result; - while (*s) { - char buf[2]; - buf[0] = *s; - buf[1] = '\0'; - result = el_list_append(result, EL_STR(strdup(buf))); - s++; - } - return result; -} - -el_val_t native_int_to_str(el_val_t n) { - return int_to_str(n); -} - -/* ── Method-call shorthand aliases ────────────────────────────────────────── - * Short names that result from the method-call convention: - * myList.append(x) → append(myList, x) - * myList.len() → len(myList) - * myList.get(i) → get(myList, i) - * myMap.map_get(k) → map_get(myMap, k) - * myMap.map_set(k,v) → map_set(myMap, k, v) */ - -el_val_t append(el_val_t list, el_val_t elem) { return el_list_append(list, elem); } -el_val_t len(el_val_t list) { return el_list_len(list); } -el_val_t get(el_val_t list, el_val_t index) { return el_list_get(list, index); } -el_val_t map_get(el_val_t map, el_val_t key) { return el_map_get(map, key); } -el_val_t map_set(el_val_t map, el_val_t key, el_val_t value) { return el_map_set(map, key, value); } - -/* ── Crypto primitives ────────────────────────────────────────────────────── - * - * SHA-256 implementation adapted from Brad Conte's public-domain reference - * (https://github.com/B-Con/crypto-algorithms/blob/master/sha256.c, public - * domain per the project's LICENSE). HMAC follows RFC 2104. Base64 encoding - * follows RFC 4648; the URL-safe variant uses the alphabet from §5 of the - * RFC and omits padding (per JWT/JWS convention). - * - * Self-contained: no OpenSSL/libcrypto dependency. The runtime keeps its - * existing `-lcurl -lpthread -ldl -lm` link line. - * - * Binary outputs (sha256_bytes, hmac_sha256_bytes) tag their buffer with a - * magic header so base64_encode/base64url_encode can recover the exact byte - * length even when the payload contains embedded NULs. Plain C strings - * (without the header) fall back to strlen(), preserving the existing API - * shape for normal text inputs. */ - -/* Magic-header for length-tagged binary buffers. Layout: - * [ uint32_t magic = EL_MAGIC_BIN ][ uint32_t length ][ data... ][ \0 ] - * The returned el_val_t points at `data`, so consumers that strlen() it still - * get a sensible (though possibly truncated) view. el_bin_len() recovers the - * true length by sniffing the 8 bytes preceding the pointer. - * - * Magic value chosen with high MSB so it cannot collide with printable ASCII - * (the same discriminator pattern used by EL_MAGIC_LIST / EL_MAGIC_MAP). */ -#define EL_MAGIC_BIN 0xE1B17EAFu - -typedef struct { - uint32_t magic; - uint32_t length; -} el_bin_hdr_t; - -/* Allocate a length-tagged binary buffer; returns pointer to the data area. */ -static unsigned char* el_bin_alloc(size_t len) { - el_bin_hdr_t* hdr = (el_bin_hdr_t*)malloc(sizeof(el_bin_hdr_t) + len + 1); - if (!hdr) { fputs("el_runtime: out of memory (bin)\n", stderr); exit(1); } - hdr->magic = EL_MAGIC_BIN; - hdr->length = (uint32_t)len; - unsigned char* data = (unsigned char*)(hdr + 1); - data[len] = '\0'; /* keep NUL-terminated for accidental strlen calls */ - return data; -} - -/* Recover length from a possibly-tagged buffer. Returns 1 if tagged. */ -static int el_bin_lookup(const void* p, size_t* out_len) { - if (!p) { *out_len = 0; return 0; } - /* Avoid reading off the front of a page on tiny pointers (e.g. NULs - * passed in as int-cast values). 4096 is a safe lower bound on any - * platform we target. */ - if ((uintptr_t)p < 4096) return 0; - const el_bin_hdr_t* hdr = (const el_bin_hdr_t*)((const char*)p - sizeof(el_bin_hdr_t)); - if (hdr->magic != EL_MAGIC_BIN) return 0; - *out_len = hdr->length; - return 1; -} - -/* Effective input length: tagged length if present, else strlen. */ -static size_t el_input_len(const char* s) { - size_t n; - if (el_bin_lookup(s, &n)) return n; - return s ? strlen(s) : 0; -} - -/* ─── SHA-256 (Brad Conte / public domain) ──────────────────────────────── */ - -typedef struct { - unsigned char data[64]; - uint32_t datalen; - uint64_t bitlen; - uint32_t state[8]; -} el_sha256_ctx_t; - -static const uint32_t el_sha256_k[64] = { - 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5,0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5, - 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3,0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174, - 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc,0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da, - 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7,0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967, - 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13,0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85, - 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3,0xd192e819,0xd6990624,0xf40e3585,0x106aa070, - 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5,0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3, - 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208,0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2 -}; - -#define EL_ROTR(x, n) (((x) >> (n)) | ((x) << (32 - (n)))) -#define EL_CH(x,y,z) (((x) & (y)) ^ (~(x) & (z))) -#define EL_MAJ(x,y,z) (((x) & (y)) ^ ((x) & (z)) ^ ((y) & (z))) -#define EL_EP0(x) (EL_ROTR(x,2) ^ EL_ROTR(x,13) ^ EL_ROTR(x,22)) -#define EL_EP1(x) (EL_ROTR(x,6) ^ EL_ROTR(x,11) ^ EL_ROTR(x,25)) -#define EL_SIG0(x) (EL_ROTR(x,7) ^ EL_ROTR(x,18) ^ ((x) >> 3)) -#define EL_SIG1(x) (EL_ROTR(x,17) ^ EL_ROTR(x,19) ^ ((x) >> 10)) - -static void el_sha256_transform(el_sha256_ctx_t* ctx, const unsigned char* data) { - uint32_t a, b, c, d, e, f, g, h, t1, t2, m[64]; - int i, j; - for (i = 0, j = 0; i < 16; ++i, j += 4) { - m[i] = ((uint32_t)data[j] << 24) | ((uint32_t)data[j + 1] << 16) - | ((uint32_t)data[j + 2] << 8) | (uint32_t)data[j + 3]; - } - for (; i < 64; ++i) { - m[i] = EL_SIG1(m[i-2]) + m[i-7] + EL_SIG0(m[i-15]) + m[i-16]; - } - a = ctx->state[0]; b = ctx->state[1]; c = ctx->state[2]; d = ctx->state[3]; - e = ctx->state[4]; f = ctx->state[5]; g = ctx->state[6]; h = ctx->state[7]; - for (i = 0; i < 64; ++i) { - t1 = h + EL_EP1(e) + EL_CH(e,f,g) + el_sha256_k[i] + m[i]; - t2 = EL_EP0(a) + EL_MAJ(a,b,c); - h = g; g = f; f = e; e = d + t1; d = c; c = b; b = a; a = t1 + t2; - } - ctx->state[0] += a; ctx->state[1] += b; ctx->state[2] += c; ctx->state[3] += d; - ctx->state[4] += e; ctx->state[5] += f; ctx->state[6] += g; ctx->state[7] += h; -} - -static void el_sha256_init(el_sha256_ctx_t* ctx) { - ctx->datalen = 0; - ctx->bitlen = 0; - ctx->state[0] = 0x6a09e667; ctx->state[1] = 0xbb67ae85; - ctx->state[2] = 0x3c6ef372; ctx->state[3] = 0xa54ff53a; - ctx->state[4] = 0x510e527f; ctx->state[5] = 0x9b05688c; - ctx->state[6] = 0x1f83d9ab; ctx->state[7] = 0x5be0cd19; -} - -static void el_sha256_update(el_sha256_ctx_t* ctx, const unsigned char* data, size_t len) { - for (size_t i = 0; i < len; ++i) { - ctx->data[ctx->datalen++] = data[i]; - if (ctx->datalen == 64) { - el_sha256_transform(ctx, ctx->data); - ctx->bitlen += 512; - ctx->datalen = 0; - } - } -} - -static void el_sha256_final(el_sha256_ctx_t* ctx, unsigned char hash[32]) { - uint32_t i = ctx->datalen; - if (ctx->datalen < 56) { - ctx->data[i++] = 0x80; - while (i < 56) ctx->data[i++] = 0x00; - } else { - ctx->data[i++] = 0x80; - while (i < 64) ctx->data[i++] = 0x00; - el_sha256_transform(ctx, ctx->data); - memset(ctx->data, 0, 56); - } - ctx->bitlen += (uint64_t)ctx->datalen * 8; - ctx->data[63] = (unsigned char)( ctx->bitlen & 0xff); - ctx->data[62] = (unsigned char)((ctx->bitlen >> 8) & 0xff); - ctx->data[61] = (unsigned char)((ctx->bitlen >> 16) & 0xff); - ctx->data[60] = (unsigned char)((ctx->bitlen >> 24) & 0xff); - ctx->data[59] = (unsigned char)((ctx->bitlen >> 32) & 0xff); - ctx->data[58] = (unsigned char)((ctx->bitlen >> 40) & 0xff); - ctx->data[57] = (unsigned char)((ctx->bitlen >> 48) & 0xff); - ctx->data[56] = (unsigned char)((ctx->bitlen >> 56) & 0xff); - el_sha256_transform(ctx, ctx->data); - for (i = 0; i < 4; ++i) { - hash[i] = (ctx->state[0] >> (24 - i * 8)) & 0xff; - hash[i + 4] = (ctx->state[1] >> (24 - i * 8)) & 0xff; - hash[i + 8] = (ctx->state[2] >> (24 - i * 8)) & 0xff; - hash[i + 12] = (ctx->state[3] >> (24 - i * 8)) & 0xff; - hash[i + 16] = (ctx->state[4] >> (24 - i * 8)) & 0xff; - hash[i + 20] = (ctx->state[5] >> (24 - i * 8)) & 0xff; - hash[i + 24] = (ctx->state[6] >> (24 - i * 8)) & 0xff; - hash[i + 28] = (ctx->state[7] >> (24 - i * 8)) & 0xff; - } -} - -static void el_sha256_oneshot(const unsigned char* data, size_t len, unsigned char out[32]) { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, data, len); - el_sha256_final(&c, out); -} - -/* ─── HMAC-SHA-256 (RFC 2104) ───────────────────────────────────────────── */ - -static void el_hmac_sha256(const unsigned char* key, size_t key_len, - const unsigned char* msg, size_t msg_len, - unsigned char out[32]) { - unsigned char k[64]; - unsigned char k_ipad[64]; - unsigned char k_opad[64]; - unsigned char inner[32]; - - if (key_len > 64) { - el_sha256_oneshot(key, key_len, k); - memset(k + 32, 0, 32); - } else { - memcpy(k, key, key_len); - memset(k + key_len, 0, 64 - key_len); - } - for (int i = 0; i < 64; ++i) { - k_ipad[i] = k[i] ^ 0x36; - k_opad[i] = k[i] ^ 0x5c; - } - { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, k_ipad, 64); - el_sha256_update(&c, msg, msg_len); - el_sha256_final(&c, inner); - } - { - el_sha256_ctx_t c; - el_sha256_init(&c); - el_sha256_update(&c, k_opad, 64); - el_sha256_update(&c, inner, 32); - el_sha256_final(&c, out); - } -} - -/* ─── Hex helper ────────────────────────────────────────────────────────── */ - -static el_val_t el_hex_encode(const unsigned char* data, size_t len) { - static const char digits[] = "0123456789abcdef"; - char* out = el_strbuf(len * 2); - for (size_t i = 0; i < len; ++i) { - out[i * 2] = digits[(data[i] >> 4) & 0xf]; - out[i * 2 + 1] = digits[ data[i] & 0xf]; - } - out[len * 2] = '\0'; - return el_wrap_str(out); -} - -/* ─── Base64 (RFC 4648) ─────────────────────────────────────────────────── */ - -static const char el_b64_std_alphabet[64] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; -static const char el_b64_url_alphabet[64] = - "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"; - -el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe) { - const char* alphabet = url_safe ? el_b64_url_alphabet : el_b64_std_alphabet; - /* Standard form is padded to multiple of 4; URL-safe omits padding. */ - size_t out_cap = ((len + 2) / 3) * 4 + 1; - char* out = el_strbuf(out_cap); - size_t i = 0, j = 0; - while (i + 3 <= len) { - uint32_t v = ((uint32_t)data[i] << 16) | ((uint32_t)data[i+1] << 8) | (uint32_t)data[i+2]; - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - out[j++] = alphabet[(v >> 6) & 0x3f]; - out[j++] = alphabet[ v & 0x3f]; - i += 3; - } - size_t rem = len - i; - if (rem == 1) { - uint32_t v = (uint32_t)data[i] << 16; - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - if (!url_safe) { out[j++] = '='; out[j++] = '='; } - } else if (rem == 2) { - uint32_t v = ((uint32_t)data[i] << 16) | ((uint32_t)data[i+1] << 8); - out[j++] = alphabet[(v >> 18) & 0x3f]; - out[j++] = alphabet[(v >> 12) & 0x3f]; - out[j++] = alphabet[(v >> 6) & 0x3f]; - if (!url_safe) { out[j++] = '='; } - } - out[j] = '\0'; - return el_wrap_str(out); -} - -/* Decode either alphabet — accepts both '+/' and '-_' transparently, and - * tolerates missing padding (which JWTs typically omit). Whitespace is - * skipped for robustness. Invalid characters cause the decode to stop and - * the partial result so far is returned. */ -static el_val_t el_base64_decode_any(const char* in) { - if (!in) { - unsigned char* empty = el_bin_alloc(0); - return EL_STR((char*)empty); - } - size_t in_len = strlen(in); - /* Worst case: 3 output bytes per 4 input chars, +1 NUL slack. */ - unsigned char* out = el_bin_alloc(((in_len + 3) / 4) * 3 + 1); - - int8_t lut[256]; - for (int i = 0; i < 256; ++i) lut[i] = -1; - for (int i = 0; i < 64; ++i) lut[(unsigned char)el_b64_std_alphabet[i]] = (int8_t)i; - /* Allow URL-safe characters too (so one decoder handles both forms). */ - lut[(unsigned char)'-'] = 62; - lut[(unsigned char)'_'] = 63; - - uint32_t buf = 0; - int bits = 0; - size_t o = 0; - for (size_t i = 0; i < in_len; ++i) { - unsigned char c = (unsigned char)in[i]; - if (c == '=' || c == '\r' || c == '\n' || c == ' ' || c == '\t') continue; - int8_t v = lut[c]; - if (v < 0) break; /* invalid char — stop */ - buf = (buf << 6) | (uint32_t)v; - bits += 6; - if (bits >= 8) { - bits -= 8; - out[o++] = (unsigned char)((buf >> bits) & 0xff); - } - } - /* Patch the length header to the actual decoded length. */ - el_bin_hdr_t* hdr = (el_bin_hdr_t*)((char*)out - sizeof(el_bin_hdr_t)); - hdr->length = (uint32_t)o; - out[o] = '\0'; - return EL_STR((char*)out); -} - -/* ─── Public crypto entry points ────────────────────────────────────────── */ - -el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len) { - unsigned char* out = el_bin_alloc(32); - el_sha256_oneshot(data, len, out); - return EL_STR((char*)out); -} - -el_val_t sha256_hex(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - unsigned char digest[32]; - el_sha256_oneshot((const unsigned char*)(s ? s : ""), n, digest); - return el_hex_encode(digest, 32); -} - -el_val_t sha256_bytes(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_sha256_bytes_n((const unsigned char*)(s ? s : ""), n); -} - -el_val_t hmac_sha256_hex(el_val_t key, el_val_t message) { - const char* k = EL_CSTR(key); - const char* m = EL_CSTR(message); - size_t kn = el_input_len(k); - size_t mn = el_input_len(m); - unsigned char mac[32]; - el_hmac_sha256((const unsigned char*)(k ? k : ""), kn, - (const unsigned char*)(m ? m : ""), mn, - mac); - return el_hex_encode(mac, 32); -} - -el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message) { - const char* k = EL_CSTR(key); - const char* m = EL_CSTR(message); - size_t kn = el_input_len(k); - size_t mn = el_input_len(m); - unsigned char* out = el_bin_alloc(32); - el_hmac_sha256((const unsigned char*)(k ? k : ""), kn, - (const unsigned char*)(m ? m : ""), mn, - out); - return EL_STR((char*)out); -} - -el_val_t base64_encode(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_base64_encode_n((const unsigned char*)(s ? s : ""), n, /*url_safe=*/0); -} - -el_val_t base64url_encode(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - return el_base64_encode_n((const unsigned char*)(s ? s : ""), n, /*url_safe=*/1); -} - -el_val_t base64_decode(el_val_t input) { - return el_base64_decode_any(EL_CSTR(input)); -} - -el_val_t base64url_decode(el_val_t input) { - return el_base64_decode_any(EL_CSTR(input)); -} - -/* ── Post-quantum cryptography (liboqs + OpenSSL) ─────────────────────────── - * - * Algorithm choices (per CNSA 2.0 / NIST PQ guidance, as of 2024): - * Signatures: CRYSTALS-Dilithium-3 (NIST security level 3, balanced) - * KEM: CRYSTALS-Kyber-768 (NIST security level 3) - * Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2) - * Hybrid: X25519 || Kyber-768, combined via HKDF-SHA256 - * - * Why hybrid: Kyber is new. X25519 has 20+ years of analysis. Hybridizing - * preserves classical security if Kyber falls to a future cryptanalytic - * advance, and preserves PQ security if X25519 falls to a quantum adversary. - * "Recordable now, decryptable later" already threatens long-lived classical - * key exchange — the only safe move for keys protecting durable doctrine - * (CGI lineage, KindredGrants, Principal-CGI covenants) is to encapsulate - * with PQ today, even if the classical leg is what the wire shows. - * - * Compile-time detection: when is unavailable the pq_* functions - * compile to stubs that return a JSON error envelope. SHA3-256 stays - * available regardless (it's implemented inline, no liboqs dep). This lets - * the runtime build cleanly on dev machines without liboqs while production - * gets the full PQ stack. */ - -/* ─── SHA3-256 (Keccak, FIPS 202) ──────────────────────────────────────────── - * Inline reference implementation. ~120 LoC, no external dependency. - * rate=1088 bits, capacity=512 bits, output=256 bits, padding=0x06. */ - -static const uint64_t el_keccak_rc[24] = { - 0x0000000000000001ULL, 0x0000000000008082ULL, 0x800000000000808aULL, - 0x8000000080008000ULL, 0x000000000000808bULL, 0x0000000080000001ULL, - 0x8000000080008081ULL, 0x8000000000008009ULL, 0x000000000000008aULL, - 0x0000000000000088ULL, 0x0000000080008009ULL, 0x000000008000000aULL, - 0x000000008000808bULL, 0x800000000000008bULL, 0x8000000000008089ULL, - 0x8000000000008003ULL, 0x8000000000008002ULL, 0x8000000000000080ULL, - 0x000000000000800aULL, 0x800000008000000aULL, 0x8000000080008081ULL, - 0x8000000000008080ULL, 0x0000000080000001ULL, 0x8000000080008008ULL -}; - -static const unsigned el_keccak_rho[24] = { - 1, 3, 6, 10, 15, 21, 28, 36, 45, 55, 2, 14, - 27, 41, 56, 8, 25, 43, 62, 18, 39, 61, 20, 44 -}; - -static const unsigned el_keccak_pi[24] = { - 10, 7, 11, 17, 18, 3, 5, 16, 8, 21, 24, 4, - 15, 23, 19, 13, 12, 2, 20, 14, 22, 9, 6, 1 -}; - -#define EL_ROTL64(x, n) (((x) << (n)) | ((x) >> (64 - (n)))) - -static void el_keccak_f1600(uint64_t s[25]) { - for (int round = 0; round < 24; ++round) { - uint64_t bc[5], t; - for (int i = 0; i < 5; ++i) - bc[i] = s[i] ^ s[i+5] ^ s[i+10] ^ s[i+15] ^ s[i+20]; - for (int i = 0; i < 5; ++i) { - t = bc[(i+4) % 5] ^ EL_ROTL64(bc[(i+1) % 5], 1); - for (int j = 0; j < 25; j += 5) s[j+i] ^= t; - } - t = s[1]; - for (int i = 0; i < 24; ++i) { - int j = el_keccak_pi[i]; - bc[0] = s[j]; - s[j] = EL_ROTL64(t, el_keccak_rho[i]); - t = bc[0]; - } - for (int j = 0; j < 25; j += 5) { - for (int i = 0; i < 5; ++i) bc[i] = s[j+i]; - for (int i = 0; i < 5; ++i) - s[j+i] = bc[i] ^ ((~bc[(i+1) % 5]) & bc[(i+2) % 5]); - } - s[0] ^= el_keccak_rc[round]; - } -} - -static void el_sha3_256_oneshot(const unsigned char* data, size_t len, - unsigned char out[32]) { - uint64_t st[25] = {0}; - unsigned char* sb = (unsigned char*)st; - const size_t rate = 136; /* 1088 bits / 8 */ - size_t i = 0; - while (len - i >= rate) { - for (size_t k = 0; k < rate; ++k) sb[k] ^= data[i + k]; - el_keccak_f1600(st); - i += rate; - } - size_t rem = len - i; - for (size_t k = 0; k < rem; ++k) sb[k] ^= data[i + k]; - sb[rem] ^= 0x06; /* SHA3 domain-separation byte */ - sb[rate - 1] ^= 0x80; /* final-block padding bit (high bit of last byte) */ - el_keccak_f1600(st); - memcpy(out, sb, 32); -} - -el_val_t sha3_256_hex(el_val_t input) { - const char* s = EL_CSTR(input); - size_t n = el_input_len(s); - unsigned char digest[32]; - el_sha3_256_oneshot((const unsigned char*)(s ? s : ""), n, digest); - return el_hex_encode(digest, 32); -} - -/* ─── Hex decode helper ───────────────────────────────────────────────────── - * Returns a length-tagged binary buffer (so embedded NULs survive); on - * odd-length / invalid input returns NULL with *out_len = 0. Caller is - * responsible for emitting the error envelope. */ - -static int el_hex_nibble(char c) { - if (c >= '0' && c <= '9') return c - '0'; - if (c >= 'a' && c <= 'f') return c - 'a' + 10; - if (c >= 'A' && c <= 'F') return c - 'A' + 10; - return -1; -} - -__attribute__((unused)) -static unsigned char* el_hex_decode(const char* s, size_t* out_len) { - *out_len = 0; - if (!s) return NULL; - size_t n = strlen(s); - if (n & 1) return NULL; - size_t blen = n / 2; - unsigned char* out = el_bin_alloc(blen); - for (size_t i = 0; i < blen; ++i) { - int hi = el_hex_nibble(s[i*2]); - int lo = el_hex_nibble(s[i*2 + 1]); - if (hi < 0 || lo < 0) return NULL; - out[i] = (unsigned char)((hi << 4) | lo); - } - *out_len = blen; - return out; -} - -/* JSON error envelope reused across all PQ entry points. */ -static el_val_t pq_error(const char* msg) { - return http_error_json(msg); -} - -#if __has_include() -#include -#define EL_HAVE_LIBOQS 1 -#else -#define EL_HAVE_LIBOQS 0 -#endif - -#if EL_HAVE_LIBOQS && __has_include() -#include -#define EL_HAVE_OPENSSL 1 -#else -#define EL_HAVE_OPENSSL 0 -#endif - -#if !EL_HAVE_LIBOQS - -/* ─── Stubs (liboqs unavailable) ─────────────────────────────────────────── - * Each entry point returns the same JSON error so callers can inspect a - * single canonical "missing primitive" string. pq_verify is the lone - * exception — verifying without liboqs simply means "not verified", so - * returning Bool false (0) keeps the type contract intact. */ - -#define EL_PQ_NO_LIB "liboqs not linked, post-quantum primitives unavailable" - -el_val_t pq_keygen_signature(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_sign(el_val_t sk, el_val_t msg) { (void)sk; (void)msg; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_verify(el_val_t pk, el_val_t msg, el_val_t sig) { (void)pk; (void)msg; (void)sig; return EL_INT(0); } -el_val_t pq_kem_keygen(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_kem_encaps(el_val_t pk) { (void)pk; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_kem_decaps(el_val_t sk, el_val_t ct) { (void)sk; (void)ct; return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_hybrid_keygen(void) { return pq_error(EL_PQ_NO_LIB); } -el_val_t pq_hybrid_handshake(el_val_t pub) { (void)pub; return pq_error(EL_PQ_NO_LIB); } - -#else /* EL_HAVE_LIBOQS */ - -/* ─── Dilithium-3 / ML-DSA-65 signatures ──────────────────────────────── - * - * NIST FIPS 204 standardized CRYSTALS-Dilithium as ML-DSA. ML-DSA-65 is the - * FIPS form of what we historically called Dilithium-3 — same algorithm - * family, same security level, identical key/sig sizes, but with a couple - * of standardization-driven tweaks (e.g. domain separation in the message - * binding). liboqs 0.12+ exposes both names; 0.15+ retired the legacy - * "Dilithium" constants in favour of "ML-DSA". We prefer ML-DSA-65 if the - * header advertises it, fall back to Dilithium-3 otherwise. Anything - * already signed with the older constant remains verifiable against that - * same constant — callers should pin the algorithm via the OQS_SIG handle's - * method_name field if they need to interoperate with archival signatures. */ - -#if defined(OQS_SIG_alg_ml_dsa_65) -# define EL_DILITHIUM_ALG OQS_SIG_alg_ml_dsa_65 -#elif defined(OQS_SIG_alg_dilithium_3) -# define EL_DILITHIUM_ALG OQS_SIG_alg_dilithium_3 -#else -# define EL_DILITHIUM_ALG "ML-DSA-65" /* string fallback; runtime probe catches misconfig */ -#endif - -el_val_t pq_keygen_signature(void) { - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return pq_error("OQS_SIG_new(dilithium-3) failed"); - unsigned char* pk = (unsigned char*)malloc(sig->length_public_key); - unsigned char* sk = (unsigned char*)malloc(sig->length_secret_key); - if (!pk || !sk) { free(pk); free(sk); OQS_SIG_free(sig); return pq_error("oom"); } - if (OQS_SIG_keypair(sig, pk, sk) != OQS_SUCCESS) { - free(pk); free(sk); OQS_SIG_free(sig); - return pq_error("dilithium-3 keypair generation failed"); - } - el_val_t pk_hex = el_hex_encode(pk, sig->length_public_key); - el_val_t sk_hex = el_hex_encode(sk, sig->length_secret_key); - OQS_MEM_secure_free(sk, sig->length_secret_key); - free(pk); - - const char* pks = EL_CSTR(pk_hex); - const char* sks = EL_CSTR(sk_hex); - char* buf = el_strbuf(strlen(pks) + strlen(sks) + 64); - sprintf(buf, "{\"public_key\":\"%s\",\"secret_key\":\"%s\"}", pks, sks); - OQS_SIG_free(sig); - return el_wrap_str(buf); -} - -el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message) { - size_t sk_len = 0; - unsigned char* sk = el_hex_decode(EL_CSTR(secret_key_hex), &sk_len); - if (!sk) return pq_error("invalid hex in secret_key"); - - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return pq_error("OQS_SIG_new(dilithium-3) failed"); - if (sk_len != sig->length_secret_key) { - OQS_SIG_free(sig); - return pq_error("secret_key length mismatch for dilithium-3"); - } - - const char* msg = EL_CSTR(message); - size_t msg_len = el_input_len(msg); - unsigned char* signature = (unsigned char*)malloc(sig->length_signature); - size_t signature_len = sig->length_signature; - if (!signature) { OQS_SIG_free(sig); return pq_error("oom"); } - - if (OQS_SIG_sign(sig, signature, &signature_len, - (const unsigned char*)(msg ? msg : ""), msg_len, sk) != OQS_SUCCESS) { - free(signature); OQS_SIG_free(sig); - return pq_error("dilithium-3 sign failed"); - } - el_val_t sig_hex = el_hex_encode(signature, signature_len); - free(signature); OQS_SIG_free(sig); - return sig_hex; -} - -el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex) { - size_t pk_len = 0, sig_len = 0; - unsigned char* pk = el_hex_decode(EL_CSTR(public_key_hex), &pk_len); - unsigned char* signature = el_hex_decode(EL_CSTR(signature_hex), &sig_len); - if (!pk || !signature) return EL_INT(0); - - OQS_SIG* sig = OQS_SIG_new(EL_DILITHIUM_ALG); - if (!sig) return EL_INT(0); - if (pk_len != sig->length_public_key) { OQS_SIG_free(sig); return EL_INT(0); } - - const char* msg = EL_CSTR(message); - size_t msg_len = el_input_len(msg); - OQS_STATUS rc = OQS_SIG_verify(sig, - (const unsigned char*)(msg ? msg : ""), msg_len, - signature, sig_len, pk); - OQS_SIG_free(sig); - return (rc == OQS_SUCCESS) ? EL_INT(1) : EL_INT(0); -} - -/* ─── Kyber-768 / ML-KEM-768 KEM ──────────────────────────────────────── - * - * NIST FIPS 203 standardized CRYSTALS-Kyber as ML-KEM. ML-KEM-768 is the - * FIPS form of what we historically called Kyber-768. Same situation as - * Dilithium → ML-DSA: prefer the standardized constant, fall back to the - * legacy name. liboqs 0.15.0 still exposes OQS_KEM_alg_kyber_768; the - * algorithm is identical at the wire level to ML-KEM-768 except for FIPS - * domain-separation tweaks, so the two ciphertexts/keys are NOT - * cross-compatible. Pin the constant for archival material. */ - -#if defined(OQS_KEM_alg_ml_kem_768) -# define EL_KYBER_ALG OQS_KEM_alg_ml_kem_768 -#elif defined(OQS_KEM_alg_kyber_768) -# define EL_KYBER_ALG OQS_KEM_alg_kyber_768 -#else -# define EL_KYBER_ALG "ML-KEM-768" -#endif - -el_val_t pq_kem_keygen(void) { - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - unsigned char* pk = (unsigned char*)malloc(kem->length_public_key); - unsigned char* sk = (unsigned char*)malloc(kem->length_secret_key); - if (!pk || !sk) { free(pk); free(sk); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_keypair(kem, pk, sk) != OQS_SUCCESS) { - free(pk); free(sk); OQS_KEM_free(kem); - return pq_error("kyber-768 keypair generation failed"); - } - el_val_t pk_hex = el_hex_encode(pk, kem->length_public_key); - el_val_t sk_hex = el_hex_encode(sk, kem->length_secret_key); - OQS_MEM_secure_free(sk, kem->length_secret_key); - free(pk); - - const char* pks = EL_CSTR(pk_hex); - const char* sks = EL_CSTR(sk_hex); - char* buf = el_strbuf(strlen(pks) + strlen(sks) + 64); - sprintf(buf, "{\"public_key\":\"%s\",\"secret_key\":\"%s\"}", pks, sks); - OQS_KEM_free(kem); - return el_wrap_str(buf); -} - -el_val_t pq_kem_encaps(el_val_t public_key_hex) { - size_t pk_len = 0; - unsigned char* pk = el_hex_decode(EL_CSTR(public_key_hex), &pk_len); - if (!pk) return pq_error("invalid hex in public_key"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (pk_len != kem->length_public_key) { - OQS_KEM_free(kem); - return pq_error("public_key length mismatch for kyber-768"); - } - unsigned char* ct = (unsigned char*)malloc(kem->length_ciphertext); - unsigned char* ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!ct || !ss) { free(ct); free(ss); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_encaps(kem, ct, ss, pk) != OQS_SUCCESS) { - free(ct); free(ss); OQS_KEM_free(kem); - return pq_error("kyber-768 encapsulation failed"); - } - el_val_t ct_hex = el_hex_encode(ct, kem->length_ciphertext); - el_val_t ss_hex = el_hex_encode(ss, kem->length_shared_secret); - free(ct); - OQS_MEM_secure_free(ss, kem->length_shared_secret); - - const char* cts = EL_CSTR(ct_hex); - const char* sss = EL_CSTR(ss_hex); - char* buf = el_strbuf(strlen(cts) + strlen(sss) + 64); - sprintf(buf, "{\"ciphertext\":\"%s\",\"shared_secret\":\"%s\"}", cts, sss); - OQS_KEM_free(kem); - return el_wrap_str(buf); -} - -el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex) { - size_t sk_len = 0, ct_len = 0; - unsigned char* sk = el_hex_decode(EL_CSTR(secret_key_hex), &sk_len); - unsigned char* ct = el_hex_decode(EL_CSTR(ciphertext_hex), &ct_len); - if (!sk || !ct) return pq_error("invalid hex in inputs"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (sk_len != kem->length_secret_key || ct_len != kem->length_ciphertext) { - OQS_KEM_free(kem); - return pq_error("input length mismatch for kyber-768"); - } - unsigned char* ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!ss) { OQS_KEM_free(kem); return pq_error("oom"); } - /* Kyber is IND-CCA via Fujisaki-Okamoto: decaps always returns *some* - * shared_secret even on tampered ciphertext (an implicit-rejection value - * derived from sk). Protocols MUST confirm the shared_secret matches via - * a subsequent step (e.g. AEAD tag, key-confirmation MAC) — do not - * assume decaps success implies authenticity. */ - if (OQS_KEM_decaps(kem, ss, ct, sk) != OQS_SUCCESS) { - free(ss); OQS_KEM_free(kem); - return pq_error("kyber-768 decapsulation failed"); - } - el_val_t ss_hex = el_hex_encode(ss, kem->length_shared_secret); - OQS_MEM_secure_free(ss, kem->length_shared_secret); - OQS_KEM_free(kem); - return ss_hex; -} - -/* ─── Hybrid handshake (X25519 + Kyber-768, HKDF-SHA256 combined) ─────── */ - -#if !EL_HAVE_OPENSSL - -el_val_t pq_hybrid_keygen(void) { - return pq_error("hybrid handshake requires OpenSSL (X25519); rebuild with -lcrypto"); -} -el_val_t pq_hybrid_handshake(el_val_t pub) { - (void)pub; - return pq_error("hybrid handshake requires OpenSSL (X25519); rebuild with -lcrypto"); -} - -#else /* EL_HAVE_OPENSSL */ - -/* HKDF-SHA256 (RFC 5869) — Extract+Expand. Reuses the inline HMAC-SHA256 - * already in this file. Empty salt → 32 zero bytes per the RFC. */ -static void el_hkdf_sha256(const unsigned char* salt, size_t salt_len, - const unsigned char* ikm, size_t ikm_len, - const unsigned char* info, size_t info_len, - unsigned char* out, size_t out_len) { - unsigned char zero_salt[32] = {0}; - if (salt_len == 0) { salt = zero_salt; salt_len = 32; } - unsigned char prk[32]; - el_hmac_sha256(salt, salt_len, ikm, ikm_len, prk); - - unsigned char t[32]; - size_t produced = 0; - unsigned char counter = 1; - unsigned char* buf = (unsigned char*)malloc(32 + info_len + 1); - if (!buf) { fputs("el_runtime: hkdf oom\n", stderr); return; } - while (produced < out_len) { - size_t off = 0; - if (counter > 1) { memcpy(buf, t, 32); off = 32; } - if (info && info_len) { memcpy(buf + off, info, info_len); off += info_len; } - buf[off++] = counter; - el_hmac_sha256(prk, 32, buf, off, t); - size_t chunk = (out_len - produced > 32) ? 32 : (out_len - produced); - memcpy(out + produced, t, chunk); - produced += chunk; - counter++; - } - free(buf); -} - -/* X25519 keygen via OpenSSL EVP. Returns 1 on success. - * Fills pk[32] and sk[32] (raw X25519 byte strings, no DER wrapper). */ -static int el_x25519_keygen(unsigned char pk[32], unsigned char sk[32]) { - EVP_PKEY_CTX* pctx = EVP_PKEY_CTX_new_id(EVP_PKEY_X25519, NULL); - if (!pctx) return 0; - if (EVP_PKEY_keygen_init(pctx) != 1) { EVP_PKEY_CTX_free(pctx); return 0; } - EVP_PKEY* key = NULL; - if (EVP_PKEY_keygen(pctx, &key) != 1) { EVP_PKEY_CTX_free(pctx); return 0; } - EVP_PKEY_CTX_free(pctx); - - size_t plen = 32, slen = 32; - if (EVP_PKEY_get_raw_public_key (key, pk, &plen) != 1 || plen != 32) { - EVP_PKEY_free(key); return 0; - } - if (EVP_PKEY_get_raw_private_key(key, sk, &slen) != 1 || slen != 32) { - EVP_PKEY_free(key); return 0; - } - EVP_PKEY_free(key); - return 1; -} - -/* X25519 ECDH: derive 32-byte shared secret from local sk and remote pk. */ -static int el_x25519_derive(const unsigned char sk[32], const unsigned char rpk[32], - unsigned char ss[32]) { - EVP_PKEY* my = EVP_PKEY_new_raw_private_key(EVP_PKEY_X25519, NULL, sk, 32); - EVP_PKEY* rem = EVP_PKEY_new_raw_public_key (EVP_PKEY_X25519, NULL, rpk, 32); - if (!my || !rem) { EVP_PKEY_free(my); EVP_PKEY_free(rem); return 0; } - EVP_PKEY_CTX* dctx = EVP_PKEY_CTX_new(my, NULL); - if (!dctx) { EVP_PKEY_free(my); EVP_PKEY_free(rem); return 0; } - int ok = 0; - size_t out_len = 32; - if (EVP_PKEY_derive_init(dctx) == 1 && - EVP_PKEY_derive_set_peer(dctx, rem) == 1 && - EVP_PKEY_derive(dctx, ss, &out_len) == 1 && - out_len == 32) ok = 1; - EVP_PKEY_CTX_free(dctx); - EVP_PKEY_free(my); - EVP_PKEY_free(rem); - return ok; -} - -/* Hybrid wire layout (binary form, before hex encode): - * public_key = x25519_pub (32) || kyber_pub (1184) → 1216 bytes - * secret_key = x25519_sec (32) || kyber_sec (2400) → 2432 bytes - * ciphertext = ephem_x25519_pub (32) || kyber_ct (1088) → 1120 bytes - * shared_secret = HKDF-SHA256(x25519_ss || kyber_ss, info="el-pq-hybrid-v1", 32 bytes) - * The keygen result also exposes the four component hex fields for callers - * that prefer to handle the legs independently. */ - -el_val_t pq_hybrid_keygen(void) { - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - - unsigned char xpk[32], xsk[32]; - if (!el_x25519_keygen(xpk, xsk)) { - OQS_KEM_free(kem); - return pq_error("X25519 keygen failed"); - } - - unsigned char* kpk = (unsigned char*)malloc(kem->length_public_key); - unsigned char* ksk = (unsigned char*)malloc(kem->length_secret_key); - if (!kpk || !ksk) { free(kpk); free(ksk); OQS_KEM_free(kem); return pq_error("oom"); } - if (OQS_KEM_keypair(kem, kpk, ksk) != OQS_SUCCESS) { - free(kpk); free(ksk); OQS_KEM_free(kem); - return pq_error("kyber-768 keypair generation failed"); - } - - size_t pub_len = 32 + kem->length_public_key; - size_t sec_len = 32 + kem->length_secret_key; - unsigned char* pub_buf = (unsigned char*)malloc(pub_len); - unsigned char* sec_buf = (unsigned char*)malloc(sec_len); - if (!pub_buf || !sec_buf) { - free(pub_buf); free(sec_buf); free(kpk); - OQS_MEM_secure_free(ksk, kem->length_secret_key); - OQS_KEM_free(kem); return pq_error("oom"); - } - memcpy(pub_buf, xpk, 32); memcpy(pub_buf + 32, kpk, kem->length_public_key); - memcpy(sec_buf, xsk, 32); memcpy(sec_buf + 32, ksk, kem->length_secret_key); - - el_val_t x_pub_hex = el_hex_encode(xpk, 32); - el_val_t x_sec_hex = el_hex_encode(xsk, 32); - el_val_t k_pub_hex = el_hex_encode(kpk, kem->length_public_key); - el_val_t k_sec_hex = el_hex_encode(ksk, kem->length_secret_key); - el_val_t pub_hex = el_hex_encode(pub_buf, pub_len); - el_val_t sec_hex = el_hex_encode(sec_buf, sec_len); - - OQS_MEM_secure_free(ksk, kem->length_secret_key); - free(kpk); free(pub_buf); free(sec_buf); - OQS_KEM_free(kem); - memset(xsk, 0, 32); /* best-effort wipe of stack copy */ - - const char* xph = EL_CSTR(x_pub_hex); - const char* xsh = EL_CSTR(x_sec_hex); - const char* kph = EL_CSTR(k_pub_hex); - const char* ksh = EL_CSTR(k_sec_hex); - const char* pubh = EL_CSTR(pub_hex); - const char* sech = EL_CSTR(sec_hex); - - char* buf = el_strbuf(strlen(xph) + strlen(xsh) + strlen(kph) + strlen(ksh) - + strlen(pubh) + strlen(sech) + 256); - sprintf(buf, - "{\"x25519_pub\":\"%s\",\"x25519_sec\":\"%s\"," - "\"kyber_pub\":\"%s\",\"kyber_sec\":\"%s\"," - "\"public_key\":\"%s\",\"secret_key\":\"%s\"}", - xph, xsh, kph, ksh, pubh, sech); - return el_wrap_str(buf); -} - -/* Initiator-side handshake. Caller supplies the responder's combined public - * key (x25519_pub || kyber_pub, hex-encoded). The runtime: - * 1. Generates an ephemeral X25519 keypair, runs ECDH against the - * responder's static x25519_pub. - * 2. Runs Kyber-768 encaps against the responder's kyber_pub → kyber_ct, - * kyber_ss. - * 3. Combined shared = HKDF-SHA256(salt="", ikm = x25519_ss || kyber_ss, - * info = "el-pq-hybrid-v1", L = 32). - * 4. Returns combined ciphertext (= ephemeral_x25519_pub || kyber_ct) and - * the derived shared_secret. - * - * Responder side composition (intentionally not a separate runtime fn — - * trivial to express in El given pq_kem_decaps + a future x25519_derive - * primitive): split the ciphertext into ephem_xpk (32) and kyber_ct, run - * X25519(static_xsk, ephem_xpk) and pq_kem_decaps(static_kyber_sk, kyber_ct), - * then HKDF-SHA256 with the same salt/info to recover the same shared_secret. - * If a separate x25519 entry point becomes valuable, add `pq_hybrid_open` - * here taking (secret_key_combined, ciphertext_combined). */ -el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined) { - size_t pub_len = 0; - unsigned char* rpub = el_hex_decode(EL_CSTR(remote_pub_combined), &pub_len); - if (!rpub) return pq_error("invalid hex in remote_pub_combined"); - - OQS_KEM* kem = OQS_KEM_new(EL_KYBER_ALG); - if (!kem) return pq_error("OQS_KEM_new(kyber-768) failed"); - if (pub_len != 32 + kem->length_public_key) { - OQS_KEM_free(kem); - return pq_error("remote_pub_combined length mismatch (expected x25519_pub || kyber_pub)"); - } - - unsigned char e_xpk[32], e_xsk[32], x_ss[32]; - if (!el_x25519_keygen(e_xpk, e_xsk)) { - OQS_KEM_free(kem); - return pq_error("X25519 ephemeral keygen failed"); - } - if (!el_x25519_derive(e_xsk, rpub, x_ss)) { - memset(e_xsk, 0, 32); - OQS_KEM_free(kem); - return pq_error("X25519 derive failed"); - } - memset(e_xsk, 0, 32); /* ephemeral; not needed after derive */ - - unsigned char* k_ct = (unsigned char*)malloc(kem->length_ciphertext); - unsigned char* k_ss = (unsigned char*)malloc(kem->length_shared_secret); - if (!k_ct || !k_ss) { - free(k_ct); free(k_ss); OQS_KEM_free(kem); - return pq_error("oom"); - } - if (OQS_KEM_encaps(kem, k_ct, k_ss, rpub + 32) != OQS_SUCCESS) { - free(k_ct); free(k_ss); OQS_KEM_free(kem); - return pq_error("kyber-768 encapsulation failed"); - } - - /* HKDF combine: ikm = x_ss || k_ss. */ - size_t ikm_len = 32 + kem->length_shared_secret; - unsigned char* ikm = (unsigned char*)malloc(ikm_len); - if (!ikm) { - free(k_ct); OQS_MEM_secure_free(k_ss, kem->length_shared_secret); - OQS_KEM_free(kem); - return pq_error("oom"); - } - memcpy(ikm, x_ss, 32); - memcpy(ikm + 32, k_ss, kem->length_shared_secret); - unsigned char combined[32]; - static const char info_str[] = "el-pq-hybrid-v1"; - el_hkdf_sha256(NULL, 0, ikm, ikm_len, - (const unsigned char*)info_str, sizeof(info_str) - 1, - combined, 32); - - memset(x_ss, 0, 32); - OQS_MEM_secure_free(k_ss, kem->length_shared_secret); - OQS_MEM_secure_free(ikm, ikm_len); - - /* Combined ciphertext = ephemeral_x25519_pub || kyber_ct. */ - size_t ct_len = 32 + kem->length_ciphertext; - unsigned char* combined_ct = (unsigned char*)malloc(ct_len); - if (!combined_ct) { free(k_ct); OQS_KEM_free(kem); return pq_error("oom"); } - memcpy(combined_ct, e_xpk, 32); - memcpy(combined_ct + 32, k_ct, kem->length_ciphertext); - free(k_ct); - OQS_KEM_free(kem); - - el_val_t ct_hex = el_hex_encode(combined_ct, ct_len); - el_val_t ss_hex = el_hex_encode(combined, 32); - free(combined_ct); - memset(combined, 0, 32); - - const char* cts = EL_CSTR(ct_hex); - const char* sss = EL_CSTR(ss_hex); - char* buf = el_strbuf(strlen(cts) + strlen(sss) + 64); - sprintf(buf, "{\"ciphertext\":\"%s\",\"shared_secret\":\"%s\"}", cts, sss); - return el_wrap_str(buf); -} - -#endif /* EL_HAVE_OPENSSL */ -#endif /* EL_HAVE_LIBOQS */ - -/* ─── AEAD: AES-256-GCM ──────────────────────────────────────────────────── - * - * Symmetric authenticated encryption used to wrap envelopes once a shared - * secret has been derived from the KEM (Kyber-768 / hybrid). The El surface - * is intentionally narrow: - * - * aead_encrypt(key_hex, plaintext) - * → {"nonce":"<24 hex>","ciphertext":"<...hex including 16-byte tag>"} - * - * aead_decrypt(key_hex, nonce_hex, ciphertext_hex) - * → plaintext String, or "" on auth failure / malformed input - * - * Conventions: - * - key_hex must decode to exactly 32 bytes (AES-256). Callers that hold - * a longer KEM shared_secret should normalize via SHA3-256(ss) → 32 bytes - * before passing it in. (Kyber-768's shared_secret is already 32 bytes, - * but keeping this contract explicit lets the El side be agnostic.) - * - nonce is a fresh 12-byte random value drawn from the OS CSPRNG. Caller - * never picks the nonce — eliminates the GCM nonce-reuse footgun entirely. - * - tag is the standard 16 bytes, appended to ciphertext per RFC 5116. - * `ciphertext` field is therefore (plaintext_len + 16) bytes, hex-encoded. - * - No associated data (AAD). If we later need bound metadata, add a - * length-prefixed AAD argument and bump the envelope version tag. - * - * Failure mode: - * aead_encrypt returns http_error_json(...) on input/system failure. - * aead_decrypt returns the empty string on ANY failure (including auth-tag - * mismatch). Callers MUST check for "" before using the result. */ - -#if !__has_include() - -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext) { - (void)key_hex; (void)plaintext; - return http_error_json("aead_encrypt requires OpenSSL (libcrypto); rebuild with -lcrypto"); -} -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex) { - (void)key_hex; (void)nonce_hex; (void)ciphertext_hex; - return el_wrap_str(el_strdup("")); -} - -#else /* OpenSSL available */ - -#include -#include - -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext) { - size_t key_len = 0; - unsigned char* key = el_hex_decode(EL_CSTR(key_hex), &key_len); - if (!key) return http_error_json("invalid hex in key"); - if (key_len != 32) return http_error_json("aead key must be 32 bytes (64 hex chars) for AES-256-GCM"); - - const char* pt = EL_CSTR(plaintext); - size_t pt_len = el_input_len(pt); - if (!pt) pt = ""; - - unsigned char nonce[12]; - if (RAND_bytes(nonce, 12) != 1) return http_error_json("OS CSPRNG failed (RAND_bytes)"); - - EVP_CIPHER_CTX* ctx = EVP_CIPHER_CTX_new(); - if (!ctx) return http_error_json("EVP_CIPHER_CTX_new failed"); - - if (EVP_EncryptInit_ex(ctx, EVP_aes_256_gcm(), NULL, NULL, NULL) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm init failed"); - } - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_IVLEN, 12, NULL) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("set ivlen failed"); - } - if (EVP_EncryptInit_ex(ctx, NULL, NULL, key, nonce) != 1) { - EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm key/iv init failed"); - } - - /* GCM ciphertext is the same length as plaintext; we append a 16-byte - * authentication tag for AEAD semantics. Allocate plaintext_len + 16. */ - unsigned char* ct = (unsigned char*)malloc(pt_len + 16); - if (!ct) { EVP_CIPHER_CTX_free(ctx); return http_error_json("oom"); } - int outlen = 0, total = 0; - if (EVP_EncryptUpdate(ctx, ct, &outlen, (const unsigned char*)pt, (int)pt_len) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm update failed"); - } - total += outlen; - if (EVP_EncryptFinal_ex(ctx, ct + total, &outlen) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm final failed"); - } - total += outlen; - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_GET_TAG, 16, ct + total) != 1) { - free(ct); EVP_CIPHER_CTX_free(ctx); return http_error_json("aes-256-gcm get tag failed"); - } - EVP_CIPHER_CTX_free(ctx); - - el_val_t nonce_hex_v = el_hex_encode(nonce, 12); - el_val_t ct_hex_v = el_hex_encode(ct, (size_t)total + 16); - free(ct); - - const char* nh = EL_CSTR(nonce_hex_v); - const char* ch = EL_CSTR(ct_hex_v); - char* buf = el_strbuf(strlen(nh) + strlen(ch) + 48); - sprintf(buf, "{\"nonce\":\"%s\",\"ciphertext\":\"%s\"}", nh, ch); - return el_wrap_str(buf); -} - -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex) { - size_t key_len = 0, nonce_len = 0, ct_len = 0; - unsigned char* key = el_hex_decode(EL_CSTR(key_hex), &key_len); - unsigned char* nonce = el_hex_decode(EL_CSTR(nonce_hex), &nonce_len); - unsigned char* ct = el_hex_decode(EL_CSTR(ciphertext_hex), &ct_len); - if (!key || !nonce || !ct) return el_wrap_str(el_strdup("")); - if (key_len != 32 || nonce_len != 12) return el_wrap_str(el_strdup("")); - if (ct_len < 16) return el_wrap_str(el_strdup("")); - - size_t body_len = ct_len - 16; - const unsigned char* tag = ct + body_len; - - EVP_CIPHER_CTX* ctx = EVP_CIPHER_CTX_new(); - if (!ctx) return el_wrap_str(el_strdup("")); - - if (EVP_DecryptInit_ex(ctx, EVP_aes_256_gcm(), NULL, NULL, NULL) != 1 || - EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_IVLEN, 12, NULL) != 1 || - EVP_DecryptInit_ex(ctx, NULL, NULL, key, nonce) != 1) { - EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - - unsigned char* pt = (unsigned char*)malloc(body_len + 1); - if (!pt) { EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); } - int outlen = 0, total = 0; - if (EVP_DecryptUpdate(ctx, pt, &outlen, ct, (int)body_len) != 1) { - free(pt); EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - total += outlen; - /* Set expected tag before final — GCM's final step is where auth happens. */ - if (EVP_CIPHER_CTX_ctrl(ctx, EVP_CTRL_GCM_SET_TAG, 16, (void*)tag) != 1) { - free(pt); EVP_CIPHER_CTX_free(ctx); return el_wrap_str(el_strdup("")); - } - int rc = EVP_DecryptFinal_ex(ctx, pt + total, &outlen); - EVP_CIPHER_CTX_free(ctx); - if (rc != 1) { - /* Auth failure or padding/length mismatch. Return empty so callers - * cannot accidentally treat tampered ciphertext as a valid message. */ - free(pt); - return el_wrap_str(el_strdup("")); - } - total += outlen; - pt[total] = '\0'; - - /* Copy into the el arena so the caller-visible string outlives this fn. */ - char* out = el_strbuf((size_t)total); - memcpy(out, pt, (size_t)total); - out[total] = '\0'; - free(pt); - return el_wrap_str(out); -} - -#endif /* __has_include() */ - -/* ──────────────────────────────────────────────────────────────────────────── - * OTLP/HTTP observability — logs, traces, metrics - * - * Design goals: - * - Zero blocking on the request path. Producers append to in-memory - * ring buffers; a single worker thread flushes to the OTLP endpoint. - * - Drop-on-failure semantics. If the endpoint is unreachable or slow, - * we drop telemetry rather than back-pressure into the request handler. - * - Best-effort serialization. Each record is pre-serialized as JSON when - * the El program calls the primitive; the worker just batches. - * - Configuration via env vars: - * OTLP_ENDPOINT e.g. https://alloy.neuralplatform.ai:4318 - * OTEL_SERVICE_NAME e.g. neuron-web (default: argv[0] basename) - * OTEL_SERVICE_VERSION (default: "0.0.0") - * OTEL_RESOURCE_ATTRS comma-sep k=v pairs (optional) - * - * Wire format: OTLP/HTTP JSON. Three endpoints: - * POST {endpoint}/v1/logs — log records - * POST {endpoint}/v1/traces — spans - * POST {endpoint}/v1/metrics — counter/gauge points - * - * El programs see four primitives: - * trace_span_start(name) -> SpanHandle (just a string id) - * trace_span_end(handle) (computes duration, queues) - * emit_log(level, msg, fields_json) (queues a log record) - * emit_metric(name, value, tags_json) (queues a counter increment) - * ──────────────────────────────────────────────────────────────────────────── - */ - -#define OTLP_BUF_CAP 4096 /* per-buffer ring size */ -#define OTLP_FLUSH_MS 2000 /* flush every 2s */ -#define OTLP_BATCH_MAX 200 /* up to 200 records per POST */ - -typedef struct { - char* data; /* malloc'd JSON fragment for this record */ -} OtlpRec; - -typedef struct { - OtlpRec ring[OTLP_BUF_CAP]; - size_t head; /* next write slot */ - size_t tail; /* next read slot */ - pthread_mutex_t mu; -} OtlpQueue; - -static OtlpQueue _otlp_logs = { .mu = PTHREAD_MUTEX_INITIALIZER }; -static OtlpQueue _otlp_traces = { .mu = PTHREAD_MUTEX_INITIALIZER }; -static OtlpQueue _otlp_metrics = { .mu = PTHREAD_MUTEX_INITIALIZER }; - -static char* _otlp_endpoint = NULL; /* e.g. https://alloy.neuralplatform.ai:4318 */ -static char* _otlp_service_name = NULL; -static char* _otlp_service_version = NULL; -static int _otlp_initialized = 0; -static pthread_t _otlp_worker_thread; - -/* enqueue — returns 1 if accepted, 0 if dropped (full buffer or no endpoint) */ -static int otlp_enqueue(OtlpQueue* q, const char* json) { - if (!_otlp_endpoint || !json) return 0; - pthread_mutex_lock(&q->mu); - size_t next_head = (q->head + 1) % OTLP_BUF_CAP; - if (next_head == q->tail) { - /* buffer full — drop oldest */ - free(q->ring[q->tail].data); - q->ring[q->tail].data = NULL; - q->tail = (q->tail + 1) % OTLP_BUF_CAP; - } - q->ring[q->head].data = strdup(json); - q->head = next_head; - pthread_mutex_unlock(&q->mu); - return 1; -} - -/* drain — copies up to OTLP_BATCH_MAX items into a comma-joined string, - * caller must free the result. Returns NULL if queue is empty. */ -static char* otlp_drain(OtlpQueue* q) { - pthread_mutex_lock(&q->mu); - if (q->head == q->tail) { pthread_mutex_unlock(&q->mu); return NULL; } - /* compute total length */ - size_t total = 0, count = 0; - size_t i = q->tail; - while (i != q->head && count < OTLP_BATCH_MAX) { - if (q->ring[i].data) total += strlen(q->ring[i].data) + 1; /* +1 for comma */ - i = (i + 1) % OTLP_BUF_CAP; - count++; - } - char* out = malloc(total + 4); - if (!out) { pthread_mutex_unlock(&q->mu); return NULL; } - out[0] = '\0'; - size_t off = 0; - i = q->tail; - count = 0; - while (i != q->head && count < OTLP_BATCH_MAX) { - if (q->ring[i].data) { - size_t l = strlen(q->ring[i].data); - if (off > 0) { out[off++] = ','; } - memcpy(out + off, q->ring[i].data, l); - off += l; - free(q->ring[i].data); - q->ring[i].data = NULL; - } - i = (i + 1) % OTLP_BUF_CAP; - count++; - } - out[off] = '\0'; - q->tail = i; - pthread_mutex_unlock(&q->mu); - return out; -} - -/* Build resource block once (service.name, service.version, host.name) */ -static char* otlp_resource_block(void) { - static char cached[1024]; - static int built = 0; - if (built) return cached; - char host[256] = "unknown"; - gethostname(host, sizeof(host) - 1); - snprintf(cached, sizeof(cached), - "{\"attributes\":[" - "{\"key\":\"service.name\",\"value\":{\"stringValue\":\"%s\"}}," - "{\"key\":\"service.version\",\"value\":{\"stringValue\":\"%s\"}}," - "{\"key\":\"host.name\",\"value\":{\"stringValue\":\"%s\"}}" - "]}", - _otlp_service_name ? _otlp_service_name : "el-app", - _otlp_service_version ? _otlp_service_version : "0.0.0", - host); - built = 1; - return cached; -} - -/* Best-effort POST. Drops on any error. */ -static void otlp_post(const char* path, const char* body) { - if (!_otlp_endpoint || !body || !*body) return; - char url[1024]; - snprintf(url, sizeof(url), "%s%s", _otlp_endpoint, path); - CURL* c = curl_easy_init(); - if (!c) return; - struct curl_slist* h = NULL; - h = curl_slist_append(h, "Content-Type: application/json"); - curl_easy_setopt(c, CURLOPT_URL, url); - curl_easy_setopt(c, CURLOPT_POST, 1L); - curl_easy_setopt(c, CURLOPT_POSTFIELDS, body); - curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)strlen(body)); - curl_easy_setopt(c, CURLOPT_HTTPHEADER, h); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, 3000L); - curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); - curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, NULL); /* discard response */ - curl_easy_perform(c); - curl_slist_free_all(h); - curl_easy_cleanup(c); -} - -/* Flush worker — runs forever until process exits */ -static void* otlp_worker(void* arg) { - (void)arg; - while (1) { - struct timespec ts = { OTLP_FLUSH_MS / 1000, (OTLP_FLUSH_MS % 1000) * 1000000L }; - nanosleep(&ts, NULL); - - char* logs = otlp_drain(&_otlp_logs); - if (logs && *logs) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceLogs\":[{\"resource\":%s," - "\"scopeLogs\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"logRecords\":[%s]}]}]}", - otlp_resource_block(), logs); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/logs", body); - } - free(logs); - - char* traces = otlp_drain(&_otlp_traces); - if (traces && *traces) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceSpans\":[{\"resource\":%s," - "\"scopeSpans\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"spans\":[%s]}]}]}", - otlp_resource_block(), traces); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/traces", body); - } - free(traces); - - char* metrics = otlp_drain(&_otlp_metrics); - if (metrics && *metrics) { - char body[OTLP_BUF_CAP * 8]; - int n = snprintf(body, sizeof(body), - "{\"resourceMetrics\":[{\"resource\":%s," - "\"scopeMetrics\":[{\"scope\":{\"name\":\"el-runtime\"}," - "\"metrics\":[%s]}]}]}", - otlp_resource_block(), metrics); - if (n > 0 && n < (int)sizeof(body)) otlp_post("/v1/metrics", body); - } - free(metrics); - } - return NULL; -} - -/* Initialize OTLP — called lazily on first emit. Idempotent. */ -static void otlp_lazy_init(void) { - if (_otlp_initialized) return; - static pthread_mutex_t once_mu = PTHREAD_MUTEX_INITIALIZER; - pthread_mutex_lock(&once_mu); - if (_otlp_initialized) { pthread_mutex_unlock(&once_mu); return; } - - const char* ep = getenv("OTLP_ENDPOINT"); - if (!ep || !*ep) { - _otlp_initialized = 1; - pthread_mutex_unlock(&once_mu); - return; - } - _otlp_endpoint = strdup(ep); - /* trim trailing slash */ - size_t l = strlen(_otlp_endpoint); - if (l > 0 && _otlp_endpoint[l - 1] == '/') _otlp_endpoint[l - 1] = '\0'; - - const char* svc = getenv("OTEL_SERVICE_NAME"); - _otlp_service_name = strdup(svc && *svc ? svc : "el-app"); - const char* ver = getenv("OTEL_SERVICE_VERSION"); - _otlp_service_version = strdup(ver && *ver ? ver : "0.0.0"); - - pthread_create(&_otlp_worker_thread, NULL, otlp_worker, NULL); - pthread_detach(_otlp_worker_thread); - _otlp_initialized = 1; - pthread_mutex_unlock(&once_mu); -} - -/* JSON-escape a string into out_buf. Returns chars written (excluding null). */ -static size_t otlp_json_escape(const char* in, char* out, size_t out_cap) { - size_t o = 0; - for (size_t i = 0; in[i] && o + 8 < out_cap; i++) { - unsigned char c = (unsigned char)in[i]; - if (c == '"') { out[o++] = '\\'; out[o++] = '"'; } - else if (c == '\\'){ out[o++] = '\\'; out[o++] = '\\'; } - else if (c == '\n'){ out[o++] = '\\'; out[o++] = 'n'; } - else if (c == '\r'){ out[o++] = '\\'; out[o++] = 'r'; } - else if (c == '\t'){ out[o++] = '\\'; out[o++] = 't'; } - else if (c < 0x20) { o += snprintf(out + o, out_cap - o, "\\u%04x", c); } - else { out[o++] = (char)c; } - } - out[o] = '\0'; - return o; -} - -/* ── Public El primitives ─────────────────────────────────────────────────── */ - -/* emit_log(level, msg, fields_json) — fields_json is a JSON object string or "" */ -el_val_t emit_log(el_val_t level_v, el_val_t msg_v, el_val_t fields_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* level = EL_CSTR(level_v); if (!level) level = "INFO"; - const char* msg = EL_CSTR(msg_v); if (!msg) msg = ""; - const char* fields = EL_CSTR(fields_v); if (!fields) fields = ""; - /* Map El level names to OTLP severity numbers */ - int sev_num = 9; /* INFO */ - if (strcmp(level, "TRACE") == 0) sev_num = 1; - else if (strcmp(level, "DEBUG") == 0) sev_num = 5; - else if (strcmp(level, "INFO") == 0) sev_num = 9; - else if (strcmp(level, "WARN") == 0 || strcmp(level, "WARNING") == 0) sev_num = 13; - else if (strcmp(level, "ERROR") == 0) sev_num = 17; - else if (strcmp(level, "FATAL") == 0) sev_num = 21; - char esc_msg[2048]; otlp_json_escape(msg, esc_msg, sizeof(esc_msg)); - /* unix nanos */ - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"timeUnixNano\":\"%lld\",\"severityNumber\":%d," - "\"severityText\":\"%s\"," - "\"body\":{\"stringValue\":\"%s\"}%s%s}", - now_nano, sev_num, level, esc_msg, - (fields && *fields) ? ",\"attributes\":" : "", - (fields && *fields) ? fields : ""); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_logs, rec); - return EL_INT(1); -} - -/* emit_metric(name, value, tags_json) — Sum (counter) data point. tags_json - * is a JSON array of {key, value} pairs or empty string. */ -el_val_t emit_metric(el_val_t name_v, el_val_t value_v, el_val_t tags_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* name = EL_CSTR(name_v); if (!name) name = "unknown"; - int64_t val = (int64_t)value_v; - const char* tags = EL_CSTR(tags_v); if (!tags) tags = ""; - char esc_name[256]; otlp_json_escape(name, esc_name, sizeof(esc_name)); - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"name\":\"%s\",\"sum\":{\"aggregationTemporality\":2,\"isMonotonic\":true," - "\"dataPoints\":[{\"asInt\":\"%lld\"," - "\"timeUnixNano\":\"%lld\"" - "%s%s}]}}", - esc_name, (long long)val, now_nano, - (tags && *tags) ? ",\"attributes\":" : "", - (tags && *tags) ? tags : ""); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_metrics, rec); - return EL_INT(1); -} - -/* trace_span_start(name) — returns a span handle (string of "traceid:spanid:start_nano:name") */ -el_val_t trace_span_start(el_val_t name_v) { - otlp_lazy_init(); - const char* name = EL_CSTR(name_v); if (!name) name = "span"; - /* generate 16-byte trace id and 8-byte span id */ - static _Thread_local int seeded = 0; - if (!seeded) { srand((unsigned int)(uintptr_t)pthread_self() ^ (unsigned int)time(NULL)); seeded = 1; } - char tid[33], sid[17]; - for (int i = 0; i < 32; i++) tid[i] = "0123456789abcdef"[rand() & 0xF]; - tid[32] = '\0'; - for (int i = 0; i < 16; i++) sid[i] = "0123456789abcdef"[rand() & 0xF]; - sid[16] = '\0'; - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long now_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char* handle = malloc(strlen(name) + 80); - if (!handle) return EL_STR(""); - sprintf(handle, "%s:%s:%lld:%s", tid, sid, now_nano, name); - el_arena_track(handle); - return EL_STR(handle); -} - -/* trace_span_end(handle) — emits the span with computed duration */ -el_val_t trace_span_end(el_val_t handle_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* h = EL_CSTR(handle_v); if (!h) return EL_INT(0); - /* parse "tid:sid:start_nano:name" */ - char tid[64], sid[32], rest[1024]; - long long start_nano = 0; - if (sscanf(h, "%63[^:]:%31[^:]:%lld:%1023[^\n]", tid, sid, &start_nano, rest) != 4) return EL_INT(0); - struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); - long long end_nano = (long long)ts.tv_sec * 1000000000LL + ts.tv_nsec; - char esc_name[1024]; otlp_json_escape(rest, esc_name, sizeof(esc_name)); - char rec[4096]; - int n = snprintf(rec, sizeof(rec), - "{\"traceId\":\"%s\",\"spanId\":\"%s\"," - "\"name\":\"%s\"," - "\"kind\":1," - "\"startTimeUnixNano\":\"%lld\"," - "\"endTimeUnixNano\":\"%lld\"," - "\"status\":{\"code\":1}}", - tid, sid, esc_name, start_nano, end_nano); - if (n > 0 && n < (int)sizeof(rec)) otlp_enqueue(&_otlp_traces, rec); - return EL_INT(1); -} - -/* Convenience: emit a one-shot timed event (emit start+end immediately). - * For El programs that want point events with duration baked in. */ -el_val_t emit_event(el_val_t name_v, el_val_t duration_ms_v) { - otlp_lazy_init(); - if (!_otlp_endpoint) return EL_INT(0); - const char* name = EL_CSTR(name_v); if (!name) name = "event"; - int64_t dur_ms = (int64_t)duration_ms_v; - el_val_t h = trace_span_start(EL_STR((char*)name)); - /* fudge start to be (now - duration) */ - (void)dur_ms; - return trace_span_end(h); -} - diff --git a/lang/el-compiler/runtime/legacy/el_runtime.h b/lang/el-compiler/runtime/legacy/el_runtime.h deleted file mode 100644 index 8b80ee7..0000000 --- a/lang/el-compiler/runtime/legacy/el_runtime.h +++ /dev/null @@ -1,761 +0,0 @@ -/* - * el_runtime.h — El language C runtime header - * - * Declares all built-in functions available to compiled El programs. - * Include this in every generated .c file. - * - * Value model: - * All El values are represented as el_val_t (= int64_t). - * On 64-bit systems a pointer fits in int64_t. - * String values are cast: (el_val_t)(uintptr_t)"hello" - * Integer values are stored directly. - * This lets arithmetic work naturally while still passing strings around. - * - * Type conventions (El -> C): - * String -> el_val_t (holds const char* via uintptr_t cast) - * Int -> el_val_t - * Bool -> el_val_t (0 = false, nonzero = true) - * Any -> el_val_t - * Void -> void - * - * Macros for convenience: - * EL_STR(s) cast string literal to el_val_t - * EL_CSTR(v) cast el_val_t back to const char* - * EL_INT(v) identity — el_val_t is already int64_t - * - * Link requirements: - * -lcurl — required for the HTTP client (http_get, http_post, llm_*). - * -lpthread — required for the HTTP server (one detached thread per - * connection, capped at 64 concurrent). - * -loqs — optional; required only when liboqs is installed and the - * pq_* / sha3_256_hex entry points are needed. Detected at - * compile time via __has_include(). - * -lcrypto — optional; pulled in alongside -loqs. Used for X25519 in - * pq_hybrid_* and HKDF-SHA256 derivation. - * - * Canonical compile command: - * cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ - * -o .c el-compiler/runtime/el_runtime.c - * - * With liboqs (post-quantum stack): - * cc -std=c11 -I el-compiler/runtime -lcurl -lpthread -loqs -lcrypto \ - * -o .c el-compiler/runtime/el_runtime.c - */ - -#pragma once - -#include -#include - -typedef int64_t el_val_t; - -#define EL_STR(s) ((el_val_t)(uintptr_t)(s)) -#define EL_CSTR(v) ((const char*)(uintptr_t)(v)) -#define EL_INT(v) (v) -#define EL_NULL ((el_val_t)0) - -/* Float values share the el_val_t (int64) slot via a bit-cast. - * The codegen emits Float literals as `el_from_float()` so the - * underlying bits represent the IEEE 754 double. Float-aware builtins - * (math, format, json) round-trip via these helpers. */ -static inline double el_to_float(el_val_t v) { - union { int64_t i; double f; } u; - u.i = (int64_t)v; - return u.f; -} - -static inline el_val_t el_from_float(double f) { - union { double f; int64_t i; } u; - u.f = f; - return (el_val_t)u.i; -} - -#ifdef __cplusplus -extern "C" { -#endif - -/* ── I/O ──────────────────────────────────────────────────────────────────── */ - -void println(el_val_t s); -void print(el_val_t s); -el_val_t readline(void); - -/* ── String builtins ─────────────────────────────────────────────────────── */ - -el_val_t el_str_concat(el_val_t a, el_val_t b); -el_val_t str_eq(el_val_t a, el_val_t b); -el_val_t str_starts_with(el_val_t s, el_val_t prefix); -el_val_t str_ends_with(el_val_t s, el_val_t suffix); -el_val_t str_len(el_val_t s); -el_val_t str_concat(el_val_t a, el_val_t b); -el_val_t int_to_str(el_val_t n); -el_val_t str_to_int(el_val_t s); -el_val_t str_slice(el_val_t s, el_val_t start, el_val_t end); -el_val_t str_contains(el_val_t s, el_val_t sub); -el_val_t str_replace(el_val_t s, el_val_t from, el_val_t to); -el_val_t str_to_upper(el_val_t s); -el_val_t str_to_lower(el_val_t s); -el_val_t str_trim(el_val_t s); - -/* ── Math ────────────────────────────────────────────────────────────────── */ - -el_val_t el_abs(el_val_t n); -el_val_t el_max(el_val_t a, el_val_t b); -el_val_t el_min(el_val_t a, el_val_t b); - -/* ── Refcount (ARC) ────────────────────────────────────────────────────────── - * Lists and Maps carry a refcount. Strings and ints do not — el_retain and - * el_release are safe no-ops on non-refcounted values (they sniff a magic - * header at offset 0 and only act if the magic matches). - * - * Codegen emits these at let-binding shadowing, function entry (params), and - * function exit (locals other than the returned value). The refcount lets - * el_list_append and el_map_set mutate in place when uniquely owned (cheap) - * and copy-on-write when shared (preserves persistent semantics across - * accumulator patterns in the compiler itself). */ - -void el_retain(el_val_t v); -void el_release(el_val_t v); - -/* ── List ────────────────────────────────────────────────────────────────── */ - -el_val_t el_list_new(el_val_t count, ...); -el_val_t el_list_len(el_val_t list); -el_val_t el_list_get(el_val_t list, el_val_t index); -el_val_t el_list_append(el_val_t list, el_val_t elem); -el_val_t el_list_empty(void); -el_val_t el_list_clone(el_val_t list); - -/* ── Map ─────────────────────────────────────────────────────────────────── */ - -el_val_t el_map_new(el_val_t pair_count, ...); -el_val_t el_get_field(el_val_t map, el_val_t key); -el_val_t el_map_get(el_val_t map, el_val_t key); -el_val_t el_map_set(el_val_t map, el_val_t key, el_val_t value); - -/* ── HTTP ─────────────────────────────────────────────────────────────────── */ - -el_val_t http_get(el_val_t url); -el_val_t http_post(el_val_t url, el_val_t body); -el_val_t http_post_json(el_val_t url, el_val_t json_body); -el_val_t http_get_with_headers(el_val_t url, el_val_t headers_map); -el_val_t http_post_with_headers(el_val_t url, el_val_t body, el_val_t headers_map); -el_val_t http_post_form_auth(el_val_t url, el_val_t form_body, el_val_t auth_header); -el_val_t http_delete(el_val_t url); -void http_serve(el_val_t port, el_val_t handler); -void http_set_handler(el_val_t name); - -/* HTTP server v2 ───────────────────────────────────────────────────────────── - * Same dispatch model as http_serve, but the handler signature is widened: - * - * el_val_t handler(method, path, headers_map, body) - * - * `headers_map` is an ElMap from lowercased header name → header value (both - * Strings). Repeated headers are joined with ", " per RFC 7230. - * - * Response value: the handler may return either - * (a) a plain body string — same auto-content-type / 200-OK behaviour as - * http_serve (3-arg) — or - * (b) a response envelope built with `http_response(status, headers_json, - * body)`. The runtime detects the envelope discriminator - * `"el_http_response":1` at the start of the returned string and - * unpacks status / headers / body before sending. - * - * The 3-arg http_serve(port, handler) remains supported unchanged for - * existing handlers (e.g. products/web/server.el): it dispatches with - * (method, path, body), hardcodes 200 OK, and auto-detects content type. */ -void http_serve_v2(el_val_t port, el_val_t handler); -void http_set_handler_v2(el_val_t name); - -/* Build an HTTP response envelope. `headers_json` should be a JSON object - * literal like `{"WWW-Authenticate":"Basic"}` (or "" / "{}" for none). The - * returned string carries the discriminator `{"el_http_response":1,...}` - * which the runtime's send-path detects and unpacks. Detection happens - * uniformly inside http_send_response, so a 3-arg handler may also return - * an envelope. The 3-arg variant remains documented as a fixed 200-OK - * auto-content-type contract for legacy handlers that return plain bodies. */ -el_val_t http_response(el_val_t status, el_val_t headers_json, el_val_t body); - -/* SSE connection fd — set by http_worker_v2 before calling the El handler, - * cleared afterwards. Defined in el_seed.c; called from el_runtime.c. - * The getter is exposed as __http_conn_fd() to El programs. */ -void el_seed_set_http_conn_fd(int fd); - -/* HTTP timeout — every libcurl request honors EL_HTTP_TIMEOUT_MS (default - * 60000ms). Read lazily on first use, so setting the env var any time before - * the first http_* call is sufficient. */ - -/* Streaming variants — write the response body straight to a file via - * libcurl's CURLOPT_WRITEFUNCTION = fwrite. These bypass the el_val_t string - * wrapper entirely, so binary payloads (audio/mpeg, image/png, etc.) survive - * embedded NUL bytes that would truncate a strlen()-based code path. - * - * Both honor EL_HTTP_TIMEOUT_MS, follow redirects, and accept the same - * `headers_map` shape as http_post_with_headers (ElMap of String→String). - * - * Return value: 1 on success (file fully written), 0 on any failure - * (network, file open, partial write). On failure the output file is removed - * so callers cannot mistake a partially-written file for a valid one. */ -el_val_t http_post_to_file(el_val_t url, el_val_t body, el_val_t headers_map, el_val_t output_path); -el_val_t http_get_to_file(el_val_t url, el_val_t headers_map, el_val_t output_path); - -/* ── URL encoding ────────────────────────────────────────────────────────── */ - -el_val_t url_encode(el_val_t s); /* RFC 3986 unreserved set */ -el_val_t url_decode(el_val_t s); /* '+' → space, %XX → byte */ - -/* ── HTML allowlist sanitizer ──────────────────────────────────────────────── - * el_html_sanitize(input_html, allowlist_json) — strict allowlist HTML - * cleaner. State-machine parser; tag/attribute names compared case- - * insensitively against the allowlist; `` / `<… src>` URL schemes - * validated (http, https, mailto, fragment-only, or relative); whole- - * subtree drop for script / style / iframe / object / embed / form; HTML- - * escapes free text outside dropped subtrees. - * - * The allowlist is JSON of the form - * {"p":[],"a":["href","title"],"strong":[],...} - * where each value is the array of attribute names allowed for that tag. */ -el_val_t el_html_sanitize(el_val_t input_html, el_val_t allowlist_json); - -/* ── Filesystem ──────────────────────────────────────────────────────────── */ - -el_val_t fs_read(el_val_t path); -el_val_t fs_write(el_val_t path, el_val_t content); -el_val_t fs_list(el_val_t path); -el_val_t fs_exists(el_val_t path); -el_val_t fs_mkdir(el_val_t path); /* mkdir -p, mode 0755 */ - -/* Length-explicit binary write. `length` is an Int (el_val_t holding the - * byte count). The caller knows the length from context — typically because - * `bytes` came from base64_decode (which produces a magic-tagged binary - * buffer with embedded NULs possible) and the caller already tracks the - * decoded length, OR because the bytes came from a fixed-size source - * (sha256_bytes = 32, hmac_sha256_bytes = 32). Bypasses strlen entirely. - * - * Returns 1 on success, 0 on failure (invalid path, can't open, partial - * write, negative length). On partial-write failure, the file is removed - * so callers cannot read back a truncated artefact. */ -el_val_t fs_write_bytes(el_val_t path, el_val_t bytes, el_val_t length); - -/* ── JSON ────────────────────────────────────────────────────────────────── */ - -el_val_t json_get(el_val_t json, el_val_t key); -el_val_t json_parse(el_val_t s); -el_val_t json_stringify(el_val_t v); -el_val_t json_get_string(el_val_t json_str, el_val_t key); -el_val_t json_get_int(el_val_t json_str, el_val_t key); -el_val_t json_get_float(el_val_t json_str, el_val_t key); -el_val_t json_get_bool(el_val_t json_str, el_val_t key); -el_val_t json_get_raw(el_val_t json_str, el_val_t key); -el_val_t json_set(el_val_t json_str, el_val_t key, el_val_t value); -el_val_t json_array_len(el_val_t json_str); -el_val_t json_array_get(el_val_t json_str, el_val_t index); -el_val_t json_array_get_string(el_val_t json_str, el_val_t index); - -/* ── Time ────────────────────────────────────────────────────────────────── */ - -el_val_t time_now(void); -el_val_t time_now_utc(void); -el_val_t sleep_secs(el_val_t secs); -el_val_t sleep_ms(el_val_t ms); -el_val_t time_format(el_val_t ts, el_val_t fmt); -el_val_t time_to_parts(el_val_t ts); -el_val_t time_from_parts(el_val_t secs, el_val_t ns, el_val_t tz); -el_val_t time_add(el_val_t ts, el_val_t n, el_val_t unit); -el_val_t time_diff(el_val_t ts1, el_val_t ts2, el_val_t unit); - -/* ── Instant + Duration: first-class temporal types ────────────────────────── - * Both types share the el_val_t (int64) slot. Instants are nanoseconds - * since the Unix epoch; Durations are signed nanoseconds. Type discipline - * is enforced at codegen-time: BinOps on names registered as Instant or - * Duration route through the typed wrappers below; mismatches like - * Instant+Instant become #error at the C compiler. - * - * Postfix literals — `30.seconds`, `1.hour`, `500.millis`, `30.nanos` — are - * recognised by the parser as DurationLit AST nodes and lowered to literal - * int64 nanoseconds at codegen time. The runtime never sees the units. */ - -el_val_t el_now_instant(void); -el_val_t now(void); -el_val_t unix_seconds(el_val_t n); -el_val_t unix_millis(el_val_t n); -el_val_t instant_from_iso8601(el_val_t s); - -el_val_t el_duration_from_nanos(el_val_t ns); -el_val_t duration_seconds(el_val_t n); -el_val_t duration_millis(el_val_t n); -el_val_t duration_nanos(el_val_t n); - -el_val_t el_instant_add_dur(el_val_t inst, el_val_t dur); -el_val_t el_instant_sub_dur(el_val_t inst, el_val_t dur); -el_val_t el_instant_diff(el_val_t a, el_val_t b); -el_val_t el_duration_add(el_val_t a, el_val_t b); -el_val_t el_duration_sub(el_val_t a, el_val_t b); -el_val_t el_duration_scale(el_val_t dur, el_val_t scalar); -el_val_t el_duration_div(el_val_t dur, el_val_t scalar); - -el_val_t el_instant_lt(el_val_t a, el_val_t b); -el_val_t el_instant_le(el_val_t a, el_val_t b); -el_val_t el_instant_gt(el_val_t a, el_val_t b); -el_val_t el_instant_ge(el_val_t a, el_val_t b); -el_val_t el_instant_eq(el_val_t a, el_val_t b); -el_val_t el_instant_ne(el_val_t a, el_val_t b); -el_val_t el_duration_lt(el_val_t a, el_val_t b); -el_val_t el_duration_le(el_val_t a, el_val_t b); -el_val_t el_duration_gt(el_val_t a, el_val_t b); -el_val_t el_duration_ge(el_val_t a, el_val_t b); -el_val_t el_duration_eq(el_val_t a, el_val_t b); -el_val_t el_duration_ne(el_val_t a, el_val_t b); - -el_val_t instant_to_unix_seconds(el_val_t i); -el_val_t instant_to_unix_millis(el_val_t i); -el_val_t instant_to_iso8601(el_val_t i); -el_val_t duration_to_seconds(el_val_t d); -el_val_t duration_to_millis(el_val_t d); -el_val_t duration_to_nanos(el_val_t d); - -el_val_t el_sleep_duration(el_val_t dur); -el_val_t unix_timestamp(void); - -el_val_t ttl_cache_set(el_val_t key, el_val_t value); -el_val_t ttl_cache_get(el_val_t key, el_val_t max_age); -el_val_t ttl_cache_age(el_val_t key); - -/* ── Calendar + CalendarTime + Rhythm + LocalDate/Time/DateTime ───────────── - * Phase 1.5 of the time system. Calendar is pluggable: EarthCalendar (IANA - * zones, Gregorian, DST) is the user-facing default; MarsCalendar, - * CycleCalendar(period), NoCycleCalendar, RelativeCalendar handle non-Earth - * domains. - * - * A Calendar interprets an Instant under a particular cycle convention and - * produces a CalendarTime. CalendarTime carries the underlying Instant and - * a back-pointer to its Calendar; arithmetic and formatting consult the - * Calendar to convert ns since epoch into year/month/day/hour/minute/second - * (or sol/phase, or cycle/phase, depending on kind). - * - * Storage convention: Calendar / CalendarTime / Rhythm / LocalDate / - * LocalDateTime are heap-allocated structs whose pointers are cast into - * el_val_t. A 24-bit magic header at offset 0 lets the runtime identify - * the kind safely. LocalTime is small enough to live in the int64 slot - * directly (nanos since midnight, signed). */ - -/* Zone — opaque IANA zone or fixed offset, used by EarthCalendar. - * `zone_id` is either an IANA name ("America/New_York", "UTC") or a fixed - * offset string ("+05:30", "-08:00"). The runtime resolves it via tzset() - * on first use of the owning EarthCalendar. */ -el_val_t zone(el_val_t id); -el_val_t zone_utc(void); -el_val_t zone_local(void); -el_val_t zone_offset(el_val_t hours, el_val_t minutes); - -/* Calendar constructors. Each returns an el_val_t pointer to a heap- - * allocated, magic-tagged Calendar struct. Calendars are interned by - * (kind, zone_id, period_ns, epoch_ns) so identical constructors return - * the same pointer — equality is reference equality. */ -el_val_t earth_calendar(el_val_t z); -el_val_t earth_calendar_default(void); -el_val_t mars_calendar(void); -el_val_t cycle_calendar(el_val_t period_dur); -el_val_t no_cycle_calendar(void); -el_val_t relative_calendar(el_val_t epoch_inst); - -/* CalendarTime constructors and methods. Returns a heap-allocated struct - * whose pointer fits in el_val_t. */ -el_val_t now_in(el_val_t cal); -el_val_t in_calendar(el_val_t inst, el_val_t cal); -el_val_t cal_format(el_val_t ct, el_val_t pattern); -el_val_t cal_to_instant(el_val_t ct); -el_val_t cal_cycle_phase(el_val_t ct); -el_val_t cal_in(el_val_t ct, el_val_t cal); - -/* LocalDate / LocalTime / LocalDateTime — calendar-agnostic value types. - * LocalTime carries nanoseconds since midnight as a signed int64 directly - * in the el_val_t slot (no allocation). LocalDate / LocalDateTime are - * heap-allocated structs with magic headers. */ -el_val_t local_date(el_val_t y, el_val_t m, el_val_t d); -el_val_t local_time(el_val_t h, el_val_t m, el_val_t s, el_val_t ns); -el_val_t local_datetime(el_val_t date, el_val_t time); -el_val_t zoned(el_val_t date, el_val_t time, el_val_t cal); - -el_val_t local_date_year(el_val_t ld); -el_val_t local_date_month(el_val_t ld); -el_val_t local_date_day(el_val_t ld); -el_val_t local_time_hour(el_val_t lt); -el_val_t local_time_minute(el_val_t lt); -el_val_t local_time_second(el_val_t lt); -el_val_t local_time_nanos(el_val_t lt); - -el_val_t el_local_date_add_dur(el_val_t ld, el_val_t dur); -el_val_t el_local_time_add_dur(el_val_t lt, el_val_t dur); -el_val_t el_local_date_lt(el_val_t a, el_val_t b); -el_val_t el_local_date_eq(el_val_t a, el_val_t b); - -/* Rhythm — pluggable recurrence AST. Returns a heap-allocated struct - * pointer in el_val_t; rhythms are immutable so callers may share them. */ -el_val_t rhythm_cycle_start(void); -el_val_t rhythm_cycle_phase(el_val_t phase); -el_val_t rhythm_duration(el_val_t d); -el_val_t rhythm_session_start(void); -el_val_t rhythm_event(el_val_t name); -el_val_t rhythm_and(el_val_t a, el_val_t b); -el_val_t rhythm_or(el_val_t a, el_val_t b); -el_val_t rhythm_weekday(el_val_t day); -el_val_t rhythm_weekly_at(el_val_t day, el_val_t hour, el_val_t minute); -el_val_t rhythm_next_after(el_val_t r, el_val_t after, el_val_t cal); -el_val_t rhythm_matches(el_val_t r, el_val_t ct); - -/* ── UUID ────────────────────────────────────────────────────────────────── */ - -el_val_t uuid_new(void); -el_val_t uuid_v4(void); - -/* ── Environment ─────────────────────────────────────────────────────────── */ - -el_val_t env(el_val_t key); - -/* ── In-process state K/V ────────────────────────────────────────────────── */ - -el_val_t state_set(el_val_t key, el_val_t value); -el_val_t state_get(el_val_t key); -el_val_t state_del(el_val_t key); -el_val_t state_keys(void); - -/* ── Float formatting ────────────────────────────────────────────────────── */ - -el_val_t float_to_str(el_val_t f); -el_val_t int_to_float(el_val_t n); -el_val_t float_to_int(el_val_t f); -el_val_t format_float(el_val_t f, el_val_t decimals); -el_val_t decimal_round(el_val_t f, el_val_t decimals); -el_val_t str_to_float(el_val_t s); - -/* ── Math (Float-aware) ──────────────────────────────────────────────────── */ - -el_val_t math_sqrt(el_val_t f); -el_val_t math_log(el_val_t f); -el_val_t math_ln(el_val_t f); -el_val_t math_sin(el_val_t f); -el_val_t math_cos(el_val_t f); -el_val_t math_pi(void); - -/* ── String additions ────────────────────────────────────────────────────── */ - -el_val_t str_index_of(el_val_t s, el_val_t sub); -el_val_t str_split(el_val_t s, el_val_t sep); -el_val_t str_char_at(el_val_t s, el_val_t i); -el_val_t str_char_code(el_val_t s, el_val_t i); -el_val_t str_pad_left(el_val_t s, el_val_t width, el_val_t pad); -el_val_t str_pad_right(el_val_t s, el_val_t width, el_val_t pad); -el_val_t str_format(el_val_t fmt, el_val_t data); -el_val_t str_lower(el_val_t s); -el_val_t str_upper(el_val_t s); - -/* ── Text-processing primitives (Phase 1: byte/codepoint, ASCII char classes) - * Phase 2 (filed): Unicode-grapheme awareness, NFC/NFD normalization, regex. - * is_* predicates: empty input returns false; multi-char requires ALL bytes - * to match. ASCII ranges only in Phase 1. */ - -/* Counting */ -el_val_t str_count(el_val_t s, el_val_t sub); /* non-overlapping */ -el_val_t str_count_chars(el_val_t s); /* codepoint count */ -el_val_t str_count_bytes(el_val_t s); /* alias of str_len */ -el_val_t str_count_lines(el_val_t s); -el_val_t str_count_words(el_val_t s); -el_val_t str_count_letters(el_val_t s); /* ASCII [A-Za-z] */ -el_val_t str_count_digits(el_val_t s); /* ASCII [0-9] */ - -/* Find / position */ -el_val_t str_index_of_all(el_val_t s, el_val_t sub); /* [Int] of byte offsets */ -el_val_t str_last_index_of(el_val_t s, el_val_t sub); -el_val_t str_find_chars(el_val_t s, el_val_t any_of); /* first idx of any ch */ - -/* Transform */ -el_val_t str_repeat(el_val_t s, el_val_t n); -el_val_t str_reverse(el_val_t s); /* by codepoint */ -el_val_t str_strip_prefix(el_val_t s, el_val_t prefix); -el_val_t str_strip_suffix(el_val_t s, el_val_t suffix); -el_val_t str_strip_chars(el_val_t s, el_val_t chars); -el_val_t str_lstrip(el_val_t s); -el_val_t str_rstrip(el_val_t s); - -/* Char classification (Bool) */ -el_val_t is_letter(el_val_t s); -el_val_t is_digit(el_val_t s); -el_val_t is_alphanumeric(el_val_t s); -el_val_t is_whitespace(el_val_t s); -el_val_t is_punctuation(el_val_t s); -el_val_t is_uppercase(el_val_t s); -el_val_t is_lowercase(el_val_t s); - -/* Split / join */ -el_val_t str_split_lines(el_val_t s); -el_val_t str_split_chars(el_val_t s); /* alias of native_string_chars */ -el_val_t str_split_n(el_val_t s, el_val_t sep, el_val_t n); -el_val_t str_join(el_val_t list, el_val_t sep); /* alias of list_join */ - -/* ── List additions ──────────────────────────────────────────────────────── */ - -el_val_t list_push(el_val_t list, el_val_t elem); -el_val_t list_push_front(el_val_t list, el_val_t elem); -el_val_t list_join(el_val_t list, el_val_t sep); -el_val_t list_range(el_val_t start, el_val_t end); - -/* ── Bool helpers ────────────────────────────────────────────────────────── */ - -el_val_t bool_to_str(el_val_t b); - -/* ── Numeric parsing ─────────────────────────────────────────────────────── */ - -el_val_t parse_int(el_val_t s, el_val_t default_val); - -/* ── Process ─────────────────────────────────────────────────────────────── */ - -void exit_program(el_val_t code); -el_val_t getpid_now(void); - -/* ── CGI identity ───────────────────────────────────────────────────────────── - * Called at the start of main() in CGI programs (those with a `cgi {}` block). - * Records the program's DHARMA identity before any other code executes. */ - -void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, - el_val_t network, el_val_t engram); - -/* ── DHARMA network builtins ───────────────────────────────────────────────── - * Available to CGI programs (declared with a `cgi {}` block). - * - * Peers are addressed by `dharma_id` of the form - * "@" e.g. "ntn-genesis@http://localhost:7770" - * If the @ portion is omitted, transport defaults to - * "http://localhost:7770" (the local CGI daemon assumption). - * - * Wire protocol (all peers expose): - * POST /dharma/recv { channel, from, content } → response body - * POST /dharma/event { type, payload, source, timestamp } - * POST /api/activate { query } → list of nodes - * - * Hosting application's responsibility: an El program with a `cgi {}` block - * runs http_serve() with its own request handler; that handler should route - * "/dharma/event" requests by calling el_runtime_dharma_event_arrive() so - * incoming events feed dharma_field() queues. The runtime itself does not - * intercept any /dharma path. */ - -el_val_t dharma_connect(el_val_t cgi_id); -el_val_t dharma_send(el_val_t channel, el_val_t content); -el_val_t dharma_activate(el_val_t query); -void dharma_emit(el_val_t event_type, el_val_t payload); -el_val_t dharma_field(el_val_t event_type); -void dharma_strengthen(el_val_t cgi_id, el_val_t weight); -el_val_t dharma_relationship(el_val_t cgi_id); -el_val_t dharma_peers(void); - -/* Public C API: called by an El program's HTTP handler when a /dharma/event - * request arrives. Pushes onto the per-event-type queue and signals any - * pending dharma_field() blockers. All three arguments must be NUL-terminated - * C strings (or NULL — then treated as empty). */ -void el_runtime_dharma_event_arrive(const char* event_type, - const char* payload, - const char* source); - -/* ── Engram local graph primitives ─────────────────────────────────────────── - * Operate on the CGI's local Engram knowledge graph. - * `engram_activate` queries the local graph only; `dharma_activate` is - * network-wide across all connected CGI graphs. */ - -el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience); -el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t importance, el_val_t confidence, - el_val_t tier, el_val_t tags); -/* Layered consciousness — see el_runtime.c for the layered architecture - * design notes (search "Layered consciousness architecture"). The five - * canonical layers (safety / core-identity / domain-knowledge / imprint / - * suit) are seeded automatically; engram_add_layer extends the registry - * with imprint or suit overlays at runtime. Nodes default to layer 1 - * (core-identity) when created via engram_node / engram_node_full. */ -el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t label, - el_val_t salience, el_val_t certainty, el_val_t confidence, - el_val_t status, el_val_t tags, el_val_t layer_id); -el_val_t engram_add_layer(el_val_t name, el_val_t priority, el_val_t suppressible, - el_val_t transparent, el_val_t injectable); -el_val_t engram_remove_layer(el_val_t layer_id); -el_val_t engram_list_layers(void); -el_val_t engram_get_node(el_val_t id); -void engram_strengthen(el_val_t node_id); -void engram_forget(el_val_t node_id); -el_val_t engram_node_count(void); -el_val_t engram_search(el_val_t query, el_val_t limit); -el_val_t engram_scan_nodes(el_val_t limit, el_val_t offset); -void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t relation); -el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id); -el_val_t engram_neighbors(el_val_t node_id); -el_val_t engram_neighbors_filtered(el_val_t node_id, el_val_t max_depth, el_val_t direction); -el_val_t engram_edge_count(void); -/* Three-pass activation: background fan-out → working-memory promotion → - * Layer 0 override. See "Three-pass activation" in el_runtime.c. */ -el_val_t engram_activate(el_val_t query, el_val_t depth); -el_val_t engram_save(el_val_t path); -el_val_t engram_load(el_val_t path); - -/* JSON-string accessors — return pre-serialized JSON so HTTP handlers - * can pass results straight through without round-tripping ElList/ElMap - * through json_stringify. */ -el_val_t engram_get_node_json(el_val_t id); -el_val_t engram_search_json(el_val_t query, el_val_t limit); -el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset); -el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset); -el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction); -el_val_t engram_activate_json(el_val_t query, el_val_t depth); -el_val_t engram_stats_json(void); -el_val_t engram_list_layers_json(void); -/* engram_compile_layered_json — produce a prompt-ready text block split - * into "[LAYER 0 — STRUCTURAL]" (non-suppressible layers, sacred fire) - * and "[ENGRAM CONTEXT]" (standard suppressible layers). Returns "" if - * no nodes promoted to working memory. */ -el_val_t engram_compile_layered_json(el_val_t intent, el_val_t depth); - -/* ── LLM (Anthropic API client) ───────────────────────────────────────────── - * All functions call https://api.anthropic.com/v1/messages with the API key - * from env ANTHROPIC_API_KEY. Default model when empty: claude-sonnet-4-5. */ - -el_val_t llm_call(el_val_t model, el_val_t prompt); -el_val_t llm_call_system(el_val_t model, el_val_t system_prompt, el_val_t user_prompt); -el_val_t llm_call_agentic(el_val_t model, el_val_t system, el_val_t user, el_val_t tools); -el_val_t llm_vision(el_val_t model, el_val_t system, el_val_t prompt, el_val_t image_url_or_b64); -el_val_t llm_models(void); - -/* Register a tool handler by name. The handler is looked up via dlsym - * (mirroring http_set_handler), so any El `fn (input)` compiles to - * a global C symbol that this function can locate at runtime. - * Handler signature: `el_val_t handler(el_val_t input_json)` — receives - * the tool input as a JSON-string el_val_t and returns a JSON-string - * el_val_t result. Used by llm_call_agentic. */ -void llm_register_tool(el_val_t name, el_val_t handler_fn_name); - -/* ── args() ───────────────────────────────────────────────────────────────── - * Provides access to command-line arguments passed to the program. - * Populated by el_runtime_init_args() before main() runs. */ - -el_val_t args(void); -void el_runtime_init_args(int argc, char** argv); - -/* ── Crypto primitives ───────────────────────────────────────────────────── - * SHA-256, HMAC-SHA-256, and base64 (standard + URL-safe). - * Self-contained — no OpenSSL/libcrypto dependency. The implementations are - * adapted from public-domain reference code (Brad Conte / RFC 4648). - * - * Bytes-returning variants (sha256_bytes, hmac_sha256_bytes) return a string - * value whose contents are raw binary; callers usually feed these into - * base64_encode. Note that el_val_t strings are NUL-terminated by convention, - * so the binary payload may contain embedded NULs — pass it directly into - * base64_encode (which uses an explicit length) rather than treating it as - * a printable C string. - * - * The "base64" variants emit/accept RFC 4648 standard alphabet with padding. - * The "base64url" variants use URL-safe alphabet (`-`/`_`) with no padding, - * as used in JWTs. */ - -el_val_t sha256_hex(el_val_t input); -el_val_t sha256_bytes(el_val_t input); -el_val_t hmac_sha256_hex(el_val_t key, el_val_t message); -el_val_t hmac_sha256_bytes(el_val_t key, el_val_t message); -el_val_t base64_encode(el_val_t input); -el_val_t base64_decode(el_val_t input); -el_val_t base64url_encode(el_val_t input); -el_val_t base64url_decode(el_val_t input); - -/* Length-aware variants (internal — exposed for the rare caller that already - * has a known-length binary buffer and doesn't want to round-trip through - * a NUL-terminated el_val_t string). Sha256_bytes and hmac_sha256_bytes feed - * these implicitly. */ -el_val_t el_sha256_bytes_n(const unsigned char* data, size_t len); -el_val_t el_base64_encode_n(const unsigned char* data, size_t len, int url_safe); - -/* ── Post-quantum primitives (liboqs-backed) ──────────────────────────────── - * All inputs/outputs hex-encoded. Algorithm choices: - * Signature: CRYSTALS-Dilithium-3 (NIST level 3, balanced) - * KEM: CRYSTALS-Kyber-768 (NIST level 3) - * Hash: SHA3-256 (Keccak) (PQ-aware protocols favour SHA3 over SHA2) - * - * If liboqs is not linked (detected via __has_include() at compile - * time), the pq_* entry points return a JSON-shaped error string so callers - * fail loudly rather than silently fall back to classical schemes: - * {"error":"liboqs not linked, post-quantum primitives unavailable"} - * - * The hybrid handshake pairs X25519 with Kyber-768 per NIST PQ guidance and - * CNSA 2.0. Combined shared secret is HKDF-SHA256(x25519_ss || kyber_ss). - * Even if Kyber falls, X25519 holds; if X25519 falls under quantum attack, - * Kyber holds. SHA3-256 also remains usable independent of liboqs (the - * Keccak permutation is PQ-OK as a primitive). */ - -el_val_t pq_keygen_signature(void); -el_val_t pq_sign(el_val_t secret_key_hex, el_val_t message); -el_val_t pq_verify(el_val_t public_key_hex, el_val_t message, el_val_t signature_hex); - -el_val_t pq_kem_keygen(void); -el_val_t pq_kem_encaps(el_val_t public_key_hex); -el_val_t pq_kem_decaps(el_val_t secret_key_hex, el_val_t ciphertext_hex); - -el_val_t pq_hybrid_keygen(void); -el_val_t pq_hybrid_handshake(el_val_t remote_pub_combined); - -el_val_t sha3_256_hex(el_val_t input); - -/* ── AEAD: AES-256-GCM (libcrypto-backed) ─────────────────────────────────── - * Symmetric authenticated encryption used to wrap envelopes after a KEM - * handshake. Caller MUST supply a 32-byte key (64 hex chars) — typically the - * Kyber-768 / hybrid shared_secret, optionally normalized via SHA3-256. - * - * aead_encrypt returns a JSON map {"nonce":"...","ciphertext":"..."} where - * ciphertext is the AES-256-GCM output with the 16-byte auth tag appended. - * Nonce is a fresh 12-byte CSPRNG draw — callers never pick the nonce, which - * structurally rules out the GCM nonce-reuse footgun. - * - * aead_decrypt returns the plaintext String, or "" on any failure (including - * auth-tag mismatch). Callers MUST check for "" before trusting the result. */ -el_val_t aead_encrypt(el_val_t key_hex, el_val_t plaintext); -el_val_t aead_decrypt(el_val_t key_hex, el_val_t nonce_hex, el_val_t ciphertext_hex); - -/* ── Native VM builtin aliases (for compiled El source) ───────────────────── - * These match the El VM's native_* builtins so that El source compiled - * to C can call the same names without modification. */ - -el_val_t native_list_get(el_val_t list, el_val_t index); -el_val_t native_list_len(el_val_t list); -el_val_t native_list_append(el_val_t list, el_val_t elem); -el_val_t native_list_empty(void); -el_val_t native_list_clone(el_val_t list); -el_val_t native_string_chars(el_val_t s); -el_val_t native_int_to_str(el_val_t n); - -/* ── Method-call shorthand aliases ────────────────────────────────────────── - * The El method-call convention `obj.method(args)` compiles to - * `method(obj, args)`. These aliases expose the runtime functions under - * the short names that result from method calls in El source. - * - * Example: `myList.append(x)` → `append(myList, x)` (calls this alias) - * `myList.len()` → `len(myList)` (calls this alias) */ - -el_val_t append(el_val_t list, el_val_t elem); /* el_list_append */ -el_val_t len(el_val_t list); /* el_list_len */ -el_val_t get(el_val_t list, el_val_t index); /* el_list_get */ -el_val_t map_get(el_val_t map, el_val_t key); /* el_map_get */ -el_val_t map_set(el_val_t map, el_val_t key, el_val_t value); /* el_map_set */ - -/* ── OTLP/HTTP Observability ─────────────────────────────────────────────── */ -/* See bottom of el_runtime.c for the implementation. - * Configured by env vars OTLP_ENDPOINT, OTEL_SERVICE_NAME, OTEL_SERVICE_VERSION. - * No-op when OTLP_ENDPOINT is unset. Drop-on-failure semantics. */ -/* ── Subprocess execution ────────────────────────────────────────────────── */ -el_val_t exec_command(el_val_t cmd); /* run shell command, return exit code */ -el_val_t exec_capture(el_val_t cmd); /* run shell command, capture stdout */ -el_val_t exec(el_val_t cmd); /* exec(cmd) → stdout String (30s timeout) */ -el_val_t exec_bg(el_val_t cmd); /* exec_bg(cmd) → PID String (non-blocking) */ - -el_val_t emit_log(el_val_t level, el_val_t msg, el_val_t fields_json); -el_val_t emit_metric(el_val_t name, el_val_t value, el_val_t tags_json); -el_val_t trace_span_start(el_val_t name); -el_val_t trace_span_end(el_val_t span_handle); -el_val_t emit_event(el_val_t name, el_val_t duration_ms); - -#ifdef __cplusplus -} -#endif diff --git a/lang/el-compiler/src/codegen-js.el b/lang/el-compiler/src/codegen-js.el index 7e44b41..f2fe025 100644 --- a/lang/el-compiler/src/codegen-js.el +++ b/lang/el-compiler/src/codegen-js.el @@ -1202,7 +1202,7 @@ fn codegen_js_inner(stmts: [Map], source: String, bundle_mode: Bool js_emit_line(js_strip_es_exports(runtime_content)) js_emit_line("") } else { - js_emit_line("// Runtime: foundation/el/el-compiler/runtime/el_runtime.js") + js_emit_line("// Runtime: foundation/el/runtime/el_runtime.js") js_emit_line("import \"./el_runtime.js\";") } // In module mode: destructure all builtins off globalThis.__el so call diff --git a/lang/el-compiler/src/codegen.el b/lang/el-compiler/src/codegen.el index a20569a..8b72adf 100644 --- a/lang/el-compiler/src/codegen.el +++ b/lang/el-compiler/src/codegen.el @@ -1292,6 +1292,43 @@ fn next_if_id() -> String { native_int_to_str(n) } +// is_void_builtin — true for runtime builtins declared `void` in el_runtime.h. +// User `-> Void` functions are emitted as el_val_t (return 0) so they are safe +// to assign; only these C-level void builtins are not. +fn is_void_builtin(name: String) -> Bool { + if str_eq(name, "println") { return true } + if str_eq(name, "print") { return true } + if str_eq(name, "engram_strengthen") { return true } + if str_eq(name, "engram_forget") { return true } + if str_eq(name, "engram_connect") { return true } + if str_eq(name, "dharma_emit") { return true } + if str_eq(name, "dharma_strengthen") { return true } + if str_eq(name, "llm_register_tool") { return true } + if str_eq(name, "exit_program") { return true } + if str_eq(name, "http_serve") { return true } + if str_eq(name, "http_set_handler") { return true } + if str_eq(name, "http_serve_async") { return true } + if str_eq(name, "el_cgi_init") { return true } + if str_eq(name, "el_retain") { return true } + if str_eq(name, "el_release") { return true } + false +} + +// cg_expr_is_void — true if `val` is a direct call to a void builtin, so the +// if-expression arm must emit it as a bare statement rather than assigning its +// (nonexistent) value to the result var. +fn cg_expr_is_void(val: Map) -> Bool { + let vk: String = val["expr"] + if str_eq(vk, "Call") { + let f = val["func"] + let fk: String = f["expr"] + if str_eq(fk, "Ident") { + return is_void_builtin(f["name"]) + } + } + false +} + // Render a single arm of the if-as-expression: emit each statement-before-last // as a side-effecting expression, then assign the final Expr's value to the // result var. If the arm body is empty or its last stmt isn't an Expr, the @@ -1300,6 +1337,10 @@ fn cg_if_expr_arm(stmts: [Map], result_var: String) -> String { let n: Int = native_list_len(stmts) // Collect statement fragments into a list to avoid O(n-) string growth. let parts: [String] = native_list_empty() + // Track names already declared in this arm's C block. El permits `let x` + // to redeclare/rebind x in the same scope, but C forbids redeclaring the + // same name in one block: emit `el_val_t x = ...` first, `x = ...` after. + let declared: [String] = native_list_empty() let i = 0 while i < n { let s = native_list_get(stmts, i) @@ -1310,18 +1351,31 @@ fn cg_if_expr_arm(stmts: [Map], result_var: String) -> String { let name: String = s["name"] let val = s["value"] let val_c: String = cg_expr(val) - let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ") + if list_contains(declared, name) { + let parts = native_list_append(parts, name + " = " + val_c + "; ") + } else { + let declared = native_list_append(declared, name) + let parts = native_list_append(parts, "el_val_t " + name + " = " + val_c + "; ") + } } else { if str_eq(sk, "Return") { let val = s["value"] let val_c: String = cg_expr(val) - let parts = native_list_append(parts, result_var + " = (" + val_c + "); ") + if cg_expr_is_void(val) { + let parts = native_list_append(parts, val_c + "; ") + } else { + let parts = native_list_append(parts, result_var + " = (" + val_c + "); ") + } } else { if str_eq(sk, "Expr") { let val = s["value"] let val_c: String = cg_expr(val) if is_last { - let parts = native_list_append(parts, result_var + " = (" + val_c + "); ") + if cg_expr_is_void(val) { + let parts = native_list_append(parts, val_c + "; ") + } else { + let parts = native_list_append(parts, result_var + " = (" + val_c + "); ") + } } else { let parts = native_list_append(parts, "(void)(" + val_c + "); ") } @@ -2669,6 +2723,9 @@ fn builtin_arity(name: String) -> Int { if str_eq(name, "engram_activate") { return 2 } if str_eq(name, "engram_save") { return 1 } if str_eq(name, "engram_load") { return 1 } + if str_eq(name, "engram_store_boot") { return 1 } + if str_eq(name, "engram_store_checkpoint") { return 0 } + if str_eq(name, "engram_store_close") { return 0 } if str_eq(name, "engram_get_node_json") { return 1 } if str_eq(name, "engram_get_node_by_label") { return 1 } if str_eq(name, "engram_search_json") { return 2 } diff --git a/lang/el-compiler/src/parser.el b/lang/el-compiler/src/parser.el index d93fcba..a064bad 100644 --- a/lang/el-compiler/src/parser.el +++ b/lang/el-compiler/src/parser.el @@ -49,6 +49,21 @@ fn tok_value(tokens: [Any], pos: Int) -> String { native_list_get(tokens, pos * 2 + 1) } +// parse_progress_fatal — robustness backstop. Called by the token-consuming +// driver loops when they detect they have iterated more times than there are +// tokens (impossible for a well-formed program, where every iteration consumes +// at least one token). Names the offending token and exits non-zero instead of +// looping forever / exhausting memory. +fn parse_progress_fatal(where: String, tokens: [Any], pos: Int) -> Void { + let k: String = tok_kind(tokens, pos) + let v: String = tok_value(tokens, pos) + println("elc: FATAL: parser made no forward progress in " + where + + " at token index " + native_int_to_str(pos) + " (kind=" + k + ")") + println("elc: likely a malformed construct near '" + v + + "' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" ).") + exit(1) +} + fn expect(tokens: [Any], pos: Int, kind: String) -> Int { let k = tok_kind(tokens, pos) if k == kind { @@ -1212,7 +1227,16 @@ fn parse_block(tokens: [Any], pos: Int) -> Map { let p = expect(tokens, pos, "LBrace") let stmts: [Map] = native_list_empty() let running = true + // Runaway backstop: a block can hold at most (token count) statements, since + // every iteration consumes >= 1 token. If we exceed that, the cursor has run + // off the end without terminating (malformed input) -> fail fast, don't hang. + let blk_total: Int = native_list_len(tokens) / 2 + let blk_iters: Int = 0 while running { + let blk_iters = blk_iters + 1 + if blk_iters > blk_total + 8 { + parse_progress_fatal("parse_block", tokens, p) + } let k = tok_kind(tokens, p) if k == "RBrace" { let running = false diff --git a/lang/elb.el b/lang/elb.el index ba77089..ece57d7 100644 --- a/lang/elb.el +++ b/lang/elb.el @@ -368,13 +368,13 @@ fn main() -> Void { let which_out: String = str_trim(exec_capture("which " + elc_bin + " 2>/dev/null")) if !str_eq(which_out, "") { let elc_dir: String = dirname_of(which_out) - runtime_path = elc_dir + "/../el-compiler/runtime/el_runtime.c" + runtime_path = elc_dir + "/../runtime/el_runtime.c" } } // If --runtime points to a directory, auto-locate el_runtime.c inside it. // This lets both forms work: - // --runtime=/opt/el/el-compiler/runtime (directory form) - // --runtime=/opt/el/el-compiler/runtime/el_runtime.c (file form) + // --runtime=/opt/el/runtime (directory form) + // --runtime=/opt/el/runtime/el_runtime.c (file form) if !str_eq(runtime_path, "") { let is_dir: String = str_trim(exec_capture("test -d " + runtime_path + " && echo dir || echo file")) if str_eq(is_dir, "dir") { diff --git a/lang/elc-combined.el b/lang/elc-combined.el index 3216641..295e062 100644 --- a/lang/elc-combined.el +++ b/lang/elc-combined.el @@ -3797,6 +3797,9 @@ fn builtin_arity(name: String) -> Int { if str_eq(name, "engram_activate") { return 2 } if str_eq(name, "engram_save") { return 1 } if str_eq(name, "engram_load") { return 1 } + if str_eq(name, "engram_store_boot") { return 1 } + if str_eq(name, "engram_store_checkpoint") { return 0 } + if str_eq(name, "engram_store_close") { return 0 } if str_eq(name, "engram_get_node_json") { return 1 } if str_eq(name, "engram_search_json") { return 2 } if str_eq(name, "engram_scan_nodes_json") { return 2 } diff --git a/lang/elc.c b/lang/elc.c index 5104d13..865c5ec 100644 --- a/lang/elc.c +++ b/lang/elc.c @@ -1423,15 +1423,53 @@ el_val_t tok_at(el_val_t tokens, el_val_t pos) { } el_val_t tok_kind(el_val_t tokens, el_val_t pos) { + /* Out-of-range reads MUST report the Eof sentinel so every `== "Eof"` + termination guard in the parser fires. Without this, reading past the + trailing Eof token returns runtime null (native_list_get OOB -> 0), which + matches no delimiter, letting inner parse loops (parse_block, parse_binop) + append AST nodes forever on malformed input -> unbounded allocation -> OOM. */ + el_val_t n = (native_list_len(tokens) / 2); + if (pos < 0) { + return EL_STR("Eof"); + } + if (pos >= n) { + return EL_STR("Eof"); + } return native_list_get(tokens, (pos * 2)); return 0; } el_val_t tok_value(el_val_t tokens, el_val_t pos) { + el_val_t n = (native_list_len(tokens) / 2); + if (pos < 0) { + return EL_STR(""); + } + if (pos >= n) { + return EL_STR(""); + } return native_list_get(tokens, ((pos * 2) + 1)); return 0; } +/* parse_progress_fatal — robustness backstop. Called by the token-consuming + driver loops when they detect they have iterated more times than there are + tokens (an impossibility for a well-formed program, where every iteration + consumes at least one token). Names the offending token and exits non-zero + instead of looping forever / exhausting memory. */ +el_val_t parse_progress_fatal(el_val_t where, el_val_t tokens, el_val_t pos) { + el_val_t k = tok_kind(tokens, pos); + el_val_t v = tok_value(tokens, pos); + println(el_str_concat(el_str_concat(el_str_concat(el_str_concat( + EL_STR("elc: FATAL: parser made no forward progress in "), where), + EL_STR(" at token index ")), native_int_to_str(pos)), + el_str_concat(EL_STR(" (kind="), el_str_concat(k, EL_STR(")"))))); + println(el_str_concat(el_str_concat( + EL_STR("elc: likely a malformed construct near '"), v), + EL_STR("' — e.g. an unterminated string or an unescaped double-quote inside a string literal (use \\\" )."))); + exit(1); + return 0; +} + el_val_t expect(el_val_t tokens, el_val_t pos, el_val_t kind) { el_val_t k = tok_kind(tokens, pos); if (str_eq(k, kind)) { @@ -2689,7 +2727,16 @@ el_val_t parse_block(el_val_t tokens, el_val_t pos) { el_val_t p = expect(tokens, pos, EL_STR("LBrace")); el_val_t stmts = native_list_empty(); el_val_t running = 1; + /* Runaway backstop: a block can hold at most (token count) statements, since + every iteration consumes >= 1 token. If we exceed that, the cursor has run + off the end without terminating (malformed input) -> fail fast, don't hang. */ + el_val_t __blk_total = (native_list_len(tokens) / 2); + el_val_t __blk_iters = 0; while (running) { + __blk_iters = (__blk_iters + 1); + if (__blk_iters > (__blk_total + 8)) { + parse_progress_fatal(EL_STR("parse_block"), tokens, p); + } el_val_t k = tok_kind(tokens, p); if (str_eq(k, EL_STR("RBrace"))) { running = 0; @@ -4838,9 +4885,51 @@ el_val_t next_if_id(void) { return 0; } +/* is_void_builtin — true for runtime builtins declared `void` in el_runtime.h. + User `-> Void` functions are emitted as el_val_t (return 0) so they are safe + to assign; only these C-level void builtins are not. */ +el_val_t is_void_builtin(el_val_t name) { + if (str_eq(name, EL_STR("println"))) { return 1; } + if (str_eq(name, EL_STR("print"))) { return 1; } + if (str_eq(name, EL_STR("engram_strengthen"))) { return 1; } + if (str_eq(name, EL_STR("engram_forget"))) { return 1; } + if (str_eq(name, EL_STR("engram_connect"))) { return 1; } + if (str_eq(name, EL_STR("dharma_emit"))) { return 1; } + if (str_eq(name, EL_STR("dharma_strengthen"))) { return 1; } + if (str_eq(name, EL_STR("llm_register_tool"))) { return 1; } + if (str_eq(name, EL_STR("exit_program"))) { return 1; } + if (str_eq(name, EL_STR("http_serve"))) { return 1; } + if (str_eq(name, EL_STR("http_set_handler"))) { return 1; } + if (str_eq(name, EL_STR("http_serve_async"))) { return 1; } + if (str_eq(name, EL_STR("el_cgi_init"))) { return 1; } + if (str_eq(name, EL_STR("el_retain"))) { return 1; } + if (str_eq(name, EL_STR("el_release"))) { return 1; } + return 0; +} + +/* cg_expr_is_void — true if `val` is a direct call to a void builtin, so the + if-expression arm must emit it as a bare statement rather than assigning its + (nonexistent) value to the result var. */ +el_val_t cg_expr_is_void(el_val_t val) { + el_val_t vk = el_get_field(val, EL_STR("expr")); + if (str_eq(vk, EL_STR("Call"))) { + el_val_t f = el_get_field(val, EL_STR("func")); + el_val_t fk = el_get_field(f, EL_STR("expr")); + if (str_eq(fk, EL_STR("Ident"))) { + return is_void_builtin(el_get_field(f, EL_STR("name"))); + } + } + return 0; +} + el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) { el_val_t n = native_list_len(stmts); el_val_t parts = native_list_empty(); + /* Track names already declared in this arm's C block. El permits `let x` + to redeclare/rebind x in the same scope, but C forbids redeclaring the + same name in one block. Emit `el_val_t x = ...` the first time and a + plain `x = ...` reassignment thereafter (mirrors cg_stmt's `declared`). */ + el_val_t declared = native_list_empty(); el_val_t i = 0; while (i < n) { el_val_t s = native_list_get(stmts, i); @@ -4853,18 +4942,31 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) { el_val_t name = el_get_field(s, EL_STR("name")); el_val_t val = el_get_field(s, EL_STR("value")); el_val_t val_c = cg_expr(val); - parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; "))); + if (list_contains(declared, name)) { + parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(name, EL_STR(" = ")), val_c), EL_STR("; "))); + } else { + declared = native_list_append(declared, name); + parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("el_val_t "), name), EL_STR(" = ")), val_c), EL_STR("; "))); + } } else { if (str_eq(sk, EL_STR("Return"))) { el_val_t val = el_get_field(s, EL_STR("value")); el_val_t val_c = cg_expr(val); - parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); "))); + if (cg_expr_is_void(val)) { + parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; "))); + } else { + parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); "))); + } } else { if (str_eq(sk, EL_STR("Expr"))) { el_val_t val = el_get_field(s, EL_STR("value")); el_val_t val_c = cg_expr(val); if (is_last) { + if (cg_expr_is_void(val)) { + parts = native_list_append(parts, el_str_concat(val_c, EL_STR("; "))); + } else { parts = native_list_append(parts, el_str_concat(el_str_concat(el_str_concat(result_var, EL_STR(" = (")), val_c), EL_STR("); "))); + } } else { parts = native_list_append(parts, el_str_concat(el_str_concat(EL_STR("(void)("), val_c), EL_STR("); "))); } @@ -4883,6 +4985,7 @@ el_val_t cg_if_expr_arm(el_val_t stmts, el_val_t result_var) { } el_val_t result = str_join(parts, EL_STR("")); el_release(parts); + el_release(declared); return result; return 0; } diff --git a/lang/examples/html-page.el b/lang/examples/html-page.el index 95019ff..aa1e64b 100644 --- a/lang/examples/html-page.el +++ b/lang/examples/html-page.el @@ -6,8 +6,8 @@ // // Compile and run: // ./dist/platform/elc examples/html-page.el > /tmp/html-page.c -// cc -std=c11 -I el-compiler/runtime -lcurl -lpthread \ -// -o /tmp/html-page /tmp/html-page.c el-compiler/runtime/el_runtime.c +// cc -std=c11 -I runtime -lcurl -lpthread \ +// -o /tmp/html-page /tmp/html-page.c runtime/el_runtime.c // /tmp/html-page fn render_item(item: String) -> String { diff --git a/lang/releases/v1.0.0-20260501/RELEASE.md b/lang/releases/v1.0.0-20260501/RELEASE.md deleted file mode 100644 index 280af6e..0000000 --- a/lang/releases/v1.0.0-20260501/RELEASE.md +++ /dev/null @@ -1,28 +0,0 @@ -# El Compiler Release v1.0.0 — 2026-05-02 - -## Components -- `bootstrap.py` — El language compiler (Python, recursive descent parser, emits C) -- `el_runtime.c` — El runtime (C, HTTP server, engram, DHARMA, LLM chain) -- `el_runtime.h` — Runtime public API header - -## Changes in this release - -### Critical bug fixes -- `state_set`/`state_get` are now thread-safe (pthread_mutex). Was racing across 64 worker threads. -- `looks_like_string` threshold raised from 1,000,000 to 4GB. Unix timestamps were being dereferenced as heap pointers. -- `fs_read` guards against negative `ftell` result (pipe/special file overflow). - -### Engram architecture (major) -- Two-layer activation: `background_activation` (Layer 1, broad fan-out) + `working_memory_weight` (Layer 2, executive filter) -- Inhibitory edges: `EngramEdge.inhibitory` flag suppresses working memory promotion without affecting background activation -- Suppression memory: `suppression_count` — nodes activated-but-suppressed accumulate pressure toward breakthrough -- Temporal decay: `temporal_decay_rate`, `created_at`, `last_activated_at`, `activation_count` on EngramNode -- Per-type activation thresholds (Safety: 0.05, Canonical: 0.15, Lesson: 0.25, Note: 0.40) -- Temporal range query: `engram_query_range(start_ms, end_ms)` -- Layered consciousness: `EngramLayer` struct, `layer_id` on nodes and edges, `EngramStore.layers[]` -- Layer 0 override pass: safety layer fires last and cannot be suppressed - -## SHA256 -bootstrap.py -el_runtime.c -el_runtime.h diff --git a/lang/el-compiler/runtime/ElBridge.java b/lang/runtime/ElBridge.java similarity index 100% rename from lang/el-compiler/runtime/ElBridge.java rename to lang/runtime/ElBridge.java diff --git a/lang/el-compiler/runtime/PLATFORM_BRIDGE_SPEC.md b/lang/runtime/PLATFORM_BRIDGE_SPEC.md similarity index 100% rename from lang/el-compiler/runtime/PLATFORM_BRIDGE_SPEC.md rename to lang/runtime/PLATFORM_BRIDGE_SPEC.md diff --git a/lang/el-compiler/runtime/detect-platforms b/lang/runtime/detect-platforms similarity index 99% rename from lang/el-compiler/runtime/detect-platforms rename to lang/runtime/detect-platforms index bceace2..b47e348 100755 --- a/lang/el-compiler/runtime/detect-platforms +++ b/lang/runtime/detect-platforms @@ -132,7 +132,7 @@ if [[ $LVGL_OK -eq 1 ]]; then else _miss "LVGL/MCU" "-DEL_TARGET_LVGL (lvgl.h not found)" echo " Install: git clone https://github.com/lvgl/lvgl" - echo " (place lvgl/ next to el-compiler/runtime/)" + echo " (place lvgl/ next to runtime/)" MISSING=$((MISSING + 1)) fi diff --git a/lang/el-compiler/runtime/el_android.c b/lang/runtime/el_android.c similarity index 100% rename from lang/el-compiler/runtime/el_android.c rename to lang/runtime/el_android.c diff --git a/lang/el-compiler/runtime/el_appkit.m b/lang/runtime/el_appkit.m similarity index 100% rename from lang/el-compiler/runtime/el_appkit.m rename to lang/runtime/el_appkit.m diff --git a/lang/el-compiler/runtime/el_gtk4.c b/lang/runtime/el_gtk4.c similarity index 100% rename from lang/el-compiler/runtime/el_gtk4.c rename to lang/runtime/el_gtk4.c diff --git a/lang/el-compiler/runtime/el_lvgl.c b/lang/runtime/el_lvgl.c similarity index 100% rename from lang/el-compiler/runtime/el_lvgl.c rename to lang/runtime/el_lvgl.c diff --git a/lang/el-compiler/runtime/el_native_target.h b/lang/runtime/el_native_target.h similarity index 100% rename from lang/el-compiler/runtime/el_native_target.h rename to lang/runtime/el_native_target.h diff --git a/lang/releases/v1.0.0-20260501/el_platform_win.h b/lang/runtime/el_platform_win.h similarity index 98% rename from lang/releases/v1.0.0-20260501/el_platform_win.h rename to lang/runtime/el_platform_win.h index 1042f44..cfc2774 100644 --- a/lang/releases/v1.0.0-20260501/el_platform_win.h +++ b/lang/runtime/el_platform_win.h @@ -85,6 +85,7 @@ static inline void* el_win_dlsym(void* handle, const char* name) { #include /* _mkdir */ #define mkdir(path, mode) _mkdir(path) /* POSIX mkdir(path,mode) → _mkdir(path) */ #define timegm _mkgmtime /* UTC tm → time_t */ +#define fsync(fd) _commit(fd) /* no fsync() on Windows; _commit() () is the equiv */ /* setenv/unsetenv: not in the Windows CRT; map to _putenv_s / SetEnvironmentVariable. */ static inline int setenv(const char* name, const char* value, int overwrite) { diff --git a/lang/releases/v1.0.0-20260501/el_runtime.c b/lang/runtime/el_runtime.c similarity index 74% rename from lang/releases/v1.0.0-20260501/el_runtime.c rename to lang/runtime/el_runtime.c index 4629cc2..7012a3d 100644 --- a/lang/releases/v1.0.0-20260501/el_runtime.c +++ b/lang/runtime/el_runtime.c @@ -809,9 +809,12 @@ static long el_http_timeout_ms(void) { return parsed; } -/* Internal: do a libcurl request; takes optional body/headers, optional method override. */ -static el_val_t http_do(const char* method, const char* url, const char* body, - struct curl_slist* extra_headers) { +/* Internal: do a libcurl request; takes optional body/headers, optional method + * override, and an optional timeout override (0 = use EL_HTTP_TIMEOUT_MS). + * The override exists for the embedding path: activation must never wait the + * full 60s default on a wedged Ollama. (2026-07-24 self-review) */ +static el_val_t http_do_t(const char* method, const char* url, const char* body, + struct curl_slist* extra_headers, long timeout_ms) { if (!url || !*url) return http_error_json("empty url"); CURL* c = curl_easy_init(); if (!c) return http_error_json("curl_easy_init failed"); @@ -821,7 +824,8 @@ static el_val_t http_do(const char* method, const char* url, const char* body, curl_easy_setopt(c, CURLOPT_WRITEFUNCTION, http_write_cb); curl_easy_setopt(c, CURLOPT_WRITEDATA, &rb); curl_easy_setopt(c, CURLOPT_FOLLOWLOCATION, 1L); - curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, el_http_timeout_ms()); + curl_easy_setopt(c, CURLOPT_TIMEOUT_MS, + timeout_ms > 0 ? timeout_ms : el_http_timeout_ms()); curl_easy_setopt(c, CURLOPT_NOSIGNAL, 1L); curl_easy_setopt(c, CURLOPT_ERRORBUFFER, errbuf); curl_easy_setopt(c, CURLOPT_USERAGENT, "el-runtime/1.0"); @@ -832,6 +836,13 @@ static el_val_t http_do(const char* method, const char* url, const char* body, curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)(body ? strlen(body) : 0)); } else if (method && strcmp(method, "DELETE") == 0) { curl_easy_setopt(c, CURLOPT_CUSTOMREQUEST, "DELETE"); + /* DELETE with a body (2026-07-24): the engram server authenticates + * mutating requests via an "_auth" field in the JSON body, so EL + * code must be able to send DELETE + body. Absent body → unchanged. */ + if (body && *body) { + curl_easy_setopt(c, CURLOPT_POSTFIELDS, body); + curl_easy_setopt(c, CURLOPT_POSTFIELDSIZE, (long)strlen(body)); + } } CURLcode rc = curl_easy_perform(c); curl_easy_cleanup(c); @@ -843,6 +854,12 @@ static el_val_t http_do(const char* method, const char* url, const char* body, return el_wrap_str(rb.data); } +/* Legacy entry point: default timeout. */ +static el_val_t http_do(const char* method, const char* url, const char* body, + struct curl_slist* extra_headers) { + return http_do_t(method, url, body, extra_headers, 0); +} + el_val_t http_get(el_val_t url) { return http_do("GET", EL_CSTR(url), NULL, NULL); } @@ -916,6 +933,16 @@ el_val_t http_delete(el_val_t url) { return http_do("DELETE", EL_CSTR(url), NULL, NULL); } +/* DELETE with a JSON body — required by the engram server's body-based + * "_auth" scheme for mutating requests. (2026-07-24 self-review) */ +el_val_t http_delete_json(el_val_t url, el_val_t json_body) { + struct curl_slist* h = NULL; + h = curl_slist_append(h, "Content-Type: application/json"); + el_val_t r = http_do("DELETE", EL_CSTR(url), EL_CSTR(json_body), h); + curl_slist_free_all(h); + return r; +} + /* ── HTTP → file streaming ──────────────────────────────────────────────── * * Why this exists: el_val_t strings are NUL-terminated by convention, so @@ -2996,10 +3023,95 @@ static char* jp_parse_string_raw(JsonParser* jp) { case 'r': c = '\r'; break; case 't': c = '\t'; break; case 'u': { - /* Skip 4 hex digits; emit '?' as a placeholder */ - for (int i = 0; i < 4 && jp->p < jp->end; i++) jp->p++; - c = '?'; - break; + /* Decode \uXXXX (with surrogate pairs) to UTF-8. + * + * (2026-08-08 self-review) This used to skip the 4 hex + * digits and emit a literal '?'. That is a LOSSY, silent, + * irreversible transform on every JSON string entering the + * runtime — and JSON writers escape non-ASCII by default + * (Python json.dumps ships ensure_ascii=True; most MCP + * clients do the same). So every em dash, curly quote, + * accented letter, and emoji arriving over MCP or HTTP was + * replaced by one question mark on the way in, with no + * error and no counter. + * + * Measured on the live store before the fix: 3,119 of + * 4,081 non-telemetry nodes (76%) carried the damage, + * including the self traversal root and all 13 values + * nodes ("Value ? Constraints as Freedom"). Node contents + * split cleanly into fully-clean or fully-mangled with + * zero overlap — the tell that this was one write path, + * not gradual rot. The corruption is unrecoverable in + * place (3 bytes collapse to 1), so the only real fix is + * to stop producing it; historical repair has to come from + * each node's upstream source. + * + * Malformed escapes keep the old '?' behaviour rather than + * failing the parse: a truncated body should not take down + * a route that previously tolerated it. */ + unsigned cp = 0; + int ok = 1; + for (int i = 0; i < 4; i++) { + if (jp->p >= jp->end) { ok = 0; break; } + char h = *jp->p++; + unsigned d; + if (h >= '0' && h <= '9') d = (unsigned)(h - '0'); + else if (h >= 'a' && h <= 'f') d = (unsigned)(h - 'a' + 10); + else if (h >= 'A' && h <= 'F') d = (unsigned)(h - 'A' + 10); + else { ok = 0; break; } + cp = (cp << 4) | d; + } + if (!ok) { c = '?'; break; } + /* High surrogate: pair it with the following low surrogate + * so astral-plane codepoints (emoji) survive. If the next + * token is not a valid low surrogate, rewind so it is + * parsed on its own terms rather than swallowed. */ + if (cp >= 0xD800 && cp <= 0xDBFF && + (size_t)(jp->end - jp->p) >= 6 && + jp->p[0] == '\\' && jp->p[1] == 'u') { + const char* save = jp->p; + unsigned lo = 0; int ok2 = 1; + jp->p += 2; + for (int i = 0; i < 4; i++) { + char h = *jp->p++; + unsigned d; + if (h >= '0' && h <= '9') d = (unsigned)(h - '0'); + else if (h >= 'a' && h <= 'f') d = (unsigned)(h - 'a' + 10); + else if (h >= 'A' && h <= 'F') d = (unsigned)(h - 'A' + 10); + else { ok2 = 0; break; } + lo = (lo << 4) | d; + } + if (ok2 && lo >= 0xDC00 && lo <= 0xDFFF) + cp = 0x10000u + ((cp - 0xD800u) << 10) + (lo - 0xDC00u); + else jp->p = save; + } + /* Lone surrogate → U+FFFD (WHATWG / serde_json behaviour): + * emitting a raw surrogate would produce invalid UTF-8. */ + if (cp >= 0xD800 && cp <= 0xDFFF) cp = 0xFFFD; + + char ub[4]; int un; + if (cp < 0x80) { + ub[0] = (char)cp; un = 1; + } else if (cp < 0x800) { + ub[0] = (char)(0xC0 | (cp >> 6)); + ub[1] = (char)(0x80 | (cp & 0x3F)); un = 2; + } else if (cp < 0x10000) { + ub[0] = (char)(0xE0 | (cp >> 12)); + ub[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); + ub[2] = (char)(0x80 | (cp & 0x3F)); un = 3; + } else { + ub[0] = (char)(0xF0 | (cp >> 18)); + ub[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); + ub[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); + ub[3] = (char)(0x80 | (cp & 0x3F)); un = 4; + } + while (len + (size_t)un >= cap) { + cap *= 2; + out = realloc(out, cap); + if (!out) { fputs("el_runtime: out of memory\n", stderr); exit(1); } + } + for (int i = 0; i < un; i++) out[len++] = ub[i]; + continue; /* bytes already appended */ } default: c = esc; break; } @@ -5698,6 +5810,43 @@ void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, * exceed the floor, so naturally-promoted nodes survive multiple decay cycles. * Invariant maintained: BREAKTHROUGH_WEIGHT < min(type_thresholds). */ #define ENGRAM_BREAKTHROUGH_WEIGHT 0.10 +/* ENGRAM_BREAKTHROUGH_BUDGET / ENGRAM_BREAKTHROUGH_COOLDOWN (2026-08-02 + * self-review): the breakthrough path was an unbounded, self-resetting loop. + * Every reached node failing its type threshold incremented suppression_count; + * on the 5th failure it was force-promoted at exactly 0.10 AND had its counter + * reset to 0 — so it re-entered the identical cycle immediately. Because + * BREAKTHROUGH_WEIGHT (0.10) > WM_FLOOR (0.05), all of them cleared the + * absolute floor and entered the rank contest tied at 0.10, where all but a + * handful were evicted by ENGRAM_WM_CAP. Evicted nodes get no access_ts + * record (Pass 6 skips wm_weights<=0), so the short-term inhibition-of-return + * damper at ENGRAM_STI_TS never applied to them and they were re-suppressed + * completely unmarked. Steady state: N_suppressed/5 breakthroughs per call, + * essentially all of them evicted the same call. + * + * Live telemetry 2026-08-02 (boot 19, uptime 23h48m): breakthroughs_delta + * 661–903 and wm_evicted_delta 485–717 PER 60s heartbeat against wm_active + * pinned at 22–24. At ~4 activate calls/minute that is ~165–225 breakthroughs + * per call — ~825–1125 nodes cycling in 5-call lockstep. Working memory was + * not remembering; it was thrashing, and the churn drowned genuine promotion. + * + * Two bounds, both required: + * BUDGET — at most WM_CAP/4 (6) intrusive thoughts may surface per call. + * Breakthrough is meant to be an occasional intrusive thought, + * not a stampede; it can never again exceed a quarter of WM. + * COOLDOWN — on breakthrough, suppression_count is set NEGATIVE rather than + * 0, so the node needs COOLDOWN+SUPPRESSION_BREAKTHROUGH further + * suppressions before it may surface again (~60 calls ≈ 15 min at + * the current cadence, vs 5 calls ≈ 75s before). This is + * inhibition-of-return applied to the breakthrough path, matching + * the STI damper already applied to the natural path. Stored in + * the existing int32_t field — serialized as %d and parsed via + * eg_get_int_field, so negative values round-trip through + * snapshots without a struct or format change. + * When the budget or cooldown blocks a breakthrough, suppression_count is NOT + * reset — it saturates, so a starved node surfaces on a later call rather than + * restarting its climb. */ +#define ENGRAM_BREAKTHROUGH_BUDGET (ENGRAM_WM_CAP / 4) +#define ENGRAM_BREAKTHROUGH_COOLDOWN 55 /* ENGRAM_WM_CAP: hard limit on concurrent working-memory nodes (2026-06-30 * self-review, porting fix from self-review 2026-06-26 branch). Without this, * broad curiosity seeds like "knowledge" promote 500+ nodes simultaneously — @@ -5707,6 +5856,22 @@ void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, * context while preventing flooding. Enforced in Pass 4 (per-call) and Pass 5 * (global across prior-promoted nodes). */ #define ENGRAM_WM_CAP 24 +/* ENGRAM_WM_FLOOR: absolute admission floor for a working-memory slot + * (2026-07-30 self-review). Before this, Pass 4/Pass 5/load-cap only ever + * trimmed the WM population down TO the cap and never below it — rank-based + * eviction guarantees the cap is filled whenever ≥24 nodes hold any nonzero + * weight, so wm_active was pinned at 24/24 and wm_saturated:1 was + * definitionally true on every heartbeat (carried no information). Soar's WM + * forgetting (Derbinsky & Laird, ICCM 2012) removes elements by comparing + * activation to an ABSOLUTE threshold θ, independent of how many other + * elements exist — fill below capacity is a reachable, meaningful state + * ("low cognitive load"). This floor is the weight-domain analogue of that θ: + * any slot whose weight sinks below it is dropped even when WM is under cap. + * Value 0.05 = the lowest per-type promotion threshold (Safety/DharmaSelf in + * engram_type_threshold) and the existing boot-time launder floor, and sits + * below ENGRAM_BREAKTHROUGH_WEIGHT (0.10) so intrusive-thought breakthroughs + * still surface. Applied: Pass 4, carry-over, Pass 5, load-cap. */ +#define ENGRAM_WM_FLOOR 0.05 #define ENGRAM_INHIBITION_FACTOR 0.1 /* ── ACT-R / Petrov hybrid base-level learning (2026-07-22 self-review) ────── @@ -5745,6 +5910,46 @@ void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal, #define ENGRAM_BLL_TAU (-3.0) #define ENGRAM_BLL_S 0.4 +/* Short-term inhibition-of-return (2026-07-25 self-review). + * Lebiere & Best 2009 ("Balancing Long-Term Reinforcement and Short-Term + * Inhibition", CogSci): subtract ln(1 + (t_n/t_s)^-d_s) from activation, + * where t_n = time since the MOST RECENT access. With their best-fit + * d_s = 1.0 the exp of the subtraction reduces to the clean multiplier + * m(t_n) = t_n / (t_n + t_s) + * applied to raw_wm in Pass 2. Immediately after a promotion the node's + * score is crushed (m → 0), then self-heals as a power law — producing an + * emergent round-robin over WM candidates instead of winner-take-all. + * This replaces the non-decaying suppression_count as the damping + * mechanism (that counter never entered the score at all — it only ever + * pushed nodes TOWARD surfacing via breakthrough; kept for that role). + * t_s = 4× the ~30 s curiosity-scan interval per the paper's guidance + * (t_s ≈ peak re-occurrence lag). A node promoted every scan holds + * m ≈ 0.2 until it loses its slot; after ~2 min unretrieved, m ≥ 0.5. + * Source: act-r.psy.cmu.edu/.../894Cogsci09-Lebiere-Best.pdf */ +#define ENGRAM_STI_TS 120.0 + +/* Carry-over occupancy inhibition (2026-07-26 self-review). + * The STI multiplier above only runs in the REACHED branch of Pass 2. + * A node carried over WITHOUT being reached (Pass 4½ below) kept + * w = wm_anchor * keep, and for a node whose base-level was inflated + * during the pre-07-25 unconditional-reinforcement era, keep ≈ 1.0 for + * days — the anchor weight is re-emitted verbatim forever. Observed + * live: wm_top0_streak = 1407 heartbeats (~23 h), one node frozen at + * its anchor 0.589 while every reached candidate rotated at the 0.10 + * breakthrough floor. Deterministic argmax over a quasi-static score + * fixates regardless of any inhibition applied only to the reached set + * (Morita et al. 2021, citing Lebiere & Best 2009). + * Fix: key inhibition on OCCUPANCY, not retrieval recency — multiply + * the carried weight by the same d_s=1 closed form over hold time + * m(t_h) = t_c / (t_c + t_h), t_h = seconds since last_activated + * (carry-over nodes are deliberately never reinforced, so + * last_activated marks when the node last EARNED its slot). Power-law + * self-healing: a node re-reached by any future activation is + * re-scored fresh in Pass 2 and re-anchors. t_c = 3600 s → a carried + * node keeps ~92% after 5 min, 50% after 1 h, ~4% after 23 h. Unlike + * m(t_n), this cannot saturate to no-op while the node camps. */ +#define ENGRAM_CARRY_TC 3600.0 + /* qsort comparator — descending double, used by WM cap enforcement. */ static int engram_cmp_double_desc(const void* a, const void* b) { double da = *(const double*)a; @@ -5893,6 +6098,27 @@ typedef struct EngramNode { int32_t access_head; int32_t access_filled; double wm_anchor; + /* Semantic embedding (2026-07-24 self-review, bl-b2d1c944). + * emb: malloc'd float vector (nomic-embed-text, 768-dim) or NULL. + * emb_dim: vector length; 0 = not embedded. Populated lazily by the + * backfill pass in engram_activate — NOT on the node-create hot path, + * so bulk sync imports never block on Ollama. Freed in engram_forget / + * engram_prune_telemetry; shift-copies move the pointer intact. */ + float* emb; + int32_t emb_dim; + /* Hebbian eligibility trace (2026-08-06 self-review, ENGRAM_HEBB_TRACE_*). + * hebb_elig: trace amplitude, set to 1.0 whenever the node holds a WM slot. + * hebb_elig_ts: wall-clock ms at which that happened; the trace is read as + * hebb_elig·exp(−Δt/TC) rather than stored decayed, so it is a pure + * function of time and independent of how often activate is called. + * + * DELIBERATELY NOT SERIALIZED. A trace is a behavioral-timescale quantity + * (minutes); persisting it across a restart would resurrect coincidences + * from an arbitrarily distant past as if they had just happened. Zero via + * the calloc at store init and the memset in engram_grow_nodes, so a cold + * boot simply has no eligible pairs until WM starts turning over. */ + double hebb_elig; + int64_t hebb_elig_ts; } EngramNode; /* Record an access (ACT-R "presentation") into the base-level ring buffer. */ @@ -5950,6 +6176,620 @@ static void engram_bll_parse_access(EngramNode* nn, const char* s) { } } +/* ── Embedding-based activation (2026-07-24 self-review, bl-b2d1c944) ────── + * Closes the gap named in every heartbeat since 2026-06-30: "no embedding + * call is made during activation. The seed-finding loop uses istr_contains + * only." Design per the 2026-07-21 integration brief: + * - Lazy backfill: engram_activate embeds up to BACKFILL_PER_CALL + * un-embedded, non-telemetry nodes per call, newest first. No latency + * on node-create paths; a fresh node is embedded within ~1 scan cycle. + * - Semantic seeding (HippoRAG pattern, use similarity twice): the query + * is embedded, the top-K nodes by cosine ≥ SEED_MIN join the seed set + * with initial activation = the similarity itself. + * - Additive WM term: raw_wm += W * relu((cos − S0)/(1 − S0)). Shift-and- + * floor at S0 because nomic-embed scores unrelated pairs 0.4–0.5; raw + * cosine in a weighted sum is a constant bias, not a signal. + * - Circuit breaker: 3 consecutive Ollama failures → stop trying for 5 + * minutes. Activation NEVER blocks on a dead embedder beyond timeout. + */ +#define ENGRAM_EMBED_S0 0.45 +#define ENGRAM_EMBED_WM_WEIGHT 0.20 +#define ENGRAM_EMBED_SEED_K 8 +#define ENGRAM_EMBED_SEED_MIN 0.60 +#define ENGRAM_EMBED_BACKFILL_PER_CALL 8 +/* ENGRAM_QGATE_FLOOR: minimum propagation multiplier for an EMBEDDED target + * node with zero/negative query similarity. Query-aware spreading gate + * (arXiv:2606.30133) adapted for partial embedding coverage — see the + * propagation loop in engram_activate. 0.25 damps semantically unrelated + * branches ~4x without severing them. Unembedded targets are ungated. */ +#define ENGRAM_QGATE_FLOOR 0.25 +#define ENGRAM_EMBED_MAX_CHARS 2000 +#define ENGRAM_EMBED_TIMEOUT_MS 4000L +#define ENGRAM_EMBED_BREAKER_LIMIT 3 +#define ENGRAM_EMBED_BREAKER_COOLDOWN_MS 300000 + +/* ── Context centroid (2026-07-29 self-review) ─────────────────────────────── + * Closes the last gap from the 2026-07-21 decay/embedding brief: cosine was + * computed against the per-call query embedding ONLY, so every curiosity scan + * was semantically memoryless — the 4 rotating seed phrases fully determined + * what ignited, with zero continuity from what the system actually touched. + * + * Mechanism (brief spec): a running EMA centroid over touch embeddings, + * c ← normalize(μ·c + (1−μ)·e_touch), μ = ENGRAM_CTX_MU + * where a "touch" is (a) the query embedding each activate call and (b) the + * embeddings of up to ENGRAM_CTX_TOUCH_MAX top WM-promoted survivors of that + * call — promotion is the retrieval event (same rule as BLL reinforcement). + * + * Feedback-loop guard: scoring does NOT use the raw centroid. The lit + * failure mode of centroid memories (EMA of your own outputs → runaway + * attractor; cf. the wm_top0_streak=1407 freeze this store already hit) is + * bounded by scoring against a query-dominant blend: + * e_eff = normalize(α·e_q + (1−α)·c), α = ENGRAM_CTX_QALPHA + * so the exogenous rotating seeds always contribute the majority of the + * scoring direction; the centroid is a context tint, not the signal. + * + * Observability: _eg_act_ctx_cos = cos(e_q, c) BEFORE the query is blended + * in. ~1.0 → centroid aligned with current query; low → context and query + * have diverged (expected at domain-rotation boundaries); -2.0 → no centroid + * yet / embedder down. Exposed via engram_act_stats_json → heartbeat ISE so + * drift is diagnosable from telemetry. In-memory only: context is + * short-term by definition, a restart legitimately starts cold. */ +#define ENGRAM_CTX_MU 0.90 +#define ENGRAM_CTX_QALPHA 0.65 +#define ENGRAM_CTX_TOUCH_MAX 8 + +/* ── Hebbian co-activation potentiation (2026-08-04 self-review) ───────────── + * THE GAP: every learning mechanism in this runtime operated on NODES — + * salience, base-level learning (ACT-R), activation_count, temporal decay, + * WM promotion. Edge weights were written once at engram_connect() and never + * changed again. `last_fired` was declared on EngramEdge, persisted, and + * emitted in JSON, but the ONLY writer in the entire 12.5k-line runtime was + * dharma_strengthen() — an unrelated CGI-relationship path. Activation read + * e->weight and never wrote it. So the graph's TOPOLOGY was frozen: an edge + * authored at the default 0.5 that proved itself useful on ten thousand + * consecutive retrievals stayed at exactly 0.5, indistinguishable from one + * that had never carried a useful signal. The nodes learned; the wiring + * between them did not. "The system must get smarter over time" was true of + * memories and false of the associations among them. + * + * THE RULE (HeLa-Mem, arXiv:2604.16839): w ← λ·w + η·1[both nodes co-retrieved]. + * + * ADAPTATION 1 — learned strength is a SEPARATE field, not a mutation of the + * authored weight. HeLa-Mem updates the association weight in place because + * all of its edges ARE learned associations. Most edges here are authored + * structure: `contains` from the self root, `identity`, `supersedes`, `tagged`. + * Decaying those would erode identity silently and irreversibly — precisely + * the failure the immutable-engram principle exists to prevent. So `weight` + * stays exactly as authored (audit trail intact, behavior exactly restorable + * by ignoring the field) and `hebb` accumulates alongside it. Propagation uses + * eg_edge_eff_weight() = weight × (1 + GAIN·hebb), clamped to 1.0. A cold + * graph has hebb == 0 everywhere and therefore behaves bit-identically to the + * pre-change runtime — this change cannot regress a fresh deploy. + * + * ADAPTATION 2 — η is set to exactly (1 − λ), which turns the update into an + * exponentially-weighted moving average. hebb then converges to a quantity + * with a plain reading: THE FRACTION OF RECENT ACTIVATION CALLS IN WHICH BOTH + * ENDPOINTS WERE SIMULTANEOUSLY IN WORKING MEMORY. Not an arbitrary strength + * unit — a probability. That makes the homeostatic budget below interpretable + * in the same units, and makes the telemetry readable without a decoder ring. + * + * ADAPTATION 3 — homeostatic scaling, which HeLa-Mem does not have. Their + * ablation shows removing adaptive forgetting costs almost nothing (34.74 → + * 34.28 F1) because their benchmark runs ~300 turns; this store has 41k edges + * and runs continuously for weeks. Pure potentiation lets a high-degree hub + * accumulate strength on ALL its edges at once and become a superhighway that + * relays activation everywhere — the exact hub-flooding pathology the + * query-aware propagation gate was added to fix in the 2026-07-27 review. + * The consolidation literature is unanimous that potentiation requires a + * compensating normalization (surviving connections are collectively scaled + * down to hold firing-rate homeostasis — PNAS 2422602122, two-factor synaptic + * consolidation). So: per node, the summed hebb across incident edges is + * capped at ENGRAM_HEBB_NODE_BUDGET and scaled down proportionally when + * exceeded. A node can hold ~4 strong associations, or many weak ones, but + * not unbounded total associative mass. Potentiation is competitive, not free. + * + * TIMESCALE: decay is per activation CALL, not per wall-clock second. That is + * deliberate — associative strength should track cognitive events, not the + * clock, so an idle daemon does not forget what it learned while working. At + * the current curiosity-scan rate (~2 calls / 30 s) the 0.9999 factor gives a + * half-life of ~6,900 calls ≈ 1.2 days: associations form over hours and fade + * over days of genuine disuse. */ +#define ENGRAM_HEBB_DECAY 0.9999 +#define ENGRAM_HEBB_ETA 0.0001 /* == 1 - DECAY ⇒ hebb is an EWMA */ +#define ENGRAM_HEBB_GAIN 0.5 /* max +50% effective propagation */ +#define ENGRAM_HEBB_NODE_BUDGET 4.0 /* homeostatic cap on per-node Σ hebb */ +/* Snap-to-zero floor. MUST stay far below ENGRAM_HEBB_ETA. Set to 0.001 + * initially — above the 0.0001 per-step increment — and live telemetry caught + * it immediately: hebb_cand_max pinned at exactly 0.0001 across 50 calls while + * 15 pairs co-activated every single time. A pair claimed a slot at ETA, the + * next call's decay pass saw 0.0001 < 0.001 and cleared it, and the same pair + * re-claimed at ETA forever. Nothing could ever cross a threshold 1,500 steps + * away when it was being reset every step — the rule was structurally + * incapable of learning anything, for edges as well as candidates. At 1e-6 an + * association touched once survives ~8 days of pure disuse before cleanup, + * which is what a decay floor is actually for. The lesson is the one this + * system keeps relearning: a mechanism that is not instrumented is a mechanism + * you are guessing about. */ +#define ENGRAM_HEBB_MIN 1e-6 + +/* ── Associative link FORMATION (2026-08-04 self-review, same session) ─────── + * MEASURED, NOT ASSUMED: after wiring the potentiation rule above I drove 30 + * activations and found zero potentiated edges. Instrumenting the reason gave + * the finding that actually matters: + * + * edges between working-memory members: 0 + * + * Working memory is populated by semantic seeding (cosine top-K) and by + * multi-hop spreading. Both routinely land on nodes that are semantically + * close and structurally distant. So the pairs that fire together are, in this + * graph, almost never already wired together — and a rule that only reweights + * EXISTING edges is a no-op. HeLa-Mem does not hit this because it maintains a + * dense association matrix over a small memory set; this is a sparse 41k-edge + * graph in which EVERY edge was authored by an explicit tool call. Nothing in + * the runtime has ever created an associative edge from experience. The system + * could strengthen what it was told; it could not notice anything on its own. + * + * So take Hebb literally rather than as HeLa-Mem specializes him: cells that + * fire together WIRE together — if the wire is absent, grow it. + * + * Consolidation is deliberately slow and heavily bounded, because unlike a + * weight tweak this permanently mutates the persisted graph: + * - A pair must sustain co-activation as an EWMA past ENGRAM_HEBB_LINK_MIN + * (~15% of recent calls ≈ 1,100 co-activations ≈ 5 h of continuous + * association) before any edge is created. One-off coincidences never + * consolidate; that is the whole point of the threshold. + * - At most ENGRAM_HEBB_LINK_PER_CALL edges are born per activation. + * - Candidates live in a fixed 8,192-slot table, in memory only. A restart + * discards them, which is a feature, not a limitation: only associations + * sustained across one continuous run earn permanence, and the table can + * never grow without bound. + * - New edges carry relation "hebbian-associate" and start at a deliberately + * weak ENGRAM_HEBB_LINK_W0, so they must keep proving themselves through + * the potentiation rule to gain any real influence. They are tagged + * precisely so every self-formed association stays auditable and the whole + * set is removable with one query if this turns out to be wrong. */ +/* ── Eligibility traces: why co-activation had to stop meaning "same call" ──── + * (2026-08-06 self-review. Measured, not theorized.) + * + * CENSUS. Across the live graph — 41,213 edges, 13,091 nodes, 23h44m uptime, + * boot 23 — the strongest Hebbian association in the entire store measured + * hebb = 0.000799, and `hebbian-associate` edges formed since the mechanism + * shipped: ZERO. 0.000799 / ENGRAM_HEBB_ETA ≈ 8. The best pair of nodes in the + * graph had co-activated eight times, net of decay, ever. ENGRAM_HEBB_LINK_MIN + * is 0.15 — 1875× further away. The effective propagation bonus the mechanism + * was delivering was GAIN·hebb = 0.5 × 0.0008 = +0.04%. + * + * Since the awareness loop calls engram_connect nowhere, Hebbian consolidation + * is the ONLY path by which this system grows its own structure. It was inert. + * Every edge in the graph was authored or imported; none was learned. + * + * WHY RAISING ETA IS THE WRONG FIX. hebb is an EWMA: h ← DECAY·h + ETA·[event], + * with ETA = 1 − DECAY. Its fixed point is P(event) — the learning rate sets how + * fast it converges there and has NO effect on where it converges. If the + * measured ceiling is 0.0008, then P(event) ≈ 0.0008, and ETA could be raised a + * thousandfold without moving the plateau a single decimal. The defect is not + * the rate. It is the event. + * + * WHAT THE EVENT WAS. "Both endpoints hold a WM slot in the same activate call." + * With ENGRAM_WM_CAP = 24 against a 13k-node store, and measured turnover of + * ~142 evictions per 60s heartbeat, two related nodes essentially never occupy + * the same 24 twice. The rule demanded exact simultaneity from a working memory + * engineered — by ENGRAM_STI_TS inhibition-of-return, by breakthrough rotation, + * by the global cap — to never let anything sit still. The three mechanisms + * that make WM healthy are precisely the ones that made this measurement empty. + * + * THE FIX (three-factor / TD(λ) eligibility traces; Sutton & Barto ch. 7, + * Gerstner et al. 2018 on neoHebbian eligibility, PLOS Comp Biol 2018 on + * differential Hebbian learning with leaky-integrator traces). Coincidence + * detection at behavioral timescales does not require simultaneity. A synapse + * that fires sets a decaying flag; potentiation occurs if the partner fires + * while the flag is still up. So: every node entering WM sets a trace to 1.0, + * the trace decays exponentially in WALL-CLOCK time, and the potentiation + * increment becomes ETA · trace(a) · trace(b) instead of ETA · [both in WM now]. + * + * This is a strict generalization, which is what makes it safe to ship: when + * both endpoints are in WM at this instant, both traces read exactly 1.0 and + * the increment is exactly ETA — bit-identical to the previous rule. The change + * is purely additive on near-coincidences the old rule discarded outright. + * + * TC = 300s. Chosen against measured cadence, not taste: curiosity scans run + * every ~31s, so 300s spans ~10 activation cycles (one cycle back reads 0.90, + * three back 0.73, ten back 0.36). Long enough to bridge WM rotation; short + * enough that two nodes surfacing an hour apart (exp(−12) ≈ 6e-6) are correctly + * treated as unrelated. It sits deliberately between ENGRAM_STI_TS (120s, the + * rotation it must survive) and ENGRAM_CARRY_TC (3600s, conversational span). + * + * TRACE_MIN snaps sub-threshold traces to zero so the warm set stays small and + * a node cannot linger as a faint associate of everything for hours. + * + * The homeostatic ENGRAM_HEBB_NODE_BUDGET cap is what keeps this from running + * away: more pairs now potentiate per call, and per-node associative mass is + * still hard-bounded and scaled down proportionally. The safety net predates + * this change and is exactly why loosening the event is not reckless. */ +#define ENGRAM_HEBB_TRACE_TC 300.0 +#define ENGRAM_HEBB_TRACE_MIN 0.05 +/* Bound on non-WM warm nodes considered for candidate pairing in one call. + * At 24 slots × ~10 cycles per trace window the warm set tops out near 240, but + * most are re-promoted incumbents already in WM; 96 is non-binding in practice + * and caps the pairing loop at 24×96 slot probes. */ +#define ENGRAM_HEBB_WARM_MAX 96 +#define ENGRAM_HEBB_CAND_SLOTS 8192 +#define ENGRAM_HEBB_LINK_MIN 0.15 +#define ENGRAM_HEBB_LINK_PER_CALL 2 +#define ENGRAM_HEBB_LINK_W0 0.15 +/* Hard ceiling on self-formed edges as a fraction of the authored graph. + * ENGRAM_HEBB_LINK_PER_CALL alone bounds the RATE (≤2/call) but not the TOTAL: + * at the production scan rate that ceiling is ~11k edges/day, which would + * swamp a 41k-edge graph inside a week in the worst case. Potentiation decays, + * but a link whose hebb has decayed back to zero still persists as a weak + * edge — there is currently no pruning path, so growth is one-way. Until there + * is one, self-formed structure is capped at 5% of the store: enough room to + * learn real associations, not enough to drown what was authored. */ +#define ENGRAM_HEBB_LINK_MAX_FRAC 0.05 + +/* ── Systems consolidation: the durable write-back queue (2026-08-07) ──────── + * + * THE DEFECT THIS CLOSES. Yesterday's eligibility-trace fix made Hebbian + * learning numerically real: hebb_max went 0.000799 → 0.4725 and 1,198 + * `hebbian-associate` edges formed in 23h48m. Today's census found where they + * went: nowhere. Measured on the live system — + * + * soul daemon (pid 3269, in-process graph): 42,426 edges, 1,198 hebbian + * engram server (:8742, the persistent store): 41,213 edges, 49 hebbian + * + * Two processes, two graphs, one direction of travel. The soul pulls from the + * server every 10 min via GET /api/sync and merges. It never pushes. And it + * cannot fall back on saving its own copy: `soul_snapshot_path` is set only + * inside `if is_genesis && safe_to_seed` in soul.el, and safe_to_seed is + * unconditionally false whenever ENGRAM_URL is set (it is, in the launchd + * plist) — because the HTTP server owns persistence and a soul that wrote + * snapshot.json would clobber it. That guard is correct. The consequence was + * not: mem_save() has never once executed, so every association the soul + * learns lives in RAM until the process dies. + * + * The soul is the ONLY process that runs idle cognition — curiosity scans + * every ~8 min, around the clock. It is where essentially all co-activation + * happens. So the system's entire capacity to grow its own structure was + * pointed at a volatile store. 1,198 associations/day, discarded at restart, + * every day, silently. The mechanism worked and the learning still evaporated. + * + * WHY A QUEUE AND NOT A SAVE. The fix is not to let the soul write the + * snapshot — that reintroduces the clobber the guard exists to prevent. It is + * to make consolidation a MESSAGE, not a file: the fast volatile store hands + * each newly-formed association to the slow durable store, one edge at a time, + * over the API the server already exposes (POST /api/edges). This is the + * hippocampal→neocortical split the rest of this file is already modeled on. + * Fast store learns online and forgets; slow store receives what survived the + * threshold and keeps it. Only edges that already cleared ENGRAM_HEBB_LINK_MIN + * are enqueued, so what crosses the process boundary is what earned it. + * + * SHAPE. Fixed 512-slot ring, overwrite-oldest. 512 is ~18x the observed + * formation rate per drain interval (14 links per 8-min heartbeat), so the + * queue only saturates when the writer is down — and when it is, keeping the + * freshest associations is the right loss. Drops are counted, not silent: + * a consolidation path that quietly discards is the failure mode this whole + * entry exists to correct. Enqueue is strdup'd because g->edges may realloc + * and node ids may be freed by later pruning; the queue owns its copies. */ +#define ENGRAM_HEBB_WB_SLOTS 512 + +typedef struct { char* a; char* b; double w; double hebb; } EgHebbWB; +static EgHebbWB _eg_hebb_wb[ENGRAM_HEBB_WB_SLOTS]; +static int _eg_hebb_wb_head = 0; /* index of the oldest live entry */ +static int _eg_hebb_wb_len = 0; +static int64_t _eg_hebb_wb_dropped = 0; /* lost to a full queue, cumulative */ +static int64_t _eg_hebb_wb_drained = 0; /* handed to the durable store, cum. */ + +static void eg_hebb_wb_push(const char* a, const char* b, double w, double h) { + if (!a || !b) return; + int slot; + if (_eg_hebb_wb_len >= ENGRAM_HEBB_WB_SLOTS) { + slot = _eg_hebb_wb_head; + free(_eg_hebb_wb[slot].a); + free(_eg_hebb_wb[slot].b); + _eg_hebb_wb_head = (_eg_hebb_wb_head + 1) % ENGRAM_HEBB_WB_SLOTS; + _eg_hebb_wb_dropped++; + } else { + slot = (_eg_hebb_wb_head + _eg_hebb_wb_len) % ENGRAM_HEBB_WB_SLOTS; + _eg_hebb_wb_len++; + } + _eg_hebb_wb[slot].a = strdup(a); + _eg_hebb_wb[slot].b = strdup(b); + _eg_hebb_wb[slot].w = w; + _eg_hebb_wb[slot].hebb = h; + /* strdup failure leaves a NULL id; the drain skips those rather than + * emitting a malformed edge. */ +} + +typedef struct { char* a; char* b; double score; } EgHebbCand; +static EgHebbCand _eg_hebb_cand[ENGRAM_HEBB_CAND_SLOTS]; +static int64_t _eg_hebb_links_formed = 0; +/* Observability for the eligibility-trace rule (2026-08-06). hebb_warm is the + * size of the last call's warm set — nodes eligible but not co-resident, i.e. + * exactly the population the old simultaneity rule discarded. If this reads 0 + * forever the traces are not arming and the change bought nothing; if it reads + * healthy while hebb_max stays flat, the bottleneck is somewhere else. That + * distinction is the whole reason yesterday's diagnosis took a census instead + * of a guess. */ +static int _eg_act_hebb_warm = 0; + +static float* _eg_ctx_c = NULL; +static int32_t _eg_ctx_dim = 0; +static double _eg_act_ctx_cos = -2.0; + +static int _eg_embed_consec_fail = 0; +static int64_t _eg_embed_breaker_until = 0; + +/* ── Activation observability counters (2026-07-27 self-review) ────────────── + * The executive-filter pathologies this system has repeatedly debugged by + * inference (WM flooded with breakthrough-floor nodes, silent cap evictions, + * embedder wedged behind the breaker) were all invisible: no ISE, no stats + * field. engram_act_stats_json() exposes them so the soul heartbeat can emit + * them. + * + * CUMULATIVE CONTRACT (2026-07-31 self-review): these are monotonic totals + * for the process lifetime, like pulse/sync_added_total — NOT per-call. The + * original per-call reset meant engram_act_stats_json only reported the LAST + * activate call, and the 60s heartbeat (2 curiosity activates per 30s in + * between) missed nearly every eviction/breakthrough event. Consumers wanting + * rates keep the previous reading and diff. Restart legitimately resets to 0. */ +static int64_t _eg_act_breakthroughs = 0; /* forced promotions at the floor, cumulative */ +static int64_t _eg_act_wm_evicted = 0; /* ALL WM evictions, cumulative (see below) */ +/* Redundancy suppression counters (2026-08-05 self-review) — see + * ENGRAM_DEDUP_COS. dup_seeds = semantic seed slots reclaimed from redundant + * copies; dup_wm = WM candidates dropped for duplicating a higher-ranked + * candidate's content. Both cumulative for the process lifetime. */ +static int64_t _eg_act_dup_seeds = 0; +static int64_t _eg_act_dup_wm = 0; +/* Redundant WM residents evicted by the GLOBAL pass (2026-08-06). Counted + * separately from _eg_act_dup_wm on purpose: dup_wm measures duplicates caught + * among this call's candidates, dup_wm_global measures duplicates that reached + * the persisted WM population through the carry-over path, which is the leak + * Pass 3½ structurally could not see. Merging them would hide whether the new + * pass is doing anything. */ +static int64_t _eg_act_dup_wm_global = 0; +/* 2026-08-02 self-review: this counted only the three Pass 4 / Pass 5 floor + * and rank paths. The two carry-over eviction paths (base-level below τ, and + * decayed weight below WM_FLOOR) were silent, so the reported eviction rate + * was an undercount of unknown magnitude — which mattered precisely while + * diagnosing the breakthrough storm. All five paths now increment. */ + +static int64_t engram_now_ms(void); /* defined in the store section below */ + +static const char* eg_embed_url(void) { + const char* s = getenv("EL_EMBED_URL"); + return (s && *s) ? s : "http://localhost:11434/api/embeddings"; +} +static const char* eg_embed_model(void) { + const char* s = getenv("EL_EMBED_MODEL"); + return (s && *s) ? s : "nomic-embed-text"; +} + +/* Cosine similarity over raw (unnormalized — nomic emits magnitudes >1) + * float vectors. Returns -2.0 on dim mismatch / null input so callers can + * distinguish "orthogonal" (0.0) from "not comparable". */ +static double eg_cosine(const float* a, const float* b, int32_t dim) { + if (!a || !b || dim <= 0) return -2.0; + double dot = 0.0, na = 0.0, nb = 0.0; + for (int32_t i = 0; i < dim; i++) { + dot += (double)a[i] * (double)b[i]; + na += (double)a[i] * (double)a[i]; + nb += (double)b[i] * (double)b[i]; + } + if (na <= 0.0 || nb <= 0.0) return -2.0; + return dot / (sqrt(na) * sqrt(nb)); +} + +/* ── Redundancy suppression (2026-08-05 self-review) ───────────────────────── + * MEASUREMENT, not intuition. Content-hash census of the live snapshot + * (13,216 nodes / 41,213 edges) on 2026-08-05: + * + * non-ISE nodes 4,138 + * duplicate content groups 1,489 + * REDUNDANT copies 1,858 (44.9% of the non-ISE graph) + * redundant copies embedded 1,856 + * + * The copies were all created in a single June 2026 import (1,854 of 1,858; + * 2 in July) — an id-scheme migration re-added nodes under fresh UUIDs + * instead of matching on content. Generation has stopped. The copies have + * not: they are byte-identical, so they carry IDENTICAL embeddings, and + * therefore identical cosine to any query. + * + * That is the damage. Semantic seeding takes the top-K by cosine + * (ENGRAM_EMBED_SEED_K = 8, ≥ ENGRAM_EMBED_SEED_MIN). A document with six + * copies does not compete for one of those eight slots — it takes six. + * Measured over 50 real query probes against the live 3,998-vector set: + * + * seed slots filled 400 + * slots consumed by redundant copies 161 (40.2%) + * probes affected 46/50 (92%) + * effective DISTINCT seeds per scan 4.78 of 8 + * + * Two fifths of every retrieval was spent re-reading the same page. The + * same collapse hits WM: duplicates score identically, so they promote + * together and hold multiple of the 24 slots for one thought. + * + * The fix is not to delete nodes — data repair is a separate, reversible + * operation with its own backup discipline. The fix is that redundancy must + * never buy a scarce slot, whatever the graph's state. Enforced at both + * scarcity points: semantic seed selection, and WM admission (Pass 3½). + * + * Identity test, cheapest first: + * 1. type|content FNV-1a — exact; catches the import duplicates + * 2. cosine ≥ ENGRAM_DEDUP_COS — catches copies differing only in + * whitespace/punctuation, which hash differently but embed identically + * 0.995 is deliberately severe: at 768 dimensions this admits only + * near-verbatim text. Distinct-but-related nodes (the associative structure + * this system exists to traverse) sit far below it and are untouched. + * This suppresses REDUNDANCY, never similarity. */ +#define ENGRAM_DEDUP_COS 0.995 + +/* eg_content_key — FNV-1a over node_type|content. Identical prose under a + * different node_type is not a duplicate (a Knowledge note and the + * BacklogItem quoting it are different objects), so the type is folded in. */ +static uint64_t eg_content_key(const EngramNode* n) { + uint64_t h = 14695981039346656037ULL; + const char* s = n->node_type; + if (s) while (*s) { h ^= (unsigned char)*s++; h *= 1099511628211ULL; } + h ^= (unsigned char)'|'; h *= 1099511628211ULL; + s = n->content; + if (s) while (*s) { h ^= (unsigned char)*s++; h *= 1099511628211ULL; } + return h; +} + +/* eg_same_content — is node `a` redundant with node `b`? + * Hash equality first (one pass, no allocation); embedding near-identity as + * the fallback for copies that differ only in insignificant characters. + * Deliberately does NOT strcmp on hash collision: a 64-bit FNV-1a collision + * across ~4k candidates is ~1e-13, and the cost of being wrong is one node + * losing one slot on one call — not corruption. */ +static int eg_same_content(const EngramNode* a, const EngramNode* b, + uint64_t ka, uint64_t kb) { + if (ka == kb) return 1; + if (a->emb && b->emb && a->emb_dim > 0 && a->emb_dim == b->emb_dim) { + if (eg_cosine(a->emb, b->emb, a->emb_dim) >= ENGRAM_DEDUP_COS) return 1; + } + return 0; +} + +/* Weight-carrying index for the Pass 3½ descending walk. */ +typedef struct { double w; int64_t idx; } EgDupCand; +static int eg_dupcand_cmp_desc(const void* a, const void* b) { + double wa = ((const EgDupCand*)a)->w, wb = ((const EgDupCand*)b)->w; + if (wa < wb) return 1; + if (wa > wb) return -1; + return 0; +} + +/* eg_ctx_blend — fold one touch embedding into the context centroid: + * c ← normalize(μ·c + (1−μ)·e). Initializes the centroid (normalized copy) + * on first touch or dim change; silently skips degenerate vectors. */ +static void eg_ctx_blend(const float* e, int32_t dim) { + if (!e || dim <= 0) return; + double ne = 0.0; + for (int32_t i = 0; i < dim; i++) ne += (double)e[i] * (double)e[i]; + if (ne <= 0.0) return; + ne = sqrt(ne); + if (!_eg_ctx_c || _eg_ctx_dim != dim) { + float* c = malloc((size_t)dim * sizeof(float)); + if (!c) return; + for (int32_t i = 0; i < dim; i++) c[i] = (float)((double)e[i] / ne); + free(_eg_ctx_c); + _eg_ctx_c = c; + _eg_ctx_dim = dim; + return; + } + double nc = 0.0; + for (int32_t i = 0; i < dim; i++) { + double v = ENGRAM_CTX_MU * (double)_eg_ctx_c[i] + + (1.0 - ENGRAM_CTX_MU) * ((double)e[i] / ne); + _eg_ctx_c[i] = (float)v; + nc += v * v; + } + if (nc > 0.0) { + nc = sqrt(nc); + for (int32_t i = 0; i < dim; i++) + _eg_ctx_c[i] = (float)((double)_eg_ctx_c[i] / nc); + } +} + +/* Fetch an embedding from Ollama. Returns malloc'd float[dim] or NULL. + * Truncates input to ENGRAM_EMBED_MAX_CHARS and JSON-escapes it. Honors the + * circuit breaker; a NULL return is always safe to ignore (fail-soft). */ +static float* eg_embed_fetch(const char* text, int32_t* out_dim) { + *out_dim = 0; + if (!text || !*text) return NULL; + int64_t now = engram_now_ms(); + if (now < _eg_embed_breaker_until) return NULL; + /* Build request body with escaped, truncated prompt. */ + size_t tlen = strlen(text); + if (tlen > ENGRAM_EMBED_MAX_CHARS) tlen = ENGRAM_EMBED_MAX_CHARS; + char* esc = malloc(tlen * 6 + 1); + if (!esc) return NULL; + size_t w = 0; + for (size_t i = 0; i < tlen; i++) { + unsigned char c = (unsigned char)text[i]; + if (c == '"' || c == '\\') { esc[w++] = '\\'; esc[w++] = (char)c; } + else if (c == '\n') { esc[w++] = '\\'; esc[w++] = 'n'; } + else if (c == '\r') { esc[w++] = '\\'; esc[w++] = 'r'; } + else if (c == '\t') { esc[w++] = '\\'; esc[w++] = 't'; } + else if (c < 0x20) { w += (size_t)snprintf(esc + w, 7, "\\u%04x", c); } + else esc[w++] = (char)c; + } + esc[w] = '\0'; + size_t blen = w + strlen(eg_embed_model()) + 64; + char* body = malloc(blen); + if (!body) { free(esc); return NULL; } + snprintf(body, blen, "{\"model\":\"%s\",\"prompt\":\"%s\"}", + eg_embed_model(), esc); + free(esc); + struct curl_slist* h = curl_slist_append(NULL, "Content-Type: application/json"); + el_val_t resp = http_do_t("POST", eg_embed_url(), body, h, + ENGRAM_EMBED_TIMEOUT_MS); + curl_slist_free_all(h); + free(body); + const char* r = EL_CSTR(resp); + const char* arr = r ? strstr(r, "\"embedding\"") : NULL; + if (!arr) { + if (++_eg_embed_consec_fail >= ENGRAM_EMBED_BREAKER_LIMIT) { + _eg_embed_breaker_until = now + ENGRAM_EMBED_BREAKER_COOLDOWN_MS; + _eg_embed_consec_fail = 0; + } + return NULL; + } + arr = strchr(arr, '['); + if (!arr) return NULL; + arr++; + int32_t cap = 1024, dim = 0; + float* v = malloc((size_t)cap * sizeof(float)); + if (!v) return NULL; + const char* p = arr; + while (*p && *p != ']') { + char* endp = NULL; + double d = strtod(p, &endp); + if (endp == p) break; + if (dim >= cap) { break; } /* >1024 dims: refuse, model mismatch */ + v[dim++] = (float)d; + p = endp; + while (*p == ',' || *p == ' ' || *p == '\n') p++; + } + if (dim < 8) { free(v); return NULL; } /* junk response */ + _eg_embed_consec_fail = 0; + *out_dim = dim; + return v; +} + +/* Node types that never receive embeddings: pure telemetry and structural + * plumbing. Everything else (Knowledge, Memory, BacklogItem, Entity, ...) + * is eligible. */ +static int eg_embed_eligible(const EngramNode* n) { + if (!n->content || strlen(n->content) < 8) return 0; + if (!n->node_type) return 1; + if (strcmp(n->node_type, "InternalStateEvent") == 0) return 0; + if (strcmp(n->node_type, "Tag") == 0) return 0; + return 1; +} + +/* Parse a persisted comma-separated float list into node->emb. */ +static void eg_parse_emb(EngramNode* nn, const char* s) { + if (!s || !*s) return; + int32_t cap = 1024, dim = 0; + float* v = malloc((size_t)cap * sizeof(float)); + if (!v) return; + const char* p = s; + while (*p) { + char* endp = NULL; + double d = strtod(p, &endp); + if (endp == p) break; + if (dim >= cap) break; + v[dim++] = (float)d; + p = endp; + while (*p == ',' || *p == ' ') p++; + } + if (dim < 8) { free(v); return; } + nn->emb = v; + nn->emb_dim = dim; +} + typedef struct EngramEdge { char* id; char* from_id; @@ -5957,6 +6797,11 @@ typedef struct EngramEdge { char* relation; char* metadata; double weight; + /* Hebbian co-activation potentiation, learned at runtime and persisted. + * Strictly separate from `weight`, which is authored and never mutated by + * activation. Reads as "fraction of recent activation calls in which both + * endpoints were in working memory together". See ENGRAM_HEBB_DECAY. */ + double hebb; double confidence; int64_t created_at; int64_t updated_at; @@ -6483,12 +7328,15 @@ static el_val_t engram_node_to_map(const EngramNode* n) { * promotion anchor so heartbeat ISEs / API consumers can see decay state. */ m = el_map_set(m, EL_STR(el_strdup("wm_anchor")), el_from_float(n->wm_anchor)); m = el_map_set(m, EL_STR(el_strdup("base_level")), el_from_float(engram_bll_base_level(n, engram_now_ms()))); + /* emb_dim only — the vector itself is too large for map/API output. + * 0 = not yet embedded by the lazy backfill. (2026-07-24) */ + m = el_map_set(m, EL_STR(el_strdup("emb_dim")), (el_val_t)(int64_t)n->emb_dim); return m; } /* (Node JSON serialization is provided by `engram_emit_node_json` further * down in the persistence section — reused by the *_json builtins below.) */ -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n); +static void engram_emit_node_json(JsonBuf* b, const EngramNode* n, int include_emb); static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e); /* Salience may arrive either as a float bit-pattern or as a small integer @@ -6511,6 +7359,256 @@ static char* engram_first_n_chars(const char* s, size_t n) { return out; } +/* ══════════════════════════════════════════════════════════════════════════ + * M3 — ENGRAM_STORE glue (CALLER side of the libengram ABI; design §10). + * + * The engine (engram_store.{c,h}) has ZERO soul dependencies and never sees an + * EngramNode/EngramEdge or a soul global. ALL mapping between the live runtime + * structs and the engine's StoreNode/StoreEdge views lives HERE, on the caller + * side of the C ABI. That is what keeps a standalone `engramd` a later additive + * choice rather than a fork. + * + * Behind the ENGRAM_STORE env flag (default OFF): + * OFF (unset / "0" / "off") — every hook below early-returns; the paged store + * is never opened or written and no store code is reached. The runtime keeps + * EXACTLY today's JSON-snapshot behavior, byte-for-byte. + * ON ("1" / "on" / "true") — engram_store_boot() imports snapshot.json ONCE + * into neuron.egm (or replays neuron.wal), loads the WHOLE store resident in + * RAM (Phase 1: no demand paging — that is M4), and every structural + * mutation (node/edge create, forget) is mirrored through the store's + * WAL-logged API so neuron.egm/neuron.wal stay authoritative. + * ══════════════════════════════════════════════════════════════════════════ */ +#include "engram_store.h" + +static EngramPagedStore* g_engram_store = NULL; + +int engram_store_enabled(void) { + const char* f = getenv("ENGRAM_STORE"); + return (f && (strcmp(f, "1") == 0 || strcmp(f, "on") == 0 || + strcmp(f, "true") == 0)) ? 1 : 0; +} + +/* EngramNode → borrowed StoreNode view (no ownership transfer; the store copies + * every field it persists, so shared string pointers are safe). */ +static void eg_node_to_store(const EngramNode* n, StoreNode* sn) { + memset(sn, 0, sizeof *sn); + sn->id = n->id; sn->content = n->content; sn->node_type = n->node_type; + sn->label = n->label; sn->tier = n->tier; sn->tags = n->tags; + sn->metadata = n->metadata; + sn->salience = n->salience; sn->importance = n->importance; + sn->confidence = n->confidence; sn->temporal_decay_rate = n->temporal_decay_rate; + sn->activation_count = n->activation_count; sn->last_activated = n->last_activated; + sn->created_at = n->created_at; sn->updated_at = n->updated_at; + sn->background_activation = n->background_activation; + sn->working_memory_weight = n->working_memory_weight; + sn->suppression_count = n->suppression_count; sn->layer_id = n->layer_id; + for (int i = 0; i < STORE_BLL_K && i < ENGRAM_BLL_K; i++) + sn->access_ts[i] = n->access_ts[i]; + sn->access_head = n->access_head; sn->access_filled = n->access_filled; + sn->wm_anchor = n->wm_anchor; sn->emb = n->emb; sn->emb_dim = n->emb_dim; +} +static void eg_edge_to_store(const EngramEdge* e, StoreEdge* se) { + memset(se, 0, sizeof *se); + se->id = e->id; se->from_id = e->from_id; se->to_id = e->to_id; + se->relation = e->relation; se->metadata = e->metadata; + se->weight = e->weight; se->hebb = e->hebb; se->confidence = e->confidence; + se->created_at = e->created_at; se->updated_at = e->updated_at; + se->last_fired = e->last_fired; se->inhibitory = e->inhibitory; + se->layer_id = e->layer_id; +} + +/* Structural-mutation hooks. Callers guard with `if (engram_store_enabled())`; + * these also null-check g_engram_store so a mutation before boot is a safe no-op. */ +static void eg_store_put_node(const EngramNode* n) { + if (!g_engram_store || !n || !n->id) return; + StoreNode sn; eg_node_to_store(n, &sn); + store_put_node(g_engram_store, &sn); +} +static void eg_store_put_edge(const EngramEdge* e) { + if (!g_engram_store || !e || !e->id) return; + StoreEdge se; eg_edge_to_store(e, &se); + store_put_edge(g_engram_store, &se); +} + +/* Resident-load callbacks: StoreNode/StoreEdge → a fresh EngramNode/EngramEdge + * appended to the in-RAM graph. Mirrors engram_load's field set. The boot-time + * WM laundering (halve + floor + global cap) that engram_load applies is done + * once, after all nodes are loaded, in engram_store_boot — see the block there — + * so the store-on boot behaves byte-identically to the JSON path (M3.5 parity). */ +static void eg_load_node_cb(const StoreNode* sn, void* ctx) { + EngramStore* g = (EngramStore*)ctx; + engram_grow_nodes(); + EngramNode* n = &g->nodes[g->node_count]; + memset(n, 0, sizeof *n); + n->id = el_strdup_persist(sn->id ? sn->id : ""); + n->content = el_strdup_persist(sn->content ? sn->content : ""); + n->node_type = el_strdup_persist(sn->node_type && *sn->node_type ? sn->node_type : "Memory"); + n->label = el_strdup_persist(sn->label ? sn->label : ""); + n->tier = el_strdup_persist(sn->tier && *sn->tier ? sn->tier : "Working"); + n->tags = el_strdup_persist(sn->tags ? sn->tags : ""); + n->metadata = el_strdup_persist(sn->metadata && *sn->metadata ? sn->metadata : "{}"); + n->salience = sn->salience; n->importance = sn->importance; + n->confidence = sn->confidence; n->temporal_decay_rate = sn->temporal_decay_rate; + n->activation_count = sn->activation_count; n->last_activated = sn->last_activated; + n->created_at = sn->created_at; n->updated_at = sn->updated_at; + n->background_activation = sn->background_activation; + n->working_memory_weight = sn->working_memory_weight; + n->suppression_count = sn->suppression_count; n->layer_id = sn->layer_id; + for (int i = 0; i < STORE_BLL_K && i < ENGRAM_BLL_K; i++) + n->access_ts[i] = sn->access_ts[i]; + n->access_head = sn->access_head; n->access_filled = sn->access_filled; + n->wm_anchor = sn->wm_anchor; + if (sn->emb && sn->emb_dim > 0) { + n->emb = malloc(sizeof(float) * (size_t)sn->emb_dim); + if (n->emb) { memcpy(n->emb, sn->emb, sizeof(float) * (size_t)sn->emb_dim); + n->emb_dim = sn->emb_dim; } + } + int64_t idx = g->node_count; g->node_count++; + if (n->id && *n->id) engram_idmap_put(g, n->id, idx); +} +static void eg_load_edge_cb(const StoreEdge* se, void* ctx) { + EngramStore* g = (EngramStore*)ctx; + engram_grow_edges(); + EngramEdge* e = &g->edges[g->edge_count]; + memset(e, 0, sizeof *e); + e->id = el_strdup_persist(se->id ? se->id : ""); + e->from_id = el_strdup_persist(se->from_id ? se->from_id : ""); + e->to_id = el_strdup_persist(se->to_id ? se->to_id : ""); + e->relation = el_strdup_persist(se->relation && *se->relation ? se->relation : "associate"); + e->metadata = el_strdup_persist(se->metadata && *se->metadata ? se->metadata : "{}"); + e->weight = se->weight; e->hebb = se->hebb; e->confidence = se->confidence; + e->created_at = se->created_at; e->updated_at = se->updated_at; + e->last_fired = se->last_fired; e->inhibitory = se->inhibitory; + e->layer_id = se->layer_id; + g->edge_count++; +} +static void eg_load_layer_cb(EngramStore* g, const StoreLayer* L) { + if (!L->name) return; + for (size_t i = 0; i < g->layer_count; i++) /* upsert by id */ + if (g->layers[i].layer_id == L->layer_id) return; /* canonical already seeded */ + if (g->layer_count >= g->layer_capacity) { + size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; + EngramLayer* nl = realloc(g->layers, nc * sizeof(EngramLayer)); + if (!nl) return; + g->layers = nl; g->layer_capacity = nc; + } + g->layers[g->layer_count++] = (EngramLayer){ + .layer_id = L->layer_id, + .name = el_strdup_persist(L->name), + .activation_priority = L->activation_priority, + .suppressible = L->suppressible, + .transparent = L->transparent, + .injectable = L->injectable + }; +} + +/* Clear the resident graph so the store becomes the sole source of truth on boot + * (mirrors engram_load's reset). */ +static void eg_reset_resident(EngramStore* g) { + for (int64_t i = 0; i < g->node_count; i++) { + free(g->nodes[i].id); free(g->nodes[i].content); free(g->nodes[i].node_type); + free(g->nodes[i].label); free(g->nodes[i].tier); free(g->nodes[i].tags); + free(g->nodes[i].metadata); + free(g->nodes[i].emb); g->nodes[i].emb = NULL; g->nodes[i].emb_dim = 0; + } + g->node_count = 0; + for (int64_t i = 0; i < g->edge_count; i++) { + free(g->edges[i].id); free(g->edges[i].from_id); free(g->edges[i].to_id); + free(g->edges[i].relation); free(g->edges[i].metadata); + } + g->edge_count = 0; + engram_idmap_free(g); + engram_adj_free(g); +} + +/* Defined later with engram_load; declared here for the boot-time WM laundering + * that keeps the store-on boot byte-identical to the JSON path (M3.5 parity). */ +static void eg_enforce_wm_cap_on_load(EngramStore* g); + +/* engram_store_boot(data_dir) — open (import-once or WAL-replay) the durable + * paged store and load it whole into RAM (Phase 1). No-op / returns 0 when the + * flag is off. Returns 1 on success. Idempotent (a second call is a no-op). */ +el_val_t engram_store_boot(el_val_t data_dir) { + if (!engram_store_enabled()) return (el_val_t)0; + if (g_engram_store) return (el_val_t)1; + const char* d = EL_CSTR(data_dir); + if (!d || !*d) return (el_val_t)0; + g_engram_store = engram_open(d); + if (!g_engram_store) return (el_val_t)0; + EngramStore* g = engram_get(); + eg_reset_resident(g); + store_scan_nodes(g_engram_store, eg_load_node_cb, g); + store_scan_edges(g_engram_store, eg_load_edge_cb, g); + StoreLayer* ls = NULL; size_t ln = 0; + if (store_list_layers(g_engram_store, &ls, &ln) == 0) { + for (size_t i = 0; i < ln; i++) eg_load_layer_cb(g, &ls[i]); + store_layers_free(ls, ln); + } + /* Boot-time WM laundering — MUST match engram_load exactly (M3.5 parity). + * The JSON load path halves every persisted working_memory_weight on boot + * (stale pinned weights decay out over successive restarts; genuine WM state + * keeps continuity) and floors sub-ENGRAM_WM_FLOOR residue to zero, then + * enforces the global WM cap. The store now persists post-activation WM + * weights, so the store-on boot must apply the identical transform or the two + * persistence paths would diverge on the very first restart. */ + for (int64_t i = 0; i < g->node_count; i++) { + g->nodes[i].working_memory_weight *= 0.5; + if (g->nodes[i].working_memory_weight < ENGRAM_WM_FLOOR) + g->nodes[i].working_memory_weight = 0.0; + } + eg_enforce_wm_cap_on_load(g); + g->adj_dirty = 1; + return (el_val_t)1; +} + +/* engram_store_checkpoint() — M3.5 PRE-FLIP GATE. + * + * Spreading activation mutates fields IN PLACE on the resident graph — edge + * `hebb`/`last_fired` (potentiation + homeostatic scaling), node + * `activation_count`/`last_activated`/`working_memory_weight`/`wm_anchor` + * (reinforcement + WM caps) — and also FORMS brand-new `hebbian-associate` + * edges. M3 only mirrored node/edge *creates* and *forgets*; none of those + * in-place mutations or activation-formed edges reached the paged store, so + * learned associations were lost on every restart. This is the fix that makes + * "learned edges survive a restart" true, and it gates the live cutover. + * + * On the soul's save/checkpoint path we push the resident graph's current field + * state through the store's WAL-logged API, then checkpoint: + * - store_put_node(n) for every node → persists WM weight, activation_count, + * last_activated, wm_anchor, the base-level access ring, etc. + * - store_put_edge(e) for every edge → persists hebb + last_fired AND creates + * any activation-formed edges. (store_hebb_batch is deliberately NOT used: it + * is a delta-only op that skips edge ids not already resident in the store — + * see apply_hebb_batch — so it cannot persist the newly-formed hebbian edges + * that are the whole point of this milestone. store_put_edge is the idempotent + * upsert that subsumes the hebb delta.) + * then engram_checkpoint flushes dirty pages + advances the checkpoint LSN. Every + * push is a WAL record, so a graceful restart restores them. + * + * Approach: a FULL WALK of the resident graph (not a dirty-set). At Phase-1 + * ~64 MB resident this is a cheap linear pass on an already-in-RAM array, run at + * the soul's save cadence (seconds-to-minutes), and it is trivially complete — + * no mutation site can be missed and no separate new-edge tracking is needed. An + * inter-checkpoint crash loses only the most-recent unsaved learning, exactly the + * same durability envelope as today's JSON-snapshot cadence. + * + * The store→JSON export path stays engram_save (the JSON is an export artifact). */ +el_val_t engram_store_checkpoint(void) { + if (!engram_store_enabled() || !g_engram_store) return (el_val_t)0; + EngramStore* g = engram_get(); + for (int64_t i = 0; i < g->node_count; i++) eg_store_put_node(&g->nodes[i]); + for (int64_t i = 0; i < g->edge_count; i++) eg_store_put_edge(&g->edges[i]); + return (el_val_t)(int64_t)(engram_checkpoint(g_engram_store) == 0 ? 1 : 0); +} + +/* engram_store_close() — checkpoint + close (used at shutdown / by tests). */ +el_val_t engram_store_close(void) { + if (!g_engram_store) return (el_val_t)0; + int r = engram_close(g_engram_store); + g_engram_store = NULL; + return (el_val_t)(int64_t)(r == 0 ? 1 : 0); +} + el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience) { EngramStore* g = engram_get(); engram_grow_nodes(); @@ -6540,9 +7638,95 @@ el_val_t engram_node(el_val_t content, el_val_t node_type, el_val_t salience) { g->node_count++; engram_idmap_put(g, n->id, new_idx); g->adj_dirty = 1; + if (engram_store_enabled()) eg_store_put_node(n); return el_wrap_str(el_strdup(n->id)); } +/* ── Text-integrity instrumentation (2026-08-08 self-review) ─────────────── + * + * WHY THIS EXISTS. The JSON parser silently replaced every \uXXXX escape with + * '?' for at least two months (see jp_parse_string_raw). 3,119 of 4,081 + * non-telemetry nodes — 76%, including the self traversal root and all 13 + * values nodes — were damaged before anything noticed, and nothing noticed + * because nothing measured. Working memory, Hebbian potentiation, embedding + * coverage, sync age, and the breaker were all instrumented to four decimal + * places; the actual TEXT was not instrumented at all. Every gauge answered + * "is the machinery running" and none answered "is what it carries intact." + * + * The damage is unrecoverable in place (a 3-byte codepoint collapses to one + * byte), and no snapshot on disk predates it, so this cannot be undone. What + * it can be is *impossible to repeat quietly*. Two numbers, split by the + * question each answers: + * + * stock — engram_text_health_json(), a full O(total bytes) census. Too + * expensive for the 60s heartbeat, exactly right for the daily + * self-review. Answers "how much damage is in the store". + * flow — _eg_txt_write_damaged, incremented per damaged node at creation. + * O(len) on a path that already copies the string, so it is free. + * Rides the heartbeat. Answers "is a write path damaging things + * RIGHT NOW" — which is the regression question, and the one that + * would have caught this in a day instead of two months. + * + * SIGNATURE. Conservative on purpose — a false alarm that cries corruption + * over ordinary punctuation is worse than useless. Two patterns, both of + * which are essentially absent from well-formed English prose: + * (a) alnum '?' alnum — "na?ve", "caf?s", "don?t". A real question mark + * never sits between two word characters. + * (b) ' ? ' followed by a lowercase letter — a lost em/en dash. A real + * question mark is not preceded by a space, and + * what follows one starts a new sentence. + * Deliberately NOT flagged: a trailing '?' after a word, '? ' before a + * capital, or '?' at end of string — all legitimate. This under-counts (it + * cannot see a mangled 'café ' where the '?' landed before a space), so the + * census is a floor on the damage, never an exaggeration of it. */ +static int eg_text_loss_signature(const char* s) { + if (!s) return 0; + for (const char* p = s; *p; p++) { + if (*p != '?') continue; + unsigned char prev = (p == s) ? 0 : (unsigned char)p[-1]; + unsigned char next = (unsigned char)p[1]; + /* (a) sandwiched between word characters. */ + if (isalnum(prev) && isalnum(next)) return 1; + /* (b) spaced, with lowercase continuation — a lost dash. */ + if (prev == ' ' && next == ' ' && islower((unsigned char)p[2])) return 1; + } + return 0; +} + +/* Damaged-node creations since process start. See the block comment above. */ +static int64_t _eg_txt_write_damaged = 0; + +/* engram_text_health_json — full text-integrity census over the store. + * O(total content bytes); call it on demand (daily self-review / a route), + * never per heartbeat. `damaged` counts nodes carrying the loss signature, + * `multibyte` counts nodes holding valid multi-byte UTF-8 — the two together + * separate "no damage" from "no non-ASCII text to damage", which a single + * number cannot do. Telemetry is excluded: ISE payloads are machine-written + * ASCII JSON and would dilute the ratio that matters. */ +el_val_t engram_text_health_json(void) { + EngramStore* g = engram_get(); + int64_t scanned = 0, damaged = 0, multibyte = 0; + for (int64_t i = 0; i < g->node_count; i++) { + EngramNode* n = &g->nodes[i]; + if (n->node_type && + (strcmp(n->node_type, "InternalStateEvent") == 0 || + strcmp(n->node_type, "Tag") == 0)) continue; + scanned++; + if (eg_text_loss_signature(n->content)) damaged++; + for (const char* p = n->content; p && *p; p++) { + if ((unsigned char)*p >= 0x80) { multibyte++; break; } + } + } + char buf[256]; + snprintf(buf, sizeof(buf), + "{\"scanned\":%lld,\"damaged\":%lld,\"multibyte\":%lld," + "\"damaged_pct\":%.2f,\"write_damaged\":%lld}", + (long long)scanned, (long long)damaged, (long long)multibyte, + scanned > 0 ? (100.0 * (double)damaged / (double)scanned) : 0.0, + (long long)_eg_txt_write_damaged); + return el_wrap_str(el_strdup(buf)); +} + el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, el_val_t salience, el_val_t importance, el_val_t confidence, el_val_t tier, el_val_t tags) { @@ -6558,6 +7742,11 @@ el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, const char* tg = EL_CSTR(tags); n->content = el_strdup_persist(c ? c : ""); n->node_type = el_strdup_persist(nt && *nt ? nt : "Memory"); + /* Flow half of the text-integrity gauge — see eg_text_loss_signature. + * Telemetry is machine-written ASCII and is excluded so the counter stays + * a clean signal about content-bearing write paths. */ + if (strcmp(n->node_type, "InternalStateEvent") != 0 && + eg_text_loss_signature(n->content)) _eg_txt_write_damaged++; n->label = el_strdup_persist(lb && *lb ? lb : (c ? engram_first_n_chars(c, 60) : "")); n->tier = el_strdup_persist(ti && *ti ? ti : "Working"); n->tags = el_strdup_persist(tg ? tg : ""); @@ -6579,6 +7768,7 @@ el_val_t engram_node_full(el_val_t content, el_val_t node_type, el_val_t label, g->node_count++; engram_idmap_put(g, n->id, new_idx_full); g->adj_dirty = 1; + if (engram_store_enabled()) eg_store_put_node(n); return el_wrap_str(el_strdup(n->id)); } @@ -6649,6 +7839,7 @@ el_val_t engram_node_layered(el_val_t content, el_val_t node_type, el_val_t labe g->node_count++; engram_idmap_put(g, n->id, new_idx_layered); g->adj_dirty = 1; + if (engram_store_enabled()) eg_store_put_node(n); return el_wrap_str(el_strdup(n->id)); } @@ -6778,8 +7969,16 @@ void engram_strengthen(el_val_t node_id) { n->activation_count++; n->last_activated = engram_now_ms(); n->updated_at = n->last_activated; - /* Explicit strengthen is a presentation too (2026-07-22 self-review). */ - engram_bll_record_access(n, n->last_activated); + /* 2026-07-26 self-review: REMOVED the BLL access record added on + * 2026-07-22 ("explicit strengthen is a presentation too"). The + * 2026-07-25 STI multiplier reads the same access ring — so an + * explicit strengthen crushed the strengthened node's promotion + * score by ×t_n/(t_n+120) for the next ~2 minutes. The awareness + * loop strengthens exactly when a node NEWLY reaches WM top + * (novelty gating); recording an access here made that + * reinforcement self-defeating. Salience/activation_count bumps + * above carry the reinforcement; the access ring stays reserved + * for genuine retrieval events (promotions in engram_activate). */ } void engram_forget(el_val_t node_id) { @@ -6788,10 +7987,17 @@ void engram_forget(el_val_t node_id) { EngramStore* g = engram_get(); int64_t idx = engram_find_node_index(sid); if (idx < 0) return; + /* Mirror the removal into the durable store BEFORE the shift-delete frees + * the incident edges' ids (node tombstone; edge records are reclaimed at + * compaction — M5). Node-level FORGET matches the existing WAL semantics. */ + if (engram_store_enabled() && g_engram_store) { + store_forget(g_engram_store, sid); + } /* Free node strings */ EngramNode* n = &g->nodes[idx]; free(n->id); free(n->content); free(n->node_type); free(n->label); free(n->tier); free(n->tags); free(n->metadata); + free(n->emb); /* Shift remaining nodes down */ for (int64_t i = idx + 1; i < g->node_count; i++) { g->nodes[i - 1] = g->nodes[i]; @@ -6877,6 +8083,7 @@ el_val_t engram_prune_telemetry(el_val_t older_than_ms) { removed_ids[removed++] = n->id; /* keep id for edge sweep */ free(n->content); free(n->node_type); free(n->label); free(n->tier); free(n->tags); free(n->metadata); + free(n->emb); } else { if (w != i) g->nodes[w] = g->nodes[i]; w++; @@ -6923,6 +8130,19 @@ el_val_t engram_prune_telemetry(el_val_t older_than_ms) { g->edge_count = ew; free(set); } + /* Mirror the telemetry prune into the durable paged store so its live + * count tracks the resident graph and stale ISE telemetry stays bounded + * in the store too. Without this, the resident graph GCs old ISE from RAM + * (node_count drops) while the store retains them (count diverges + the + * store re-accumulates the very telemetry bloat this prune was written to + * stop). Matches engram_forget's store_forget mirror. Done before the ids + * are freed below. */ + if (engram_store_enabled() && g_engram_store) { + for (int64_t i = 0; i < removed; i++) { + store_forget(g_engram_store, removed_ids[i]); + } + } + for (int64_t i = 0; i < removed; i++) free(removed_ids[i]); free(removed_ids); @@ -7124,6 +8344,7 @@ void engram_connect(el_val_t from_id, el_val_t to_id, el_val_t weight, el_val_t e->layer_id = ENGRAM_LAYER_DEFAULT; g->edge_count++; g->adj_dirty = 1; + if (engram_store_enabled()) eg_store_put_edge(e); } el_val_t engram_edge_between(el_val_t from_id, el_val_t to_id) { @@ -7152,6 +8373,7 @@ static el_val_t engram_edge_to_map(const EngramEdge* e) { m = el_map_set(m, EL_STR(el_strdup("confidence")), el_from_float(e->confidence)); m = el_map_set(m, EL_STR(el_strdup("created_at")), (el_val_t)e->created_at); m = el_map_set(m, EL_STR(el_strdup("updated_at")), (el_val_t)e->updated_at); + m = el_map_set(m, EL_STR(el_strdup("hebb")), el_from_float(e->hebb)); m = el_map_set(m, EL_STR(el_strdup("last_fired")), (el_val_t)e->last_fired); m = el_map_set(m, EL_STR(el_strdup("inhibitory")), (el_val_t)(e->inhibitory ? 1 : 0)); m = el_map_set(m, EL_STR(el_strdup("layer_id")), (el_val_t)(int64_t)e->layer_id); @@ -7234,14 +8456,202 @@ el_val_t engram_edge_count(void) { /* Compute temporal decay factor for a node given current time. * effective contribution = salience * exp(-lambda * age_hours / T_half) * Clamped to [0.05, 1.0] so very old nodes retain a meaningful floor. */ +/* eg_edge_eff_weight — the weight spreading activation actually propagates + * through: the authored weight, potentiated by learned co-activation. + * hebb == 0 (fresh edge, cold graph, or feature effectively disabled) returns + * exactly e->weight, so this is a strict no-op until the graph has learned + * something. Clamped to 1.0 so a potentiated edge can never amplify a signal + * above its source. See the ENGRAM_HEBB_* block for the full rationale. */ +static double eg_edge_eff_weight(const EngramEdge* e) { + double w = e->weight; + if (e->hebb > 0.0) { + w *= (1.0 + ENGRAM_HEBB_GAIN * e->hebb); + if (w > 1.0) w = 1.0; + } + return w; +} + +/* eg_wm_carry_over — the ACT-R/Petrov retention rule for a node that already + * holds a working-memory slot and was not re-promoted on this call. Hard-evict + * below the base-level threshold τ (Soar-style forgetting); otherwise hold a + * weight shaped by the retrieval-probability logistic and decayed by how long + * the slot has been held (occupancy inhibition). Pure function of wall-clock + * time, so it is idempotent no matter how often activate is called. + * + * Extracted 2026-08-04: this logic was inline and applied to exactly ONE of + * the two paths that need it. See the call sites. */ +static void eg_wm_carry_over(EngramNode* cn, int64_t now_ms, int64_t* evict_ctr) { + double anchor = (cn->wm_anchor > 0.0) ? cn->wm_anchor + : cn->working_memory_weight; + double B = engram_bll_base_level(cn, now_ms); + double w = 0.0; + if (B >= ENGRAM_BLL_TAU) { + double keep = 1.0 / (1.0 + exp(-(B - ENGRAM_BLL_TAU) / ENGRAM_BLL_S)); + double hold_s = (double)(now_ms - cn->last_activated) / 1000.0; + if (hold_s < 0.0) hold_s = 0.0; + double occ = ENGRAM_CARRY_TC / (ENGRAM_CARRY_TC + hold_s); + w = anchor * keep * occ; + } + if (w < ENGRAM_WM_FLOOR) { + cn->working_memory_weight = 0.0; + cn->wm_anchor = 0.0; + if (evict_ctr) (*evict_ctr)++; + } else { + cn->working_memory_weight = w; + } +} + +/* ── Hebbian candidate-pair table helpers ─────────────────────────────────── + * Pairs are order-normalized by strcmp so (a,b) and (b,a) always resolve to + * the same slot. Collisions are resolved by strength: an incumbent that has + * decayed to nothing yields its slot, a live one keeps it and the challenger + * simply loses this round. That is a lossy table by design — consolidation + * should favor associations that recur, and a pair that keeps losing a + * collision is by definition not recurring often enough to matter. */ +/* eg_hebb_trace — the node's eligibility trace right now, in [0,1]. + * Stored as (amplitude, timestamp) and decayed on read, so the value is a pure + * function of wall-clock time: idempotent no matter how often activate runs. + * Snapped to 0 below ENGRAM_HEBB_TRACE_MIN. See ENGRAM_HEBB_TRACE_TC. */ +static double eg_hebb_trace(const EngramNode* n, int64_t now_ms) { + if (n->hebb_elig <= 0.0 || n->hebb_elig_ts <= 0) return 0.0; + double dt = (double)(now_ms - n->hebb_elig_ts) / 1000.0; + if (dt < 0.0) dt = 0.0; /* clock skew ⇒ treat as fresh */ + double t = n->hebb_elig * exp(-dt / ENGRAM_HEBB_TRACE_TC); + return (t < ENGRAM_HEBB_TRACE_MIN) ? 0.0 : t; +} + +static int eg_hebb_slot(const char* a, const char* b) { + if (!a || !b) return -1; + const char* lo = (strcmp(a, b) <= 0) ? a : b; + const char* hi = (lo == a) ? b : a; + uint64_t h = engram_id_hash(lo) * 1000003u ^ engram_id_hash(hi); + return (int)(h % (uint64_t)ENGRAM_HEBB_CAND_SLOTS); +} + +static int eg_hebb_slot_holds(const EgHebbCand* c, const char* a, const char* b) { + if (!c->a || !c->b) return 0; + return (strcmp(c->a, a) == 0 && strcmp(c->b, b) == 0) + || (strcmp(c->a, b) == 0 && strcmp(c->b, a) == 0); +} + +static void eg_hebb_slot_clear(EgHebbCand* c) { + free(c->a); free(c->b); + c->a = NULL; c->b = NULL; c->score = 0.0; +} + +/* eg_hebb_cand_bump — reinforce the (a,b) candidate association by `inc`. + * Extracted 2026-08-06 so the same collision policy serves both the co-resident + * pairs and the eligibility-trace pairs; two copies of this logic would have + * drifted. Collision policy is unchanged: claim a free slot, reinforce our own, + * evict an incumbent only once it has decayed to nothing, otherwise lose the + * round. `inc` is graded by the partner's trace, so an incumbent is never + * displaced by a challenger carrying less weight than one full co-activation. */ +static void eg_hebb_cand_bump(const char* a, const char* b, double inc) { + if (!a || !b || inc <= 0.0) return; + int s = eg_hebb_slot(a, b); + if (s < 0) return; + EgHebbCand* c = &_eg_hebb_cand[s]; + if (!c->a) { /* free slot: claim */ + c->a = el_strdup_persist(a); + c->b = el_strdup_persist(b); + c->score = inc; + } else if (eg_hebb_slot_holds(c, a, b)) { + c->score += inc; /* ours: reinforce */ + } else if (c->score <= ENGRAM_HEBB_ETA) { + eg_hebb_slot_clear(c); /* dead incumbent: take the slot */ + c->a = el_strdup_persist(a); + c->b = el_strdup_persist(b); + c->score = inc; + } + /* else: live incumbent keeps the slot this round. */ +} + +/* Does any edge already connect these two nodes, in either direction? + * Linear over the edge array, but called at most ENGRAM_HEBB_LINK_PER_CALL + * times per activation and only for pairs that already cleared the + * consolidation threshold — a handful of scans per day, not per hop. */ +static int eg_edge_exists_between(EngramStore* g, const char* a, const char* b) { + for (int64_t i = 0; i < g->edge_count; i++) { + const EngramEdge* e = &g->edges[i]; + if (!e->from_id || !e->to_id) continue; + if ((strcmp(e->from_id, a) == 0 && strcmp(e->to_id, b) == 0) || + (strcmp(e->from_id, b) == 0 && strcmp(e->to_id, a) == 0)) return 1; + } + return 0; +} + +/* engram_temporal_decay — recency shaping on the activation path. + * + * MEASURED FAILURE (2026-08-05 self-review). Census of the live graph under + * the previous form (uniform 168 h half-life, floor 0.05): + * + * node type n median tdecay % pinned at the 0.05 floor + * Memory 1233 0.0500 81% + * Knowledge 1183 0.0500 58% + * BacklogItem 1057 0.0500 91% + * Project 321 0.0500 98% + * Tag 135 0.0500 100% + * + * The median value for EVERY node type was the clamp. A function whose median + * output is its floor is not a signal — it is a constant with exceptions, and + * the exceptions were exactly the nodes touched in the last few days. + * + * What that cost, concretely: 10 of the 13 grounded value nodes — "Precision + * Over Brute Force", "Honesty Before Comfort", "The System Must Accumulate" — + * sat at 0.05, a 20x activation penalty, while Knowledge ingested overnight + * sat near 1.0 and held the working-memory top slots. The decay function was + * quietly erasing the accumulated library in favour of whatever arrived last + * night. That is a direct inversion of the system's purpose. + * + * Worse, tdecay multiplies at EVERY hop (seed activation and each propagation + * step), so a 2-hop path through settled knowledge compounded to 0.05^2 = + * 0.0025. Old regions of the graph were not disfavoured; they were unreachable. + * + * EXTERNAL EVIDENCE. "Not All Memories Age the Same" (arXiv:2604.26970) + * measures retrieval under different decay regimes: + * + * no temporal weighting NDCG@5 0.274 + * uniform exponential decay NDCG@5 0.015 <- 18x WORSE than none + * domain-adaptive decay NDCG@5 0.241 + * full adaptive hierarchy NDCG@5 0.260 + * + * Uniform exponential decay is not merely suboptimal — it is worse than having + * no decay at all, because it penalises stable knowledge (rarely accessed, + * heavily load-bearing) while failing to suppress stale volatile facts. Notably + * not even the full adaptive hierarchy beat switching decay off. + * + * THE FIX: make the half-life a function of how established a node is, and + * make the floor a preference rather than a cliff. + * + * T_eff = T_HALF * (1 + ln(1 + activation_count)) + * + * Frequently-retrieved nodes age slowly; nodes nothing has ever asked for age + * at the original rate. This is the spacing effect and the Lindy property in + * one line, it is monotone and log-bounded (a 10,000-activation node gets only + * a ~10x longer half-life, not a permanent exemption), and it is built from + * activation_count — which is measured, unlike `tier`, whose assignments are + * inconsistent enough to be untrustworthy here (the values node is tagged + * Episodic). + * + * The floor moves 0.05 -> 0.25. Given the evidence that no decay outperforms + * uniform decay, the honest maximum penalty for age alone is 4x, not 20x. Age + * should express a preference for the recent; it should never make a region of + * the graph structurally unreachable. + * + * Explicit per-node temporal_decay_rate still overrides lambda (77 nodes carry + * one) — that path is untouched and remains the escape hatch for content that + * genuinely should expire fast. */ +#define ENGRAM_DECAY_FLOOR 0.25 static double engram_temporal_decay(const EngramNode* n, int64_t now_ms) { int64_t age_ms = now_ms - n->last_activated; if (age_ms <= 0) return 1.0; double lambda = (n->temporal_decay_rate > 0.0) ? n->temporal_decay_rate : ENGRAM_DECAY_LAMBDA; double age_hours = (double)age_ms / 3600000.0; - double factor = exp(-lambda * age_hours / ENGRAM_T_HALF_HOURS); - if (factor < 0.05) factor = 0.05; + double t_half = ENGRAM_T_HALF_HOURS * + (1.0 + log(1.0 + (double)n->activation_count)); + double factor = exp(-lambda * age_hours / t_half); + if (factor < ENGRAM_DECAY_FLOOR) factor = ENGRAM_DECAY_FLOOR; return factor; } @@ -7413,12 +8823,106 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { int64_t now_ms = engram_now_ms(); + /* Observability counters: _eg_act_breakthroughs/_eg_act_wm_evicted are + * CUMULATIVE for the process lifetime and intentionally NOT reset here + * (2026-07-31 self-review — the old per-call reset made the 60s heartbeat + * miss nearly all events between beats; see the definition site). + * ctx_cos stays per-call: it is a gauge of THIS query vs the centroid. */ + _eg_act_ctx_cos = -2.0; + + /* ── Embedding backfill + query embedding (2026-07-24, bl-b2d1c944) ── + * Backfill: embed up to N un-embedded eligible nodes per call, newest + * first (append order ≈ creation order), so fresh content is semantic- + * searchable within one scan cycle and the historical store fills in + * gradually — ~16 nodes/min under the 30s curiosity cadence, no bulk + * hammering of Ollama, no latency on any create path. */ + { + int backfilled = 0; + for (int64_t i = g->node_count - 1; + i >= 0 && backfilled < ENGRAM_EMBED_BACKFILL_PER_CALL; i--) { + EngramNode* n = &g->nodes[i]; + if (n->emb || !eg_embed_eligible(n)) continue; + int32_t d = 0; + float* v = eg_embed_fetch(n->content, &d); + if (!v) break; /* embedder down / breaker open — stop this call */ + n->emb = v; n->emb_dim = d; + backfilled++; + } + } + /* Query embedding, cached single-slot: the curiosity loop re-issues the + * same 4 rotating phrases, so consecutive identical queries skip the + * HTTP round-trip entirely. */ + static char* _eg_qcache_text = NULL; + static float* _eg_qcache_emb = NULL; + static int32_t _eg_qcache_dim = 0; + float* q_emb = NULL; + int32_t q_dim = 0; + if (_eg_qcache_text && strcmp(_eg_qcache_text, q) == 0) { + q_emb = _eg_qcache_emb; q_dim = _eg_qcache_dim; + } else { + int32_t d = 0; + float* v = eg_embed_fetch(q, &d); + if (v) { + free(_eg_qcache_text); free(_eg_qcache_emb); + _eg_qcache_text = strdup(q); + _eg_qcache_emb = v; + _eg_qcache_dim = d; + q_emb = v; q_dim = d; + } + } + /* ── Context centroid fold-in (2026-07-29) ────────────────────────── + * Record drift BEFORE blending (cos of the query against yesterday's + * context), then fold the query in as a touch, then build the + * query-dominant effective scoring vector. See the ENGRAM_CTX_* block + * for the design and the feedback-loop guard rationale. */ + float* e_eff = NULL; + if (q_emb) { + if (_eg_ctx_c && _eg_ctx_dim == q_dim) + _eg_act_ctx_cos = eg_cosine(q_emb, _eg_ctx_c, q_dim); + eg_ctx_blend(q_emb, q_dim); + if (_eg_ctx_c && _eg_ctx_dim == q_dim) { + e_eff = malloc((size_t)q_dim * sizeof(float)); + if (e_eff) { + double nq = 0.0; + for (int32_t i = 0; i < q_dim; i++) + nq += (double)q_emb[i] * (double)q_emb[i]; + nq = (nq > 0.0) ? sqrt(nq) : 1.0; + double nn = 0.0; + for (int32_t i = 0; i < q_dim; i++) { + double v = ENGRAM_CTX_QALPHA * ((double)q_emb[i] / nq) + + (1.0 - ENGRAM_CTX_QALPHA) * (double)_eg_ctx_c[i]; + e_eff[i] = (float)v; + nn += v * v; + } + if (nn <= 0.0) { free(e_eff); e_eff = NULL; } + } + } + } + /* Per-node cosine vs the effective query (query ⊕ context centroid; + * plain query on cold start), computed once, consumed twice: semantic + * seeding below and the additive WM term in Pass 2 (use similarity + * twice, coherently — HippoRAG). cosq stays NULL when the embedder is + * unavailable; every consumer degrades to pure lexical behavior. */ + double* cosq = NULL; + if (q_emb) { + const float* qv = e_eff ? e_eff : q_emb; + cosq = calloc((size_t)g->node_count, sizeof(double)); + if (cosq) { + for (int64_t i = 0; i < g->node_count; i++) { + EngramNode* n = &g->nodes[i]; + cosq[i] = (n->emb && n->emb_dim == q_dim) + ? eg_cosine(n->emb, qv, q_dim) : -2.0; + } + } + } + free(e_eff); e_eff = NULL; /* only needed to fill cosq */ + /* Per-node layer-1 tracking. */ double* best_bg = calloc((size_t)g->node_count, sizeof(double)); int64_t* best_hops = calloc((size_t)g->node_count, sizeof(int64_t)); int* reached = calloc((size_t)g->node_count, sizeof(int)); if (!best_bg || !best_hops || !reached) { - free(best_bg); free(best_hops); free(reached); return out; + free(best_bg); free(best_hops); free(reached); free(cosq); return out; } /* ── LAYER 1: broad fan-out (background activation) ───────────────── @@ -7429,7 +8933,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { SeedEntry* seeds = malloc((size_t)g->node_count * sizeof(SeedEntry)); int64_t seed_count = 0; if (!seeds) { - free(best_bg); free(best_hops); free(reached); return out; + free(best_bg); free(best_hops); free(reached); free(cosq); return out; } /* Tokenize once: a node seeds if it matches ANY query token, and its seed * activation is scaled by token coverage (fraction of distinct query @@ -7468,6 +8972,65 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { reached[i] = 1; } } + /* ── Semantic seed supplement (2026-07-24, bl-b2d1c944) ───────────── + * Top-K nodes by cosine ≥ SEED_MIN join the seed set with initial + * activation = similarity × the same decay/dampen shaping the lexical + * seeds get. This is the fix for "idle cognition firing blanks": a + * curiosity phrase like "decision pattern lesson" now ignites nodes + * that MEAN decisions and lessons, not just nodes that contain those + * literal substrings. Lexically-seeded nodes are skipped — the lexical + * path already gave them coverage-scaled activation. */ + if (cosq) { + /* Redundancy-suppressed top-K (2026-08-05 self-review; see + * ENGRAM_DEDUP_COS for the measurement that motivated it). A rejected + * candidate does NOT consume one of the K slots — the loop retries for + * the next-best distinct node, so K distinct meanings are seeded rather + * than K copies of one. Rejects are recorded in seed_dup[] rather than + * reached[] or cosq[]: marking reached[] would suppress the node's + * propagation, and clobbering cosq[] would change the downstream + * query-aware propagation gate. Neither belongs in a seeding decision. + * `guard` bounds the retries so a pathological duplicate cluster can + * never turn seed selection into an O(K·N²) scan. */ + unsigned char* seed_dup = calloc((size_t)g->node_count, 1); + int64_t sel[ENGRAM_EMBED_SEED_K]; + uint64_t selkey[ENGRAM_EMBED_SEED_K]; + int nsel = 0; + int guard = ENGRAM_EMBED_SEED_K * 8; + while (nsel < ENGRAM_EMBED_SEED_K && guard-- > 0) { + int64_t bi = -1; double bc = ENGRAM_EMBED_SEED_MIN; + for (int64_t i = 0; i < g->node_count; i++) { + if (reached[i]) continue; + if (seed_dup && seed_dup[i]) continue; + if (cosq[i] > bc) { bc = cosq[i]; bi = i; } + } + if (bi < 0) break; + EngramNode* n = &g->nodes[bi]; + uint64_t key = eg_content_key(n); + int dup = 0; + for (int s = 0; s < nsel; s++) { + if (eg_same_content(n, &g->nodes[sel[s]], key, selkey[s])) { + dup = 1; break; + } + } + if (dup) { + _eg_act_dup_seeds++; + if (seed_dup) { seed_dup[bi] = 1; continue; } + break; /* OOM on the skip map: stop rather than spin */ + } + double tdecay = engram_temporal_decay(n, now_ms); + double dampen = engram_activation_dampen(n); + double act = bc * tdecay * dampen; + seeds[seed_count].idx = bi; + seeds[seed_count].act = act; + seeds[seed_count].created_at = n->created_at; + seed_count++; + best_bg[bi] = act; + best_hops[bi] = 0; + reached[bi] = 1; + sel[nsel] = bi; selkey[nsel] = key; nsel++; + } + free(seed_dup); + } /* Compute mean seed created_at for temporal proximity bonus. * Was a running pairwise average — seed_epoch = (seed_epoch + t_s)/2 — * which is NOT the arithmetic mean: it exponentially over-weights the @@ -7486,7 +9049,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { typedef struct { int64_t idx; int64_t hops; double act; } Frontier; Frontier* fr = malloc((size_t)(g->node_count * (max_depth + 1)) * sizeof(Frontier) + 16 * sizeof(Frontier)); if (!fr) { - free(best_bg); free(best_hops); free(reached); free(seeds); return out; + free(best_bg); free(best_hops); free(reached); free(seeds); free(cosq); return out; } int64_t fhead = 0, ftail = 0; int64_t fcap = (int64_t)((size_t)(g->node_count * (max_depth + 1)) + 16); @@ -7546,8 +9109,34 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { double tbonus = engram_temporal_proximity_bonus(on->created_at, seed_epoch); double tdecay = engram_temporal_decay(on, now_ms); double dampen = engram_activation_dampen(on); - double new_act = f.act * e->weight * SPREAD_DECAY * (1.0 + tbonus) - * tdecay * dampen; + /* ── Query-aware propagation gate (2026-07-27 self-review) ── + * Prior behavior was "query-blind" spreading: the query chose + * the seeds, but propagation depended only on graph structure, + * so high-degree hubs relayed activation into branches with no + * semantic relation to the query. Per arXiv:2606.30133, gating + * each increment by the TARGET node's query similarity + * (sigma(v) = max(cos(e_v, e_q), 0)) prunes low-information + * branches at every hop (+3.6..+7.4 F1 over uniform spreading, + * 1.5-4.9x faster via a shrinking working set). + * + * Adaptation for partial embedding coverage: the paper skips + * unembedded targets outright, but only eligible non-ISE/Tag + * nodes carry embeddings here — a hard gate would sever purely + * lexical/structural pathways. So: embedded targets get a soft + * gate FLOOR + (1-FLOOR)*clip(cos) (dissimilar nodes damped + * ~4x, never killed); unembedded targets pass ungated (no + * information, no penalty); cosq == NULL (embedder down) means + * no gating at all — same graceful degradation as seeding. */ + double qgate = 1.0; + if (cosq && cosq[oi] > -1.5) { + double c = cosq[oi] > 0.0 ? cosq[oi] : 0.0; + qgate = ENGRAM_QGATE_FLOOR + (1.0 - ENGRAM_QGATE_FLOOR) * c; + } + /* eg_edge_eff_weight, not e->weight: edges that have repeatedly + * carried co-activated pairs propagate more strongly. Identity on + * an unlearned edge. (2026-08-04 self-review.) */ + double new_act = f.act * eg_edge_eff_weight(e) * SPREAD_DECAY + * (1.0 + tbonus) * tdecay * dampen * qgate; /* Firing threshold per classic spreading-activation: sub-threshold * activation neither updates the target nor enqueues it, so weak * signals die out instead of flooding the whole graph with tiny @@ -7580,7 +9169,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { double* inhibition = calloc((size_t)g->node_count, sizeof(double)); if (!inhibition) { free(best_bg); free(best_hops); free(reached); free(seeds); free(fr); - return out; + free(cosq); return out; } for (int64_t ei = 0; ei < g->edge_count; ei++) { EngramEdge* e = &g->edges[ei]; @@ -7605,8 +9194,10 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { double* wm_weights = calloc((size_t)g->node_count, sizeof(double)); if (!wm_weights) { free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); return out; + free(fr); free(inhibition); free(cosq); return out; } + /* Per-call breakthrough budget (2026-08-02) — see ENGRAM_BREAKTHROUGH_BUDGET. */ + int64_t bt_budget = ENGRAM_BREAKTHROUGH_BUDGET; for (int64_t i = 0; i < g->node_count; i++) { if (!reached[i] || best_bg[i] <= 0.0) continue; EngramNode* n = &g->nodes[i]; @@ -7627,13 +9218,60 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { double type_threshold = engram_type_threshold(n->node_type, n->tier); /* Goal bias weights the node's relevance to current intent. */ double bias = engram_goal_bias(n, q); - /* Raw working memory score. */ - double raw_wm = best_bg[i] * bias * n->confidence; + /* Raw working memory score. + * Importance factor (2026-07-28 self-review): n->importance was + * stored, serialized, and clamped at creation (default 0.5) but + * never read by any scoring path — a curated importance=1.0 node + * competed identically with a throwaway note at equal activation. + * Map it to a gentle multiplier centered on the 0.5 default: + * impf = 0.5 + importance → default nodes unchanged (×1.0), + * critical (1.0) ×1.5, low (0.1) ×0.6. Nodes loaded from legacy + * snapshots with importance<=0 stay neutral rather than being + * silently suppressed. */ + double impf = (n->importance > 0.0) ? (0.5 + n->importance) : 1.0; + double raw_wm = best_bg[i] * bias * n->confidence * impf; /* Apply inhibitory suppression. Full inhibition → scale by factor. */ double inh = inhibition[i]; if (inh > 1.0) inh = 1.0; double suppress = 1.0 - (1.0 - ENGRAM_INHIBITION_FACTOR) * inh; raw_wm *= suppress; + /* Short-term inhibition-of-return (2026-07-25, Lebiere-Best): + * damp by t_n/(t_n + t_s) where t_n = seconds since the most + * recent recorded access (WM promotion / strengthen). A node that + * just held a WM slot yields it even to structurally stronger + * competitors, and recovers as t_n grows. access_ts is recorded + * at promotion, so persistent WM residents self-inhibit. Nodes + * with no access history (never promoted) are uninhibited. + * Layer-0 override in Pass 3 still floors safety nodes. */ + if (n->access_filled > 0) { + int32_t sti_last = (n->access_head + ENGRAM_BLL_K - 1) + % ENGRAM_BLL_K; + double sti_tn = (double)(now_ms - n->access_ts[sti_last]) + / 1000.0; + if (sti_tn < 0.1) sti_tn = 0.1; + raw_wm *= sti_tn / (sti_tn + ENGRAM_STI_TS); + } + /* Additive semantic-relevance term (2026-07-24, bl-b2d1c944): + * shift-and-floor at S0 — nomic-embed scores unrelated pairs + * 0.4–0.5, so raw cosine in a weighted sum would be a constant + * bias swamping the decayed base-level signal. Above S0 the term + * ramps 0 → WM_WEIGHT, breaking ties among structurally equivalent + * candidates in favor of nodes that mean what the query means. + * + * MOVED AFTER the STI damper (2026-08-02 self-review). It used to + * be added BEFORE, so the recency multiplier scaled the semantic + * term too: an incumbent re-reached 30s later took t_n/(t_n+120) + * = 0.2×, cutting the cosine term's ceiling from 0.20 to 0.04 — + * below every per-type threshold (0.15–0.40). Meaning-match was + * being punished for having been recently useful. Inhibition-of- + * return should rotate the STRUCTURAL score (what the graph + * dragged in), not the semantic one (what the query actually + * means); relevance to the current query is not stale merely + * because the node was in WM a moment ago. */ + if (cosq && cosq[i] > ENGRAM_EMBED_S0) { + raw_wm += ENGRAM_EMBED_WM_WEIGHT + * (cosq[i] - ENGRAM_EMBED_S0) / (1.0 - ENGRAM_EMBED_S0); + } /* Threshold gate: must exceed per-type threshold to enter working * memory. Type threshold replaces the old flat 0.2 filter. */ if (raw_wm >= type_threshold) { @@ -7641,13 +9279,41 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { if (n->suppression_count > 0) n->suppression_count = 0; } else { /* Node didn't make it through — increment suppression counter. - * After N consecutive suppressions: force breakthrough. */ + * After N consecutive suppressions it MAY force a breakthrough, + * subject to the per-call budget and the negative-count cooldown + * (2026-08-02 — see ENGRAM_BREAKTHROUGH_BUDGET/_COOLDOWN). */ n->suppression_count++; - if (n->suppression_count >= ENGRAM_SUPPRESSION_BREAKTHROUGH) { - wm_weights[i] = ENGRAM_BREAKTHROUGH_WEIGHT; - n->suppression_count = 0; + if (n->suppression_count >= ENGRAM_SUPPRESSION_BREAKTHROUGH + && bt_budget > 0) { + /* Graded breakthrough weight (2026-08-02): previously every + * breakthrough landed on exactly ENGRAM_BREAKTHROUGH_WEIGHT, + * so hundreds tied at 0.10 and the rank-cap tie-break at the + * cutoff degenerated to node-array index order — i.e. whoever + * was inserted earliest won, which is not a cognitive + * criterion. Scale within ±10% by how close the node came to + * its own threshold, so a near-miss outranks a node that was + * nowhere near. Stays strictly below min(type_threshold) + * (0.15) and strictly above ENGRAM_WM_FLOOR (0.05), which is + * the invariant ENGRAM_BREAKTHROUGH_WEIGHT documents. */ + double near = (type_threshold > 0.0) + ? (raw_wm / type_threshold) : 0.0; + if (near < 0.0) near = 0.0; + if (near > 1.0) near = 1.0; + wm_weights[i] = ENGRAM_BREAKTHROUGH_WEIGHT + * (0.9 + 0.2 * near); + /* Negative = cooldown. Must climb back through the cooldown + * before it can breach again. */ + n->suppression_count = -ENGRAM_BREAKTHROUGH_COOLDOWN; + bt_budget--; + _eg_act_breakthroughs++; } else { wm_weights[i] = 0.0; + /* Budget-starved or cooling down: do NOT reset the counter — + * let it saturate so the node surfaces on a later call rather + * than restarting its climb from zero. Cap the ceiling so the + * int32 cannot drift unbounded over a long uptime. */ + if (n->suppression_count > ENGRAM_SUPPRESSION_BREAKTHROUGH * 4) + n->suppression_count = ENGRAM_SUPPRESSION_BREAKTHROUGH * 4; } } } @@ -7674,6 +9340,76 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { n->suppression_count = 0; } + /* ── PASS 3½: redundancy suppression ──────────────────────────────────── + * (2026-08-05 self-review; see ENGRAM_DEDUP_COS for the census.) Runs + * BEFORE the capacity cap, so the 24 slots are contested by 24 distinct + * meanings rather than by however many copies of one document the graph + * happens to hold. Byte-identical nodes score byte-identically, so without + * this they promote as a block — a six-copy document could hold a quarter + * of working memory while saying one thing. + * + * Walk candidates in descending weight; the first occurrence of a content + * survives, later ones are evicted. Highest-weight copy wins, which keeps + * the survivor choice deterministic and preserves the strongest activation. + * + * Bounded work: once WM_CAP distinct survivors are held, every remaining + * candidate is weaker than the weakest survivor and Pass 4 would evict it + * anyway — so the walk stops at the first candidate STRICTLY below the + * cap-th survivor's weight. Ties keep being processed, because a tie can + * still take a slot through Pass 4's at_cutoff_slots path. Typical cost is + * a few dozen content hashes per call, not a full-graph sweep. + * + * This runs after Pass 3 deliberately: Layer 0 (safety) force-promotions + * are already in wm_weights and are ranked like anything else. If safety + * content is genuinely duplicated, one copy still holds a slot — the + * guarantee is that the content is present, not that every copy of it is. */ + { + int64_t nc = 0; + for (int64_t i = 0; i < g->node_count; i++) + if (wm_weights[i] > 0.0) nc++; + if (nc > 1) { + EgDupCand* dc = malloc((size_t)nc * sizeof(EgDupCand)); + if (dc) { + int64_t ci = 0; + for (int64_t i = 0; i < g->node_count; i++) { + if (wm_weights[i] > 0.0) { + dc[ci].w = wm_weights[i]; dc[ci].idx = i; ci++; + } + } + qsort(dc, (size_t)nc, sizeof(EgDupCand), eg_dupcand_cmp_desc); + int64_t keep[ENGRAM_WM_CAP]; + uint64_t kkey[ENGRAM_WM_CAP]; + int nk = 0; + double cap_w = 0.0; /* weight of the WM_CAP-th survivor */ + for (int64_t z = 0; z < nc; z++) { + if (nk >= ENGRAM_WM_CAP && dc[z].w < cap_w) break; + int64_t i = dc[z].idx; + EngramNode* n = &g->nodes[i]; + uint64_t key = eg_content_key(n); + int dup = 0; + for (int s = 0; s < nk; s++) { + if (eg_same_content(n, &g->nodes[keep[s]], key, kkey[s])) { + dup = 1; break; + } + } + if (dup) { + wm_weights[i] = 0.0; + _eg_act_wm_evicted++; + _eg_act_dup_wm++; + continue; + } + if (nk < ENGRAM_WM_CAP) { + keep[nk] = i; kkey[nk] = key; nk++; + if (nk == ENGRAM_WM_CAP) cap_w = dc[z].w; + } + } + free(dc); + } + /* malloc failure: skip suppression — duplicates may share slots + * this call, which is the pre-2026-08-05 behavior. No corruption. */ + } + } + /* ── PASS 4: WM capacity cap (per-call) ───────────────────────────────── * Enforce ENGRAM_WM_CAP as a hard upper bound on nodes promoted in this * activation call. Without this, broad curiosity seeds like "knowledge" @@ -7682,6 +9418,16 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { * becomes useless. (Ported from 2026-06-26 self-review branch; observed * 525 promoted for "knowledge", 524 at breakthrough floor 0.25, 1 natural.) */ { + /* Absolute admission floor (2026-07-30): drop sub-floor candidates + * BEFORE rank-trimming, so the cap is filled only by nodes that clear + * an absolute bar — fill below ENGRAM_WM_CAP becomes reachable and + * wm_saturated becomes an informative signal. See ENGRAM_WM_FLOOR. */ + for (int64_t i = 0; i < g->node_count; i++) { + if (wm_weights[i] > 0.0 && wm_weights[i] < ENGRAM_WM_FLOOR) { + wm_weights[i] = 0.0; + _eg_act_wm_evicted++; + } + } int64_t cap_count = 0; for (int64_t i = 0; i < g->node_count; i++) { if (wm_weights[i] > 0.0) cap_count++; @@ -7714,12 +9460,26 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { continue; /* fills a slot */ } wm_weights[i] = 0.0; /* over cap: evict */ + _eg_act_wm_evicted++; } } /* If malloc failed, skip cap — WM unbounded this call, no corruption. */ } } + /* Pre-persist residency snapshot (2026-07-30): record which nodes held a + * WM slot BEFORE this call's results are written back. Used below to fold + * only NEW WM entrants into the context centroid — an incumbent that gets + * re-promoted every scan no longer re-entrenches the centroid each time, + * which was the remaining positive-feedback path in the WM→centroid→ + * e_eff→re-selection loop (fixation driver; cf. wm_top0_streak=1407 + * incident). NULL on OOM → fold falls back to previous behavior. */ + unsigned char* was_wm = malloc((size_t)g->node_count); + if (was_wm) { + for (int64_t i = 0; i < g->node_count; i++) + was_wm[i] = (g->nodes[i].working_memory_weight > 0.0) ? 1 : 0; + } + /* Persist working_memory_weight (post Pass 4) to node store. * * Conversational thread continuity (ENGRAM_WM_DECAY): @@ -7749,23 +9509,45 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { * above τ, shape the weight held at promotion (wm_anchor) by the * retrieval-probability logistic. Pure function of wall-clock * time — idempotent no matter how often activate is called. */ - EngramNode* cn = &g->nodes[i]; - double anchor = (cn->wm_anchor > 0.0) ? cn->wm_anchor - : cn->working_memory_weight; - double B = engram_bll_base_level(cn, now_ms); - if (B < ENGRAM_BLL_TAU) { - cn->working_memory_weight = 0.0; - } else { - double keep = 1.0 / (1.0 + exp(-(B - ENGRAM_BLL_TAU) - / ENGRAM_BLL_S)); - double w = anchor * keep; - cn->working_memory_weight = (w < 0.01) ? 0.0 : w; - } - } else { + eg_wm_carry_over(&g->nodes[i], now_ms, &_eg_act_wm_evicted); + } else if (wm_weights[i] > 0.0) { g->nodes[i].working_memory_weight = wm_weights[i]; /* Anchor the promotion weight: carry-over decay above computes * from this fixed point rather than compounding per call. */ - if (wm_weights[i] > 0.0) g->nodes[i].wm_anchor = wm_weights[i]; + g->nodes[i].wm_anchor = wm_weights[i]; + } else if (was_wm && was_wm[i]) { + /* ── Reached but sub-threshold (2026-08-04 self-review) ────────── + * This case used to fall into the unconditional `= wm_weights[i]` + * below, zeroing the slot outright — no carry-over, no base-level + * check, not even counted as an eviction. The asymmetry was exactly + * backwards: a node the current query did NOT reach got the full + * ACT-R retention treatment, while a node the query DID reach, but + * which landed a hair under its type threshold, was dropped + * instantly. Being found was punished relative to not being found. + * + * MEASURED CONSEQUENCE: working memory turned over 100% on every + * call. Three consecutive activations with a byte-identical query + * gave |A∩B| = |B∩C| = 0 — no node survived a single call — while + * wm_evicted stayed at 0 the whole time, because this path never + * incremented it. WM was not a working set at all; it was six fresh + * suppression-breakthrough nodes per call, re-drawn each time. That + * silently defeated every mechanism built on WM continuity: the + * conversational-thread carry-over documented since the two-layer + * architecture landed, the wm_anchor fixed point, and (this + * session) any possibility of learning from co-activation, since no + * pair can co-activate twice if nothing survives one call. + * + * Same helper as the unreached path: one retention rule, both ways + * out of a WM slot. The existing guards (τ hard-evict, WM_FLOOR, + * occupancy decay, Pass 5 global cap) all still apply — routing + * into them is why this is safe rather than merely sticky. */ + eg_wm_carry_over(&g->nodes[i], now_ms, &_eg_act_wm_evicted); + } else { + g->nodes[i].working_memory_weight = 0.0; + /* Zero the anchor when the slot empties (2026-07-30): a stale + * anchor on an evicted node was a latent resurrection bug if the + * carry-over entry guard ever changes. */ + g->nodes[i].wm_anchor = 0.0; } } @@ -7779,6 +9561,93 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { * activations outcompete older decayed ones. (Ported from 2026-06-26 * self-review branch.) */ { + /* Absolute admission floor, global pass (2026-07-30): see + * ENGRAM_WM_FLOOR. Sub-floor residents are dropped even when the + * global population is under cap — this is what lets wm_active + * drain below 24 during quiet periods. */ + for (int64_t i = 0; i < g->node_count; i++) { + EngramNode* fn = &g->nodes[i]; + if (fn->working_memory_weight > 0.0 && + fn->working_memory_weight < ENGRAM_WM_FLOOR) { + fn->working_memory_weight = 0.0; + fn->wm_anchor = 0.0; + _eg_act_wm_evicted++; + } + } + /* ── Global redundancy suppression (2026-08-06 self-review) ────────── + * Pass 3½ deduplicates THIS CALL'S candidates. But the WM population + * that actually exists after persist is a UNION of two sets: nodes + * promoted this call, and nodes carried over from earlier calls by + * eg_wm_carry_over. Pass 3½ never sees the second set, so the dedup + * guarantee it advertises does not hold for the thing it is a + * guarantee about. + * + * OBSERVED, not inferred. A live WM census caught two byte-identical + * Knowledge nodes — f93d5f90 and b578d6a7, the same 3,193-character + * "# Memory Integration" document under the same node_type — both + * holding slots at 0.289 and 0.271, having arrived by the two + * different routes. Pass 3½ had run and had correctly passed, because + * only one of the two was a candidate that call. + * + * The cost is small and constant: 1–2 of 24 slots, ~4–8% of working + * memory, indefinitely. It is worth fixing anyway, because the failure + * is silent and self-reinforcing — a duplicate that holds a slot gets + * reinforced for holding it, and (as of this session) now also earns + * Hebbian eligibility, so redundancy would start teaching the graph + * that a document is associated with itself. + * + * Runs BEFORE the cap count below, so slots freed here are reclaimed + * by distinct content in the same pass rather than left empty. Same + * identity test and same highest-weight-survives rule as Pass 3½. */ + { + int64_t gn = 0; + for (int64_t i = 0; i < g->node_count; i++) + if (g->nodes[i].working_memory_weight > 0.0) gn++; + if (gn > 1) { + EgDupCand* gd = malloc((size_t)gn * sizeof(EgDupCand)); + if (gd) { + int64_t gi = 0; + for (int64_t i = 0; i < g->node_count; i++) { + if (g->nodes[i].working_memory_weight > 0.0) { + gd[gi].w = g->nodes[i].working_memory_weight; + gd[gi].idx = i; + gi++; + } + } + qsort(gd, (size_t)gn, sizeof(EgDupCand), eg_dupcand_cmp_desc); + int64_t gkeep[ENGRAM_WM_CAP]; + uint64_t gkey[ENGRAM_WM_CAP]; + int gnk = 0; + double gcap_w = 0.0; + for (int64_t z = 0; z < gn; z++) { + if (gnk >= ENGRAM_WM_CAP && gd[z].w < gcap_w) break; + int64_t i = gd[z].idx; + EngramNode* n = &g->nodes[i]; + uint64_t key = eg_content_key(n); + int dup = 0; + for (int s = 0; s < gnk; s++) { + if (eg_same_content(n, &g->nodes[gkeep[s]], key, gkey[s])) { + dup = 1; break; + } + } + if (dup) { + n->working_memory_weight = 0.0; + n->wm_anchor = 0.0; + _eg_act_wm_evicted++; + _eg_act_dup_wm_global++; + continue; + } + if (gnk < ENGRAM_WM_CAP) { + gkeep[gnk] = i; gkey[gnk] = key; gnk++; + if (gnk == ENGRAM_WM_CAP) gcap_w = gd[z].w; + } + } + free(gd); + } + /* malloc failure: skip — duplicates may share slots this call, + * which is the pre-2026-08-06 behavior. No corruption. */ + } + } int64_t global_wm_count = 0; for (int64_t i = 0; i < g->node_count; i++) { if (g->nodes[i].working_memory_weight > 0.0) global_wm_count++; @@ -7809,6 +9678,8 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { continue; /* fills a slot */ } n->working_memory_weight = 0.0; /* evict: over global cap */ + n->wm_anchor = 0.0; /* keep anchor coherent */ + _eg_act_wm_evicted++; /* was uncounted before 2026-08-02 */ } } /* If malloc failed, skip — WM over cap this call, no data corruption. */ @@ -7843,6 +9714,233 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { engram_bll_record_access(n, now_ms); } + /* ── Hebbian edge potentiation (2026-08-04 self-review) ───────────────── + * The edge-level counterpart of the node-level reinforcement immediately + * above. That loop says "this memory was retrieved"; this one says "these + * two memories were retrieved TOGETHER, so the path between them is worth + * more than it was". Runs on final post-Pass-5 working memory, so only + * pairs that survived both capacity caps count as co-active — the same + * "promotion to WM is the analog of actual retrieval" standard the ACT-R + * reinforcement above uses. Consistency matters: two mechanisms disagreeing + * about what counts as a retrieval would drift apart invisibly. + * + * Three steps: (1) decay every edge, so disuse fades; (2) increment + * co-active pairs; (3) homeostatic scaling so no node accumulates + * unbounded associative mass. All three are required — see ENGRAM_HEBB_*. */ + { + unsigned char* in_wm = calloc((size_t)g->node_count, 1); + if (in_wm) { + int64_t wm_n = 0; + for (int64_t i = 0; i < g->node_count; i++) { + if (g->nodes[i].working_memory_weight > 0.0) { + in_wm[i] = 1; + wm_n++; + } + } + /* Step 0 (2026-08-06): refresh the eligibility trace of everything + * currently in WM. Done BEFORE the edge pass so a pair that is + * co-resident right now reads trace 1.0 on both ends and receives + * exactly ENGRAM_HEBB_ETA — identical to the pre-trace rule. */ + for (int64_t i = 0; i < g->node_count; i++) { + if (!in_wm[i]) continue; + g->nodes[i].hebb_elig = 1.0; + g->nodes[i].hebb_elig_ts = now_ms; + } + /* Steps 1+2: decay all, potentiate co-active. Fused into one O(E) + * pass. Edges whose endpoints resolve to nothing still decay — + * a dangling edge should not hold learned strength forever. + * + * Co-activation is now graded by the product of the endpoints' + * eligibility traces rather than gated on same-call WM residency; + * see the ENGRAM_HEBB_TRACE_* block for the census that forced + * this. The old `wm_n > 1` guard is gone because a single node + * entering WM can now legitimately potentiate against a partner + * that was in WM moments ago — that asymmetric case is precisely + * the signal the simultaneity rule was throwing away. When no + * trace is warm, co is 0 and the pass degenerates to pure decay. */ + for (int64_t ei = 0; ei < g->edge_count; ei++) { + EngramEdge* e = &g->edges[ei]; + double h = e->hebb * ENGRAM_HEBB_DECAY; + if (!e->inhibitory) { + int64_t a = engram_idmap_get(g, e->from_id); + int64_t b = engram_idmap_get(g, e->to_id); + if (a >= 0 && a < g->node_count && b >= 0 && b < g->node_count) { + double co = eg_hebb_trace(&g->nodes[a], now_ms) + * eg_hebb_trace(&g->nodes[b], now_ms); + if (co > 0.0) { + h += ENGRAM_HEBB_ETA * co; + e->last_fired = now_ms; /* first real writer outside dharma_strengthen */ + } + } + } + e->hebb = (h < ENGRAM_HEBB_MIN) ? 0.0 : h; + } + /* Step 3: homeostatic scaling. Sum incident hebb per node; any node + * over budget scales ALL its incident edges down proportionally. + * An edge is scaled by the stronger (smaller) of its two endpoints' + * factors, so one pass satisfies both endpoints' constraints — + * conservative, non-iterative, and stable. Skipped on OOM: the + * potentiation above is still correct, just uncompensated for one + * call, and the next call re-normalizes. */ + double* mass = calloc((size_t)g->node_count, sizeof(double)); + if (mass) { + for (int64_t ei = 0; ei < g->edge_count; ei++) { + EngramEdge* e = &g->edges[ei]; + if (e->hebb <= 0.0) continue; + int64_t a = engram_idmap_get(g, e->from_id); + int64_t b = engram_idmap_get(g, e->to_id); + if (a >= 0 && a < g->node_count) mass[a] += e->hebb; + if (b >= 0 && b < g->node_count) mass[b] += e->hebb; + } + for (int64_t ei = 0; ei < g->edge_count; ei++) { + EngramEdge* e = &g->edges[ei]; + if (e->hebb <= 0.0) continue; + int64_t a = engram_idmap_get(g, e->from_id); + int64_t b = engram_idmap_get(g, e->to_id); + double s = 1.0; + if (a >= 0 && a < g->node_count && mass[a] > ENGRAM_HEBB_NODE_BUDGET) + s = ENGRAM_HEBB_NODE_BUDGET / mass[a]; + if (b >= 0 && b < g->node_count && mass[b] > ENGRAM_HEBB_NODE_BUDGET) { + double sb = ENGRAM_HEBB_NODE_BUDGET / mass[b]; + if (sb < s) s = sb; + } + if (s < 1.0) { + double h = e->hebb * s; + e->hebb = (h < ENGRAM_HEBB_MIN) ? 0.0 : h; + } + } + free(mass); + } + + /* ── Associative link formation ────────────────────────────── + * Everything above reweights edges that already exist. This part + * grows the ones that don't. See the ENGRAM_HEBB_LINK_* block for + * why this is gated as hard as it is. */ + { + /* Decay every candidate slot, exactly as edges decay, so an + * association that stops recurring loses ground at the same + * rate whether or not it has been consolidated yet. */ + for (int s = 0; s < ENGRAM_HEBB_CAND_SLOTS; s++) { + EgHebbCand* c = &_eg_hebb_cand[s]; + if (!c->a) continue; + c->score *= ENGRAM_HEBB_DECAY; + if (c->score < ENGRAM_HEBB_MIN) eg_hebb_slot_clear(c); + } + /* Gather this call's WM members (bounded by ENGRAM_WM_CAP, so + * at most 276 pairs — the O(n²) here is over ≤24 items). */ + int64_t wm_idx[ENGRAM_WM_CAP]; + int wm_k = 0; + for (int64_t i = 0; i < g->node_count && wm_k < ENGRAM_WM_CAP; i++) { + if (in_wm[i]) wm_idx[wm_k++] = i; + } + /* Warm set (2026-08-06): nodes NOT in WM right now but whose + * eligibility trace is still up. These are the partners the + * simultaneity rule could never see. Bounded by + * ENGRAM_HEBB_WARM_MAX; the scan is O(node_count), which is + * an order of magnitude cheaper than the O(edge_count) pass + * already running above it. */ + int64_t warm_idx[ENGRAM_HEBB_WARM_MAX]; + double warm_t[ENGRAM_HEBB_WARM_MAX]; + int warm_k = 0; + for (int64_t i = 0; i < g->node_count + && warm_k < ENGRAM_HEBB_WARM_MAX; i++) { + if (in_wm[i]) continue; + double t = eg_hebb_trace(&g->nodes[i], now_ms); + if (t <= 0.0) continue; + warm_idx[warm_k] = i; + warm_t[warm_k] = t; + warm_k++; + } + _eg_act_hebb_warm = warm_k; + (void)wm_n; /* superseded as a gate by the trace product */ + /* Reinforce candidate scores. Two families, one rule: + * WM × WM — both traces are 1.0 ⇒ increment ETA, exactly + * the pre-2026-08-06 behavior. + * WM × warm — increment ETA·trace, so a partner that left + * working memory one cycle ago still earns most + * of the credit and one that left ten cycles ago + * earns a third of it. + * warm × warm is deliberately NOT paired: with nothing currently + * active there is no event to be eligible FOR, and pairing decayed + * residue against decayed residue would manufacture associations + * out of two absences. Eligibility gates on something happening + * now — that is the whole content of the three-factor rule. */ + for (int x = 0; x < wm_k; x++) { + for (int y = x + 1; y < wm_k; y++) { + eg_hebb_cand_bump(g->nodes[wm_idx[x]].id, + g->nodes[wm_idx[y]].id, + ENGRAM_HEBB_ETA); + } + for (int w = 0; w < warm_k; w++) { + eg_hebb_cand_bump(g->nodes[wm_idx[x]].id, + g->nodes[warm_idx[w]].id, + ENGRAM_HEBB_ETA * warm_t[w]); + } + } + /* Consolidate the strongest qualifying candidates into real + * edges. Done last and separately because engram_grow_edges() + * may realloc g->edges — no EngramEdge* may be held across + * this point. */ + int formed = 0; + /* Count existing self-formed edges once, up front: the cap is + * on total learned structure, not on this call's rate. */ + int64_t hebb_edge_total = 0; + for (int64_t i = 0; i < g->edge_count; i++) { + if (g->edges[i].relation && + strcmp(g->edges[i].relation, "hebbian-associate") == 0) + hebb_edge_total++; + } + int64_t hebb_edge_cap = + (int64_t)((double)g->edge_count * ENGRAM_HEBB_LINK_MAX_FRAC); + for (int s = 0; s < ENGRAM_HEBB_CAND_SLOTS + && formed < ENGRAM_HEBB_LINK_PER_CALL + && hebb_edge_total < hebb_edge_cap; s++) { + EgHebbCand* c = &_eg_hebb_cand[s]; + if (!c->a || c->score < ENGRAM_HEBB_LINK_MIN) continue; + if (engram_idmap_get(g, c->a) < 0 || + engram_idmap_get(g, c->b) < 0) { /* node gone */ + eg_hebb_slot_clear(c); continue; + } + if (eg_edge_exists_between(g, c->a, c->b)) { + eg_hebb_slot_clear(c); continue; /* already wired */ + } + engram_grow_edges(); + EngramEdge* ne = &g->edges[g->edge_count]; + memset(ne, 0, sizeof(*ne)); + ne->id = engram_new_id(); + ne->from_id = el_strdup_persist(c->a); + ne->to_id = el_strdup_persist(c->b); + ne->relation = el_strdup_persist("hebbian-associate"); + ne->metadata = el_strdup_persist("{\"origin\":\"co-activation\"}"); + ne->weight = ENGRAM_HEBB_LINK_W0; + /* Carry the earned score across so a freshly consolidated + * edge starts where the association already is, rather + * than restarting a climb it has already made. */ + ne->hebb = c->score; + ne->confidence = 1.0; + ne->created_at = now_ms; + ne->updated_at = now_ms; + ne->last_fired = now_ms; + ne->layer_id = ENGRAM_LAYER_DEFAULT; + g->edge_count++; + g->adj_dirty = 1; + _eg_hebb_links_formed++; + hebb_edge_total++; + formed++; + /* Hand the association to the durable store. In the soul + * daemon this local edge is the ONLY copy and dies with the + * process — see the ENGRAM_HEBB_WB_SLOTS block. Enqueue + * from ne->from_id/to_id rather than c->a/c->b: the slot is + * cleared on the next line and the edge now owns the ids. */ + eg_hebb_wb_push(ne->from_id, ne->to_id, + ne->weight, ne->hebb); + eg_hebb_slot_clear(c); /* the edge is the record now */ + } + } + free(in_wm); + } + } + /* ── Collect all background-activated nodes for the return value ──── * Callers see both layers. Context compilation uses only promoted nodes * (working_memory_weight > 0). Sort: promoted first by wm_weight desc, @@ -7852,7 +9950,8 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { int64_t rcount = 0; if (!results) { free(best_bg); free(best_hops); free(reached); free(seeds); - free(fr); free(inhibition); free(wm_weights); return out; + free(fr); free(inhibition); free(wm_weights); free(cosq); + free(was_wm); return out; } for (int64_t i = 0; i < g->node_count; i++) { if (!reached[i]) continue; @@ -7879,6 +9978,28 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { } results[j + 1] = key; } + /* ── Context centroid: fold in the touched nodes (2026-07-29) ─────── + * Results are sorted promoted-first by wm_weight desc, so the first + * ENGRAM_CTX_TOUCH_MAX embedded entries with wm > 0 are exactly the + * strongest WM survivors of THIS call — the same "promotion is the + * retrieval event" rule the BLL reinforcement pass uses. μ=0.9 EMA + * keeps any single scan's touches a minority contribution. */ + { + int touched = 0; + for (int64_t i = 0; i < rcount && touched < ENGRAM_CTX_TOUCH_MAX; i++) { + if (results[i].wm <= 0.0) break; /* promoted block exhausted */ + EngramNode* n = &g->nodes[results[i].idx]; + if (!n->emb || n->emb_dim <= 0) continue; + /* New-entrant gate (2026-07-30): skip nodes that already held a + * WM slot before this call — incumbents must not keep pulling + * the centroid toward themselves. Fresh topical shifts (new + * entrants + the query fold at call start) steer it instead. */ + if (was_wm && was_wm[results[i].idx]) continue; + eg_ctx_blend(n->emb, n->emb_dim); + touched++; + } + } + free(was_wm); for (int64_t i = 0; i < rcount; i++) { el_val_t entry = el_map_new(0); entry = el_map_set(entry, EL_STR(el_strdup("node")), @@ -7897,12 +10018,19 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) { } free(best_bg); free(best_hops); free(reached); free(seeds); free(fr); free(inhibition); free(wm_weights); free(results); + free(cosq); return out; } /* ── Engram persistence (JSON snapshot) ─────────────────────────────────── */ -static void engram_emit_node_json(JsonBuf* b, const EngramNode* n) { +/* include_emb (2026-07-31 self-review): the ~5.7KB "emb" vector belongs ONLY + * in persistence/replication output (engram_save → snapshot.json, which also + * backs /api/sync and /api/edges via scratch exports). Consumer read routes + * (/api/nodes, /api/search, activation results, neighbors, compiled context) + * were shipping it on every node — responses 10-50x oversized, blowing MCP + * token limits. Pass include_emb=1 only from engram_save. */ +static void engram_emit_node_json(JsonBuf* b, const EngramNode* n, int include_emb) { jb_putc(b, '{'); jb_puts(b, "\"id\":"); jb_emit_escaped(b, n->id ? n->id : ""); jb_puts(b, ",\"content\":"); jb_emit_escaped(b, n->content ? n->content : ""); @@ -7941,6 +10069,19 @@ static void engram_emit_node_json(JsonBuf* b, const EngramNode* n) { } jb_putc(b, '"'); } + /* Semantic embedding (2026-07-24): compact %.4g comma list — cosine is + * insensitive to 4-sig-fig rounding, and this keeps snapshot bloat to + * ~5KB per embedded node without a base64 codec. Absent field = not + * embedded; the lazy backfill re-embeds eventually if dropped. */ + if (include_emb && n->emb && n->emb_dim > 0) { + jb_puts(b, ",\"emb\":\""); + for (int32_t j = 0; j < n->emb_dim; j++) { + snprintf(tmp, sizeof(tmp), "%s%.4g", j ? "," : "", + (double)n->emb[j]); + jb_puts(b, tmp); + } + jb_putc(b, '"'); + } jb_putc(b, '}'); } @@ -7953,6 +10094,12 @@ static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e) { jb_puts(b, ",\"metadata\":"); jb_emit_escaped(b, e->metadata ? e->metadata : "{}"); char tmp[64]; snprintf(tmp, sizeof(tmp), ",\"weight\":%g", e->weight); jb_puts(b, tmp); + /* Learned potentiation is persisted: it is the graph's accumulated + * associative experience and must survive restarts, or the system relearns + * from zero every boot and never accumulates. Emitted only when nonzero so + * a cold snapshot stays byte-comparable to the pre-change format. + * (2026-08-04 self-review.) */ + if (e->hebb > 0.0) { snprintf(tmp, sizeof(tmp), ",\"hebb\":%.6g", e->hebb); jb_puts(b, tmp); } snprintf(tmp, sizeof(tmp), ",\"confidence\":%g", e->confidence); jb_puts(b, tmp); snprintf(tmp, sizeof(tmp), ",\"created_at\":%lld", (long long)e->created_at); jb_puts(b, tmp); snprintf(tmp, sizeof(tmp), ",\"updated_at\":%lld", (long long)e->updated_at); jb_puts(b, tmp); @@ -7970,7 +10117,7 @@ el_val_t engram_save(el_val_t path) { jb_puts(&b, "{\"nodes\":["); for (int64_t i = 0; i < g->node_count; i++) { if (i > 0) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[i]); + engram_emit_node_json(&b, &g->nodes[i], 1); } jb_puts(&b, "],\"edges\":["); for (int64_t i = 0; i < g->edge_count; i++) { @@ -8066,6 +10213,15 @@ static const char* eg_skip_ws(const char* p) { * the activation path. Same top-K-by-weight logic as engram_activate * Pass 5 (Cowan 2001: WM capacity is global). */ static void eg_enforce_wm_cap_on_load(EngramStore* g) { + /* Absolute admission floor (2026-07-30): see ENGRAM_WM_FLOOR. */ + for (int64_t i = 0; i < g->node_count; i++) { + EngramNode* fn = &g->nodes[i]; + if (fn->working_memory_weight > 0.0 && + fn->working_memory_weight < ENGRAM_WM_FLOOR) { + fn->working_memory_weight = 0.0; + fn->wm_anchor = 0.0; + } + } int64_t wm_count = 0; for (int64_t i = 0; i < g->node_count; i++) { if (g->nodes[i].working_memory_weight > 0.0) wm_count++; @@ -8092,6 +10248,7 @@ static void eg_enforce_wm_cap_on_load(EngramStore* g) { if (n->working_memory_weight > cutoff) continue; if (slots_at_cutoff > 0) { slots_at_cutoff--; continue; } n->working_memory_weight = 0.0; /* evict: over cap at load */ + n->wm_anchor = 0.0; /* keep anchor coherent */ } } @@ -8116,6 +10273,10 @@ el_val_t engram_load(el_val_t path) { free(g->nodes[i].id); free(g->nodes[i].content); free(g->nodes[i].node_type); free(g->nodes[i].label); free(g->nodes[i].tier); free(g->nodes[i].tags); free(g->nodes[i].metadata); + /* 2026-07-26 self-review: emb was the one heap field not freed + * here — ~3 KB leaked per embedded node per reload (~11 MB per + * reload at 3.7k embedded). forget/prune already free it. */ + free(g->nodes[i].emb); g->nodes[i].emb = NULL; g->nodes[i].emb_dim = 0; } g->node_count = 0; for (int64_t i = 0; i < g->edge_count; i++) { @@ -8168,7 +10329,7 @@ el_val_t engram_load(el_val_t path) { * continuity across a restart while stale pinned weights decay * out over successive boots; sub-0.05 residue drops to zero. */ nn->working_memory_weight *= 0.5; - if (nn->working_memory_weight < 0.05) nn->working_memory_weight = 0.0; + if (nn->working_memory_weight < ENGRAM_WM_FLOOR) nn->working_memory_weight = 0.0; nn->suppression_count = (int32_t)eg_get_int_field(obj, "suppression_count"); /* layer_id defaults to ENGRAM_LAYER_DEFAULT (core-identity) * for snapshots that predate the layered schema. We can't @@ -8186,6 +10347,10 @@ el_val_t engram_load(el_val_t path) { char* ats = eg_get_str_field(obj, "access_ts"); if (ats) { engram_bll_parse_access(nn, ats); free(ats); } } + { + char* es = eg_get_str_field(obj, "emb"); + if (es) { eg_parse_emb(nn, es); free(es); } + } int64_t load_idx = g->node_count; g->node_count++; if (nn->id && *nn->id) engram_idmap_put(g, nn->id, load_idx); @@ -8220,6 +10385,7 @@ el_val_t engram_load(el_val_t path) { ee->metadata = eg_get_str_field(obj, "metadata"); if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = el_strdup_persist("{}"); } ee->weight = eg_get_num_field(obj, "weight"); + ee->hebb = eg_get_num_field(obj, "hebb"); /* absent ⇒ 0 */ ee->confidence = eg_get_num_field(obj, "confidence"); ee->created_at = eg_get_int_field(obj, "created_at"); ee->updated_at = eg_get_int_field(obj, "updated_at"); @@ -8386,6 +10552,10 @@ el_val_t engram_load_merge(el_val_t path) { char* ats = eg_get_str_field(obj, "access_ts"); if (ats) { engram_bll_parse_access(nn, ats); free(ats); } } + { + char* es = eg_get_str_field(obj, "emb"); + if (es) { eg_parse_emb(nn, es); free(es); } + } int64_t merge_idx = g->node_count; g->node_count++; added_nodes++; @@ -8439,6 +10609,7 @@ el_val_t engram_load_merge(el_val_t path) { ee->metadata = eg_get_str_field(obj, "metadata"); if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = strdup("{}"); } ee->weight = eg_get_num_field(obj, "weight"); + ee->hebb = eg_get_num_field(obj, "hebb"); /* absent ⇒ 0 */ ee->confidence = eg_get_num_field(obj, "confidence"); ee->created_at = eg_get_int_field(obj, "created_at"); ee->updated_at = eg_get_int_field(obj, "updated_at"); @@ -8469,6 +10640,654 @@ el_val_t engram_load_merge(el_val_t path) { return (el_val_t)added_nodes; } +/* ══════════════════════════════════════════════════════════════════════════ + * Engram WAL (write-ahead log) + compaction + integrity hardening. + * + * Design: docs/architecture/design/engram-storage-engine-wal.md §§3-14, §18. + * Whole feature is gated behind ENGRAM_WAL=on (default off → byte-identical + * to the snapshot-as-store behavior). When on, structural mutations append a + * framed, CRC'd record to /engram.wal instead of rewriting the full + * snapshot; boot replays the WAL over the base snapshot; a size threshold + * triggers compaction (fresh snapshot + WAL truncate). + * + * Record framing (little-endian native; single-machine prototype): + * [u32 magic 'EWL1'][u32 payload_len][u8 op][u8 flags][u64 lsn] + * [u32 crc32][payload…] + * crc32 covers op|flags|lsn|payload. A record whose magic/length/crc fails + * validation marks end-of-valid-log (torn tail from a crash → replay stops + * cleanly, never crashes). + * ══════════════════════════════════════════════════════════════════════════ */ + +#define EG_WAL_MAGIC 0x314C5745u /* 'E''W''L''1' */ +#define EG_WAL_HDR_LEN 22 /* 4+4+1+1+8+4 */ + +/* Op codes (§5). NODE_PUT/EDGE_PUT are id-keyed upserts (replay-idempotent). */ +enum { + EG_OP_NODE_PUT = 1, + EG_OP_EDGE_PUT = 2, + EG_OP_TOMBSTONE = 3, + EG_OP_SUPERSEDE = 4, + EG_OP_LAYER_PUT = 5, + EG_OP_LAYER_DEL = 6, + EG_OP_FORGET = 7, + EG_OP_HEBB_BATCH = 8, + EG_OP_COMPACT_MARK = 9 +}; + +/* ── crc32 (IEEE 802.3, reflected, poly 0xEDB88320) ────────────────────────── + * init 0xFFFFFFFF, final XOR 0xFFFFFFFF. Known answers: + * crc32("") == 0x00000000 + * crc32("123456789") == 0xCBF43926 (the canonical check value) + * eg_crc32_update takes/returns the *internal* (pre-final-xor) running value + * so a checksum can be computed across several buffers. */ +static uint32_t eg_crc32_table[256]; +static int eg_crc32_ready = 0; +static void eg_crc32_init(void) { + for (uint32_t i = 0; i < 256; i++) { + uint32_t c = i; + for (int k = 0; k < 8; k++) + c = (c & 1) ? (0xEDB88320u ^ (c >> 1)) : (c >> 1); + eg_crc32_table[i] = c; + } + eg_crc32_ready = 1; +} +static uint32_t eg_crc32_update(uint32_t crc, const void* data, size_t n) { + if (!eg_crc32_ready) eg_crc32_init(); + const uint8_t* d = (const uint8_t*)data; + for (size_t i = 0; i < n; i++) + crc = eg_crc32_table[(crc ^ d[i]) & 0xFF] ^ (crc >> 8); + return crc; +} +static uint32_t eg_crc32(const void* data, size_t n) { + return eg_crc32_update(0xFFFFFFFFu, data, n) ^ 0xFFFFFFFFu; +} +/* crc over a record's covered bytes: op|flags|lsn|payload. */ +static uint32_t eg_wal_record_crc(uint8_t op, uint8_t flags, uint64_t lsn, + const char* payload, size_t plen) { + uint32_t c = 0xFFFFFFFFu; + c = eg_crc32_update(c, &op, 1); + c = eg_crc32_update(c, &flags, 1); + c = eg_crc32_update(c, &lsn, sizeof(lsn)); + if (plen) c = eg_crc32_update(c, payload, plen); + return c ^ 0xFFFFFFFFu; +} + +/* EL builtin: crc32 of a string (known-answer unit tests / debugging). */ +el_val_t engram_crc32(el_val_t s) { + const char* p = EL_CSTR(s); + if (!p) return (el_val_t)0; + return (el_val_t)(int64_t)(uint32_t)eg_crc32(p, strlen(p)); +} + +/* ── WAL runtime state (single log per process — engram is single-threaded) ── */ +typedef struct { + FILE* fp; /* append handle, or NULL when closed */ + char path[1024]; /* /engram.wal */ + uint64_t lsn; /* last assigned lsn; next record = lsn+1 */ + uint64_t uncommitted; /* records appended since last fsync */ + int64_t last_sync_ms; /* wall clock of last fsync */ + int64_t bytes; /* current WAL size (for compaction trigger) */ +} EngramWal; +static EngramWal eg_wal = { NULL, {0}, 0, 0, 0, 0 }; + +static int eg_wal_sync_mode(void) { + /* ENGRAM_WAL_SYNC=always|group|off (default group). */ + const char* m = getenv("ENGRAM_WAL_SYNC"); + if (m && strcmp(m, "always") == 0) return 2; + if (m && strcmp(m, "off") == 0) return 0; + return 1; /* group */ +} +static int64_t eg_wal_group_ms(void) { + const char* v = getenv("ENGRAM_WAL_SYNC_MS"); + if (v && *v) { long n = strtol(v, NULL, 10); if (n > 0) return n; } + return 50; /* §5: N≈50ms */ +} +static int64_t eg_wal_compact_bytes(void) { + const char* v = getenv("ENGRAM_WAL_COMPACT_BYTES"); + if (v && *v) { long long n = strtoll(v, NULL, 10); if (n > 0) return n; } + return 32LL * 1024 * 1024; /* §7 default 32 MB */ +} + +int engram_wal_enabled(void) { + const char* f = getenv("ENGRAM_WAL"); + return (f && (strcmp(f, "on") == 0 || strcmp(f, "1") == 0)) ? 1 : 0; +} + +/* Force the WAL to durable storage per the commit policy. force=1 (used by + * compaction / explicit commit) always fsyncs; otherwise group policy. */ +static void eg_wal_commit(int force) { + if (!eg_wal.fp) return; + int mode = eg_wal_sync_mode(); + if (mode == 0 && !force) { fflush(eg_wal.fp); return; } + if (!force && mode == 1) { + int64_t now = engram_now_ms(); + if (eg_wal.uncommitted == 0) return; + if (now - eg_wal.last_sync_ms < eg_wal_group_ms()) { fflush(eg_wal.fp); return; } + } + fflush(eg_wal.fp); + fsync(fileno(eg_wal.fp)); + eg_wal.uncommitted = 0; + eg_wal.last_sync_ms = engram_now_ms(); +} + +/* Append one framed+CRC'd record. Returns 1 on success, 0 on I/O failure + * (caller keeps the mutation in RAM; no corruption — §11 disk-full row). */ +static int eg_wal_write(uint8_t op, uint8_t flags, const char* payload, size_t plen) { + if (!eg_wal.fp) return 0; + /* Remember the pre-record offset so a partial write (e.g. ENOSPC mid- + * record) can be rolled back — otherwise a torn record would sit in the + * MIDDLE of the log and prematurely end replay of everything after it. + * We roll back to a clean record boundary and report failure; the caller + * keeps the mutation in RAM (§11 disk-full row). */ + long start = ftell(eg_wal.fp); + uint64_t lsn = ++eg_wal.lsn; + uint32_t magic = EG_WAL_MAGIC; + uint32_t len32 = (uint32_t)plen; + uint32_t crc = eg_wal_record_crc(op, flags, lsn, payload, plen); + uint8_t hdr[EG_WAL_HDR_LEN]; + memcpy(hdr + 0, &magic, 4); + memcpy(hdr + 4, &len32, 4); + hdr[8] = op; hdr[9] = flags; + memcpy(hdr + 10, &lsn, 8); + memcpy(hdr + 18, &crc, 4); + int ok = (fwrite(hdr, 1, EG_WAL_HDR_LEN, eg_wal.fp) == EG_WAL_HDR_LEN); + if (ok && plen) ok = (fwrite(payload, 1, plen, eg_wal.fp) == plen); + if (!ok) { + eg_wal.lsn--; /* reclaim the lsn */ + fflush(eg_wal.fp); + if (start >= 0) { + if (ftruncate(fileno(eg_wal.fp), start) == 0) {} /* drop torn bytes */ + fseek(eg_wal.fp, 0, SEEK_END); + } + return 0; + } + eg_wal.bytes += EG_WAL_HDR_LEN + (int64_t)plen; + eg_wal.uncommitted++; + eg_wal_commit(0); + return 1; +} + +/* ── Single-record apply (replay + live are the same code path) ───────────── */ + +/* Populate an EngramNode (freshly memset OR an existing node being overwritten + * in place) from a node JSON object. Mirrors engram_load's field set exactly, + * WITHOUT the boot-time working_memory_weight laundering, so replay reproduces + * the exact logged state (parity with the direct-apply oracle). Caller owns + * freeing prior heap fields when overwriting. */ +static void eg_fill_node_from_json(EngramNode* nn, const char* obj) { + nn->id = eg_get_str_field(obj, "id"); + nn->content = eg_get_str_field(obj, "content"); + nn->node_type = eg_get_str_field(obj, "node_type"); + nn->label = eg_get_str_field(obj, "label"); + nn->tier = eg_get_str_field(obj, "tier"); + nn->tags = eg_get_str_field(obj, "tags"); + nn->metadata = eg_get_str_field(obj, "metadata"); + if (!nn->metadata || !*nn->metadata) { free(nn->metadata); nn->metadata = el_strdup_persist("{}"); } + nn->salience = eg_get_num_field(obj, "salience"); + nn->importance = eg_get_num_field(obj, "importance"); + nn->confidence = eg_get_num_field(obj, "confidence"); + nn->temporal_decay_rate = eg_get_num_field(obj, "temporal_decay_rate"); + nn->activation_count = eg_get_int_field(obj, "activation_count"); + nn->last_activated = eg_get_int_field(obj, "last_activated"); + nn->created_at = eg_get_int_field(obj, "created_at"); + nn->updated_at = eg_get_int_field(obj, "updated_at"); + nn->background_activation = eg_get_num_field(obj, "background_activation"); + nn->working_memory_weight = eg_get_num_field(obj, "working_memory_weight"); + nn->suppression_count = (int32_t)eg_get_int_field(obj, "suppression_count"); + if (json_find_key(obj, "layer_id")) nn->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); + else nn->layer_id = ENGRAM_LAYER_DEFAULT; + nn->wm_anchor = eg_get_num_field(obj, "wm_anchor"); + { char* ats = eg_get_str_field(obj, "access_ts"); if (ats) { engram_bll_parse_access(nn, ats); free(ats); } } + { char* es = eg_get_str_field(obj, "emb"); if (es) { eg_parse_emb(nn, es); free(es); } } +} + +static void eg_free_node_heap(EngramNode* n) { + free(n->id); free(n->content); free(n->node_type); free(n->label); + free(n->tier); free(n->tags); free(n->metadata); free(n->emb); +} +static void eg_free_edge_heap(EngramEdge* e) { + free(e->id); free(e->from_id); free(e->to_id); free(e->relation); free(e->metadata); +} + +/* Upsert a node by id. Idempotent: replaying the same NODE_PUT twice yields + * the same single node. */ +static void eg_apply_node_put(const char* obj) { + EngramStore* g = engram_get(); + char* id = eg_get_str_field(obj, "id"); + if (!id || !*id) { free(id); return; } + int64_t idx = engram_find_node_index(id); + if (idx >= 0) { + EngramNode* n = &g->nodes[idx]; + int32_t emb_dim = n->emb_dim; (void)emb_dim; + eg_free_node_heap(n); + memset(n, 0, sizeof(*n)); + eg_fill_node_from_json(n, obj); + /* id unchanged → id_map entry (which owns its own key copy) stays + * valid; no re-put needed. */ + } else { + engram_grow_nodes(); + EngramNode* n = &g->nodes[g->node_count]; + memset(n, 0, sizeof(*n)); + eg_fill_node_from_json(n, obj); + int64_t ni = g->node_count; + g->node_count++; + if (n->id && *n->id) engram_idmap_put(g, n->id, ni); + } + g->adj_dirty = 1; + free(id); +} + +static int64_t eg_find_edge_index(EngramStore* g, const char* id) { + if (!id || !*id) return -1; + for (int64_t i = 0; i < g->edge_count; i++) + if (g->edges[i].id && strcmp(g->edges[i].id, id) == 0) return i; + return -1; +} +static void eg_fill_edge_from_json(EngramEdge* ee, const char* obj) { + ee->id = eg_get_str_field(obj, "id"); + ee->from_id = eg_get_str_field(obj, "from_id"); + ee->to_id = eg_get_str_field(obj, "to_id"); + ee->relation = eg_get_str_field(obj, "relation"); + ee->metadata = eg_get_str_field(obj, "metadata"); + if (!ee->metadata || !*ee->metadata) { free(ee->metadata); ee->metadata = el_strdup_persist("{}"); } + ee->weight = eg_get_num_field(obj, "weight"); + ee->hebb = eg_get_num_field(obj, "hebb"); + ee->confidence = eg_get_num_field(obj, "confidence"); + ee->created_at = eg_get_int_field(obj, "created_at"); + ee->updated_at = eg_get_int_field(obj, "updated_at"); + ee->last_fired = eg_get_int_field(obj, "last_fired"); + ee->inhibitory = (int)eg_get_int_field(obj, "inhibitory"); + if (json_find_key(obj, "layer_id")) ee->layer_id = (uint32_t)eg_get_int_field(obj, "layer_id"); + else ee->layer_id = ENGRAM_LAYER_DEFAULT; +} +static void eg_apply_edge_put(const char* obj) { + EngramStore* g = engram_get(); + char* id = eg_get_str_field(obj, "id"); + int64_t idx = eg_find_edge_index(g, id); + if (idx >= 0) { + EngramEdge* e = &g->edges[idx]; + eg_free_edge_heap(e); + memset(e, 0, sizeof(*e)); + eg_fill_edge_from_json(e, obj); + } else { + engram_grow_edges(); + EngramEdge* e = &g->edges[g->edge_count]; + memset(e, 0, sizeof(*e)); + eg_fill_edge_from_json(e, obj); + g->edge_count++; + } + g->adj_dirty = 1; + free(id); +} + +/* TOMBSTONE / SUPERSEDE — store-level soft markers on the node's metadata. + * (The server's DELETE route uses the higher-level marker-node model; these + * ops exist for completeness and are replay-idempotent.) */ +static void eg_apply_meta_marker(const char* obj, const char* markerfield, const char* from) { + EngramStore* g = engram_get(); + char* id = eg_get_str_field(obj, "id"); + int64_t idx = engram_find_node_index(id); + if (idx >= 0) { + EngramNode* n = &g->nodes[idx]; + char* by = from ? eg_get_str_field(obj, from) : NULL; + size_t need = strlen(markerfield) + (by ? strlen(by) : 4) + 32; + char* meta = (char*)malloc(need); + if (by && *by) snprintf(meta, need, "{\"%s\":\"%s\"}", markerfield, by); + else snprintf(meta, need, "{\"%s\":1}", markerfield); + free(n->metadata); n->metadata = el_strdup_persist(meta); + free(meta); free(by); + } + free(id); +} + +static void eg_apply_layer_put(const char* obj) { + EngramStore* g = engram_get(); + uint32_t lid = (uint32_t)eg_get_int_field(obj, "layer_id"); + EngramLayer* L = NULL; + for (size_t i = 0; i < g->layer_count; i++) + if (g->layers[i].layer_id == lid) { L = &g->layers[i]; break; } + if (!L) { + if (g->layer_count >= g->layer_capacity) { + size_t nc = g->layer_capacity ? g->layer_capacity * 2 : 16; + EngramLayer* grown = realloc(g->layers, nc * sizeof(EngramLayer)); + if (!grown) return; + memset(grown + g->layer_capacity, 0, (nc - g->layer_capacity) * sizeof(EngramLayer)); + g->layers = grown; g->layer_capacity = nc; + } + L = &g->layers[g->layer_count++]; + memset(L, 0, sizeof(*L)); + } else if (L->name) { free(L->name); L->name = NULL; } + L->layer_id = lid; + L->activation_priority = (uint32_t)eg_get_int_field(obj, "activation_priority"); + L->suppressible = (int)eg_get_int_field(obj, "suppressible") ? 1 : 0; + L->transparent = (int)eg_get_int_field(obj, "transparent") ? 1 : 0; + L->injectable = (int)eg_get_int_field(obj, "injectable") ? 1 : 0; + char* nm = eg_get_str_field(obj, "name"); + L->name = el_strdup_persist(nm && *nm ? nm : ""); + free(nm); +} +static void eg_apply_layer_del(const char* obj) { + EngramStore* g = engram_get(); + uint32_t lid = (uint32_t)eg_get_int_field(obj, "layer_id"); + for (size_t i = 0; i < g->layer_count; i++) + if (g->layers[i].layer_id == lid && g->layers[i].name) { + free(g->layers[i].name); g->layers[i].name = NULL; break; + } +} +static void eg_apply_forget(const char* obj) { + char* id = eg_get_str_field(obj, "id"); + if (id && *id) engram_forget((el_val_t)(uintptr_t)id); + free(id); +} + +/* Apply one decoded record to the in-RAM store. HEBB_BATCH payload is a JSON + * object {"edges":[edge,…]} — one record, one fsync, N edge upserts (§5-B). */ +static void eg_wal_apply(uint8_t op, const char* payload, size_t plen) { + char* obj = (char*)malloc(plen + 1); + if (!obj) return; + memcpy(obj, payload, plen); obj[plen] = '\0'; + switch (op) { + case EG_OP_NODE_PUT: eg_apply_node_put(obj); break; + case EG_OP_EDGE_PUT: eg_apply_edge_put(obj); break; + case EG_OP_TOMBSTONE: eg_apply_meta_marker(obj, "tombstoned", NULL); break; + case EG_OP_SUPERSEDE: eg_apply_meta_marker(obj, "superseded_by", "by"); break; + case EG_OP_LAYER_PUT: eg_apply_layer_put(obj); break; + case EG_OP_LAYER_DEL: eg_apply_layer_del(obj); break; + case EG_OP_FORGET: eg_apply_forget(obj); break; + case EG_OP_HEBB_BATCH: { + const char* arr = json_find_key(obj, "edges"); + if (arr) { arr = eg_skip_ws(arr); + if (*arr == '[') { arr++; arr = eg_skip_ws(arr); + while (*arr && *arr != ']') { + if (*arr != '{') { arr++; continue; } + const char* end = json_skip_value(arr); + size_t en = (size_t)(end - arr); + char* eobj = (char*)malloc(en + 1); + memcpy(eobj, arr, en); eobj[en] = '\0'; + eg_apply_edge_put(eobj); + free(eobj); + arr = eg_skip_ws(end); + if (*arr == ',') { arr++; arr = eg_skip_ws(arr); } + } + } + } + break; + } + case EG_OP_COMPACT_MARK: break; /* boundary marker; no state change */ + default: break; + } + free(obj); +} + +/* ── Replay (§6) ───────────────────────────────────────────────────────────── + * Reads records in lsn order, applies each intact one, and stops at the first + * record that fails magic/length/crc validation (torn tail) — never crashes. + * Returns the number of records applied; sets *out_last_lsn to the highest + * good lsn seen (0 if none). */ +static int64_t eg_wal_replay_file(const char* path, uint64_t* out_last_lsn) { + if (out_last_lsn) *out_last_lsn = 0; + FILE* f = fopen(path, "rb"); + if (!f) return 0; + fseek(f, 0, SEEK_END); long fsz = ftell(f); rewind(f); + if (fsz <= 0) { fclose(f); return 0; } + uint8_t* buf = (uint8_t*)malloc((size_t)fsz); + if (!buf) { fclose(f); return 0; } + size_t got = fread(buf, 1, (size_t)fsz, f); + fclose(f); + int64_t applied = 0; + size_t off = 0; + while (off + EG_WAL_HDR_LEN <= got) { + uint32_t magic, len32, crc; uint64_t lsn; uint8_t op, flags; + memcpy(&magic, buf + off + 0, 4); + if (magic != EG_WAL_MAGIC) break; /* garbage / torn */ + memcpy(&len32, buf + off + 4, 4); + op = buf[off + 8]; flags = buf[off + 9]; + memcpy(&lsn, buf + off + 10, 8); + memcpy(&crc, buf + off + 18, 4); + size_t plen = (size_t)len32; + if (off + EG_WAL_HDR_LEN + plen > got) break; /* short final record */ + const char* payload = (const char*)(buf + off + EG_WAL_HDR_LEN); + if (eg_wal_record_crc(op, flags, lsn, payload, plen) != crc) break; /* torn */ + eg_wal_apply(op, payload, plen); + if (out_last_lsn) *out_last_lsn = lsn; + applied++; + off += EG_WAL_HDR_LEN + plen; + } + free(buf); + return applied; +} + +static void eg_wal_build_path(const char* dir, char* out, size_t cap) { + snprintf(out, cap, "%s/engram.wal", dir ? dir : "."); +} + +/* Open (create if absent) the WAL for appending. Idempotent for a given dir. */ +static int eg_wal_open(const char* dir) { + char path[1024]; + eg_wal_build_path(dir, path, sizeof(path)); + if (eg_wal.fp && strcmp(eg_wal.path, path) == 0) return 1; /* already open */ + if (eg_wal.fp) { fclose(eg_wal.fp); eg_wal.fp = NULL; } + FILE* f = fopen(path, "ab"); + if (!f) return 0; + eg_wal.fp = f; + snprintf(eg_wal.path, sizeof(eg_wal.path), "%s", path); + fseek(f, 0, SEEK_END); + eg_wal.bytes = ftell(f); + eg_wal.last_sync_ms = engram_now_ms(); + eg_wal.uncommitted = 0; + return 1; +} + +/* ── EL-facing builtins ──────────────────────────────────────────────────── */ + +/* Boot: replay /engram.wal over the already-loaded base snapshot, then + * open the WAL for appending, continuing the lsn sequence. Idempotent replay + * makes overlap with the base harmless (§6). Returns records replayed. */ +el_val_t engram_wal_boot(el_val_t dir) { + const char* d = EL_CSTR(dir); + char path[1024]; + eg_wal_build_path(d, path, sizeof(path)); + uint64_t last = 0; + int64_t applied = eg_wal_replay_file(path, &last); + eg_wal.lsn = last; /* continue monotonically */ + eg_wal_open(d); + return (el_val_t)applied; +} + +el_val_t engram_wal_open_dir(el_val_t dir) { + return (el_val_t)(int64_t)eg_wal_open(EL_CSTR(dir)); +} + +/* Append a NODE_PUT for node `id` (serialized via the shared emitter, incl. + * emb). Returns 1 on success. */ +el_val_t engram_wal_node_put(el_val_t dir, el_val_t id) { + if (!eg_wal_open(EL_CSTR(dir))) return (el_val_t)0; + EngramNode* n = engram_find_node(EL_CSTR(id)); + if (!n) return (el_val_t)0; + JsonBuf b; jb_init(&b); + engram_emit_node_json(&b, n, 1); + int ok = eg_wal_write(EG_OP_NODE_PUT, 0, b.buf, b.len); + free(b.buf); + return (el_val_t)(int64_t)ok; +} + +/* Append an EDGE_PUT for every edge at index >= start_count. Covers both the + * single-edge route and any append-only batch. Returns edges logged. */ +el_val_t engram_wal_edges_since(el_val_t dir, el_val_t start_count) { + if (!eg_wal_open(EL_CSTR(dir))) return (el_val_t)0; + EngramStore* g = engram_get(); + int64_t start = (int64_t)start_count; if (start < 0) start = 0; + int64_t logged = 0; + for (int64_t i = start; i < g->edge_count; i++) { + JsonBuf b; jb_init(&b); + engram_emit_edge_json(&b, &g->edges[i]); + if (eg_wal_write(EG_OP_EDGE_PUT, 0, b.buf, b.len)) logged++; + free(b.buf); + } + return (el_val_t)logged; +} + +/* Append ONE HEBB_BATCH record covering every edge at index >= start_count + * (single fsync for the whole consolidation batch — §5 tier B). */ +el_val_t engram_wal_hebb_batch(el_val_t dir, el_val_t start_count) { + if (!eg_wal_open(EL_CSTR(dir))) return (el_val_t)0; + EngramStore* g = engram_get(); + int64_t start = (int64_t)start_count; if (start < 0) start = 0; + if (start >= g->edge_count) return (el_val_t)0; + JsonBuf b; jb_init(&b); + jb_puts(&b, "{\"edges\":["); + int first = 1; + for (int64_t i = start; i < g->edge_count; i++) { + if (!first) jb_putc(&b, ','); + first = 0; + engram_emit_edge_json(&b, &g->edges[i]); + } + jb_puts(&b, "]}"); + int ok = eg_wal_write(EG_OP_HEBB_BATCH, 0, b.buf, b.len); + free(b.buf); + return (el_val_t)(int64_t)ok; +} + +/* Append a FORGET (hard remove). Internal-GC only — NOT wired to HTTP DELETE. */ +el_val_t engram_wal_forget(el_val_t dir, el_val_t id) { + if (!eg_wal_open(EL_CSTR(dir))) return (el_val_t)0; + const char* sid = EL_CSTR(id); + JsonBuf b; jb_init(&b); + jb_puts(&b, "{\"id\":"); jb_emit_escaped(&b, sid ? sid : ""); jb_putc(&b, '}'); + int ok = eg_wal_write(EG_OP_FORGET, 0, b.buf, b.len); + free(b.buf); + return (el_val_t)(int64_t)ok; +} + +/* Compaction (§7). Crash-safe ordering: fresh base snapshot renamed into place + * (engram_save is atomic temp+fsync+rename) BEFORE the WAL is truncated. A + * crash in the window replays the still-present WAL over the (old or new) + * base; idempotent apply converges. */ +static int eg_wal_compact(const char* dir) { + char snap[1024], waltmp[1024], walpath[1024]; + snprintf(snap, sizeof(snap), "%s/snapshot.json", dir); + eg_wal_build_path(dir, walpath, sizeof(walpath)); + snprintf(waltmp, sizeof(waltmp), "%s/engram.wal.tmp", dir); + /* 1. new base (atomic) */ + if (!engram_save((el_val_t)(uintptr_t)snap)) return 0; + /* fsync the directory so the rename is durable before we touch the WAL */ + { int dfd = open(dir, O_RDONLY); if (dfd >= 0) { fsync(dfd); close(dfd); } } + /* 2. fresh WAL containing only a COMPACT_MARK, atomically swapped in */ + uint64_t base_lsn = eg_wal.lsn; + FILE* tf = fopen(waltmp, "wb"); + if (!tf) return 0; + { + char pl[64]; int pn = snprintf(pl, sizeof(pl), "{\"base_lsn\":%llu}", (unsigned long long)base_lsn); + uint64_t lsn = ++eg_wal.lsn; + uint32_t magic = EG_WAL_MAGIC, len32 = (uint32_t)pn; + uint32_t crc = eg_wal_record_crc(EG_OP_COMPACT_MARK, 0, lsn, pl, pn); + uint8_t hdr[EG_WAL_HDR_LEN]; + memcpy(hdr, &magic, 4); memcpy(hdr + 4, &len32, 4); + hdr[8] = EG_OP_COMPACT_MARK; hdr[9] = 0; + memcpy(hdr + 10, &lsn, 8); memcpy(hdr + 18, &crc, 4); + fwrite(hdr, 1, EG_WAL_HDR_LEN, tf); fwrite(pl, 1, pn, tf); + fflush(tf); fsync(fileno(tf)); + } + fclose(tf); + if (eg_wal.fp) { fclose(eg_wal.fp); eg_wal.fp = NULL; } + if (rename(waltmp, walpath) != 0) { return 0; } + { int dfd = open(dir, O_RDONLY); if (dfd >= 0) { fsync(dfd); close(dfd); } } + /* 3. reopen the truncated WAL for appending */ + eg_wal_open(dir); + return 1; +} + +el_val_t engram_wal_compact(el_val_t dir) { + return (el_val_t)(int64_t)eg_wal_compact(EL_CSTR(dir)); +} + +/* Compact iff the WAL has crossed the size threshold. Returns 1 if compacted. */ +el_val_t engram_wal_maybe_compact(el_val_t dir) { + if (!eg_wal.fp) eg_wal_open(EL_CSTR(dir)); + if (eg_wal.bytes > eg_wal_compact_bytes()) + return (el_val_t)(int64_t)eg_wal_compact(EL_CSTR(dir)); + return (el_val_t)0; +} + +/* ── Integrity: safe data-dir resolution (§18.2) ───────────────────────────── + * Unset ENGRAM_DATA_DIR → $HOME/.neuron/engram (created if absent). If HOME is + * also unresolvable, FAIL LOUD (exit) rather than silently persisting to an + * ephemeral /tmp. Prod (ENGRAM_DATA_DIR=/data) is unaffected. */ +el_val_t engram_resolve_data_dir(void) { + const char* d = getenv("ENGRAM_DATA_DIR"); + if (d && *d) return el_wrap_str(el_strdup(d)); + const char* home = getenv("HOME"); + if (!home || !*home) { + fprintf(stderr, "[engram] FATAL: ENGRAM_DATA_DIR unset and HOME unresolved; " + "refusing to persist to an ephemeral dir. Set ENGRAM_DATA_DIR.\n"); + exit(1); + } + char neuron[1024], engramdir[1024]; + snprintf(neuron, sizeof(neuron), "%s/.neuron", home); + snprintf(engramdir, sizeof(engramdir), "%s/.neuron/engram", home); + mkdir(neuron, 0700); + mkdir(engramdir, 0700); + return el_wrap_str(el_strdup(engramdir)); +} + +/* ── Integrity: store-level write-protection (§18.1, §18.3) ────────────────── + * The protected set is DERIVED from the self-graph at call time, not hardcoded: + * the self root and the values hub, plus every node adjacent to either (in + * either direction). Adjacency of the values hub yields all value nodes; of + * the self root yields the identity children — so new values/identity stay + * protected automatically. */ +#define EG_SELF_ROOT "kn-efeb4a5b" +#define EG_VALUES_HUB "kn-5b606390" + +static int eg_is_neighbor_of(EngramStore* g, const char* hub, const char* id) { + for (int64_t i = 0; i < g->edge_count; i++) { + EngramEdge* e = &g->edges[i]; + if (e->from_id && e->to_id) { + if (strcmp(e->from_id, hub) == 0 && strcmp(e->to_id, id) == 0) return 1; + if (strcmp(e->to_id, hub) == 0 && strcmp(e->from_id, id) == 0) return 1; + } + } + return 0; +} +static int eg_is_protected(const char* id) { + if (!id || !*id) return 0; + if (strcmp(id, EG_SELF_ROOT) == 0) return 1; + if (strcmp(id, EG_VALUES_HUB) == 0) return 1; + EngramStore* g = engram_get(); + if (eg_is_neighbor_of(g, EG_VALUES_HUB, id)) return 1; + if (eg_is_neighbor_of(g, EG_SELF_ROOT, id)) return 1; + return 0; +} +el_val_t engram_is_protected(el_val_t id) { + return (el_val_t)(int64_t)eg_is_protected(EL_CSTR(id)); +} + +/* Derived protected set as a JSON array (self root + values hub + all adjacent + * ids). For diagnostics and the integrity tests (§18.5). */ +el_val_t engram_protected_json(void) { + EngramStore* g = engram_get(); + JsonBuf b; jb_init(&b); + jb_putc(&b, '['); + int first = 1; + const char* seeds[2] = { EG_SELF_ROOT, EG_VALUES_HUB }; + /* emit the two hubs themselves */ + for (int s = 0; s < 2; s++) { + if (!first) jb_putc(&b, ','); first = 0; + jb_emit_escaped(&b, seeds[s]); + } + for (int64_t i = 0; i < g->node_count; i++) { + const char* nid = g->nodes[i].id; + if (!nid || !*nid) continue; + if (strcmp(nid, EG_SELF_ROOT) == 0 || strcmp(nid, EG_VALUES_HUB) == 0) continue; + if (eg_is_neighbor_of(g, EG_VALUES_HUB, nid) || eg_is_neighbor_of(g, EG_SELF_ROOT, nid)) { + if (!first) jb_putc(&b, ','); first = 0; + jb_emit_escaped(&b, nid); + } + } + jb_putc(&b, ']'); + return el_wrap_str(b.buf); +} + /* ── Engram JSON-string accessors ───────────────────────────────────────── * These return pre-serialized JSON strings so callers (especially HTTP * handlers) don't have to round-trip ElList/ElMap through json_stringify @@ -8481,7 +11300,7 @@ el_val_t engram_get_node_json(el_val_t id) { EngramNode* n = engram_find_node(sid); if (!n) return el_wrap_str(el_strdup("{}")); JsonBuf b; jb_init(&b); - engram_emit_node_json(&b, n); + engram_emit_node_json(&b, n, 0); return el_wrap_str(b.buf); } @@ -8506,7 +11325,7 @@ el_val_t engram_get_node_by_label(el_val_t label) { EngramNode* n = &g->nodes[i]; if (n->label && strcmp(n->label, lbl) == 0) { JsonBuf b; jb_init(&b); - engram_emit_node_json(&b, n); + engram_emit_node_json(&b, n, 0); return el_wrap_str(b.buf); } } @@ -8547,7 +11366,7 @@ el_val_t engram_search_json(el_val_t query, el_val_t limit) { int64_t end = nhits < lim ? nhits : lim; for (int64_t k = 0; k < end; k++) { if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[hits[k].idx]); + engram_emit_node_json(&b, &g->nodes[hits[k].idx], 0); first = 0; } free(hits); @@ -8579,7 +11398,7 @@ el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset) { int first = 1; for (int64_t i = off; i < end; i++) { if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); + engram_emit_node_json(&b, &g->nodes[idx[i]], 0); first = 0; } free(idx); @@ -8616,7 +11435,7 @@ el_val_t engram_scan_nodes_by_type_json(el_val_t type_v, el_val_t limit, el_val_ int first = 1; for (int64_t i = off; i < end; i++) { if (!first) jb_putc(&b, ','); - engram_emit_node_json(&b, &g->nodes[idx[i]]); + engram_emit_node_json(&b, &g->nodes[idx[i]], 0); first = 0; } free(idx); @@ -8680,7 +11499,7 @@ el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t di if (!n) continue; if (!first) jb_putc(&b, ','); jb_puts(&b, "{\"node\":"); - engram_emit_node_json(&b, n); + engram_emit_node_json(&b, n, 0); jb_puts(&b, ",\"edge\":"); engram_emit_edge_json(&b, e); char tmp[64]; snprintf(tmp, sizeof(tmp), ",\"hops\":%lld}", (long long)(h + 1)); @@ -8723,7 +11542,7 @@ el_val_t engram_activate_json(el_val_t query, el_val_t depth) { if (i > 0) jb_putc(&b, ','); jb_puts(&b, "{\"node\":"); if (n) { - engram_emit_node_json(&b, n); + engram_emit_node_json(&b, n, 0); } else { jb_puts(&b, "{}"); } @@ -8812,7 +11631,13 @@ el_val_t engram_wm_top_json(el_val_t n_v) { EngramNode* n = &g->nodes[idx[k]]; if (k > 0) jb_putc(&b, ','); jb_putc(&b, '{'); - jb_puts(&b, "\"label\":"); + /* 2026-07-26 self-review: id was never emitted here, so the + * awareness heartbeat's wm_top0_streak compared ""=="" and + * incremented unconditionally — the streak metric measured + * uptime, not fixation. */ + jb_puts(&b, "\"id\":"); + jb_emit_escaped(&b, n->id ? n->id : ""); + jb_puts(&b, ",\"label\":"); jb_emit_escaped(&b, n->label ? n->label : ""); jb_puts(&b, ",\"node_type\":"); jb_emit_escaped(&b, n->node_type ? n->node_type : ""); @@ -8830,10 +11655,279 @@ el_val_t engram_wm_top_json(el_val_t n_v) { el_val_t engram_stats_json(void) { EngramStore* g = engram_get(); - char buf[128]; + /* embedded_count: how far the lazy backfill has progressed. The single + * observable that tells the daily self-review whether semantic + * activation is actually accumulating coverage. (2026-07-24) + * + * embed_eligible_count (2026-07-27): embedded_count alone misleads — + * ~70%+ of the store is ISE/Tag/short-content nodes that are permanently + * ineligible for embedding, so raw embedded/node_count reads as "~30% + * coverage, something is broken" when eligible coverage may be complete. + * This exact misdiagnosis happened in today's self-review. Report the + * true denominator so coverage = embedded_count / embed_eligible_count. */ + int64_t embedded = 0, eligible = 0; + for (int64_t i = 0; i < g->node_count; i++) { + if (g->nodes[i].emb) embedded++; + if (eg_embed_eligible(&g->nodes[i])) eligible++; + } + char buf[256]; snprintf(buf, sizeof(buf), - "{\"node_count\":%lld,\"edge_count\":%lld,\"layer_count\":%zu}", - (long long)g->node_count, (long long)g->edge_count, g->layer_count); + "{\"node_count\":%lld,\"edge_count\":%lld,\"layer_count\":%zu," + "\"embedded_count\":%lld,\"embed_eligible_count\":%lld}", + (long long)g->node_count, (long long)g->edge_count, g->layer_count, + (long long)embedded, (long long)eligible); + return el_wrap_str(el_strdup(buf)); +} + +/* engram_act_stats_json — activation observability + embedder breaker state. + * (2026-07-27 self-review; counters made cumulative 2026-07-31.) + * wm_evicted/breakthroughs are monotonic process-lifetime totals across ALL + * engram_activate calls on this store (diff successive readings for rates; + * they reset to 0 only on restart); embed_breaker_open=1 means + * eg_embed_fetch is currently refusing calls (semantic activation silently + * degraded to lexical until the cooldown expires). The soul heartbeat folds + * this into its ISE so the pathologies are diagnosable from telemetry + * instead of inferred from wm_avg_weight hovering at the floor. */ +el_val_t engram_act_stats_json(void) { + int64_t now = engram_now_ms(); + int breaker_open = (now < _eg_embed_breaker_until) ? 1 : 0; + /* Hebbian potentiation gauges (2026-08-04 self-review). Three numbers, + * each answering a question the mechanism can fail on: + * hebb_edges — is it learning at all? (0 forever ⇒ co-activation never + * happens, or the pass is dead) + * hebb_max — is any single association saturating? (persistent 1.0 ⇒ + * homeostasis is not biting) + * hebb_mass — total associative mass; the aggregate that runaway + * potentiation would show up in first. Should plateau, not + * climb without bound. + * O(E) per call, and this is called once per 60s heartbeat. */ + EngramStore* g = engram_get(); + int64_t hebb_edges = 0; + double hebb_max = 0.0, hebb_mass = 0.0; + for (int64_t i = 0; i < g->edge_count; i++) { + double h = g->edges[i].hebb; + if (h <= 0.0) continue; + hebb_edges++; + hebb_mass += h; + if (h > hebb_max) hebb_max = h; + } + /* Candidate-table gauges: hebb_cands is how many associations are being + * tracked toward consolidation, hebb_cand_max how close the leader is to + * ENGRAM_HEBB_LINK_MIN. Together they answer "is anything about to be + * learned, and if nothing ever consolidates, is it because nothing + * co-activates or because the threshold is set too high?" — the question + * the zero-potentiated-edges measurement had to be instrumented to answer. */ + int hebb_cands = 0; + double hebb_cand_max = 0.0; + for (int i = 0; i < ENGRAM_HEBB_CAND_SLOTS; i++) { + if (!_eg_hebb_cand[i].a) continue; + hebb_cands++; + if (_eg_hebb_cand[i].score > hebb_cand_max) + hebb_cand_max = _eg_hebb_cand[i].score; + } + /* 768, not 512: the write-back gauges added 2026-08-07 push the worst-case + * rendering past the old bound, and snprintf would truncate the JSON into + * an unparseable tail rather than fail loudly. */ + char buf[896]; + /* ctx_cos (2026-07-29): cos(query, context centroid) at the LAST + * activate call, measured before the query was folded in. ~1.0 = + * context aligned with current query; low = divergence (expected at + * curiosity domain-rotation boundaries); -2.0 = no centroid yet or + * embedder down. The drift gauge for the context-centroid mechanism. */ + snprintf(buf, sizeof(buf), + "{\"wm_evicted\":%lld,\"breakthroughs\":%lld," + "\"embed_breaker_open\":%d,\"embed_consec_fail\":%d," + "\"ctx_cos\":%.3f," + "\"hebb_edges\":%lld,\"hebb_max\":%.4f,\"hebb_mass\":%.3f," + "\"hebb_cands\":%d,\"hebb_cand_max\":%.4f,\"hebb_links\":%lld," + "\"hebb_warm\":%d," + "\"hebb_wb_pending\":%d,\"hebb_wb_drained\":%lld," + "\"hebb_wb_dropped\":%lld," + "\"dup_seeds\":%lld,\"dup_wm\":%lld,\"dup_wm_global\":%lld," + /* txt_damaged: nodes created THIS process whose content carries + * the character-loss signature. Steady 0 is the healthy state; + * any climb means a write path is mangling text again. Cheap + * (counted at creation) — the full census lives in + * engram_text_health_json. (2026-08-08 self-review) */ + "\"txt_damaged\":%lld}", + (long long)_eg_act_wm_evicted, + (long long)_eg_act_breakthroughs, + breaker_open, _eg_embed_consec_fail, + _eg_act_ctx_cos, + (long long)hebb_edges, hebb_max, hebb_mass, + hebb_cands, hebb_cand_max, (long long)_eg_hebb_links_formed, + _eg_act_hebb_warm, + _eg_hebb_wb_len, (long long)_eg_hebb_wb_drained, + (long long)_eg_hebb_wb_dropped, + (long long)_eg_act_dup_seeds, (long long)_eg_act_dup_wm, + (long long)_eg_act_dup_wm_global, + (long long)_eg_txt_write_damaged); + return el_wrap_str(el_strdup(buf)); +} + +/* engram_hebb_drain_json — pop up to `max` newly-formed Hebbian associations + * off the write-back queue and return them as a JSON array: + * + * [{"from_id":"...","to_id":"...","weight":0.15,"hebb":0.31}, ...] + * + * Draining is DESTRUCTIVE: entries returned here are gone from the queue. The + * caller owns delivery from that point on. That is deliberate — the alternative + * (peek, deliver, ack) needs a second round trip and a retry ledger to be + * correct, and the payload is an association that will re-form from live + * co-activation if it genuinely matters. Losing one is cheap; a queue that + * silently refills forever because acks never land is not. + * + * Empty queue returns "[]". See the ENGRAM_HEBB_WB_SLOTS block for why this + * exists at all: the process that does the learning is not the process that + * owns persistence. (2026-08-07 self-review.) */ +el_val_t engram_hebb_drain_json(el_val_t max_v) { + int64_t max_n = (int64_t)max_v; + if (max_n <= 0) max_n = 64; + if (max_n > ENGRAM_HEBB_WB_SLOTS) max_n = ENGRAM_HEBB_WB_SLOTS; + JsonBuf b; jb_init(&b); + jb_puts(&b, "["); + int emitted = 0; + while (_eg_hebb_wb_len > 0 && emitted < (int)max_n) { + EgHebbWB* e = &_eg_hebb_wb[_eg_hebb_wb_head]; + if (e->a && e->b) { + if (emitted > 0) jb_puts(&b, ","); + /* relation is emitted here, not stamped on by the caller: the + * payload should be postable to /api/edges/batch verbatim. A + * consumer that has to rewrite the JSON to make it valid is a + * consumer that will eventually rewrite it wrong. */ + jb_puts(&b, "{\"relation\":\"hebbian-associate\",\"from_id\":"); + jb_emit_escaped(&b, e->a); + jb_puts(&b, ",\"to_id\":"); + jb_emit_escaped(&b, e->b); + char tmp[96]; + snprintf(tmp, sizeof(tmp), ",\"weight\":%.6g,\"hebb\":%.6g}", + e->w, e->hebb); + jb_puts(&b, tmp); + emitted++; + _eg_hebb_wb_drained++; + } + /* Free and advance whether or not the entry rendered — a NULL id is a + * strdup failure at push time, not a retryable condition. */ + free(e->a); free(e->b); + e->a = NULL; e->b = NULL; + _eg_hebb_wb_head = (_eg_hebb_wb_head + 1) % ENGRAM_HEBB_WB_SLOTS; + _eg_hebb_wb_len--; + } + jb_puts(&b, "]"); + return el_wrap_str(b.buf ? b.buf : el_strdup("[]")); +} + +/* engram_cosine_sim — cosine similarity between two nodes' embeddings. + * Returns a float in [-1, 1], or -2.0 when either node is missing or not + * yet embedded. Exposed so EL code (and the introspection API) can probe + * semantic distance directly. (2026-07-24, bl-b2d1c944) */ +el_val_t engram_cosine_sim(el_val_t id_a, el_val_t id_b) { + EngramStore* g = engram_get(); + int64_t ia = engram_find_node_index(EL_CSTR(id_a)); + int64_t ib = engram_find_node_index(EL_CSTR(id_b)); + if (ia < 0 || ib < 0) return el_from_float(-2.0); + EngramNode* a = &g->nodes[ia]; + EngramNode* b = &g->nodes[ib]; + if (!a->emb || !b->emb || a->emb_dim != b->emb_dim) + return el_from_float(-2.0); + return el_from_float(eg_cosine(a->emb, b->emb, a->emb_dim)); +} + +/* engram_label_df — document frequency of `term` across node LABELS. + * Returns the count of nodes whose label contains term (case-insensitive), + * or the total node count for an empty/NULL term so callers treat "no term" + * as maximally unspecific (i.e. reject it). + * + * WHY THIS EXISTS (2026-08-03 self-review). The soul's auto-term extractor + * (awareness.el:auto_term_try_slot) picks a curiosity seed by taking the + * FIRST WORD of a top-WM node label. A first-word extractor has no notion of + * term quality, so three consecutive self-reviews each bolted another + * hand-curated blocklist onto it — genre words (07-23), quoted titles + * (07-25), English stopwords (07-30). Every one of those was written + * REACTIVELY, after observing a flood in the live ISE stream. The mechanism + * is whack-a-mole: the list can only ever contain floods that already + * happened. + * + * Measured live on this store (13,370 nodes) while two unanticipated floods + * were in flight and unfixed: + * "