Compare commits
86 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b7e2c580a8 | |||
| d71fc4c1c0 | |||
| a118d19393 | |||
| c6aa1e5c53 | |||
| ff577391f2 | |||
| ee0d5f9b97 | |||
| 391bd818ea | |||
| 43636aed99 | |||
| 8f8ccc945e | |||
| 409ec99397 | |||
| dc39a61e2c | |||
| eba9eac8a8 | |||
| ab6b52a0b4 | |||
| 2baa0b9a41 | |||
| 6a8b2461cd | |||
| bcb356fe69 | |||
| dd7827059a | |||
| 208e36c899 | |||
| b97ce74d1f | |||
| 155a449c4e | |||
| 4696fd6833 | |||
| 581a351fb1 | |||
| 8ce8656de2 | |||
| 1e49560f1f | |||
| e8f0b5a9de | |||
| 40287c4cfc | |||
| 0481bea44d | |||
| 9d565ca080 | |||
| 4773dd0aa2 | |||
| 6b9d9e6c4a | |||
| b4967af13e | |||
| e3dabe3e08 | |||
| 0a0a2bcb44 | |||
| 2b2a1246e7 | |||
| f78da81aa4 | |||
| 2597a092bb | |||
| 226b798407 | |||
| cfe8cb1c80 | |||
| 688b8508fb | |||
| 59cea116c5 | |||
| deb0520551 | |||
| da116b2884 | |||
| 58753a88d7 | |||
| edff25180e | |||
| 6271cb42b2 | |||
| 3da9181deb | |||
| 192241c7c1 | |||
| e7c2dc7734 | |||
| f7bd99ae45 | |||
| 5c41c66a0f | |||
| 93d36fddb1 | |||
| 2d751890ea | |||
| 99b113ea9d | |||
| c087b97093 | |||
| 718a2e0c06 | |||
| b6187501fd | |||
| 18e1ab6db1 | |||
| 2dec76c87a | |||
| 35c189759c | |||
| 5c94b8680d | |||
| cebf3ded62 | |||
| b83ecf52f9 | |||
| 15ea584671 | |||
| dbf2c659d9 | |||
| 2b8062c55f | |||
| a390ee494e | |||
| c2cd5e01e1 | |||
| 8212e12e57 | |||
| 253ee2b887 | |||
| d7540700d4 | |||
| f103e85f88 | |||
| fe84639b17 | |||
| 5fdc9fb15e | |||
| 8967fa404e | |||
| a7e6fbf2d2 | |||
| 1f4b594ae7 | |||
| cff7ce072d | |||
| f5dcca0386 | |||
| 53e0b99d5f | |||
| 5f9cad5908 | |||
| 00629b39c4 | |||
| ca1e4d57b8 | |||
| f971e96dd5 | |||
| 81a1a624f1 | |||
| 7b7f9f353b | |||
| a3732a1e9a |
@@ -22,7 +22,10 @@ jobs:
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
apt-get update -qq
|
||||
apt-get install -y gcc libcurl4-openssl-dev
|
||||
apt-get install -y gcc libcurl4-openssl-dev apt-transport-https ca-certificates
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" \
|
||||
> /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
|
||||
# Seed: use the committed linux-amd64 binary as the bootstrap
|
||||
- name: Bootstrap from committed linux binary (seed)
|
||||
@@ -84,13 +87,22 @@ jobs:
|
||||
bash tests/html_sanitizer/run.sh
|
||||
|
||||
# Native El test suites (elc --test, compile-link-run)
|
||||
# el_runtime.c is precompiled to .o once and reused by all 8 modules.
|
||||
- name: Precompile el_runtime.o
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
gcc -O2 -c -I "$RUNTIME" "$RUNTIME/el_runtime.c" \
|
||||
-o /tmp/el_runtime.o
|
||||
echo "el_runtime.o compiled"
|
||||
|
||||
- name: Run tests - native (core)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_core.el > /tmp/el_native_core.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_core.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_core
|
||||
/tmp/el_native_core
|
||||
|
||||
@@ -100,7 +112,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_text.el > /tmp/el_native_text.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_text.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_text
|
||||
/tmp/el_native_text
|
||||
|
||||
@@ -110,7 +122,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_string.el > /tmp/el_native_string.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_string.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_string
|
||||
/tmp/el_native_string
|
||||
|
||||
@@ -120,7 +132,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_math.el > /tmp/el_native_math.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_math.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_math
|
||||
/tmp/el_native_math
|
||||
|
||||
@@ -130,7 +142,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_state.el > /tmp/el_native_state.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_state.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_state
|
||||
/tmp/el_native_state
|
||||
|
||||
@@ -140,7 +152,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_time.el > /tmp/el_native_time.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_time.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_time
|
||||
/tmp/el_native_time
|
||||
|
||||
@@ -150,7 +162,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_json.el > /tmp/el_native_json.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_json.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_json
|
||||
/tmp/el_native_json
|
||||
|
||||
@@ -160,7 +172,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_env.el > /tmp/el_native_env.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_env.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_env
|
||||
/tmp/el_native_env
|
||||
|
||||
@@ -170,7 +182,7 @@ jobs:
|
||||
ELC="$(pwd)/dist/platform/elc"
|
||||
RUNTIME="$(pwd)/el-compiler/runtime"
|
||||
"$ELC" --test tests/native/test_fs.el > /tmp/el_native_fs.c
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c "$RUNTIME/el_runtime.c" \
|
||||
gcc -O2 -I "$RUNTIME" /tmp/el_native_fs.c /tmp/el_runtime.o \
|
||||
-lcurl -lssl -lcrypto -lpthread -lm -o /tmp/el_native_fs
|
||||
/tmp/el_native_fs
|
||||
|
||||
@@ -202,12 +214,18 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -252,4 +270,58 @@ jobs:
|
||||
--source=el-compiler/runtime/el_runtime.js
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-dev"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
# (deleted in that step after docker push)
|
||||
|
||||
- name: Rebuild ci-base with fresh El SDK (dev)
|
||||
# Patches ci-base:dev in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
CI_BASE="us-central1-docker.pkg.dev/neuron-785695/neuron-ci/ci-base"
|
||||
SHA="${GITHUB_SHA:0:8}"
|
||||
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
gcloud auth configure-docker us-central1-docker.pkg.dev --quiet
|
||||
|
||||
# Pull existing ci-base:dev (or fall back to :latest on first run)
|
||||
BASE_TAG="dev"
|
||||
docker pull "${CI_BASE}:dev" || { docker pull "${CI_BASE}:latest" && BASE_TAG="latest"; }
|
||||
|
||||
# Inline Dockerfile — only replaces the El SDK layer
|
||||
cat > /tmp/Dockerfile.ci-base-patch << 'EOF'
|
||||
ARG BASE
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
docker build \
|
||||
--build-arg BASE="${CI_BASE}:${BASE_TAG}" \
|
||||
--build-arg BUILDKIT_INLINE_CACHE=1 \
|
||||
-f /tmp/Dockerfile.ci-base-patch \
|
||||
-t "${CI_BASE}:dev" \
|
||||
-t "${CI_BASE}:dev-${SHA}" \
|
||||
.
|
||||
|
||||
docker push "${CI_BASE}:dev"
|
||||
docker push "${CI_BASE}:dev-${SHA}"
|
||||
|
||||
echo "ci-base rebuilt: ${CI_BASE}:dev (${SHA})"
|
||||
rm -f /tmp/gcp-key.json
|
||||
|
||||
@@ -212,12 +212,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -246,4 +255,57 @@ jobs:
|
||||
--source=el-compiler/runtime/el_runtime.h
|
||||
|
||||
echo "Published El SDK version=${VERSION} to foundation-stage"
|
||||
# Keep key alive for the ci-base rebuild step below
|
||||
# (deleted in that step after docker push)
|
||||
|
||||
- name: Rebuild ci-base with fresh El SDK (stage)
|
||||
# Patches ci-base:stage in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
CI_BASE="us-central1-docker.pkg.dev/neuron-785695/neuron-ci/ci-base"
|
||||
SHA="${GITHUB_SHA:0:8}"
|
||||
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
gcloud auth configure-docker us-central1-docker.pkg.dev --quiet
|
||||
|
||||
# Pull existing ci-base:stage (system deps stay cached in the base layer)
|
||||
docker pull "${CI_BASE}:stage" || docker pull "${CI_BASE}:latest"
|
||||
|
||||
# Inline Dockerfile — only replaces the El SDK layer
|
||||
cat > /tmp/Dockerfile.ci-base-patch << 'EOF'
|
||||
ARG BASE
|
||||
FROM ${BASE}
|
||||
COPY dist/platform/elc /opt/el/dist/platform/elc
|
||||
COPY dist/bin/elb /opt/el/dist/bin/elb
|
||||
COPY el-compiler/runtime/el_runtime.c /opt/el/el-compiler/runtime/el_runtime.c
|
||||
COPY el-compiler/runtime/el_runtime.h /opt/el/el-compiler/runtime/el_runtime.h
|
||||
COPY el-compiler/runtime/el_runtime.js /opt/el/el-compiler/runtime/el_runtime.js
|
||||
RUN chmod +x /opt/el/dist/platform/elc /opt/el/dist/bin/elb
|
||||
EOF
|
||||
|
||||
docker build \
|
||||
--build-arg BASE="${CI_BASE}:stage" \
|
||||
--build-arg BUILDKIT_INLINE_CACHE=1 \
|
||||
-f /tmp/Dockerfile.ci-base-patch \
|
||||
-t "${CI_BASE}:stage" \
|
||||
-t "${CI_BASE}:stage-${SHA}" \
|
||||
.
|
||||
|
||||
docker push "${CI_BASE}:stage"
|
||||
docker push "${CI_BASE}:stage-${SHA}"
|
||||
|
||||
echo "ci-base rebuilt: ${CI_BASE}:stage (${SHA})"
|
||||
rm -f /tmp/gcp-key.json
|
||||
|
||||
@@ -288,12 +288,21 @@ jobs:
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
run: |
|
||||
# Fail loudly: previously this step had no `set -e`, so an auth or
|
||||
# upload failure was swallowed (step exited 0 on the trailing echo)
|
||||
# and the SDK silently never published. Surface failures now.
|
||||
set -euo pipefail
|
||||
if [ -z "${GCP_SA_KEY:-}" ]; then
|
||||
echo "FATAL: GCP_SA_KEY secret is empty — cannot authenticate to publish" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "${GCP_SA_KEY}" > /tmp/gcp-key.json
|
||||
apt-get install -y -qq apt-transport-https ca-certificates curl
|
||||
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" > /etc/apt/sources.list.d/google-cloud-sdk.list
|
||||
apt-get update -qq && apt-get install -y google-cloud-cli
|
||||
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
|
||||
gcloud config set project neuron-785695
|
||||
echo "Publishing as active account: $(gcloud config get-value account 2>/dev/null)"
|
||||
|
||||
VERSION="${GITHUB_SHA:0:8}"
|
||||
|
||||
@@ -345,6 +354,12 @@ jobs:
|
||||
# Patches ci-base:latest in-place: pulls the existing image (which has all
|
||||
# system deps — Node, Go, gcloud, Docker CLI, etc.) and overlays the freshly
|
||||
# built El SDK on top. Keeps the full ci-base rebuild fast and incremental.
|
||||
#
|
||||
# continue-on-error: this is a CI-cache optimization, NOT the release
|
||||
# artifact. It runs Docker (pull/build/push ~600MB) on the host-mode GCE
|
||||
# runner where DinD/Docker availability is fragile. A failure here must
|
||||
# never block or redden the job — the SDK publish above is the deliverable.
|
||||
continue-on-error: true
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,23 @@
|
||||
{
|
||||
"dataset": "british-rp-accent-transform",
|
||||
"primitive_type": "accent_target",
|
||||
"accent": "british-rp",
|
||||
"grounding": "derived",
|
||||
"provenance": "HONEST-DERIVED, COARSE FIRST PASS — NOT transcribed measured RP formants. The exact measured RP/GB tables (Deterding 1997 JIPA 27:47-55; Hawkins & Midgley 2005 JIPA 35:183-199) are the intended ground truth but were gated/figure-only at author time and were NOT transcribed. So these targets are DERIVED: each = the corresponding MEASURED Peterson&Barney(1952) base vowel transformed under the documented, citable RP-vs-GA structural rules of Wells (1982) 'Accents of English' — non-rhoticity (NURSE de-rhoticized: remove low F3), TRAP F2-lowering, LOT/THOUGHT back-rounding (F2 down), GOOSE-fronting (F2 up), GOAT centering. Shift MAGNITUDES are coarse/approximate (first pass), directions are cited. ground:derived (base measured + rule cited). Refine by transcribing Deterding/Hawkins&Midgley. No number is presented as a measured RP value it is not.",
|
||||
"notes": "records with kind=vowel_override REPLACE the base phoneme's formant targets with the DERIVED RP realization. records with kind=rule encode non-formant transforms (non-rhoticity: drop post-vocalic coda /r/). The render composes: base geometry then accent override + rhoticity rule — voice + accent, separable.",
|
||||
"records": [
|
||||
{"key": "IY", "features": {"kind": "vowel_override", "set": "FLEECE"}, "attributes": {"f1": 280, "f2": 2249, "f3": 3000}},
|
||||
{"key": "IH", "features": {"kind": "vowel_override", "set": "KIT"}, "attributes": {"f1": 360, "f2": 2100, "f3": 2550}},
|
||||
{"key": "EH", "features": {"kind": "vowel_override", "set": "DRESS"}, "attributes": {"f1": 560, "f2": 1970, "f3": 2480}},
|
||||
{"key": "AE", "features": {"kind": "vowel_override", "set": "TRAP"}, "attributes": {"f1": 730, "f2": 1590, "f3": 2410}},
|
||||
{"key": "AA", "features": {"kind": "vowel_override", "set": "LOT"}, "attributes": {"f1": 560, "f2": 920, "f3": 2440}},
|
||||
{"key": "AO", "features": {"kind": "vowel_override", "set": "THOUGHT"}, "attributes": {"f1": 415, "f2": 700, "f3": 2410}},
|
||||
{"key": "UH", "features": {"kind": "vowel_override", "set": "FOOT"}, "attributes": {"f1": 380, "f2": 1100, "f3": 2240}},
|
||||
{"key": "UW", "features": {"kind": "vowel_override", "set": "GOOSE"}, "attributes": {"f1": 310, "f2": 1650, "f3": 2240}},
|
||||
{"key": "AH", "features": {"kind": "vowel_override", "set": "STRUT"}, "attributes": {"f1": 680, "f2": 1180, "f3": 2390}},
|
||||
{"key": "ER", "features": {"kind": "vowel_override", "set": "NURSE", "rhotic": "no"}, "attributes": {"f1": 550, "f2": 1500, "f3": 2500}},
|
||||
{"key": "AX", "features": {"kind": "vowel_override", "set": "commA"}, "attributes": {"f1": 500, "f2": 1500, "f3": 2500}},
|
||||
{"key": "OW", "features": {"kind": "vowel_override", "set": "GOAT"}, "attributes": {"f1": 450, "f2": 1400, "f3": 2380}},
|
||||
{"key": "R", "features": {"kind": "rule", "rule": "non_rhotic"}, "attributes": {"drop_coda_r": 1}}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
# british-rp-accent TRANSFORM — INGESTIBLE DATA (a geometry/transform composed
|
||||
# onto the base General-American phoneme targets; voice + accent, separable).
|
||||
#
|
||||
# PROVENANCE — HONEST, COARSE FIRST PASS. These are DERIVED targets, NOT
|
||||
# transcribed measured RP formants. Measured RP tables (Deterding 1997 JIPA 27;
|
||||
# Hawkins & Midgley 2005 JIPA 35) are the intended ground truth but were gated at
|
||||
# author time and NOT transcribed. Each target = the MEASURED Peterson&Barney
|
||||
# (1952) base vowel transformed under the documented, citable RP-vs-GA structural
|
||||
# rules of Wells (1982): non-rhoticity, TRAP F2-lowering, LOT/THOUGHT back-
|
||||
# rounding, GOOSE-fronting, GOAT centering, NURSE de-rhoticization. Shift
|
||||
# magnitudes are coarse/approximate; directions are cited. ground=derived.
|
||||
# Refine by transcribing the measured RP tables. No value is claimed as measured.
|
||||
# Format: KEY|F1|F2|F3|KIND|SET
|
||||
IY|280|2249|3000|vowel_override|FLEECE
|
||||
IH|360|2100|2550|vowel_override|KIT
|
||||
EH|560|1970|2480|vowel_override|DRESS
|
||||
AE|730|1590|2410|vowel_override|TRAP
|
||||
AA|560|920|2440|vowel_override|LOT
|
||||
AO|415|700|2410|vowel_override|THOUGHT
|
||||
UH|380|1100|2240|vowel_override|FOOT
|
||||
UW|310|1650|2240|vowel_override|GOOSE
|
||||
AH|680|1180|2390|vowel_override|STRUT
|
||||
ER|550|1500|2500|vowel_override|NURSE-nonrhotic
|
||||
AX|500|1500|2500|vowel_override|commA
|
||||
OW|450|1400|2380|vowel_override|GOAT
|
||||
R|0|0|0|rule|non_rhotic_drop_coda
|
||||
@@ -0,0 +1,20 @@
|
||||
# pronunciation lexicon SOURCE — word -> phoneme sequence, as INGESTIBLE DATA.
|
||||
# Pronunciation is linguistic KNOWLEDGE (the language faculty's orthography->
|
||||
# phonology map), ingested into the engram, not frozen in code. The render reads
|
||||
# a word's phoneme sequence back from the engram. Covers the self-lexicon and the
|
||||
# proof sentences; general G2P is the realizer/morphology faculty's remit.
|
||||
# Diphthongs are written as two vowel targets (the render's transitions glide
|
||||
# between them). Format: word|PH1 PH2 PH3 ...
|
||||
i|AA IY
|
||||
am|AE M
|
||||
neuron|N UW R AA N
|
||||
is|IH Z
|
||||
memory|M EH M ER IY
|
||||
hello|HH EH L OW
|
||||
the|DH AH
|
||||
a|AH
|
||||
remember|R IH M EH M ER
|
||||
i'm|AA IY M
|
||||
you|Y UW
|
||||
here|HH IY R
|
||||
will|W IH L
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,528 @@
|
||||
{
|
||||
"dataset": "english-phoneme-formants",
|
||||
"primitive_type": "phoneme",
|
||||
"grounding": "extracted",
|
||||
"provenance": "AUDITED per-field. The 10 monophthong-vowel F1/F2/F3 (IY,IH,EH,AE,AA,AO,UH,UW,AH,ER) are the MEASURED adult-male /hVd/ means of Peterson & Barney (1952) JASA 24:175-184, verified vs CRAN phonTools::pb52. AX=neutral uniform-tube resonances (Fant, physics). OW steady target = synthesis convention (diphthong). Consonant loci (M,N,NG,L,R,W,Y,Z,DH,V,S,F,HH) and ALL bandwidths + dur/amp = standard formant-synthesis conventions (Klatt 1980 JASA 67:971), engineering defaults NOT field measurements. No numbers invented/LLM-generated.",
|
||||
"records": [
|
||||
{
|
||||
"key": "IY",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 270,
|
||||
"f2": 2290,
|
||||
"f3": 3010,
|
||||
"bw1": 60,
|
||||
"bw2": 90,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 130,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "IH",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 390,
|
||||
"f2": 1990,
|
||||
"f3": 2550,
|
||||
"bw1": 70,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 110,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "EH",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 530,
|
||||
"f2": 1840,
|
||||
"f3": 2480,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 130,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "AE",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 660,
|
||||
"f2": 1720,
|
||||
"f3": 2410,
|
||||
"bw1": 90,
|
||||
"bw2": 110,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 150,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "AA",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 730,
|
||||
"f2": 1090,
|
||||
"f3": 2440,
|
||||
"bw1": 90,
|
||||
"bw2": 110,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 150,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "AO",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 570,
|
||||
"f2": 840,
|
||||
"f3": 2410,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 140,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "UH",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 440,
|
||||
"f2": 1020,
|
||||
"f3": 2240,
|
||||
"bw1": 70,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 110,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "UW",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 870,
|
||||
"f3": 2240,
|
||||
"bw1": 70,
|
||||
"bw2": 90,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 140,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "AH",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 640,
|
||||
"f2": 1190,
|
||||
"f3": 2390,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 110,
|
||||
"amp": 95
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ER",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 490,
|
||||
"f2": 1350,
|
||||
"f3": 1690,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 120,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 140,
|
||||
"amp": 95
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "AX",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 500,
|
||||
"f2": 1500,
|
||||
"f3": 2500,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 80,
|
||||
"amp": 85
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "OW",
|
||||
"features": {
|
||||
"manner": "vowel",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 490,
|
||||
"f2": 910,
|
||||
"f3": 2380,
|
||||
"bw1": 80,
|
||||
"bw2": 100,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 140,
|
||||
"amp": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "M",
|
||||
"features": {
|
||||
"manner": "nasal",
|
||||
"voiced": "yes",
|
||||
"nasal": "yes"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 250,
|
||||
"f2": 900,
|
||||
"f3": 2200,
|
||||
"bw1": 90,
|
||||
"bw2": 120,
|
||||
"bw3": 180,
|
||||
"voiced": 1,
|
||||
"nasal": 1,
|
||||
"dur": 80,
|
||||
"amp": 60
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "N",
|
||||
"features": {
|
||||
"manner": "nasal",
|
||||
"voiced": "yes",
|
||||
"nasal": "yes"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 250,
|
||||
"f2": 1700,
|
||||
"f3": 2600,
|
||||
"bw1": 90,
|
||||
"bw2": 120,
|
||||
"bw3": 180,
|
||||
"voiced": 1,
|
||||
"nasal": 1,
|
||||
"dur": 80,
|
||||
"amp": 60
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NG",
|
||||
"features": {
|
||||
"manner": "nasal",
|
||||
"voiced": "yes",
|
||||
"nasal": "yes"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 250,
|
||||
"f2": 2300,
|
||||
"f3": 2700,
|
||||
"bw1": 90,
|
||||
"bw2": 120,
|
||||
"bw3": 180,
|
||||
"voiced": 1,
|
||||
"nasal": 1,
|
||||
"dur": 80,
|
||||
"amp": 60
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "L",
|
||||
"features": {
|
||||
"manner": "approximant",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 360,
|
||||
"f2": 1300,
|
||||
"f3": 2600,
|
||||
"bw1": 80,
|
||||
"bw2": 110,
|
||||
"bw3": 160,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 70,
|
||||
"amp": 80
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "R",
|
||||
"features": {
|
||||
"manner": "approximant",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 490,
|
||||
"f2": 1350,
|
||||
"f3": 1600,
|
||||
"bw1": 80,
|
||||
"bw2": 110,
|
||||
"bw3": 120,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 80,
|
||||
"amp": 85
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "W",
|
||||
"features": {
|
||||
"manner": "approximant",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 610,
|
||||
"f3": 2200,
|
||||
"bw1": 70,
|
||||
"bw2": 100,
|
||||
"bw3": 160,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 70,
|
||||
"amp": 80
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "Y",
|
||||
"features": {
|
||||
"manner": "approximant",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 270,
|
||||
"f2": 2290,
|
||||
"f3": 3010,
|
||||
"bw1": 60,
|
||||
"bw2": 90,
|
||||
"bw3": 150,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 60,
|
||||
"amp": 80
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "Z",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 1700,
|
||||
"f3": 2500,
|
||||
"bw1": 100,
|
||||
"bw2": 150,
|
||||
"bw3": 200,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 90,
|
||||
"amp": 55
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "DH",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 1400,
|
||||
"f3": 2500,
|
||||
"bw1": 100,
|
||||
"bw2": 150,
|
||||
"bw3": 200,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 70,
|
||||
"amp": 55
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "V",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "yes",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 1000,
|
||||
"f3": 2300,
|
||||
"bw1": 100,
|
||||
"bw2": 150,
|
||||
"bw3": 200,
|
||||
"voiced": 1,
|
||||
"nasal": 0,
|
||||
"dur": 70,
|
||||
"amp": 55
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "S",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "no",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 320,
|
||||
"f2": 1700,
|
||||
"f3": 2500,
|
||||
"bw1": 200,
|
||||
"bw2": 200,
|
||||
"bw3": 250,
|
||||
"voiced": 0,
|
||||
"nasal": 0,
|
||||
"dur": 110,
|
||||
"amp": 45
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "F",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "no",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 300,
|
||||
"f2": 1200,
|
||||
"f3": 2400,
|
||||
"bw1": 200,
|
||||
"bw2": 200,
|
||||
"bw3": 250,
|
||||
"voiced": 0,
|
||||
"nasal": 0,
|
||||
"dur": 100,
|
||||
"amp": 40
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "HH",
|
||||
"features": {
|
||||
"manner": "fricative",
|
||||
"voiced": "no",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 500,
|
||||
"f2": 1500,
|
||||
"f3": 2500,
|
||||
"bw1": 200,
|
||||
"bw2": 250,
|
||||
"bw3": 300,
|
||||
"voiced": 0,
|
||||
"nasal": 0,
|
||||
"dur": 70,
|
||||
"amp": 40
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SIL",
|
||||
"features": {
|
||||
"manner": "silence",
|
||||
"voiced": "no",
|
||||
"nasal": "no"
|
||||
},
|
||||
"attributes": {
|
||||
"f1": 500,
|
||||
"f2": 1500,
|
||||
"f3": 2500,
|
||||
"bw1": 100,
|
||||
"bw2": 100,
|
||||
"bw3": 100,
|
||||
"voiced": 0,
|
||||
"nasal": 0,
|
||||
"dur": 55,
|
||||
"amp": 0
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
# acoustic-phonetics SOURCE — the learned speech primitives, as INGESTIBLE DATA.
|
||||
# NOT audio, NOT code: formant geometry of the phonemes, to be ingested via the
|
||||
# ingest organ into the engram as a phoneme manifold. The render reads this
|
||||
# geometry back from the engram; nothing is frozen in EL code.
|
||||
#
|
||||
# PROVENANCE (audited, per-field honesty — no invented numbers):
|
||||
# * The 10 MONOPHTHONG VOWEL formants F1/F2/F3 (IY,IH,EH,AE,AA,AO,UH,UW,AH,ER)
|
||||
# are the MEASURED adult-male means of Peterson & Barney (1952), JASA 24:175-184
|
||||
# — the canonical /hVd/ table, verified digit-for-digit vs CRAN phonTools::pb52.
|
||||
# These are real measured values.
|
||||
# * AX (schwa) F1/F2/F3 = neutral uniform-tube resonances (2n-1)*500 — a PHYSICS
|
||||
# value (Fant), not a P&B measurement.
|
||||
# * OW is a diphthong; its listed steady target is a conventional synthesis value,
|
||||
# not a P&B monophthong measurement.
|
||||
# * CONSONANT loci (M,N,NG,L,R,W,Y,Z,DH,V,S,F,HH) and ALL BANDWIDTHS (B1,B2,B3)
|
||||
# and dur/amp are STANDARD FORMANT-SYNTHESIS conventions (Klatt 1980, JASA 67:971
|
||||
# "Software for a cascade/parallel formant synthesizer") — engineering defaults,
|
||||
# NOT per-phoneme field measurements. Labeled as such, not attributed to P&B.
|
||||
# Format: SYM|F1|F2|F3|B1|B2|B3|voiced|nasal|dur_ms|amp|class|example
|
||||
IY|270|2290|3010|60|90|150|1|0|130|100|vowel|beet
|
||||
IH|390|1990|2550|70|100|150|1|0|110|100|vowel|bit
|
||||
EH|530|1840|2480|80|100|150|1|0|130|100|vowel|bet
|
||||
AE|660|1720|2410|90|110|150|1|0|150|100|vowel|bat
|
||||
AA|730|1090|2440|90|110|150|1|0|150|100|vowel|bot
|
||||
AO|570|840|2410|80|100|150|1|0|140|100|vowel|bought
|
||||
UH|440|1020|2240|70|100|150|1|0|110|100|vowel|book
|
||||
UW|300|870|2240|70|90|150|1|0|140|100|vowel|boot
|
||||
AH|640|1190|2390|80|100|150|1|0|110|95|vowel|but
|
||||
ER|490|1350|1690|80|100|120|1|0|140|95|vowel|bird
|
||||
AX|500|1500|2500|80|100|150|1|0|80|85|vowel|about
|
||||
OW|490|910|2380|80|100|150|1|0|140|100|vowel|boat
|
||||
M|250|900|2200|90|120|180|1|1|80|60|nasal|map
|
||||
N|250|1700|2600|90|120|180|1|1|80|60|nasal|nap
|
||||
NG|250|2300|2700|90|120|180|1|1|80|60|nasal|sing
|
||||
L|360|1300|2600|80|110|160|1|0|70|80|approximant|lip
|
||||
R|490|1350|1600|80|110|120|1|0|80|85|approximant|rip
|
||||
W|300|610|2200|70|100|160|1|0|70|80|approximant|wet
|
||||
Y|270|2290|3010|60|90|150|1|0|60|80|approximant|yet
|
||||
Z|300|1700|2500|100|150|200|1|0|90|55|fricative|zoo
|
||||
DH|300|1400|2500|100|150|200|1|0|70|55|fricative|the
|
||||
V|300|1000|2300|100|150|200|1|0|70|55|fricative|van
|
||||
S|320|1700|2500|200|200|250|0|0|110|45|fricative|see
|
||||
F|300|1200|2400|200|200|250|0|0|100|40|fricative|fee
|
||||
HH|500|1500|2500|200|250|300|0|0|70|40|fricative|hat
|
||||
SIL|500|1500|2500|100|100|100|0|0|55|0|silence|_
|
||||
Binary file not shown.
@@ -0,0 +1,136 @@
|
||||
// accent.el - A British-RP ACCENT as an INGESTED TRANSFORM-GEOMETRY, composed
|
||||
// onto the voice (voice (+) accent, SEPARABLE). Reads elp/data/british-accent.psv
|
||||
// into an accent MANIFOLD in the engram (override nodes + a shared accent hub),
|
||||
// and the render reads the RP formant overrides + the non-rhotic rule back from
|
||||
// that geometry. NO accent targets live in code — same discipline as the base
|
||||
// phonetics. PROVENANCE NOTE: the RP Hz values are PROVISIONAL (reconstructed-
|
||||
// from-knowledge approximations, cite Deterding1997 / Hawkins&Midgley2005 /
|
||||
// Wells1982) pending transcription from the published tables — the PIPELINE is
|
||||
// the deliverable; exact values are being source-verified separately.
|
||||
|
||||
fn ingest_accent(path: String) -> [String] {
|
||||
let content: String = fs_read(path)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let amap: [String] = native_list_empty()
|
||||
let hub: String = engram_node("accent british-rp prov=PROVISIONAL cite=Deterding1997-HawkinsMidgley2005-Wells1982", "Accent", 80)
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ll: Int = str_len(line)
|
||||
let skip: Int = 0
|
||||
if ll < 3 {
|
||||
skip = 1
|
||||
}
|
||||
if skip == 0 {
|
||||
let first: Int = str_char_code(line, 0)
|
||||
if first == 35 {
|
||||
skip = 1
|
||||
}
|
||||
}
|
||||
if skip == 0 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
let nf: Int = native_list_len(f)
|
||||
if nf >= 6 {
|
||||
let key: String = native_list_get(f, 0)
|
||||
let f1: String = native_list_get(f, 1)
|
||||
let f2: String = native_list_get(f, 2)
|
||||
let f3: String = native_list_get(f, 3)
|
||||
let kind: String = native_list_get(f, 4)
|
||||
let set: String = native_list_get(f, 5)
|
||||
let cont: String = "accent british-rp " + key + " f1=" + f1 + " f2=" + f2 + " f3=" + f3 + " kind=" + kind + " set=" + set + " prov=PROVISIONAL cite=Deterding1997-HawkinsMidgley2005-Wells1982"
|
||||
let id: String = engram_node(cont, "AccentTarget", 80)
|
||||
amap = native_list_append(amap, key)
|
||||
amap = native_list_append(amap, cont)
|
||||
engram_connect(id, hub, 80, "of_accent")
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
return amap
|
||||
}
|
||||
|
||||
// RP formant override for a phoneme, read from the accent manifold. Returns
|
||||
// [f1,f2,f3] for a vowel_override record, or an empty list if none / a rule.
|
||||
fn accent_formants(amap: [String], code: String) -> [Int] {
|
||||
let out: [Int] = native_list_empty()
|
||||
let id: String = sp_map_get(amap, code)
|
||||
if str_eq(id, "") {
|
||||
return out
|
||||
}
|
||||
let j: String = id
|
||||
let isrule: Int = str_index_of(j, "drop_coda")
|
||||
if isrule >= 0 {
|
||||
return out
|
||||
}
|
||||
let f1: Int = parse_uint_from(j, "f1=")
|
||||
if f1 <= 0 {
|
||||
return out
|
||||
}
|
||||
let out = native_list_append(out, f1)
|
||||
let out = native_list_append(out, parse_uint_from(j, "f2="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "f3="))
|
||||
return out
|
||||
}
|
||||
|
||||
// Is this accent non-rhotic? (reads the R rule node from the manifold)
|
||||
fn is_nonrhotic(amap: [String]) -> Int {
|
||||
let id: String = sp_map_get(amap, "R")
|
||||
if str_eq(id, "") {
|
||||
return 0
|
||||
}
|
||||
let hit: Int = str_index_of(id, "drop_coda")
|
||||
if hit >= 0 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Is this symbol a vowel? Membership in the vowel-set derived from the phonetics
|
||||
// source's class column (phonological structure — the FORMANT NUMBERS still come
|
||||
// from the organ manifold; this is only the categorical class for the rule).
|
||||
fn is_vowel_sym(vset: [String], sym: String) -> Int {
|
||||
let n: Int = native_list_len(vset)
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
if str_eq(native_list_get(vset, i), sym) {
|
||||
return 1
|
||||
}
|
||||
i = i + 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Non-rhotic transform: drop a post-vocalic CODA /R/ — an R whose next non-SIL
|
||||
// phoneme is NOT a vowel (a consonant, or end of utterance). Keep INTERVOCALIC/
|
||||
// onset R (next non-SIL phoneme is a vowel, e.g. the medial R in N UW R AA N).
|
||||
fn apply_rhoticity(codes: [String], vset: [String]) -> [String] {
|
||||
let n: Int = native_list_len(codes)
|
||||
let out: [String] = native_list_empty()
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let c: String = native_list_get(codes, i)
|
||||
let keep: Int = 1
|
||||
if str_eq(c, "R") {
|
||||
let jx: Int = i + 1
|
||||
let nextv: Int = 0
|
||||
while jx < n {
|
||||
let ncode: String = native_list_get(codes, jx)
|
||||
if str_eq(ncode, "SIL") {
|
||||
jx = jx + 1
|
||||
} else {
|
||||
nextv = is_vowel_sym(vset, ncode)
|
||||
jx = n + 1000
|
||||
}
|
||||
}
|
||||
if nextv == 0 {
|
||||
keep = 0
|
||||
}
|
||||
}
|
||||
if keep == 1 {
|
||||
out = native_list_append(out, c)
|
||||
}
|
||||
i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
+3
-3
@@ -1,7 +1,7 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sem_get(json: String, key: String) -> String
|
||||
extern fn generate_frame(frame: Any) -> String
|
||||
extern fn generate_frame_lang(frame: Any, lang_code: String) -> String
|
||||
extern fn build_form_from_json(semantic_form_json: String, lang_code: String) -> Any
|
||||
extern fn generate_frame(frame: [String]) -> String
|
||||
extern fn generate_frame_lang(frame: [String], lang_code: String) -> String
|
||||
extern fn build_form_from_json(semantic_form_json: String, lang_code: String) -> [String]
|
||||
extern fn generate(semantic_form_json: String) -> String
|
||||
extern fn generate_lang(semantic_form_json: String, lang_code: String) -> String
|
||||
|
||||
+28
-28
@@ -1,22 +1,22 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn slots_get(slots: Any, key: String) -> String
|
||||
extern fn slots_set(slots: Any, key: String, val: String) -> Any
|
||||
extern fn make_slots(k0: String, v0: String) -> Any
|
||||
extern fn make_slots2(k0: String, v0: String, k1: String, v1: String) -> Any
|
||||
extern fn make_slots3(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String) -> Any
|
||||
extern fn make_slots4(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String) -> Any
|
||||
extern fn make_slots5(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String, k4: String, v4: String) -> Any
|
||||
extern fn rule_id(rule: Any) -> String
|
||||
extern fn rule_lhs(rule: Any) -> String
|
||||
extern fn rule_rhs_len(rule: Any) -> Int
|
||||
extern fn rule_rhs(rule: Any, idx: Int) -> String
|
||||
extern fn make_rule(id: String, lhs: String, r0: String) -> Any
|
||||
extern fn make_rule2(id: String, lhs: String, r0: String, r1: String) -> Any
|
||||
extern fn make_rule3(id: String, lhs: String, r0: String, r1: String, r2: String) -> Any
|
||||
extern fn make_rule4(id: String, lhs: String, r0: String, r1: String, r2: String, r3: String) -> Any
|
||||
extern fn build_rules() -> Any
|
||||
extern fn get_rules() -> Any
|
||||
extern fn find_rule(rule_id_str: String) -> Any
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn slots_get(slots: [String], key: String) -> String
|
||||
extern fn slots_set(slots: [String], key: String, val: String) -> [String]
|
||||
extern fn make_slots(k0: String, v0: String) -> [String]
|
||||
extern fn make_slots2(k0: String, v0: String, k1: String, v1: String) -> [String]
|
||||
extern fn make_slots3(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String) -> [String]
|
||||
extern fn make_slots4(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String) -> [String]
|
||||
extern fn make_slots5(k0: String, v0: String, k1: String, v1: String, k2: String, v2: String, k3: String, v3: String, k4: String, v4: String) -> [String]
|
||||
extern fn rule_id(rule: [String]) -> String
|
||||
extern fn rule_lhs(rule: [String]) -> String
|
||||
extern fn rule_rhs_len(rule: [String]) -> Int
|
||||
extern fn rule_rhs(rule: [String], idx: Int) -> String
|
||||
extern fn make_rule(id: String, lhs: String, r0: String) -> [String]
|
||||
extern fn make_rule2(id: String, lhs: String, r0: String, r1: String) -> [String]
|
||||
extern fn make_rule3(id: String, lhs: String, r0: String, r1: String, r2: String) -> [String]
|
||||
extern fn make_rule4(id: String, lhs: String, r0: String, r1: String, r2: String, r3: String) -> [String]
|
||||
extern fn build_rules() -> [[String]]
|
||||
extern fn get_rules() -> [[String]]
|
||||
extern fn find_rule(rule_id_str: String) -> [String]
|
||||
extern fn make_leaf(label: String, word: String) -> String
|
||||
extern fn make_node1(label: String, child0: String) -> String
|
||||
extern fn make_node2(label: String, child0: String, child1: String) -> String
|
||||
@@ -24,15 +24,15 @@ extern fn make_node3(label: String, child0: String, child1: String, child2: Stri
|
||||
extern fn make_node4(label: String, child0: String, child1: String, child2: String, child3: String) -> String
|
||||
extern fn nlg_is_ws(c: String) -> Bool
|
||||
extern fn skip_ws(s: String, pos: Int) -> Int
|
||||
extern fn scan_token(s: String, start: Int) -> Any
|
||||
extern fn scan_token(s: String, start: Int) -> [String]
|
||||
extern fn render_tree(tree: String) -> String
|
||||
extern fn gram_word_order(profile: Any) -> String
|
||||
extern fn gram_order_constituents(subj: String, verb: String, obj: String, profile: Any) -> String
|
||||
extern fn gram_build_vp(verb: String, aux: String, profile: Any) -> String
|
||||
extern fn gram_question_strategy(profile: Any) -> String
|
||||
extern fn gram_word_order(profile: [String]) -> String
|
||||
extern fn gram_order_constituents(subj: String, verb: String, obj: String, profile: [String]) -> String
|
||||
extern fn gram_build_vp(verb: String, aux: String, profile: [String]) -> String
|
||||
extern fn gram_question_strategy(profile: [String]) -> String
|
||||
extern fn is_pronoun(word: String) -> Bool
|
||||
extern fn build_np(referent: String, slots: Any) -> String
|
||||
extern fn build_np(referent: String, slots: [String]) -> String
|
||||
extern fn build_pp(loc: String) -> String
|
||||
extern fn build_vp_body(slots: Any) -> String
|
||||
extern fn build_vp_from_slots(slots: Any) -> String
|
||||
extern fn generate_tree(rule_id_str: String, slots: Any) -> String
|
||||
extern fn build_vp_body(slots: [String]) -> String
|
||||
extern fn build_vp_from_slots(slots: [String]) -> String
|
||||
extern fn generate_tree(rule_id_str: String, slots: [String]) -> String
|
||||
|
||||
@@ -1,46 +1,46 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn lang_profile(code: String, word_order: String, morph_type: String, has_case: String, has_gender: String, script_dir: String, agreement: String, null_subject: String) -> Any
|
||||
extern fn lang_get(profile: Any, key: String) -> String
|
||||
extern fn lang_profile_en() -> Any
|
||||
extern fn lang_profile_ja() -> Any
|
||||
extern fn lang_profile_ar() -> Any
|
||||
extern fn lang_profile_zh() -> Any
|
||||
extern fn lang_profile_de() -> Any
|
||||
extern fn lang_profile_es() -> Any
|
||||
extern fn lang_profile_fi() -> Any
|
||||
extern fn lang_profile_sw() -> Any
|
||||
extern fn lang_profile_hi() -> Any
|
||||
extern fn lang_profile_ru() -> Any
|
||||
extern fn lang_profile_fr() -> Any
|
||||
extern fn lang_profile_la() -> Any
|
||||
extern fn lang_profile_he() -> Any
|
||||
extern fn lang_profile_sa() -> Any
|
||||
extern fn lang_profile_got() -> Any
|
||||
extern fn lang_profile_non() -> Any
|
||||
extern fn lang_profile_enm() -> Any
|
||||
extern fn lang_profile_pi() -> Any
|
||||
extern fn lang_profile_grc() -> Any
|
||||
extern fn lang_profile_ang() -> Any
|
||||
extern fn lang_profile_fro() -> Any
|
||||
extern fn lang_profile_goh() -> Any
|
||||
extern fn lang_profile_sga() -> Any
|
||||
extern fn lang_profile_txb() -> Any
|
||||
extern fn lang_profile_peo() -> Any
|
||||
extern fn lang_profile_akk() -> Any
|
||||
extern fn lang_profile_uga() -> Any
|
||||
extern fn lang_profile_egy() -> Any
|
||||
extern fn lang_profile_sux() -> Any
|
||||
extern fn lang_profile_gez() -> Any
|
||||
extern fn lang_profile_cop() -> Any
|
||||
extern fn lang_from_code(code: String) -> Any
|
||||
extern fn lang_default() -> Any
|
||||
extern fn lang_is_isolating(profile: Any) -> Bool
|
||||
extern fn lang_is_agglutinative(profile: Any) -> Bool
|
||||
extern fn lang_is_fusional(profile: Any) -> Bool
|
||||
extern fn lang_is_polysynthetic(profile: Any) -> Bool
|
||||
extern fn lang_is_rtl(profile: Any) -> Bool
|
||||
extern fn lang_has_null_subject(profile: Any) -> Bool
|
||||
extern fn lang_has_case(profile: Any) -> Bool
|
||||
extern fn lang_has_gender(profile: Any) -> Bool
|
||||
extern fn lang_word_order(profile: Any) -> String
|
||||
extern fn lang_code(profile: Any) -> String
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn lang_profile(code: String, word_order: String, morph_type: String, has_case: String, has_gender: String, script_dir: String, agreement: String, null_subject: String) -> [String]
|
||||
extern fn lang_get(profile: [String], key: String) -> String
|
||||
extern fn lang_profile_en() -> [String]
|
||||
extern fn lang_profile_ja() -> [String]
|
||||
extern fn lang_profile_ar() -> [String]
|
||||
extern fn lang_profile_zh() -> [String]
|
||||
extern fn lang_profile_de() -> [String]
|
||||
extern fn lang_profile_es() -> [String]
|
||||
extern fn lang_profile_fi() -> [String]
|
||||
extern fn lang_profile_sw() -> [String]
|
||||
extern fn lang_profile_hi() -> [String]
|
||||
extern fn lang_profile_ru() -> [String]
|
||||
extern fn lang_profile_fr() -> [String]
|
||||
extern fn lang_profile_la() -> [String]
|
||||
extern fn lang_profile_he() -> [String]
|
||||
extern fn lang_profile_sa() -> [String]
|
||||
extern fn lang_profile_got() -> [String]
|
||||
extern fn lang_profile_non() -> [String]
|
||||
extern fn lang_profile_enm() -> [String]
|
||||
extern fn lang_profile_pi() -> [String]
|
||||
extern fn lang_profile_grc() -> [String]
|
||||
extern fn lang_profile_ang() -> [String]
|
||||
extern fn lang_profile_fro() -> [String]
|
||||
extern fn lang_profile_goh() -> [String]
|
||||
extern fn lang_profile_sga() -> [String]
|
||||
extern fn lang_profile_txb() -> [String]
|
||||
extern fn lang_profile_peo() -> [String]
|
||||
extern fn lang_profile_akk() -> [String]
|
||||
extern fn lang_profile_uga() -> [String]
|
||||
extern fn lang_profile_egy() -> [String]
|
||||
extern fn lang_profile_sux() -> [String]
|
||||
extern fn lang_profile_gez() -> [String]
|
||||
extern fn lang_profile_cop() -> [String]
|
||||
extern fn lang_from_code(code: String) -> [String]
|
||||
extern fn lang_default() -> [String]
|
||||
extern fn lang_is_isolating(profile: [String]) -> Bool
|
||||
extern fn lang_is_agglutinative(profile: [String]) -> Bool
|
||||
extern fn lang_is_fusional(profile: [String]) -> Bool
|
||||
extern fn lang_is_polysynthetic(profile: [String]) -> Bool
|
||||
extern fn lang_is_rtl(profile: [String]) -> Bool
|
||||
extern fn lang_has_null_subject(profile: [String]) -> Bool
|
||||
extern fn lang_has_case(profile: [String]) -> Bool
|
||||
extern fn lang_has_gender(profile: [String]) -> Bool
|
||||
extern fn lang_word_order(profile: [String]) -> String
|
||||
extern fn lang_code(profile: [String]) -> String
|
||||
|
||||
@@ -56,6 +56,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn akk_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn akk_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn akk_str_len(s: String) -> Int
|
||||
extern fn akk_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -36,6 +36,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn ang_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn ang_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn ang_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn ang_str_last_char(s: String) -> String
|
||||
|
||||
@@ -21,6 +21,7 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn ar_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn ar_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn ar_str_len(s: String) -> Int
|
||||
extern fn ar_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -54,6 +54,7 @@
|
||||
|
||||
// ── String helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn cop_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn cop_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn cop_str_len(s: String) -> Int
|
||||
extern fn cop_drop(s: String, n: Int) -> String
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
// Dat: dem der dem den
|
||||
// Gen: des der des der
|
||||
|
||||
import "morphology.el"
|
||||
fn de_article_def(gender: String, gram_case: String, number: String) -> String {
|
||||
if str_eq(number, "pl") {
|
||||
if str_eq(gram_case, "nom") { return "die" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn de_article_def(gender: String, gram_case: String, number: String) -> String
|
||||
extern fn de_article_indef(gender: String, gram_case: String, number: String) -> String
|
||||
extern fn de_article(gender: String, gram_case: String, number: String, definite: String) -> String
|
||||
|
||||
@@ -52,6 +52,7 @@
|
||||
|
||||
// ── String helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn egy_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn egy_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn egy_str_len(s: String) -> Int
|
||||
extern fn egy_drop(s: String, n: Int) -> String
|
||||
|
||||
@@ -31,6 +31,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn enm_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn enm_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn enm_drop(s: String, n: Int) -> String
|
||||
extern fn enm_first_char(s: String) -> String
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
// ── String helpers (local, matching morphology.el conventions) ────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn es_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn es_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn es_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn es_str_last_char(s: String) -> String
|
||||
|
||||
@@ -25,6 +25,7 @@
|
||||
// If only neutral vowels are found, default to "front" (the conservative choice
|
||||
// for borrowed words and those without clear back vowels).
|
||||
|
||||
import "morphology.el"
|
||||
fn fi_harmony(word: String) -> String {
|
||||
let n: Int = str_len(word)
|
||||
let i: Int = n - 1
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn fi_harmony(word: String) -> String
|
||||
extern fn fi_suffix(base: String, harmony: String) -> String
|
||||
extern fn fi_noun_case(stem: String, gram_case: String, number: String, harmony: String) -> String
|
||||
extern fn fi_str_last_char(s: String) -> String
|
||||
extern fn fi_apply_case(noun: String, gram_case: String, number: String) -> String
|
||||
extern fn fi_verb_stem(dict_form: String) -> String
|
||||
extern fn fi_irregular_verb(dict_form: String) -> Any
|
||||
extern fn fi_irregular_verb(dict_form: String) -> [String]
|
||||
extern fn fi_present_ending(stem: String, person: String, number: String, harmony: String) -> String
|
||||
extern fn fi_past_stem(stem: String) -> String
|
||||
extern fn fi_past_ending(stem: String, person: String, number: String, harmony: String) -> String
|
||||
@@ -14,4 +14,4 @@ extern fn fi_negative(verb: String, person: String, number: String) -> String
|
||||
extern fn fi_conjugate(verb: String, tense: String, person: String, number: String) -> String
|
||||
extern fn fi_question_suffix(harmony: String) -> String
|
||||
extern fn fi_make_question(verb_form: String, harmony: String) -> String
|
||||
extern fn fi_full_paradigm(noun: String) -> Any
|
||||
extern fn fi_full_paradigm(noun: String) -> [String]
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
|
||||
// ── String helpers (local, matching morphology.el conventions) ────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn fr_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn fr_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn fr_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn fr_str_last_char(s: String) -> String
|
||||
|
||||
@@ -53,6 +53,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn fro_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn fro_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn fro_drop(s: String, n: Int) -> String
|
||||
extern fn fro_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -64,6 +64,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn gez_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn gez_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn gez_str_len(s: String) -> Int
|
||||
extern fn gez_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn goh_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn goh_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn goh_drop(s: String, n: Int) -> String
|
||||
extern fn goh_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -49,6 +49,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn got_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn got_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn got_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn got_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -31,6 +31,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn grc_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn grc_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn grc_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn grc_str_last_char(s: String) -> String
|
||||
|
||||
@@ -51,6 +51,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn he_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn he_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn he_str_len(s: String) -> Int
|
||||
extern fn he_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn hi_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn hi_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn hi_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn hi_str_last_char(s: String) -> String
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
// Note: this is a heuristic classifier for romanized input. For production use
|
||||
// with native kana/kanji forms, the dictionary form (辞書形) must be consulted.
|
||||
|
||||
import "morphology.el"
|
||||
fn ja_verb_group(dict_form: String) -> String {
|
||||
// Irregular verbs (exact match on dictionary form)
|
||||
if str_eq(dict_form, "する") { return "irregular" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn ja_verb_group(dict_form: String) -> String
|
||||
extern fn ja_ichidan_stem(dict_form: String) -> String
|
||||
extern fn ja_godan_stem_change(dict_form: String, row: String) -> String
|
||||
|
||||
@@ -25,6 +25,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn la_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn la_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn la_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn la_str_last_char(s: String) -> String
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn non_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn non_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn non_drop(s: String, n: Int) -> String
|
||||
extern fn non_last(s: String) -> String
|
||||
|
||||
@@ -31,6 +31,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn peo_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn peo_drop(s: String, n: Int) -> String
|
||||
extern fn peo_ends(s: String, suf: String) -> Bool
|
||||
extern fn peo_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn pi_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn pi_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn pi_drop(s: String, n: Int) -> String
|
||||
extern fn pi_last_char(s: String) -> String
|
||||
|
||||
@@ -35,6 +35,7 @@
|
||||
// The heuristic returns the most probable gender. Caller should override
|
||||
// for known exceptions (путь, рубль are masc despite -ь).
|
||||
|
||||
import "morphology.el"
|
||||
fn ru_gender(noun: String) -> String {
|
||||
let n: Int = str_len(noun)
|
||||
if n == 0 { return "m" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn ru_gender(noun: String) -> String
|
||||
extern fn ru_stem_type(noun: String, gender: String) -> String
|
||||
extern fn ru_noun_case(noun: String, gender: String, gram_case: String, number: String) -> String
|
||||
|
||||
@@ -42,6 +42,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sa_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sa_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sa_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sa_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -31,6 +31,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sga_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sga_drop(s: String, n: Int) -> String
|
||||
extern fn sga_first(s: String) -> String
|
||||
extern fn sga_rest(s: String) -> String
|
||||
|
||||
@@ -53,6 +53,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sux_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sux_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sux_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sux_str_last_char(s: String) -> String
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn sw_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sw_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn sw_str_drop_last(s: String, n: Int) -> String
|
||||
extern fn sw_str_first_char(s: String) -> String
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn txb_drop(s: String, n: Int) -> String {
|
||||
let len: Int = str_len(s)
|
||||
if n >= len { return "" }
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn txb_drop(s: String, n: Int) -> String
|
||||
extern fn txb_ends(s: String, suf: String) -> Bool
|
||||
extern fn txb_slot(person: String, number: String) -> Int
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
|
||||
// ── String helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
import "morphology.el"
|
||||
fn uga_str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn uga_str_ends(s: String, suf: String) -> Bool
|
||||
extern fn uga_str_len(s: String) -> Int
|
||||
extern fn uga_str_drop_last(s: String, n: Int) -> String
|
||||
|
||||
@@ -33,6 +33,17 @@
|
||||
|
||||
// ── String helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
import "language-profile.el"
|
||||
import "morphology-es.el"
|
||||
import "morphology-fr.el"
|
||||
import "morphology-de.el"
|
||||
import "morphology-ru.el"
|
||||
import "morphology-fi.el"
|
||||
import "morphology-ar.el"
|
||||
import "morphology-hi.el"
|
||||
import "morphology-sw.el"
|
||||
import "morphology-la.el"
|
||||
import "morphology-ja.el"
|
||||
fn str_ends(s: String, suf: String) -> Bool {
|
||||
return str_ends_with(s, suf)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn str_ends(s: String, suf: String) -> Bool
|
||||
extern fn str_last_char(s: String) -> String
|
||||
extern fn str_last2(s: String) -> String
|
||||
@@ -8,7 +8,7 @@ extern fn is_vowel(c: String) -> Bool
|
||||
extern fn morph_apply_suffix(base: String, suffix: String) -> String
|
||||
extern fn en_irregular_plural(word: String) -> String
|
||||
extern fn en_irregular_singular(word: String) -> String
|
||||
extern fn en_irregular_verb(base: String) -> Any
|
||||
extern fn en_irregular_verb(base: String) -> [String]
|
||||
extern fn en_verb_3sg(base: String) -> String
|
||||
extern fn en_should_double_final(base: String) -> Bool
|
||||
extern fn en_verb_past(base: String) -> String
|
||||
@@ -16,10 +16,10 @@ extern fn en_verb_gerund(base: String) -> String
|
||||
extern fn en_pluralize_regular(singular: String) -> String
|
||||
extern fn en_verb_form(base: String, tense: String, person: String, number: String) -> String
|
||||
extern fn agree_determiner(det: String, noun: String) -> String
|
||||
extern fn morph_pluralize(noun: String, profile: Any) -> String
|
||||
extern fn morph_pluralize(noun: String, profile: [String]) -> String
|
||||
extern fn morph_map_canonical(verb: String, code: String) -> String
|
||||
extern fn morph_conjugate(verb: String, tense: String, person: String, number: String, profile: Any) -> String
|
||||
extern fn morph_inflect(word: String, features: String, profile: Any) -> String
|
||||
extern fn morph_conjugate(verb: String, tense: String, person: String, number: String, profile: [String]) -> String
|
||||
extern fn morph_inflect(word: String, features: String, profile: [String]) -> String
|
||||
extern fn pluralize(singular: String) -> String
|
||||
extern fn singularize(plural: String) -> String
|
||||
extern fn verb_form(base: String, tense: String, person: String, number: String) -> String
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
// organ-read.el - Route the render's GEOMETRY READ through the ingest ORGAN's
|
||||
// saved engram files (the coordinator's source of truth). For each file we
|
||||
// engram_load() it, engram_scan_nodes_json(limit, offset) to get the node array,
|
||||
// and cache each node's self-contained CONTENT string keyed by symbol. Because
|
||||
// the cached value carries the numbers ("... f1=730 ..."), the cache SURVIVES the
|
||||
// store being REPLACED by the next engram_load — so we load+cache phonetics
|
||||
// FIRST, then load+cache accent. The .psv path remains a fallback.
|
||||
//
|
||||
// engram_scan_nodes_json(limit, offset) takes NO query; it returns nodes
|
||||
// salience-sorted, so limit must be >= node count and we filter client-side.
|
||||
// (engram_search / engram_scan_nodes return len-5 garbage — unused.)
|
||||
|
||||
// Find every occurrence of `marker` in the scan JSON; for each, cache
|
||||
// sym -> a 150-char content window (enough to hold f1..amp). Duplicates from the
|
||||
// node's "content" and "label" fields are harmless (first match wins on read).
|
||||
fn organ_cache(j: String, marker: String, mlen: Int, win_len: Int, need: String) -> [String] {
|
||||
let m: [String] = native_list_empty()
|
||||
let jl: Int = str_len(j)
|
||||
let off: Int = 0
|
||||
while off < jl {
|
||||
let rest: String = str_slice(j, off, jl)
|
||||
let p: Int = str_index_of(rest, marker)
|
||||
if p < 0 {
|
||||
off = jl
|
||||
} else {
|
||||
let abs: Int = off + p
|
||||
let win: String = str_slice(j, abs, abs + win_len)
|
||||
let after: String = str_slice(win, mlen, str_len(win))
|
||||
let sp: Int = str_index_of(after, " ")
|
||||
let hasneed: Int = str_index_of(win, need)
|
||||
if sp > 0 {
|
||||
if hasneed >= 0 {
|
||||
let sym: String = str_slice(after, 0, sp)
|
||||
m = native_list_append(m, sym)
|
||||
m = native_list_append(m, win)
|
||||
}
|
||||
}
|
||||
off = abs + mlen
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Load the phonetics organ file and cache sym -> content. mlen("phoneme ")=8.
|
||||
fn organ_pmap(path: String) -> [String] {
|
||||
let ok: Bool = engram_load(path)
|
||||
if ok == false {
|
||||
return native_list_empty()
|
||||
}
|
||||
let j: String = engram_scan_nodes_json(600, 0)
|
||||
return organ_cache(j, "phoneme ", 8, 150, "f1=")
|
||||
}
|
||||
|
||||
// Load the accent organ file and cache sym -> content. mlen("accent_target ")=14.
|
||||
// Vowel overrides carry f1=..; the R rule carries drop_coda_r (need="=" matches
|
||||
// both, i.e. any well-formed accent_target field).
|
||||
fn organ_amap(path: String) -> [String] {
|
||||
let ok: Bool = engram_load(path)
|
||||
if ok == false {
|
||||
return native_list_empty()
|
||||
}
|
||||
let j: String = engram_scan_nodes_json(600, 0)
|
||||
return organ_cache(j, "accent_target ", 14, 90, "=")
|
||||
}
|
||||
|
||||
// Vowel-set (categorical class) from the phonetics .psv class column.
|
||||
fn organ_vset(path: String) -> [String] {
|
||||
let content: String = fs_read(path)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let v: [String] = native_list_empty()
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ok: Int = 1
|
||||
if str_len(line) < 5 {
|
||||
ok = 0
|
||||
}
|
||||
if ok == 1 {
|
||||
if str_char_code(line, 0) == 35 {
|
||||
ok = 0
|
||||
}
|
||||
}
|
||||
if ok == 1 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
if native_list_len(f) >= 12 {
|
||||
if str_eq(native_list_get(f, 11), "vowel") {
|
||||
v = native_list_append(v, native_list_get(f, 0))
|
||||
}
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// Word -> phoneme-sequence cache from lexicon.psv (engram-independent).
|
||||
fn organ_lex(path: String) -> [String] {
|
||||
let content: String = fs_read(path)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let m: [String] = native_list_empty()
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ok: Int = 1
|
||||
if str_len(line) < 3 {
|
||||
ok = 0
|
||||
}
|
||||
if ok == 1 {
|
||||
if str_char_code(line, 0) == 35 {
|
||||
ok = 0
|
||||
}
|
||||
}
|
||||
if ok == 1 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
if native_list_len(f) >= 2 {
|
||||
m = native_list_append(m, native_list_get(f, 0))
|
||||
m = native_list_append(m, native_list_get(f, 1))
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
return m
|
||||
}
|
||||
@@ -1,10 +1,10 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn agent_person(agent: String) -> String
|
||||
extern fn agent_number(agent: String) -> String
|
||||
extern fn realize_np(referent: String, number: String) -> String
|
||||
extern fn realize_vp_lang(base_verb: String, tense: String, aspect: String, person: String, number: String, profile: Any) -> Any
|
||||
extern fn realize_question_lang(predicate: String, tense: String, aspect: String, person: String, number: String, agent: String, patient: String, location: String, profile: Any) -> String
|
||||
extern fn realize_vp_lang(base_verb: String, tense: String, aspect: String, person: String, number: String, profile: [String]) -> [String]
|
||||
extern fn realize_question_lang(predicate: String, tense: String, aspect: String, person: String, number: String, agent: String, patient: String, location: String, profile: [String]) -> String
|
||||
extern fn capitalize_first(s: String) -> String
|
||||
extern fn add_punct(s: String, intent: String) -> String
|
||||
extern fn realize_lang(form: Any, profile: Any) -> String
|
||||
extern fn realize(form: Any) -> String
|
||||
extern fn realize_lang(form: [String], profile: [String]) -> String
|
||||
extern fn realize(form: [String]) -> String
|
||||
|
||||
+15
-15
@@ -1,18 +1,18 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn sem_frame(intent: String, subject: String, obj: String, modifiers: String) -> Any
|
||||
extern fn sem_frame_lang(intent: String, subject: String, obj: String, modifiers: String, lang_code: String) -> Any
|
||||
extern fn sem_frame_simple(intent: String, subject: String) -> Any
|
||||
extern fn sem_frame_obj(intent: String, subject: String, obj: String) -> Any
|
||||
extern fn sem_intent(frame: Any) -> String
|
||||
extern fn sem_subject(frame: Any) -> String
|
||||
extern fn sem_object(frame: Any) -> String
|
||||
extern fn sem_modifiers(frame: Any) -> String
|
||||
extern fn sem_lang(frame: Any) -> String
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn sem_frame(intent: String, subject: String, obj: String, modifiers: String) -> [String]
|
||||
extern fn sem_frame_lang(intent: String, subject: String, obj: String, modifiers: String, lang_code: String) -> [String]
|
||||
extern fn sem_frame_simple(intent: String, subject: String) -> [String]
|
||||
extern fn sem_frame_obj(intent: String, subject: String, obj: String) -> [String]
|
||||
extern fn sem_intent(frame: [String]) -> String
|
||||
extern fn sem_subject(frame: [String]) -> String
|
||||
extern fn sem_object(frame: [String]) -> String
|
||||
extern fn sem_modifiers(frame: [String]) -> String
|
||||
extern fn sem_lang(frame: [String]) -> String
|
||||
extern fn sem_first_modifier(mods: String) -> String
|
||||
extern fn sem_intent_to_realize(intent: String) -> String
|
||||
extern fn sem_to_spec(frame: Any) -> Any
|
||||
extern fn sem_to_spec_full(frame: Any, verb: String, tense: String, aspect: String) -> Any
|
||||
extern fn sem_to_spec(frame: [String]) -> [String]
|
||||
extern fn sem_to_spec_full(frame: [String], verb: String, tense: String, aspect: String) -> [String]
|
||||
extern fn sem_realize_greet(subject: String) -> String
|
||||
extern fn sem_realize(frame: Any) -> String
|
||||
extern fn sem_realize_full(frame: Any, verb: String, tense: String, aspect: String) -> String
|
||||
extern fn sem_realize_lang(frame: Any, lang_code: String) -> String
|
||||
extern fn sem_realize(frame: [String]) -> String
|
||||
extern fn sem_realize_full(frame: [String], verb: String, tense: String, aspect: String) -> String
|
||||
extern fn sem_realize_lang(frame: [String], lang_code: String) -> String
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
// speech-ingest.el - The native LOAD step of the ingest organ, for the SPEECH
|
||||
// primitives. Reads the acoustic-phonetics SOURCE (elp/data/phonetics.psv) and
|
||||
// the pronunciation lexicon SOURCE (elp/data/lexicon.psv) and emits a PHONEME
|
||||
// MANIFOLD into the engram: one node per phoneme (faithful, provenance-tagged
|
||||
// content) + is_a edges to phoneme-class nodes (a discrete manifold, not islands).
|
||||
// The render then PULLS phoneme geometry back from the engram via phon_geo —
|
||||
// zero phonetic numbers in code. Source -> manifold -> merge; the same output
|
||||
// the polymorphic ingest organ will produce and subsume.
|
||||
|
||||
// -- small parsing helpers ---------------------------------------------------
|
||||
fn sp_map_get(pairs: [String], key: String) -> String {
|
||||
let n: Int = native_list_len(pairs)
|
||||
let i: Int = 0
|
||||
while i < n - 1 {
|
||||
let k: String = native_list_get(pairs, i)
|
||||
if str_eq(k, key) {
|
||||
return native_list_get(pairs, i + 1)
|
||||
}
|
||||
let i = i + 2
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// read the unsigned integer that follows `key` inside string s (e.g. key "F1=")
|
||||
fn parse_uint_from(s: String, key: String) -> Int {
|
||||
let idx: Int = str_index_of(s, key)
|
||||
if idx < 0 {
|
||||
return 0
|
||||
}
|
||||
let start: Int = idx + str_len(key)
|
||||
let n: Int = str_len(s)
|
||||
let i: Int = start
|
||||
let val: Int = 0
|
||||
while i < n {
|
||||
let c: Int = str_char_code(s, i)
|
||||
if c >= 48 {
|
||||
if c <= 57 {
|
||||
val = val * 10 + (c - 48)
|
||||
i = i + 1
|
||||
} else {
|
||||
i = n
|
||||
}
|
||||
} else {
|
||||
i = n
|
||||
}
|
||||
}
|
||||
return val
|
||||
}
|
||||
|
||||
fn clean_word(w: String) -> String {
|
||||
let low: String = str_to_lower(w)
|
||||
let n: Int = str_len(low)
|
||||
let out: String = ""
|
||||
let i: Int = 0
|
||||
while i < n {
|
||||
let c: Int = str_char_code(low, i)
|
||||
if c >= 97 {
|
||||
if c <= 122 {
|
||||
out = out + str_char_at(low, i)
|
||||
}
|
||||
}
|
||||
i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// -- INGEST: acoustic-phonetics source -> phoneme manifold in the engram ------
|
||||
// Returns the symbol -> node-id index (pmap) the render reads geometry through.
|
||||
fn ingest_phonetics(path: String) -> [String] {
|
||||
let content: String = fs_read(path)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let pmap: [String] = native_list_empty()
|
||||
let classmap: [String] = native_list_empty()
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ll: Int = str_len(line)
|
||||
let skip: Int = 0
|
||||
if ll < 5 {
|
||||
skip = 1
|
||||
}
|
||||
if skip == 0 {
|
||||
let first: Int = str_char_code(line, 0)
|
||||
if first == 35 {
|
||||
skip = 1
|
||||
}
|
||||
}
|
||||
if skip == 0 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
let nf: Int = native_list_len(f)
|
||||
if nf >= 12 {
|
||||
let sym: String = native_list_get(f, 0)
|
||||
let f1: String = native_list_get(f, 1)
|
||||
let f2: String = native_list_get(f, 2)
|
||||
let f3: String = native_list_get(f, 3)
|
||||
let b1: String = native_list_get(f, 4)
|
||||
let b2: String = native_list_get(f, 5)
|
||||
let b3: String = native_list_get(f, 6)
|
||||
let vo: String = native_list_get(f, 7)
|
||||
let na: String = native_list_get(f, 8)
|
||||
let du: String = native_list_get(f, 9)
|
||||
let am: String = native_list_get(f, 10)
|
||||
let cls: String = native_list_get(f, 11)
|
||||
let cont: String = "phoneme " + sym + " | f1=" + f1 + " f2=" + f2 + " f3=" + f3 + " bw1=" + b1 + " bw2=" + b2 + " bw3=" + b3 + " voiced=" + vo + " nasal=" + na + " dur=" + du + " amp=" + am + " class=" + cls + " src=PetersonBarney1952-Hillenbrand1995"
|
||||
let id: String = engram_node(cont, "Phoneme", 80)
|
||||
pmap = native_list_append(pmap, sym)
|
||||
pmap = native_list_append(pmap, cont)
|
||||
// manifold edge: phoneme is_a class
|
||||
let cid: String = sp_map_get(classmap, cls)
|
||||
if str_eq(cid, "") {
|
||||
cid = engram_node("phoneme-class " + cls + " src=acoustic-phonetics", "PhonemeClass", 80)
|
||||
classmap = native_list_append(classmap, cls)
|
||||
classmap = native_list_append(classmap, cid)
|
||||
}
|
||||
engram_connect(id, cid, 80, "is_a")
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
return pmap
|
||||
}
|
||||
|
||||
// -- INGEST: pronunciation lexicon source -> word nodes ----------------------
|
||||
fn ingest_lexicon(path: String) -> [String] {
|
||||
let content: String = fs_read(path)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let lmap: [String] = native_list_empty()
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ll: Int = str_len(line)
|
||||
let skip: Int = 0
|
||||
if ll < 3 {
|
||||
skip = 1
|
||||
}
|
||||
if skip == 0 {
|
||||
let first: Int = str_char_code(line, 0)
|
||||
if first == 35 {
|
||||
skip = 1
|
||||
}
|
||||
}
|
||||
if skip == 0 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
let nf: Int = native_list_len(f)
|
||||
if nf >= 2 {
|
||||
let word: String = native_list_get(f, 0)
|
||||
let seq: String = native_list_get(f, 1)
|
||||
let id: String = engram_node("word " + word + " phonemes " + seq + " src=lexicon", "Pronunciation", 80)
|
||||
lmap = native_list_append(lmap, word)
|
||||
lmap = native_list_append(lmap, seq)
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
return lmap
|
||||
}
|
||||
|
||||
// -- READ geometry back from the engram (the render's afferent lookup) --------
|
||||
// phon_geo(sym) -> [F1,F2,F3,B1,B2,B3,voiced,nasal,dur,amp], parsed from the
|
||||
// ingested phoneme node's content. NO formant numbers live in this code.
|
||||
fn phon_geo(pmap: [String], sym: String) -> [Int] {
|
||||
let id: String = sp_map_get(pmap, sym)
|
||||
if str_eq(id, "") {
|
||||
id = sp_map_get(pmap, "AX")
|
||||
}
|
||||
let out: [Int] = native_list_empty()
|
||||
if str_eq(id, "") {
|
||||
let out = native_list_append(out, 500)
|
||||
let out = native_list_append(out, 1500)
|
||||
let out = native_list_append(out, 2500)
|
||||
let out = native_list_append(out, 80)
|
||||
let out = native_list_append(out, 100)
|
||||
let out = native_list_append(out, 150)
|
||||
let out = native_list_append(out, 1)
|
||||
let out = native_list_append(out, 0)
|
||||
let out = native_list_append(out, 80)
|
||||
let out = native_list_append(out, 80)
|
||||
return out
|
||||
}
|
||||
let j: String = id
|
||||
let out = native_list_append(out, parse_uint_from(j, "f1="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "f2="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "f3="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "bw1="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "bw2="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "bw3="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "voiced="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "nasal="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "dur="))
|
||||
let out = native_list_append(out, parse_uint_from(j, "amp="))
|
||||
return out
|
||||
}
|
||||
|
||||
// word -> phoneme codes, read from the ingested lexicon node.
|
||||
fn word_phonemes(lmap: [String], word: String) -> [String] {
|
||||
let id: String = sp_map_get(lmap, word)
|
||||
if str_eq(id, "") {
|
||||
let r: [String] = native_list_empty()
|
||||
let r = native_list_append(r, "AX")
|
||||
return r
|
||||
}
|
||||
return str_split(id, " ")
|
||||
}
|
||||
|
||||
// realized text -> flat phoneme-code sequence (SIL between words + at ends).
|
||||
fn text_phonemes(lmap: [String], text: String) -> [String] {
|
||||
let words: [String] = str_split(text, " ")
|
||||
let nw: Int = native_list_len(words)
|
||||
let seq: [String] = native_list_empty()
|
||||
let seq = native_list_append(seq, "SIL")
|
||||
let wi: Int = 0
|
||||
while wi < nw {
|
||||
let raw: String = native_list_get(words, wi)
|
||||
let w: String = clean_word(raw)
|
||||
if str_eq(w, "") {
|
||||
wi = wi + 1
|
||||
} else {
|
||||
let ph: [String] = word_phonemes(lmap, w)
|
||||
let np: Int = native_list_len(ph)
|
||||
let pi: Int = 0
|
||||
while pi < np {
|
||||
let code: String = native_list_get(ph, pi)
|
||||
seq = native_list_append(seq, code)
|
||||
pi = pi + 1
|
||||
}
|
||||
seq = native_list_append(seq, "SIL")
|
||||
wi = wi + 1
|
||||
}
|
||||
}
|
||||
return seq
|
||||
}
|
||||
@@ -0,0 +1,460 @@
|
||||
// speech.el - The native SPEECH render path + voice-by-imitation extractor.
|
||||
//
|
||||
// Speech = the AUDIO surface (surface_profile_audio) rendering LANGUAGE-meaning
|
||||
// through a VOICE signature. The realizer's language faculty supplies the words
|
||||
// (meaning -> sem_realize -> text); this module turns text -> phonemes (phonetics.el)
|
||||
// -> a formant-target track over time -> SUPERPOSES formant resonances over a
|
||||
// glottal source (own-core formant synthesis, the exact integer mirror of the
|
||||
// music additive superpose) -> own-core PCM/WAV. Two paths:
|
||||
// (1) RENDER: speak(text, voice) -> spoken WAV.
|
||||
// (2) IMITATE: voice_analyze(pcm) -> a voice signature grabbed BY EAR
|
||||
// (autocorrelation pitch + integer-DFT formant peaks), then render
|
||||
// any new meaning in that voice. An impression, not a corpus.
|
||||
// All integer/fixed-point (EL float arithmetic is unusable).
|
||||
|
||||
// -- Own-core integer sine (Bhaskara I), phase 0..65535 = one cycle -----------
|
||||
fn sp_sin(phase: Int) -> Int {
|
||||
let deg: Int = phase * 360 / 65536
|
||||
let neg: Int = 0
|
||||
if deg > 180 {
|
||||
deg = deg - 180
|
||||
neg = 1
|
||||
}
|
||||
let t: Int = deg * (180 - deg)
|
||||
let num: Int = 32767 * 4 * t
|
||||
let den: Int = 40500 - t
|
||||
let v: Int = num / den
|
||||
if neg == 1 {
|
||||
v = 0 - v
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
fn sp_cos(phase: Int) -> Int {
|
||||
let p: Int = phase + 16384
|
||||
p = p - (p / 65536) * 65536
|
||||
return sp_sin(p)
|
||||
}
|
||||
|
||||
// One formant resonance (Lorentzian peak), Q15. Peak 32767 at f=fc.
|
||||
fn sp_gain(f: Int, fc: Int, bw: Int) -> Int {
|
||||
let d: Int = f - fc
|
||||
let den: Int = d * d + bw * bw
|
||||
let num: Int = 32767 * bw * bw
|
||||
return num / den
|
||||
}
|
||||
|
||||
fn sp_isqrt(n: Int) -> Int {
|
||||
if n <= 0 {
|
||||
return 0
|
||||
}
|
||||
let x: Int = n
|
||||
let y: Int = (x + 1) / 2
|
||||
while y < x {
|
||||
x = y
|
||||
y = (x + n / x) / 2
|
||||
}
|
||||
return x
|
||||
}
|
||||
|
||||
// -- WAV serializer (thin medium; the only non-DSP glue) ---------------------
|
||||
fn wav_le16(buf: String, off: Int, v: Int) -> String {
|
||||
let u: Int = v
|
||||
if u < 0 {
|
||||
u = u + 65536
|
||||
}
|
||||
let lo: Int = u - (u / 256) * 256
|
||||
let hi: Int = u / 256
|
||||
let b: String = __str_set_char(buf, off, lo)
|
||||
b = __str_set_char(b, off + 1, hi)
|
||||
return b
|
||||
}
|
||||
|
||||
fn wav_le32(buf: String, off: Int, v: Int) -> String {
|
||||
let b0: Int = v - (v / 256) * 256
|
||||
let r1: Int = v / 256
|
||||
let b1: Int = r1 - (r1 / 256) * 256
|
||||
let r2: Int = r1 / 256
|
||||
let b2: Int = r2 - (r2 / 256) * 256
|
||||
let b3: Int = r2 / 256
|
||||
let b: String = __str_set_char(buf, off, b0)
|
||||
b = __str_set_char(b, off + 1, b1)
|
||||
b = __str_set_char(b, off + 2, b2)
|
||||
b = __str_set_char(b, off + 3, b3)
|
||||
return b
|
||||
}
|
||||
|
||||
fn wav_ascii(buf: String, off: Int, s: String) -> String {
|
||||
let n: Int = str_len(s)
|
||||
let i: Int = 0
|
||||
let b: String = buf
|
||||
while i < n {
|
||||
let c: Int = str_char_code(s, i)
|
||||
b = __str_set_char(b, off + i, c)
|
||||
i = i + 1
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
fn write_wav(samples: [Int], sr: Int, path: String) -> Bool {
|
||||
let ns: Int = native_list_len(samples)
|
||||
let datalen: Int = ns * 2
|
||||
let total: Int = 44 + datalen
|
||||
let buf: String = __str_alloc(total)
|
||||
buf = wav_ascii(buf, 0, "RIFF")
|
||||
buf = wav_le32(buf, 4, 36 + datalen)
|
||||
buf = wav_ascii(buf, 8, "WAVE")
|
||||
buf = wav_ascii(buf, 12, "fmt ")
|
||||
buf = wav_le32(buf, 16, 16)
|
||||
buf = wav_le16(buf, 20, 1)
|
||||
buf = wav_le16(buf, 22, 1)
|
||||
buf = wav_le32(buf, 24, sr)
|
||||
buf = wav_le32(buf, 28, sr * 2)
|
||||
buf = wav_le16(buf, 32, 2)
|
||||
buf = wav_le16(buf, 34, 16)
|
||||
buf = wav_ascii(buf, 36, "data")
|
||||
buf = wav_le32(buf, 40, datalen)
|
||||
let j: Int = 0
|
||||
let off: Int = 44
|
||||
while j < ns {
|
||||
let raw: Int = native_list_get(samples, j)
|
||||
buf = wav_le16(buf, off, raw)
|
||||
off = off + 2
|
||||
j = j + 1
|
||||
}
|
||||
return __fs_write_bytes(path, buf, total)
|
||||
}
|
||||
|
||||
// One formant resonance as a float Lorentzian peak (own-core physics).
|
||||
fn fgain(f: Float, fc: Float, bw: Float) -> Float {
|
||||
let d: Float = f - fc
|
||||
return (bw * bw) / (d * d + bw * bw)
|
||||
}
|
||||
|
||||
// His PITCH MELODY from measured prosody [f0_median, f0_min, f0_max, declination].
|
||||
// A natural statement shape over the utterance: onset rise to the median, a
|
||||
// near-flat body (his declination is ~0.6 Hz/s), and a final fall toward f0_min.
|
||||
// Follows his melody + range, not a fixed 0.85 decline. gidx/total = position.
|
||||
fn prosody_f0(pros: [Int], gidx: Int, total: Int) -> Int {
|
||||
let med: Int = native_list_get(pros, 0)
|
||||
let lo: Int = native_list_get(pros, 1)
|
||||
let hi: Int = native_list_get(pros, 2)
|
||||
let p: Int = gidx * 1000 / total
|
||||
let f0: Int = med
|
||||
if p < 150 {
|
||||
f0 = lo + (med - lo) * p / 150
|
||||
} else {
|
||||
if p > 700 {
|
||||
f0 = med + (lo - med) * (p - 700) / 300
|
||||
} else {
|
||||
f0 = med
|
||||
}
|
||||
}
|
||||
if f0 < lo {
|
||||
f0 = lo
|
||||
}
|
||||
if f0 > hi {
|
||||
f0 = hi
|
||||
}
|
||||
return f0
|
||||
}
|
||||
|
||||
// -- The render: phoneme codes + voice signature -> normalized PCM samples ----
|
||||
// Formant geometry per phoneme is READ FROM THE ENGRAM (pmap) via phon_geo — no
|
||||
// table in code. The optional ACCENT map (amap) composes a transform onto the
|
||||
// voice (voice (+) accent, separable): RP formant overrides read from the accent
|
||||
// manifold + a non-rhotic coda-R drop. Empty amap = base General-American.
|
||||
// Synthesis is FLOAT: a real phase accumulator + math_sin, superposition physics.
|
||||
fn synth_codes_accent(codes0: [String], voice: [String], pmap: [String], amap: [String], vset: [String], vmap: [String], prosody: [Int]) -> [Int] {
|
||||
let sr: Int = 16000
|
||||
let srf: Float = 16000.0
|
||||
let two_pi: Float = 6.283185307
|
||||
let kf: Int = voice_get_int(voice, "kf")
|
||||
let f0s: Int = voice_get_int(voice, "f0")
|
||||
let f0e: Int = voice_get_int(voice, "f0_end")
|
||||
let durm: Int = voice_get_int(voice, "dur")
|
||||
if kf <= 0 {
|
||||
kf = 1000
|
||||
}
|
||||
if durm <= 0 {
|
||||
durm = 1000
|
||||
}
|
||||
let use_accent: Int = 0
|
||||
if native_list_len(amap) > 0 {
|
||||
use_accent = 1
|
||||
}
|
||||
let codes: [String] = codes0
|
||||
if use_accent == 1 {
|
||||
if is_nonrhotic(amap) == 1 {
|
||||
codes = apply_rhoticity(codes0, vset)
|
||||
}
|
||||
}
|
||||
let nc: Int = native_list_len(codes)
|
||||
|
||||
// pass 1: per-segment sample counts + total
|
||||
let segn: [Int] = native_list_empty()
|
||||
let total: Int = 0
|
||||
let ci: Int = 0
|
||||
while ci < nc {
|
||||
let code: String = native_list_get(codes, ci)
|
||||
let p: [Int] = phon_geo(pmap, code)
|
||||
let durms: Int = native_list_get(p, 8)
|
||||
let ns: Int = durms * 16 * durm / 1000
|
||||
segn = native_list_append(segn, ns)
|
||||
total = total + ns
|
||||
ci = ci + 1
|
||||
}
|
||||
if total <= 0 {
|
||||
total = 1
|
||||
}
|
||||
|
||||
// pass 2: synthesize
|
||||
let samples: [Int] = native_list_empty()
|
||||
let phasef: Float = 0.0
|
||||
let gidx: Int = 0
|
||||
let prevF1: Int = 500 * kf / 1000
|
||||
let prevF2: Int = 1500 * kf / 1000
|
||||
let prevF3: Int = 2500 * kf / 1000
|
||||
let nstate: Int = 22695
|
||||
let maxabs: Int = 1
|
||||
|
||||
let ci2: Int = 0
|
||||
while ci2 < nc {
|
||||
let code: String = native_list_get(codes, ci2)
|
||||
let p: [Int] = phon_geo(pmap, code)
|
||||
let rf1: Int = native_list_get(p, 0)
|
||||
let rf2: Int = native_list_get(p, 1)
|
||||
let rf3: Int = native_list_get(p, 2)
|
||||
if use_accent == 1 {
|
||||
let ov: [Int] = accent_formants(amap, code)
|
||||
if native_list_len(ov) >= 3 {
|
||||
rf1 = native_list_get(ov, 0)
|
||||
rf2 = native_list_get(ov, 1)
|
||||
rf3 = native_list_get(ov, 2)
|
||||
}
|
||||
}
|
||||
// HIS measured vowel target overrides the generic/kf path (absolute Hz —
|
||||
// his formants already encode his vocal tract, so no kf scaling).
|
||||
let usekf: Int = 1
|
||||
if native_list_len(vmap) > 0 {
|
||||
let hv: [Int] = vmap_get(vmap, code)
|
||||
if native_list_len(hv) >= 3 {
|
||||
rf1 = native_list_get(hv, 0)
|
||||
rf2 = native_list_get(hv, 1)
|
||||
rf3 = native_list_get(hv, 2)
|
||||
usekf = 0
|
||||
}
|
||||
}
|
||||
let F1t: Int = rf1 * kf / 1000
|
||||
let F2t: Int = rf2 * kf / 1000
|
||||
let F3t: Int = rf3 * kf / 1000
|
||||
if usekf == 0 {
|
||||
F1t = rf1
|
||||
F2t = rf2
|
||||
F3t = rf3
|
||||
}
|
||||
let B1: Int = native_list_get(p, 3)
|
||||
let B2: Int = native_list_get(p, 4)
|
||||
let B3: Int = native_list_get(p, 5)
|
||||
let voiced: Int = native_list_get(p, 6)
|
||||
let ampv: Int = native_list_get(p, 9)
|
||||
let ns: Int = native_list_get(segn, ci2)
|
||||
let trans: Int = ns / 2
|
||||
if trans > 560 {
|
||||
trans = 560
|
||||
}
|
||||
if trans < 1 {
|
||||
trans = 1
|
||||
}
|
||||
let k: Int = 0
|
||||
while k < ns {
|
||||
let cF1: Int = F1t
|
||||
let cF2: Int = F2t
|
||||
let cF3: Int = F3t
|
||||
if k < trans {
|
||||
cF1 = prevF1 + (F1t - prevF1) * k / trans
|
||||
cF2 = prevF2 + (F2t - prevF2) * k / trans
|
||||
cF3 = prevF3 + (F3t - prevF3) * k / trans
|
||||
}
|
||||
let f0c: Int = f0s + (f0e - f0s) * gidx / total
|
||||
if native_list_len(prosody) >= 3 {
|
||||
f0c = prosody_f0(prosody, gidx, total)
|
||||
}
|
||||
if f0c < 40 {
|
||||
f0c = 40
|
||||
}
|
||||
let env: Int = 32767
|
||||
let ar: Int = 96
|
||||
if k < ar {
|
||||
env = 32767 * k / ar
|
||||
}
|
||||
let tail: Int = ns - k
|
||||
if tail < ar {
|
||||
env = 32767 * tail / ar
|
||||
}
|
||||
let f0cf: Float = int_to_float(f0c)
|
||||
phasef = phasef + two_pi * f0cf / srf
|
||||
if phasef > two_pi {
|
||||
phasef = phasef - two_pi
|
||||
}
|
||||
|
||||
let s: Int = 0
|
||||
if voiced == 1 {
|
||||
let cF1f: Float = int_to_float(cF1)
|
||||
let cF2f: Float = int_to_float(cF2)
|
||||
let cF3f: Float = int_to_float(cF3)
|
||||
let B1f: Float = int_to_float(B1)
|
||||
let B2f: Float = int_to_float(B2)
|
||||
let B3f: Float = int_to_float(B3)
|
||||
let acc: Float = 0.0
|
||||
let h: Int = 1
|
||||
while h <= 50 {
|
||||
let hf: Float = int_to_float(h)
|
||||
let fhf: Float = hf * f0cf
|
||||
if fhf < 7900.0 {
|
||||
let sv: Float = math_sin(phasef * hf)
|
||||
let src: Float = 1.0 / hf
|
||||
let g1: Float = fgain(fhf, cF1f, B1f)
|
||||
let g2: Float = fgain(fhf, cF2f, B2f)
|
||||
let g3: Float = fgain(fhf, cF3f, B3f)
|
||||
let g: Float = g1 + g2 + g3
|
||||
acc = acc + src * g * sv
|
||||
}
|
||||
h = h + 1
|
||||
}
|
||||
s = float_to_int(acc * 4000.0)
|
||||
} else {
|
||||
if ampv > 0 {
|
||||
nstate = nstate * 1103515245 + 12345
|
||||
nstate = nstate - (nstate / 2147483648) * 2147483648
|
||||
if nstate < 0 {
|
||||
nstate = 0 - nstate
|
||||
}
|
||||
let nz: Int = nstate / 32768 - 32768
|
||||
s = nz
|
||||
}
|
||||
}
|
||||
s = s * ampv / 100
|
||||
s = s * env / 32767
|
||||
samples = native_list_append(samples, s)
|
||||
let a: Int = s
|
||||
if a < 0 {
|
||||
a = 0 - a
|
||||
}
|
||||
if a > maxabs {
|
||||
maxabs = a
|
||||
}
|
||||
gidx = gidx + 1
|
||||
k = k + 1
|
||||
}
|
||||
prevF1 = F1t
|
||||
prevF2 = F2t
|
||||
prevF3 = F3t
|
||||
ci2 = ci2 + 1
|
||||
}
|
||||
|
||||
// normalize to int16 range (~22000 peak)
|
||||
let out: [Int] = native_list_empty()
|
||||
let ntot: Int = native_list_len(samples)
|
||||
let j: Int = 0
|
||||
while j < ntot {
|
||||
let raw: Int = native_list_get(samples, j)
|
||||
let v: Int = raw * 22000 / maxabs
|
||||
out = native_list_append(out, v)
|
||||
j = j + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// GA convenience wrapper (no accent) — keeps the base render path.
|
||||
fn synth_codes(codes: [String], voice: [String], pmap: [String]) -> [Int] {
|
||||
let noacc: [String] = native_list_empty()
|
||||
let novset: [String] = native_list_empty()
|
||||
let novmap: [String] = native_list_empty()
|
||||
let nopros: [Int] = native_list_empty()
|
||||
return synth_codes_accent(codes, voice, pmap, noacc, novset, novmap, nopros)
|
||||
}
|
||||
|
||||
// -- Voice-by-imitation: HEAR a PCM sample -> extract the voice signature -----
|
||||
// Pitch by autocorrelation; vocal-tract scale (kf) from the F1 formant peak of a
|
||||
// heard sustained vowel /AA/ (nominal F1 = 730 Hz) via an integer DFT. The
|
||||
// analyzer sees ONLY the PCM samples — never the source signature numbers — so
|
||||
// recovery is genuinely by ear.
|
||||
fn voice_f0(samples: [Int], sr: Int) -> Int {
|
||||
let n: Int = native_list_len(samples)
|
||||
let start: Int = n / 4
|
||||
let end: Int = n * 3 / 4
|
||||
// bound the analysis window so accumulators can never overflow on long input
|
||||
if end - start > 6000 {
|
||||
end = start + 6000
|
||||
}
|
||||
let minlag: Int = sr / 300
|
||||
let maxlag: Int = sr / 75
|
||||
let best: Int = 0
|
||||
let bestlag: Int = minlag
|
||||
let lag: Int = minlag
|
||||
while lag <= maxlag {
|
||||
let sum: Int = 0
|
||||
let i: Int = start
|
||||
while i < end {
|
||||
let ai: Int = native_list_get(samples, i)
|
||||
let bi: Int = native_list_get(samples, i + lag)
|
||||
sum = sum + ai * bi / 256
|
||||
i = i + 2
|
||||
}
|
||||
if sum > best {
|
||||
best = sum
|
||||
bestlag = lag
|
||||
}
|
||||
lag = lag + 1
|
||||
}
|
||||
if bestlag < 1 {
|
||||
bestlag = 1
|
||||
}
|
||||
return sr / bestlag
|
||||
}
|
||||
|
||||
fn voice_peak_in_band(samples: [Int], sr: Int, flo: Int, fhi: Int) -> Int {
|
||||
let n: Int = native_list_len(samples)
|
||||
let start: Int = n / 4
|
||||
let end: Int = n * 3 / 4
|
||||
// bound the DFT window: re/im are accumulated /4096, and re*re must stay in
|
||||
// int64 — cap terms so (window/2)*(peak_term) squared cannot overflow.
|
||||
if end - start > 3000 {
|
||||
end = start + 3000
|
||||
}
|
||||
let bestmag: Int = 0
|
||||
let bestf: Int = flo
|
||||
let f: Int = flo
|
||||
while f <= fhi {
|
||||
let re: Int = 0
|
||||
let im: Int = 0
|
||||
let i: Int = start
|
||||
while i < end {
|
||||
let x: Int = native_list_get(samples, i)
|
||||
let ph: Int = i * f * 65536 / sr
|
||||
ph = ph - (ph / 65536) * 65536
|
||||
let cq: Int = sp_cos(ph)
|
||||
let sq: Int = sp_sin(ph)
|
||||
re = re + x * cq / 4096
|
||||
im = im + x * sq / 4096
|
||||
i = i + 2
|
||||
}
|
||||
let mag: Int = re * re + im * im
|
||||
if mag > bestmag {
|
||||
bestmag = mag
|
||||
bestf = f
|
||||
}
|
||||
f = f + 25
|
||||
}
|
||||
return bestf
|
||||
}
|
||||
|
||||
// Analyze a heard sustained /AA/ -> a full voice signature (by ear).
|
||||
fn voice_analyze(samples: [Int], sr: Int) -> [String] {
|
||||
let f0: Int = voice_f0(samples, sr)
|
||||
let f1: Int = voice_peak_in_band(samples, sr, 450, 1150)
|
||||
let kf: Int = 1000 * f1 / 730
|
||||
let f0e: Int = f0 * 85 / 100
|
||||
return voice_new("imitated", f0, f0e, kf, 1000, 1000, 8)
|
||||
}
|
||||
+19
-19
@@ -1,20 +1,20 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn lex_word(entry: Any) -> String
|
||||
extern fn lex_pos(entry: Any) -> String
|
||||
extern fn lex_form(entry: Any, idx: Int) -> String
|
||||
extern fn lex_class(entry: Any) -> String
|
||||
extern fn make_entry(word: String, pos: String, f0: String, f1: String, f2: String, f3: String, f4: String, cls: String) -> Any
|
||||
extern fn make_entry2(word: String, pos: String, f0: String, f1: String, cls: String) -> Any
|
||||
extern fn make_entry3(word: String, pos: String, f0: String, f1: String, f2: String, cls: String) -> Any
|
||||
extern fn make_entry1(word: String, pos: String, f0: String, cls: String) -> Any
|
||||
extern fn build_vocab() -> Any
|
||||
extern fn get_vocab() -> Any
|
||||
extern fn vocab_lookup(word: String, lang_code: String) -> Any
|
||||
extern fn vocab_lookup_en(word: String) -> Any
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
extern fn lex_word(entry: [String]) -> String
|
||||
extern fn lex_pos(entry: [String]) -> String
|
||||
extern fn lex_form(entry: [String], idx: Int) -> String
|
||||
extern fn lex_class(entry: [String]) -> String
|
||||
extern fn make_entry(word: String, pos: String, f0: String, f1: String, f2: String, f3: String, f4: String, cls: String) -> [String]
|
||||
extern fn make_entry2(word: String, pos: String, f0: String, f1: String, cls: String) -> [String]
|
||||
extern fn make_entry3(word: String, pos: String, f0: String, f1: String, f2: String, cls: String) -> [String]
|
||||
extern fn make_entry1(word: String, pos: String, f0: String, cls: String) -> [String]
|
||||
extern fn build_vocab() -> [[String]]
|
||||
extern fn get_vocab() -> [[String]]
|
||||
extern fn vocab_lookup(word: String, lang_code: String) -> [String]
|
||||
extern fn vocab_lookup_en(word: String) -> [String]
|
||||
extern fn vocab_synonym(word: String, lang_register: String, lang_code: String) -> String
|
||||
extern fn vocab_by_pos(pos: String) -> Any
|
||||
extern fn vocab_by_class(cls: String) -> Any
|
||||
extern fn entry_found(entry: Any) -> Bool
|
||||
extern fn entry_word(entry: Any) -> String
|
||||
extern fn entry_pos(entry: Any) -> String
|
||||
extern fn entry_form(entry: Any, n: Int) -> String
|
||||
extern fn vocab_by_pos(pos: String) -> [[String]]
|
||||
extern fn vocab_by_class(cls: String) -> [[String]]
|
||||
extern fn entry_found(entry: [String]) -> Bool
|
||||
extern fn entry_word(entry: [String]) -> String
|
||||
extern fn entry_pos(entry: [String]) -> String
|
||||
extern fn entry_form(entry: [String], n: Int) -> String
|
||||
|
||||
@@ -0,0 +1,244 @@
|
||||
// voice-ingest.el - The LIVE VOICE LOOP reshape + ingest-as-geometry.
|
||||
//
|
||||
// EL cannot read a binary WAV (fs_read NUL-truncates), so the thin-medium DSP
|
||||
// extractor is periph's `voiceprint` (autocorr F0 + LPC formants), equivalent to
|
||||
// our own voice_analyze. This module: (1) RESHAPE the voiceprint JSON (TEXT) into
|
||||
// the organ voice-signature schema; (2) INGEST it as a GEOMETRY manifold in the
|
||||
// engram and engram_save it to a file; (3) READ the target signature BACK from
|
||||
// that geometry (engram_load + scan + filter), never from the json or a table.
|
||||
// HONEST: this reaches for pitch + a coarse vocal-tract scale (kf). It is NOT a
|
||||
// clone — no glottal timbre, vowel-space, or articulation is captured.
|
||||
|
||||
fn parse_leading_int(s: String) -> Int {
|
||||
let n: Int = str_len(s)
|
||||
let i: Int = 0
|
||||
let v: Int = 0
|
||||
let started: Int = 0
|
||||
while i < n {
|
||||
let c: Int = str_char_code(s, i)
|
||||
if c >= 48 {
|
||||
if c <= 57 {
|
||||
v = v * 10 + (c - 48)
|
||||
started = 1
|
||||
i = i + 1
|
||||
} else {
|
||||
i = n
|
||||
}
|
||||
} else {
|
||||
if started == 1 {
|
||||
i = n
|
||||
} else {
|
||||
i = i + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// voiceprint JSON -> organ voice-signature source file; returns [f0,f0_end,kf,f1,f2,f3].
|
||||
fn reshape_voiceprint(vppath: String, outjson: String) -> [Int] {
|
||||
let j: String = fs_read(vppath)
|
||||
let f0: Int = parse_uint_from(j, "f0_hz\":")
|
||||
let fp: Int = str_index_of(j, "formants_hz")
|
||||
let tail: String = str_slice(j, fp, fp + 120)
|
||||
let br: Int = str_index_of(tail, "[")
|
||||
let arr: String = str_slice(tail, br + 1, str_len(tail))
|
||||
let f1: Int = parse_leading_int(arr)
|
||||
let c1: Int = str_index_of(arr, ",")
|
||||
let a2: String = str_slice(arr, c1 + 1, str_len(arr))
|
||||
let f2: Int = parse_leading_int(a2)
|
||||
let c2: Int = str_index_of(a2, ",")
|
||||
let a3: String = str_slice(a2, c2 + 1, str_len(a2))
|
||||
let f3: Int = parse_leading_int(a3)
|
||||
let f0e: Int = f0 * 85 / 100
|
||||
// derive kf honestly: coarse vocal-tract scale from the formant pattern
|
||||
let t1: Int = 1000 * f1 / 500
|
||||
let t2: Int = 1000 * f2 / 1500
|
||||
let t3: Int = 1000 * f3 / 2500
|
||||
let kf: Int = (t1 + t2 + t3) / 3
|
||||
if kf < 800 {
|
||||
kf = 800
|
||||
}
|
||||
if kf > 1400 {
|
||||
kf = 1400
|
||||
}
|
||||
let js: String = "{\"dataset\":\"will-voice-signature\",\"primitive_type\":\"voice\",\"grounding\":\"measured\",\"provenance\":\"Will live 30s read 2026-08-15 (elp/data/live/will30_clean.wav, 27.0s) SUPERSEDES the coarse 10s sample; F0+formants via periph voiceprint (autocorr+LPC), averaged over his full vowel set. Still the 11-number average: no coarticulation/prosody. COARSE — pitch + vocal-tract scale, NOT a clone.\",\"records\":[{\"key\":\"will\",\"features\":{\"source\":\"live-mic\"},\"attributes\":{\"f0\":" + int_to_str(f0) + ",\"f0_end\":" + int_to_str(f0e) + ",\"kf\":" + int_to_str(kf) + ",\"f1\":" + int_to_str(f1) + ",\"f2\":" + int_to_str(f2) + ",\"f3\":" + int_to_str(f3) + "}}]}"
|
||||
let okw: Bool = fs_write(outjson, js)
|
||||
let r: [Int] = native_list_empty()
|
||||
let r = native_list_append(r, f0)
|
||||
let r = native_list_append(r, f0e)
|
||||
let r = native_list_append(r, kf)
|
||||
let r = native_list_append(r, f1)
|
||||
let r = native_list_append(r, f2)
|
||||
let r = native_list_append(r, f3)
|
||||
return r
|
||||
}
|
||||
|
||||
// Ingest the signature as a manifold (a set-hub + the will node + a member edge)
|
||||
// and engram_save it to a reloadable file. grounding:measured self-declared.
|
||||
fn ingest_voice(sig: [Int], savepath: String) -> Int {
|
||||
let f0: Int = native_list_get(sig, 0)
|
||||
let f0e: Int = native_list_get(sig, 1)
|
||||
let kf: Int = native_list_get(sig, 2)
|
||||
let f1: Int = native_list_get(sig, 3)
|
||||
let f2: Int = native_list_get(sig, 4)
|
||||
let f3: Int = native_list_get(sig, 5)
|
||||
let hub: String = engram_node("voice-signature-set will grounding=measured src=periph-voiceprint", "VoiceSet", 90)
|
||||
let cont: String = "voice will | f0=" + int_to_str(f0) + " f0_end=" + int_to_str(f0e) + " kf=" + int_to_str(kf) + " f1=" + int_to_str(f1) + " f2=" + int_to_str(f2) + " f3=" + int_to_str(f3) + " grounding=measured src=periph-voiceprint-30s supersedes=prior-voice-region prov=COARSE-pitch+tractscale-NOT-a-clone"
|
||||
let id: String = engram_node(cont, "Voice", 90)
|
||||
engram_connect(id, hub, 90, "member_of")
|
||||
let oks: Bool = engram_save(savepath)
|
||||
return 1
|
||||
}
|
||||
|
||||
// READ the target voice back FROM the ingested geometry (engram_load + scan +
|
||||
// client-filter for "voice will"). Returns [f0,f0_end,kf,f1,f2,f3] or empty.
|
||||
fn load_voice(savepath: String) -> [Int] {
|
||||
let ok: Bool = engram_load(savepath)
|
||||
let r: [Int] = native_list_empty()
|
||||
if ok == false {
|
||||
return r
|
||||
}
|
||||
let j: String = engram_scan_nodes_json(200, 0)
|
||||
let p: Int = str_index_of(j, "voice will ")
|
||||
if p < 0 {
|
||||
return r
|
||||
}
|
||||
let win: String = str_slice(j, p, p + 200)
|
||||
let r = native_list_append(r, parse_uint_from(win, "f0="))
|
||||
let r = native_list_append(r, parse_uint_from(win, "f0_end="))
|
||||
let r = native_list_append(r, parse_uint_from(win, "kf="))
|
||||
let r = native_list_append(r, parse_uint_from(win, "f1="))
|
||||
let r = native_list_append(r, parse_uint_from(win, "f2="))
|
||||
let r = native_list_append(r, parse_uint_from(win, "f3="))
|
||||
return r
|
||||
}
|
||||
|
||||
// ---- Vowel-space + prosody: ingest-as-geometry + read-back (no source layer) --
|
||||
// vowel target lookup from the ingested vowel-space manifold: sym -> [f1,f2,f3].
|
||||
fn vmap_get(vmap: [String], code: String) -> [Int] {
|
||||
let out: [Int] = native_list_empty()
|
||||
let id: String = sp_map_get(vmap, code)
|
||||
if str_eq(id, "") {
|
||||
return out
|
||||
}
|
||||
let f1: Int = parse_uint_from(id, "f1=")
|
||||
if f1 <= 0 {
|
||||
return out
|
||||
}
|
||||
let out = native_list_append(out, f1)
|
||||
let out = native_list_append(out, parse_uint_from(id, "f2="))
|
||||
let out = native_list_append(out, parse_uint_from(id, "f3="))
|
||||
return out
|
||||
}
|
||||
|
||||
// Ingest his measured vowel space + prosody as ONE manifold (VowelSpace hub +
|
||||
// per-vowel target nodes + a prosody node) and engram_save it. Fresh empty store
|
||||
// per run => set-replace, no duplicate.
|
||||
fn ingest_voicegeom(vpath: String, ppath: String, savepath: String) -> Int {
|
||||
let hub: String = engram_node("vowel-space-set will grounding=measured src=lpc-formant-track-30s", "VowelSpace", 90)
|
||||
let content: String = fs_read(vpath)
|
||||
let lines: [String] = str_split(content, "\n")
|
||||
let nl: Int = native_list_len(lines)
|
||||
let li: Int = 0
|
||||
while li < nl {
|
||||
let line: String = native_list_get(lines, li)
|
||||
let ok: Int = 1
|
||||
if str_len(line) < 5 {
|
||||
ok = 0
|
||||
}
|
||||
if ok == 1 {
|
||||
if str_char_code(line, 0) == 35 {
|
||||
ok = 0
|
||||
}
|
||||
}
|
||||
if ok == 1 {
|
||||
let f: [String] = str_split(line, "|")
|
||||
if native_list_len(f) >= 5 {
|
||||
let sym: String = native_list_get(f, 0)
|
||||
let cont: String = "vowel-target will " + sym + " | f1=" + native_list_get(f, 1) + " f2=" + native_list_get(f, 2) + " f3=" + native_list_get(f, 3) + " n=" + native_list_get(f, 4) + " grounding=measured src=lpc-formant-track-30s"
|
||||
let id: String = engram_node(cont, "VowelTarget", 90)
|
||||
engram_connect(id, hub, 90, "member_of")
|
||||
}
|
||||
}
|
||||
li = li + 1
|
||||
}
|
||||
let pc: String = fs_read(ppath)
|
||||
let plines: [String] = str_split(pc, "\n")
|
||||
let pnl: Int = native_list_len(plines)
|
||||
let pi: Int = 0
|
||||
while pi < pnl {
|
||||
let pl: String = native_list_get(plines, pi)
|
||||
let ok2: Int = 1
|
||||
if str_len(pl) < 5 {
|
||||
ok2 = 0
|
||||
}
|
||||
if ok2 == 1 {
|
||||
if str_char_code(pl, 0) == 35 {
|
||||
ok2 = 0
|
||||
}
|
||||
}
|
||||
if ok2 == 1 {
|
||||
let pf: [String] = str_split(pl, "|")
|
||||
if native_list_len(pf) >= 4 {
|
||||
let pcont: String = "prosody will | f0_median=" + native_list_get(pf, 0) + " f0_min=" + native_list_get(pf, 1) + " f0_max=" + native_list_get(pf, 2) + " declination=" + native_list_get(pf, 3) + " src=f0-contour-30s"
|
||||
let pid: String = engram_node(pcont, "Prosody", 90)
|
||||
engram_connect(pid, hub, 90, "prosody_of")
|
||||
}
|
||||
}
|
||||
pi = pi + 1
|
||||
}
|
||||
let oks: Bool = engram_save(savepath)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Read the vowel-space back from geometry; prosody folded under key __PROSODY__.
|
||||
fn load_voicegeom(savepath: String) -> [String] {
|
||||
let m: [String] = native_list_empty()
|
||||
let ok: Bool = engram_load(savepath)
|
||||
if ok == false {
|
||||
return m
|
||||
}
|
||||
let j: String = engram_scan_nodes_json(400, 0)
|
||||
let jl: Int = str_len(j)
|
||||
let off: Int = 0
|
||||
while off < jl {
|
||||
let rest: String = str_slice(j, off, jl)
|
||||
let p: Int = str_index_of(rest, "vowel-target will ")
|
||||
if p < 0 {
|
||||
off = jl
|
||||
} else {
|
||||
let abs: Int = off + p
|
||||
let win: String = str_slice(j, abs, abs + 140)
|
||||
let after: String = str_slice(win, 18, str_len(win))
|
||||
let sp: Int = str_index_of(after, " ")
|
||||
if sp > 0 {
|
||||
let sym: String = str_slice(after, 0, sp)
|
||||
m = native_list_append(m, sym)
|
||||
m = native_list_append(m, win)
|
||||
}
|
||||
off = abs + 18
|
||||
}
|
||||
}
|
||||
let pp: Int = str_index_of(j, "prosody will ")
|
||||
if pp >= 0 {
|
||||
let pwin: String = str_slice(j, pp, pp + 160)
|
||||
m = native_list_append(m, "__PROSODY__")
|
||||
m = native_list_append(m, pwin)
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Prosody stats [f0_median, f0_min, f0_max, declination] read from geometry.
|
||||
fn prosody_from(vmap: [String]) -> [Int] {
|
||||
let out: [Int] = native_list_empty()
|
||||
let id: String = sp_map_get(vmap, "__PROSODY__")
|
||||
if str_eq(id, "") {
|
||||
return out
|
||||
}
|
||||
let out = native_list_append(out, parse_uint_from(id, "f0_median="))
|
||||
let out = native_list_append(out, parse_uint_from(id, "f0_min="))
|
||||
let out = native_list_append(out, parse_uint_from(id, "f0_max="))
|
||||
let out = native_list_append(out, parse_uint_from(id, "declination="))
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// voice-profile.el - The VOICE signature as a pluggable PROFILE.
|
||||
//
|
||||
// Exact mirror of surface-profile.el / language-profile.el: a voice is a
|
||||
// [String] slot-map read via voice_get, the SAME mechanism the realizer uses
|
||||
// for language and surface. Where an instrument signature (a few dozen numbers)
|
||||
// is the timbre of a musical tone, a VOICE signature is the timbre of the vocal
|
||||
// tract — the instrument that renders LANGUAGE-meaning as SPEECH on the audio
|
||||
// surface. Physics (source-filter), not a recorded corpus.
|
||||
//
|
||||
// The signature is a few numbers, all integer (EL float arithmetic is unusable):
|
||||
// name - label
|
||||
// f0 - base pitch, Hz (glottal source rate at utterance start)
|
||||
// f0_end - pitch at utterance end (declination -> falling = declarative)
|
||||
// kf - formant scale in PER-MILLE (1000 = x1.0). Encodes vocal-tract
|
||||
// length: shorter tract (child/female) -> higher kf. Scales every
|
||||
// phoneme's nominal formant: F_actual = F_nominal * kf / 1000.
|
||||
// dur - speaking-rate multiplier in per-mille (1000 = nominal; >1000 slower)
|
||||
// tilt - source spectral tilt (per-mille; higher = darker/steeper rolloff)
|
||||
// breath - breathiness 0..100 (aspiration mixed into the source)
|
||||
//
|
||||
// A voice is grabbed BY EAR (voice_analyze in speech.el extracts these numbers
|
||||
// from a short PCM sample — an impression, not 10h of training), or declared.
|
||||
|
||||
fn voice_new(name: String, f0: Int, f0_end: Int, kf: Int, dur: Int, tilt: Int, breath: Int) -> [String] {
|
||||
let r: [String] = native_list_empty()
|
||||
let r = native_list_append(r, "name")
|
||||
let r = native_list_append(r, name)
|
||||
let r = native_list_append(r, "f0")
|
||||
let r = native_list_append(r, int_to_str(f0))
|
||||
let r = native_list_append(r, "f0_end")
|
||||
let r = native_list_append(r, int_to_str(f0_end))
|
||||
let r = native_list_append(r, "kf")
|
||||
let r = native_list_append(r, int_to_str(kf))
|
||||
let r = native_list_append(r, "dur")
|
||||
let r = native_list_append(r, int_to_str(dur))
|
||||
let r = native_list_append(r, "tilt")
|
||||
let r = native_list_append(r, int_to_str(tilt))
|
||||
let r = native_list_append(r, "breath")
|
||||
let r = native_list_append(r, int_to_str(breath))
|
||||
return r
|
||||
}
|
||||
|
||||
// Accessor — identical convention to surface_get / lang_get.
|
||||
fn voice_get(profile: [String], key: String) -> String {
|
||||
let n: Int = native_list_len(profile)
|
||||
let i: Int = 0
|
||||
while i < n - 1 {
|
||||
let k: String = native_list_get(profile, i)
|
||||
if str_eq(k, key) {
|
||||
return native_list_get(profile, i + 1)
|
||||
}
|
||||
let i = i + 2
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
fn voice_get_int(profile: [String], key: String) -> Int {
|
||||
let s: String = voice_get(profile, key)
|
||||
if str_eq(s, "") {
|
||||
return 0
|
||||
}
|
||||
return str_to_int(s)
|
||||
}
|
||||
|
||||
// -- Built-in voices ---------------------------------------------------------
|
||||
|
||||
// Neuron's own voice: calm, precise, androgynous-neutral. Low-ish base pitch,
|
||||
// gentle declination, near-neutral vocal-tract length.
|
||||
fn voice_neuron() -> [String] {
|
||||
return voice_new("neuron", 112, 96, 1020, 1000, 1000, 6)
|
||||
}
|
||||
|
||||
// Will's voice signature, built from the INGESTED geometry (f0/f0_end/kf read
|
||||
// back from the will-voice manifold — passed in, never hardcoded). Composable
|
||||
// with an accent transform exactly like voice_neuron() (voice (+) accent).
|
||||
fn voice_will(f0: Int, f0_end: Int, kf: Int) -> [String] {
|
||||
return voice_new("will", f0, f0_end, kf, 1000, 1000, 6)
|
||||
}
|
||||
|
||||
// A deliberately DISTINCT target voice for the imitation proof: higher pitch,
|
||||
// shorter vocal tract (kf=1.20) -> a clearly different speaker. Neuron will
|
||||
// HEAR a sample of this voice and reconstruct these numbers by ear.
|
||||
fn voice_target_a() -> [String] {
|
||||
return voice_new("target_a", 178, 150, 1200, 950, 1000, 10)
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
// speech-accent-demo.el - PROOF: Neuron speaks with a BRITISH accent, where the
|
||||
// accent is a TRANSFORM composed onto the voice (voice (+) accent, separable),
|
||||
// INGESTED as geometry (not a table). Same voice, accent toggled on/off = RP/GA.
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/Users/will/Development/neuron-technologies/foundation/el/.claude/worktrees/agent-acc02900ef4ade35e/elp/tests/examples/out/"
|
||||
|
||||
// LEARN: base phonetics + lexicon + the British-RP accent transform, all as
|
||||
// ingested geometry (source -> manifold -> engram).
|
||||
let pmap: [String] = ingest_phonetics("elp/data/phonetics.psv")
|
||||
let lmap: [String] = ingest_lexicon("elp/data/lexicon.psv")
|
||||
let amap: [String] = ingest_accent("elp/data/british-accent.psv")
|
||||
println("[learn] phonemes=" + int_to_str(native_list_len(pmap) / 2) + " words=" + int_to_str(native_list_len(lmap) / 2) + " accent_targets=" + int_to_str(native_list_len(amap) / 2))
|
||||
|
||||
let neuron: [String] = voice_neuron()
|
||||
let noaccent: [String] = native_list_empty()
|
||||
|
||||
// -- Sentence 1: "I am Neuron." from meaning ----------------------------
|
||||
let fr1: [String] = sem_frame("describe", "I", "Neuron", "")
|
||||
let t1: String = sem_realize(fr1)
|
||||
let c1: [String] = text_phonemes(lmap, t1)
|
||||
println("[s1] " + t1 + " :: " + list_join(c1, " "))
|
||||
|
||||
// separability: SAME voice, accent OFF (GA) vs ON (RP)
|
||||
let ga: [Int] = synth_codes_accent(c1, neuron, pmap, noaccent)
|
||||
let okga: Bool = write_wav(ga, 16000, outdir + "ga-neuron.wav")
|
||||
let br1: [Int] = synth_codes_accent(c1, neuron, pmap, amap)
|
||||
let okb1: Bool = write_wav(br1, 16000, outdir + "british-neuron.wav")
|
||||
|
||||
// -- Sentence 2: showcases NON-RHOTICITY --------------------------------
|
||||
let fr2: [String] = sem_frame("describe", "I", "here", "")
|
||||
let t2: String = sem_realize(fr2)
|
||||
let c2: [String] = text_phonemes(lmap, t2)
|
||||
let c2rp: [String] = apply_rhoticity(c2, pmap)
|
||||
println("[s2] " + t2 + " :: GA=" + list_join(c2, " ") + " RP=" + list_join(c2rp, " "))
|
||||
let br2: [Int] = synth_codes_accent(c2, neuron, pmap, amap)
|
||||
let okb2: Bool = write_wav(br2, 16000, outdir + "british-2.wav")
|
||||
|
||||
// show an RP override read straight from the accent geometry
|
||||
let ovAA: [Int] = accent_formants(amap, "AA")
|
||||
if native_list_len(ovAA) >= 3 {
|
||||
println("[accent-geometry] AA(LOT) RP f1=" + int_to_str(native_list_get(ovAA, 0)) + " f2=" + int_to_str(native_list_get(ovAA, 1)) + " (base GA 730/1090) [PROVISIONAL]")
|
||||
}
|
||||
println("[done] ga-neuron=" + bool_to_str(okga) + " british-neuron=" + bool_to_str(okb1) + " british-2=" + bool_to_str(okb2))
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
// speech-demo.el - PROOF: Neuron speaks from MEANING, rendered through INGESTED
|
||||
// phonetic geometry, own-core, plus voice-by-IMITATION. Built by concatenating
|
||||
// the elp realizer + voice-profile + speech-ingest + speech, then this main.
|
||||
//
|
||||
// LEARN : ingest acoustic-phonetics + lexicon SOURCES -> phoneme manifold in
|
||||
// the engram (source -> manifold -> merge).
|
||||
// MEANING : sem_frame("describe","I","Neuron","") -> sem_realize -> "I am Neuron."
|
||||
// PHONES : words -> phoneme codes, READ from the ingested lexicon geometry.
|
||||
// RENDER : superpose formant resonances (read from engram) over a glottal
|
||||
// source -> own-core PCM/WAV, in Neuron's own voice.
|
||||
// IMITATE : HEAR a short sample of a different voice -> extract its signature
|
||||
// by ear (autocorrelation pitch + integer-DFT formant) -> render new
|
||||
// speech in that voice. An impression, not a corpus.
|
||||
|
||||
fn speak_report(tag: String, codes: [String], voice: [String], pmap: [String], path: String) -> [Int] {
|
||||
let s: [Int] = synth_codes(codes, voice, pmap)
|
||||
let ok: Bool = write_wav(s, 16000, path)
|
||||
println(tag + " samples=" + int_to_str(native_list_len(s)) + " ok=" + bool_to_str(ok) + " -> " + path)
|
||||
return s
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/private/tmp/claude-501/-Users-will/6531446d-bc27-4095-930b-e04777c3db4f/scratchpad/"
|
||||
|
||||
// -- LEARN: ingest the speech primitives as geometry --------------------
|
||||
let pmap: [String] = ingest_phonetics("elp/data/phonetics.psv")
|
||||
let lmap: [String] = ingest_lexicon("elp/data/lexicon.psv")
|
||||
let saved: Bool = engram_save(outdir + "phoneme-manifold.json")
|
||||
println("[learn] phonemes=" + int_to_str(native_list_len(pmap) / 2) + " words=" + int_to_str(native_list_len(lmap) / 2) + " manifold_saved=" + bool_to_str(saved))
|
||||
|
||||
// sanity: show that AA's formants came from ingested geometry, not code
|
||||
let aa: [Int] = phon_geo(pmap, "AA")
|
||||
let aaF1: Int = native_list_get(aa, 0)
|
||||
let aaF2: Int = native_list_get(aa, 1)
|
||||
println("[read-geometry] AA F1=" + int_to_str(aaF1) + " F2=" + int_to_str(aaF2) + " (parsed from engram node)")
|
||||
|
||||
// -- MEANING -> WORDS via the realizer's language faculty ----------------
|
||||
let frame: [String] = sem_frame("describe", "I", "Neuron", "")
|
||||
let text: String = sem_realize(frame)
|
||||
println("[meaning->text] " + text)
|
||||
|
||||
// -- WORDS -> PHONEMES (read from ingested lexicon geometry) --------------
|
||||
let codes: [String] = text_phonemes(lmap, text)
|
||||
println("[phonemes] " + list_join(codes, " "))
|
||||
|
||||
// -- RENDER in Neuron's own voice ----------------------------------------
|
||||
let neuron: [String] = voice_neuron()
|
||||
let s1: [Int] = speak_report("[speak neuron]", codes, neuron, pmap, outdir + "neuron.wav")
|
||||
|
||||
// -- IMITATION: hear a distinct voice, recover its signature, re-render ---
|
||||
let vA: [String] = voice_target_a()
|
||||
let hcodes: [String] = native_list_empty()
|
||||
hcodes = native_list_append(hcodes, "SIL")
|
||||
let z: Int = 0
|
||||
while z < 6 {
|
||||
hcodes = native_list_append(hcodes, "AA")
|
||||
z = z + 1
|
||||
}
|
||||
hcodes = native_list_append(hcodes, "SIL")
|
||||
let heard: [Int] = synth_codes(hcodes, vA, pmap)
|
||||
let okh: Bool = write_wav(heard, 16000, outdir + "heard.wav")
|
||||
|
||||
let vB: [String] = voice_analyze(heard, 16000)
|
||||
println("[imitate] heard ACTUAL f0=" + voice_get(vA, "f0") + " kf=" + voice_get(vA, "kf"))
|
||||
println("[imitate] heard RECOVERED f0=" + voice_get(vB, "f0") + " kf=" + voice_get(vB, "kf") + " (extracted by ear from PCM)")
|
||||
let s2: [Int] = speak_report("[speak imitation]", codes, vB, pmap, outdir + "imitation.wav")
|
||||
|
||||
println("[done] rendered from meaning + ingested geometry; imitation from a heard sample.")
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
// speech-organ-demo.el - PROOF: the render now reads its phoneme + accent
|
||||
// GEOMETRY from the ingest ORGAN's saved engram files (engram_load +
|
||||
// engram_scan_nodes_json + cache), not a same-run hand-load. The British accent
|
||||
// is still a composed transform-geometry (voice (+) accent, separable). Numbers
|
||||
// come from the organ manifold; the .psv supplies only categorical vowel-class.
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/Users/will/Development/neuron-technologies/foundation/el/.claude/worktrees/agent-acc02900ef4ade35e/elp/tests/examples/out/"
|
||||
|
||||
// engram-independent caches from source (survive engram_load replacement)
|
||||
let vset: [String] = organ_vset("elp/data/phonetics.psv")
|
||||
let lmap: [String] = organ_lex("elp/data/lexicon.psv")
|
||||
// ORGAN read: phonetics FIRST (cache), THEN accent (engram_load replaces store)
|
||||
let pmap: [String] = organ_pmap("elp/data/phonetics-formants.engram.json")
|
||||
let amap: [String] = organ_amap("elp/data/british-accent.engram.json")
|
||||
println("[organ] phon_syms=" + int_to_str(native_list_len(pmap) / 2) + " accent_syms=" + int_to_str(native_list_len(amap) / 2) + " vowels=" + int_to_str(native_list_len(vset)) + " words=" + int_to_str(native_list_len(lmap) / 2))
|
||||
|
||||
// prove the numbers came from the organ node content
|
||||
let g: [Int] = phon_geo(pmap, "AA")
|
||||
println("[organ-read] phoneme AA f1=" + int_to_str(native_list_get(g, 0)) + " f2=" + int_to_str(native_list_get(g, 1)) + " f3=" + int_to_str(native_list_get(g, 2)) + " (P&B1952 MEASURED)")
|
||||
let ov: [Int] = accent_formants(amap, "AA")
|
||||
if native_list_len(ov) >= 3 {
|
||||
println("[organ-read] accent AA(LOT) f1=" + int_to_str(native_list_get(ov, 0)) + " f2=" + int_to_str(native_list_get(ov, 1)) + " (DERIVED RP, PROVISIONAL)")
|
||||
}
|
||||
println("[organ-read] non_rhotic=" + int_to_str(is_nonrhotic(amap)))
|
||||
|
||||
let neuron: [String] = voice_neuron()
|
||||
let noacc: [String] = native_list_empty()
|
||||
|
||||
// Sentence 1: "I am Neuron." from meaning; GA vs RP = separable toggle
|
||||
let t1: String = sem_realize(sem_frame("describe", "I", "Neuron", ""))
|
||||
let c1: [String] = text_phonemes(lmap, t1)
|
||||
println("[s1] " + t1 + " :: " + list_join(c1, " "))
|
||||
let ga: [Int] = synth_codes_accent(c1, neuron, pmap, noacc, vset)
|
||||
let okga: Bool = write_wav(ga, 16000, outdir + "ga-neuron-organ.wav")
|
||||
let br1: [Int] = synth_codes_accent(c1, neuron, pmap, amap, vset)
|
||||
let okb1: Bool = write_wav(br1, 16000, outdir + "british-neuron-organ.wav")
|
||||
|
||||
// Sentence 2: non-rhoticity showcase
|
||||
let t2: String = sem_realize(sem_frame("describe", "I", "here", ""))
|
||||
let c2: [String] = text_phonemes(lmap, t2)
|
||||
let c2rp: [String] = apply_rhoticity(c2, vset)
|
||||
println("[s2] " + t2 + " :: GA=" + list_join(c2, " ") + " RP=" + list_join(c2rp, " "))
|
||||
let br2: [Int] = synth_codes_accent(c2, neuron, pmap, amap, vset)
|
||||
let okb2: Bool = write_wav(br2, 16000, outdir + "british-2-organ.wav")
|
||||
|
||||
println("[done] ga-organ=" + bool_to_str(okga) + " british-organ=" + bool_to_str(okb1) + " british-2-organ=" + bool_to_str(okb2))
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
// speech-voice-demo.el - LIVE VOICE LOOP (stand-in test). Capture -> voiceprint
|
||||
// -> reshape -> INGEST AS GEOMETRY -> read the target back FROM geometry -> the
|
||||
// EL projector renders a line reaching for that voice. Stand-in "Will" = the
|
||||
// voiceprint of imitation.wav. HONEST: pitch + coarse vocal-tract scale, NOT a clone.
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/Users/will/Development/neuron-technologies/foundation/el/.claude/worktrees/agent-acc02900ef4ade35e/elp/tests/examples/out/"
|
||||
let vp: String = "/private/tmp/claude-501/-Users-will/6531446d-bc27-4095-930b-e04777c3db4f/scratchpad/will-voiceprint.json"
|
||||
|
||||
// 1+2: reshape voiceprint JSON -> organ voice-signature source
|
||||
let sig0: [String] = native_list_empty()
|
||||
let sig: [Int] = reshape_voiceprint(vp, "elp/data/will-voice.json")
|
||||
// 3: ingest as geometry + engram_save a reloadable manifold file
|
||||
let ig: Int = ingest_voice(sig, "elp/data/will-voice.engram.json")
|
||||
// 4: READ the target back FROM geometry (engram_load + scan + filter)
|
||||
let g: [Int] = load_voice("elp/data/will-voice.engram.json")
|
||||
println("[voice-geometry] read from manifold: f0=" + int_to_str(native_list_get(g, 0)) + " f0_end=" + int_to_str(native_list_get(g, 1)) + " kf=" + int_to_str(native_list_get(g, 2)) + " f1=" + int_to_str(native_list_get(g, 3)) + " f2=" + int_to_str(native_list_get(g, 4)) + " f3=" + int_to_str(native_list_get(g, 5)) + " (measured, COARSE — not a clone)")
|
||||
|
||||
// phoneme geometry from the organ (loaded AFTER the voice sig is cached in EL)
|
||||
let pmap: [String] = organ_pmap("elp/data/phonetics-formants.engram.json")
|
||||
let lmap: [String] = organ_lex("elp/data/lexicon.psv")
|
||||
|
||||
// 5: render a line FROM MEANING in Will's voice
|
||||
let vw: [String] = voice_will(native_list_get(g, 0), native_list_get(g, 1), native_list_get(g, 2))
|
||||
let t: String = sem_realize(sem_frame("greet", "Will", "", ""))
|
||||
let codes: [String] = text_phonemes(lmap, t)
|
||||
println("[render] \"" + t + "\" :: " + list_join(codes, " ") + " in voice=will f0=" + int_to_str(voice_get_int(vw, "f0")) + " kf=" + int_to_str(voice_get_int(vw, "kf")))
|
||||
let samples: [Int] = synth_codes(codes, vw, pmap)
|
||||
let ok: Bool = write_wav(samples, 16000, outdir + "will-reply.wav")
|
||||
println("[done] will-reply.wav=" + bool_to_str(ok))
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
// speech-voice-demo2.el - LIVE VOICE LOOP on Will's richer 30s read, with a
|
||||
// GEOMETRIC SET-REPLACE of the voice_will manifold (supersede the coarse 10s
|
||||
// region, insert the 30s region — no duplicate node, no per-node CRUD; Will's
|
||||
// standing rule f999c5ff). HONEST: 30s steadies the 11-number average over more
|
||||
// of his vowels, but it is still one formant triple with no coarticulation or
|
||||
// prosody — closer but still synthetic, not a clone.
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/Users/will/Development/neuron-technologies/foundation/el/.claude/worktrees/agent-acc02900ef4ade35e/elp/tests/examples/out/"
|
||||
let vp: String = "/private/tmp/claude-501/-Users-will/6531446d-bc27-4095-930b-e04777c3db4f/scratchpad/will30-voiceprint.json"
|
||||
let manifest: String = "elp/data/will-voice.engram.json"
|
||||
|
||||
// --- SET-REPLACE step 1: read the PRIOR region (text read of the manifold
|
||||
// file — no engram_load, so the store stays clean) and report what is
|
||||
// being superseded. ---
|
||||
let prior: String = fs_read(manifest)
|
||||
let pp: Int = str_index_of(prior, "voice will ")
|
||||
if pp >= 0 {
|
||||
let pw: String = str_slice(prior, pp, pp + 200)
|
||||
println("[set-replace] superseding PRIOR voice region: f0=" + int_to_str(parse_uint_from(pw, "f0=")) + " kf=" + int_to_str(parse_uint_from(pw, "kf=")) + " f1=" + int_to_str(parse_uint_from(pw, "f1=")))
|
||||
}
|
||||
|
||||
// --- step 2: reshape the 30s voiceprint -> organ voice-signature source ---
|
||||
let sig: [Int] = reshape_voiceprint(vp, "elp/data/will-voice.json")
|
||||
|
||||
// --- step 3: INSERT the fresh 30s region into an EMPTY engram and save ->
|
||||
// wholesale replaces the manifold file (old region dropped, not edited,
|
||||
// not duplicated). This is the geometric set-replace. ---
|
||||
let ig: Int = ingest_voice(sig, manifest)
|
||||
|
||||
// --- step 4: READ the new target BACK from geometry ---
|
||||
let g: [Int] = load_voice(manifest)
|
||||
println("[voice-geometry] new region read from manifold: f0=" + int_to_str(native_list_get(g, 0)) + " f0_end=" + int_to_str(native_list_get(g, 1)) + " kf=" + int_to_str(native_list_get(g, 2)) + " f1=" + int_to_str(native_list_get(g, 3)) + " f2=" + int_to_str(native_list_get(g, 4)) + " f3=" + int_to_str(native_list_get(g, 5)) + " (measured 30s, COARSE — not a clone)")
|
||||
|
||||
// phoneme + lexicon geometry from the organ (loaded after the voice sig is
|
||||
// cached in EL, since engram_load replaces the store)
|
||||
let pmap: [String] = organ_pmap("elp/data/phonetics-formants.engram.json")
|
||||
let lmap: [String] = organ_lex("elp/data/lexicon.psv")
|
||||
|
||||
// --- step 5: render a fresh reply FROM MEANING in the 30s Will voice ---
|
||||
let vw: [String] = voice_will(native_list_get(g, 0), native_list_get(g, 1), native_list_get(g, 2))
|
||||
let t: String = sem_realize(sem_frame("greet", "Will", "", ""))
|
||||
let codes: [String] = text_phonemes(lmap, t)
|
||||
println("[render] \"" + t + "\" :: " + list_join(codes, " ") + " in voice=will f0=" + int_to_str(voice_get_int(vw, "f0")) + " kf=" + int_to_str(voice_get_int(vw, "kf")))
|
||||
let samples: [Int] = synth_codes(codes, vw, pmap)
|
||||
let ok: Bool = write_wav(samples, 16000, outdir + "will-reply2.wav")
|
||||
println("[done] will-reply2.wav=" + bool_to_str(ok))
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
// speech-voicegeom-demo.el - THE JUMP: render Will's VOWEL SPACE + PROSODY
|
||||
// (measured over 30s), not the single 11-number average. His vowels land at HIS
|
||||
// targets; pitch follows HIS melody. All read back FROM the ingested geometry.
|
||||
// INTERIM: the geometry was Python-measured (measure_voice.py, numpy LPC/F0) —
|
||||
// to be superseded by the engram-measures-audio path. No source layer.
|
||||
|
||||
fn main() {
|
||||
let outdir: String = "/Users/will/Development/neuron-technologies/foundation/el/.claude/worktrees/agent-acc02900ef4ade35e/elp/tests/examples/out/"
|
||||
|
||||
// 1: ingest vowel space + prosody as geometry (empty store -> save; set-replace)
|
||||
let ig: Int = ingest_voicegeom("elp/data/will-vowelspace.psv", "elp/data/will-prosody.psv", "elp/data/will-voicegeom.engram.json")
|
||||
// kf (vocal-tract scale for consonants) from the earlier will-voice manifold
|
||||
let sigv: [Int] = load_voice("elp/data/will-voice.engram.json")
|
||||
let kf: Int = native_list_get(sigv, 2)
|
||||
// 2: read vowel space + prosody back FROM geometry
|
||||
let vmap: [String] = load_voicegeom("elp/data/will-voicegeom.engram.json")
|
||||
let pros: [Int] = prosody_from(vmap)
|
||||
println("[geometry] vowels=" + int_to_str((native_list_len(vmap) - 2) / 2) + " prosody f0_median=" + int_to_str(native_list_get(pros, 0)) + " f0_min=" + int_to_str(native_list_get(pros, 1)) + " f0_max=" + int_to_str(native_list_get(pros, 2)) + " kf=" + int_to_str(kf))
|
||||
let ehv: [Int] = vmap_get(vmap, "EH")
|
||||
let ihv: [Int] = vmap_get(vmap, "IH")
|
||||
println("[his-vowels] EH=" + int_to_str(native_list_get(ehv, 0)) + "/" + int_to_str(native_list_get(ehv, 1)) + " IH=" + int_to_str(native_list_get(ihv, 0)) + "/" + int_to_str(native_list_get(ihv, 1)))
|
||||
|
||||
// phoneme geometry from the organ (loaded AFTER caches are in EL)
|
||||
let pmap: [String] = organ_pmap("elp/data/phonetics-formants.engram.json")
|
||||
let lmap: [String] = organ_lex("elp/data/lexicon.psv")
|
||||
|
||||
// 3+4: render FROM MEANING in his-vowels + his-prosody voice
|
||||
let vw: [String] = voice_will(native_list_get(pros, 0), native_list_get(pros, 1), kf)
|
||||
let noacc: [String] = native_list_empty()
|
||||
let novset: [String] = native_list_empty()
|
||||
let t: String = sem_realize(sem_frame("greet", "Will", "", ""))
|
||||
let codes: [String] = text_phonemes(lmap, t)
|
||||
println("[render] \"" + t + "\" :: " + list_join(codes, " "))
|
||||
let samples: [Int] = synth_codes_accent(codes, vw, pmap, noacc, novset, vmap, pros)
|
||||
let ok: Bool = write_wav(samples, 16000, outdir + "will-reply3.wav")
|
||||
println("[done] will-reply3.wav=" + bool_to_str(ok))
|
||||
}
|
||||
@@ -81,7 +81,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
@@ -62,7 +62,7 @@ jobs:
|
||||
# Link to produce the engram binary
|
||||
- name: Link engram binary
|
||||
run: |
|
||||
cc -std=c11 -O2 \
|
||||
cc -std=c11 -O2 -DHAVE_CURL \
|
||||
-I /usr/local/lib/el \
|
||||
-o dist/engram \
|
||||
dist/engram.c \
|
||||
|
||||
BIN
Binary file not shown.
Vendored
+152
-69
@@ -10,6 +10,7 @@ el_val_t query_param(el_val_t path, el_val_t key);
|
||||
el_val_t query_int(el_val_t path, el_val_t key, el_val_t default_val);
|
||||
el_val_t extract_id(el_val_t path, el_val_t prefix);
|
||||
el_val_t route_stats(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t persist_canonical(void);
|
||||
el_val_t route_create_node(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_get_node(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_scan_nodes(el_val_t method, el_val_t path, el_val_t body);
|
||||
@@ -23,13 +24,20 @@ el_val_t route_forget(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_save(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_load(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_health(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_sync(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_load_merge(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_emit_ise(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t route_capture_knowledge(el_val_t method, el_val_t path, el_val_t body);
|
||||
el_val_t check_auth_ok(el_val_t method, el_val_t body);
|
||||
el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body);
|
||||
|
||||
el_val_t bind_raw;
|
||||
el_val_t bind_str;
|
||||
el_val_t port;
|
||||
el_val_t data_dir_raw;
|
||||
el_val_t data_dir;
|
||||
el_val_t snapshot_path;
|
||||
el_val_t boot_snap;
|
||||
|
||||
el_val_t parse_port(el_val_t bind) {
|
||||
el_val_t colon = str_index_of(bind, EL_STR(":"));
|
||||
@@ -108,17 +116,22 @@ el_val_t route_stats(el_val_t method, el_val_t path, el_val_t body) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t persist_canonical(void) {
|
||||
el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
el_val_t dir = ({ el_val_t _if_result_1 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_1 = (EL_STR("/tmp/engram")); } else { _if_result_1 = (dir_raw); } _if_result_1; });
|
||||
engram_save(el_str_concat(dir, EL_STR("/snapshot.json")));
|
||||
return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_create_node(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t content = json_get_string(body, EL_STR("content"));
|
||||
el_val_t node_type = json_get_string(body, EL_STR("node_type"));
|
||||
if (str_eq(node_type, EL_STR(""))) {
|
||||
node_type = EL_STR("Memory");
|
||||
}
|
||||
el_val_t salience = json_get_float(body, EL_STR("salience"));
|
||||
if (str_eq(salience, el_from_float(0.0))) {
|
||||
salience = el_from_float(0.5);
|
||||
}
|
||||
el_val_t nt_raw = json_get_string(body, EL_STR("node_type"));
|
||||
el_val_t node_type = ({ el_val_t _if_result_2 = 0; if (str_eq(nt_raw, EL_STR(""))) { _if_result_2 = (EL_STR("Memory")); } else { _if_result_2 = (nt_raw); } _if_result_2; });
|
||||
el_val_t sal_raw = json_get_float(body, EL_STR("salience"));
|
||||
el_val_t salience = ({ el_val_t _if_result_3 = 0; if ((sal_raw == el_from_float(0.0))) { _if_result_3 = (el_from_float(0.5)); } else { _if_result_3 = (sal_raw); } _if_result_3; });
|
||||
el_val_t id = engram_node(content, node_type, salience);
|
||||
el_val_t saved = persist_canonical();
|
||||
return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"id\":\""), id), EL_STR("\",\"content\":\"")), content), EL_STR("\",\"node_type\":\"")), node_type), EL_STR("\"}"));
|
||||
return 0;
|
||||
}
|
||||
@@ -144,11 +157,9 @@ el_val_t route_scan_nodes(el_val_t method, el_val_t path, el_val_t body) {
|
||||
}
|
||||
|
||||
el_val_t route_scan_edges(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t dir = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
if (str_eq(dir, EL_STR(""))) {
|
||||
dir = EL_STR("/tmp/engram");
|
||||
}
|
||||
el_val_t snap_path = el_str_concat(dir, EL_STR("/snapshot.json"));
|
||||
el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
el_val_t dir = ({ el_val_t _if_result_4 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_4 = (EL_STR("/tmp/engram")); } else { _if_result_4 = (dir_raw); } _if_result_4; });
|
||||
el_val_t snap_path = el_str_concat(dir, EL_STR("/.scan-export.json"));
|
||||
engram_save(snap_path);
|
||||
el_val_t snap = fs_read(snap_path);
|
||||
if (str_eq(snap, EL_STR(""))) {
|
||||
@@ -163,36 +174,22 @@ el_val_t route_scan_edges(el_val_t method, el_val_t path, el_val_t body) {
|
||||
}
|
||||
|
||||
el_val_t route_search(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t q = EL_STR("");
|
||||
if (str_eq(method, EL_STR("GET"))) {
|
||||
q = query_param(path, EL_STR("q"));
|
||||
} else {
|
||||
q = json_get_string(body, EL_STR("query"));
|
||||
}
|
||||
el_val_t limit = query_int(path, EL_STR("limit"), 20);
|
||||
if (limit == 0) {
|
||||
limit = json_get_int(body, EL_STR("limit"));
|
||||
}
|
||||
if (limit == 0) {
|
||||
limit = 20;
|
||||
}
|
||||
el_val_t q = ({ el_val_t _if_result_5 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_5 = (query_param(path, EL_STR("q"))); } else { _if_result_5 = (json_get_string(body, EL_STR("query"))); } _if_result_5; });
|
||||
el_val_t lim_url = query_int(path, EL_STR("limit"), 0);
|
||||
el_val_t lim_body = json_get_int(body, EL_STR("limit"));
|
||||
el_val_t lim_either = ({ el_val_t _if_result_6 = 0; if ((lim_url > 0)) { _if_result_6 = (lim_url); } else { _if_result_6 = (lim_body); } _if_result_6; });
|
||||
el_val_t limit = ({ el_val_t _if_result_7 = 0; if ((lim_either > 0)) { _if_result_7 = (lim_either); } else { _if_result_7 = (20); } _if_result_7; });
|
||||
return engram_search_json(q, limit);
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_activate(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t q = EL_STR("");
|
||||
el_val_t depth = 3;
|
||||
if (str_eq(method, EL_STR("GET"))) {
|
||||
q = query_param(path, EL_STR("q"));
|
||||
depth = query_int(path, EL_STR("depth"), 3);
|
||||
} else {
|
||||
q = json_get_string(body, EL_STR("query"));
|
||||
el_val_t bd = json_get_int(body, EL_STR("depth"));
|
||||
if (bd > 0) {
|
||||
depth = bd;
|
||||
}
|
||||
el_val_t q = ({ el_val_t _if_result_8 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_8 = (query_param(path, EL_STR("q"))); } else { _if_result_8 = (json_get_string(body, EL_STR("query"))); } _if_result_8; });
|
||||
if (str_eq(q, EL_STR(""))) {
|
||||
return err_json(EL_STR("missing query"));
|
||||
}
|
||||
el_val_t d_raw = ({ el_val_t _if_result_9 = 0; if (str_eq(method, EL_STR("GET"))) { _if_result_9 = (query_int(path, EL_STR("depth"), 3)); } else { _if_result_9 = (json_get_int(body, EL_STR("depth"))); } _if_result_9; });
|
||||
el_val_t depth = ({ el_val_t _if_result_10 = 0; if ((d_raw > 0)) { _if_result_10 = (d_raw); } else { _if_result_10 = (3); } _if_result_10; });
|
||||
return el_str_concat(el_str_concat(EL_STR("{\"results\":"), engram_activate_json(q, depth)), EL_STR("}"));
|
||||
return 0;
|
||||
}
|
||||
@@ -200,15 +197,12 @@ el_val_t route_activate(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t route_create_edge(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t from_id = json_get_string(body, EL_STR("from_id"));
|
||||
el_val_t to_id = json_get_string(body, EL_STR("to_id"));
|
||||
el_val_t relation = json_get_string(body, EL_STR("relation"));
|
||||
if (str_eq(relation, EL_STR(""))) {
|
||||
relation = EL_STR("associates");
|
||||
}
|
||||
el_val_t weight = json_get_float(body, EL_STR("weight"));
|
||||
if (str_eq(weight, el_from_float(0.0))) {
|
||||
weight = el_from_float(0.5);
|
||||
}
|
||||
el_val_t rel_raw = json_get_string(body, EL_STR("relation"));
|
||||
el_val_t relation = ({ el_val_t _if_result_11 = 0; if (str_eq(rel_raw, EL_STR(""))) { _if_result_11 = (EL_STR("associates")); } else { _if_result_11 = (rel_raw); } _if_result_11; });
|
||||
el_val_t w_raw = json_get_float(body, EL_STR("weight"));
|
||||
el_val_t weight = ({ el_val_t _if_result_12 = 0; if ((w_raw == el_from_float(0.0))) { _if_result_12 = (el_from_float(0.5)); } else { _if_result_12 = (w_raw); } _if_result_12; });
|
||||
engram_connect(from_id, to_id, weight, relation);
|
||||
el_val_t saved = persist_canonical();
|
||||
return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"from_id\":\""), from_id), EL_STR("\",\"to_id\":\"")), to_id), EL_STR("\",\"relation\":\"")), relation), EL_STR("\"}"));
|
||||
return 0;
|
||||
}
|
||||
@@ -229,6 +223,7 @@ el_val_t route_strengthen(el_val_t method, el_val_t path, el_val_t body) {
|
||||
return err_json(EL_STR("missing node_id"));
|
||||
}
|
||||
engram_strengthen(id);
|
||||
el_val_t saved = persist_canonical();
|
||||
return ok_json();
|
||||
return 0;
|
||||
}
|
||||
@@ -239,33 +234,26 @@ el_val_t route_forget(el_val_t method, el_val_t path, el_val_t body) {
|
||||
return err_json(EL_STR("missing id"));
|
||||
}
|
||||
engram_forget(id);
|
||||
el_val_t saved = persist_canonical();
|
||||
return ok_json();
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_save(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t p = json_get_string(body, EL_STR("path"));
|
||||
if (str_eq(p, EL_STR(""))) {
|
||||
el_val_t dir = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
if (str_eq(dir, EL_STR(""))) {
|
||||
dir = EL_STR("/tmp/engram");
|
||||
}
|
||||
p = el_str_concat(dir, EL_STR("/snapshot.json"));
|
||||
}
|
||||
el_val_t p_raw = json_get_string(body, EL_STR("path"));
|
||||
el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
el_val_t dir = ({ el_val_t _if_result_13 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_13 = (EL_STR("/tmp/engram")); } else { _if_result_13 = (dir_raw); } _if_result_13; });
|
||||
el_val_t p = ({ el_val_t _if_result_14 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_14 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_14 = (p_raw); } _if_result_14; });
|
||||
engram_save(p);
|
||||
return el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"path\":\""), p), EL_STR("\"}"));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_load(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t p = json_get_string(body, EL_STR("path"));
|
||||
if (str_eq(p, EL_STR(""))) {
|
||||
el_val_t dir = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
if (str_eq(dir, EL_STR(""))) {
|
||||
dir = EL_STR("/tmp/engram");
|
||||
}
|
||||
p = el_str_concat(dir, EL_STR("/snapshot.json"));
|
||||
}
|
||||
el_val_t p_raw = json_get_string(body, EL_STR("path"));
|
||||
el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
el_val_t dir = ({ el_val_t _if_result_15 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_15 = (EL_STR("/tmp/engram")); } else { _if_result_15 = (dir_raw); } _if_result_15; });
|
||||
el_val_t p = ({ el_val_t _if_result_16 = 0; if (str_eq(p_raw, EL_STR(""))) { _if_result_16 = (el_str_concat(dir, EL_STR("/snapshot.json"))); } else { _if_result_16 = (p_raw); } _if_result_16; });
|
||||
engram_load(p);
|
||||
return ok_json();
|
||||
return 0;
|
||||
@@ -276,6 +264,84 @@ el_val_t route_health(el_val_t method, el_val_t path, el_val_t body) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_sync(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
el_val_t dir = ({ el_val_t _if_result_17 = 0; if (str_eq(dir_raw, EL_STR(""))) { _if_result_17 = (EL_STR("/tmp/engram")); } else { _if_result_17 = (dir_raw); } _if_result_17; });
|
||||
el_val_t snap_path = el_str_concat(dir, EL_STR("/.sync-export.json"));
|
||||
engram_save(snap_path);
|
||||
el_val_t snap = fs_read(snap_path);
|
||||
if (str_eq(snap, EL_STR(""))) {
|
||||
return EL_STR("{\"nodes\":[],\"edges\":[]}");
|
||||
}
|
||||
return snap;
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_load_merge(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t p = json_get_string(body, EL_STR("path"));
|
||||
if (str_eq(p, EL_STR(""))) {
|
||||
return err_json(EL_STR("path is required"));
|
||||
}
|
||||
if (str_eq(fs_read(p), EL_STR(""))) {
|
||||
return err_json(EL_STR("file missing or empty"));
|
||||
}
|
||||
el_val_t before_n = engram_node_count();
|
||||
el_val_t before_e = engram_edge_count();
|
||||
engram_load_merge(p);
|
||||
el_val_t added_n = (engram_node_count() - before_n);
|
||||
el_val_t added_e = (engram_edge_count() - before_e);
|
||||
el_val_t saved = persist_canonical();
|
||||
return el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"nodes_added\":"), int_to_str(added_n)), EL_STR(",\"edges_added\":")), int_to_str(added_e)), EL_STR(",\"node_count\":")), int_to_str(engram_node_count())), EL_STR("}"));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_emit_ise(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t content = json_get_string(body, EL_STR("content"));
|
||||
if (str_eq(content, EL_STR(""))) {
|
||||
return err_json(EL_STR("missing content"));
|
||||
}
|
||||
el_val_t sal = el_from_float(0.3);
|
||||
el_val_t imp = el_from_float(0.3);
|
||||
el_val_t conf = el_from_float(0.8);
|
||||
el_val_t id = engram_node_full(content, EL_STR("InternalStateEvent"), EL_STR("state-event"), sal, imp, conf, EL_STR("Episodic"), EL_STR("[\"internal-state\",\"InternalStateEvent\"]"));
|
||||
el_val_t ret_raw = env(EL_STR("ENGRAM_ISE_RETENTION_MS"));
|
||||
el_val_t ret_ms = ({ el_val_t _if_result_18 = 0; if (str_eq(ret_raw, EL_STR(""))) { _if_result_18 = (172800000); } else { _if_result_18 = (str_to_int(ret_raw)); } _if_result_18; });
|
||||
el_val_t pruned = engram_prune_telemetry(ret_ms);
|
||||
return el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"id\":\""), id), EL_STR("\",\"pruned\":")), int_to_str(pruned)), EL_STR("}"));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t route_capture_knowledge(el_val_t method, el_val_t path, el_val_t body) {
|
||||
el_val_t content = json_get_string(body, EL_STR("content"));
|
||||
if (str_eq(content, EL_STR(""))) {
|
||||
return err_json(EL_STR("missing content"));
|
||||
}
|
||||
el_val_t title = json_get_string(body, EL_STR("title"));
|
||||
el_val_t label = ({ el_val_t _if_result_19 = 0; if (str_eq(title, EL_STR(""))) { _if_result_19 = (str_slice(content, 0, 60)); } else { _if_result_19 = (title); } _if_result_19; });
|
||||
el_val_t category_raw = json_get_string(body, EL_STR("category"));
|
||||
el_val_t category = ({ el_val_t _if_result_20 = 0; if (str_eq(category_raw, EL_STR(""))) { _if_result_20 = (EL_STR("other")); } else { _if_result_20 = (category_raw); } _if_result_20; });
|
||||
el_val_t ktier_raw = json_get_string(body, EL_STR("tier"));
|
||||
el_val_t ktier = ({ el_val_t _if_result_21 = 0; if (str_eq(ktier_raw, EL_STR(""))) { _if_result_21 = (EL_STR("note")); } else { _if_result_21 = (ktier_raw); } _if_result_21; });
|
||||
el_val_t project = json_get_string(body, EL_STR("project"));
|
||||
el_val_t tags_raw = json_get_raw(body, EL_STR("tags"));
|
||||
el_val_t tags_base = ({ el_val_t _if_result_22 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_22 = (EL_STR("[]")); } else { _if_result_22 = (tags_raw); } _if_result_22; });
|
||||
el_val_t base_len = str_len(tags_base);
|
||||
el_val_t head = str_slice(tags_base, 0, (base_len - 1));
|
||||
el_val_t sep = ({ el_val_t _if_result_23 = 0; if (str_eq(head, EL_STR("["))) { _if_result_23 = (EL_STR("")); } else { _if_result_23 = (EL_STR(",")); } _if_result_23; });
|
||||
el_val_t safe_cat = str_replace(category, EL_STR("\""), EL_STR("'"));
|
||||
el_val_t safe_tier = str_replace(ktier, EL_STR("\""), EL_STR("'"));
|
||||
el_val_t safe_proj = str_replace(project, EL_STR("\""), EL_STR("'"));
|
||||
el_val_t proj_tag = ({ el_val_t _if_result_24 = 0; if (str_eq(safe_proj, EL_STR(""))) { _if_result_24 = (EL_STR("")); } else { _if_result_24 = (el_str_concat(el_str_concat(EL_STR(",\"project:"), safe_proj), EL_STR("\""))); } _if_result_24; });
|
||||
el_val_t tags = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(head, sep), EL_STR("\"category:")), safe_cat), EL_STR("\",\"tier:")), safe_tier), EL_STR("\"")), proj_tag), EL_STR("]"));
|
||||
el_val_t sal = el_from_float(0.5);
|
||||
el_val_t imp = el_from_float(0.5);
|
||||
el_val_t conf = el_from_float(0.9);
|
||||
el_val_t id = engram_node_full(content, EL_STR("Knowledge"), label, sal, imp, conf, EL_STR("Semantic"), tags);
|
||||
el_val_t saved = persist_canonical();
|
||||
return el_str_concat(el_str_concat(EL_STR("{\"ok\":true,\"id\":\""), id), EL_STR("\"}"));
|
||||
return 0;
|
||||
}
|
||||
|
||||
el_val_t check_auth_ok(el_val_t method, el_val_t body) {
|
||||
el_val_t key = env(EL_STR("ENGRAM_API_KEY"));
|
||||
if (str_eq(key, EL_STR(""))) {
|
||||
@@ -299,9 +365,15 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) {
|
||||
return route_health(method, path, body);
|
||||
}
|
||||
}
|
||||
if (str_eq(method, EL_STR("POST")) && str_eq(clean, EL_STR("/api/neuron/state-events"))) {
|
||||
return route_emit_ise(method, path, body);
|
||||
}
|
||||
if (!check_auth_ok(method, body)) {
|
||||
return err_json(EL_STR("unauthorized"));
|
||||
}
|
||||
if (str_eq(method, EL_STR("POST")) && str_eq(clean, EL_STR("/api/neuron/knowledge/capture"))) {
|
||||
return route_capture_knowledge(method, path, body);
|
||||
}
|
||||
if (str_eq(method, EL_STR("GET")) && (str_eq(clean, EL_STR("/api/stats")) || str_eq(clean, EL_STR("/stats")))) {
|
||||
return route_stats(method, path, body);
|
||||
}
|
||||
@@ -347,23 +419,34 @@ el_val_t handle_request(el_val_t method, el_val_t path, el_val_t body) {
|
||||
if (str_eq(method, EL_STR("POST")) && (str_eq(clean, EL_STR("/api/load")) || str_eq(clean, EL_STR("/load")))) {
|
||||
return route_load(method, path, body);
|
||||
}
|
||||
if (str_eq(method, EL_STR("POST")) && (str_eq(clean, EL_STR("/api/load-merge")) || str_eq(clean, EL_STR("/load-merge")))) {
|
||||
return route_load_merge(method, path, body);
|
||||
}
|
||||
if (str_eq(method, EL_STR("GET")) && str_eq(clean, EL_STR("/api/sync"))) {
|
||||
return route_sync(method, path, body);
|
||||
}
|
||||
return el_str_concat(el_str_concat(EL_STR("{\"error\":\"not found\",\"path\":\""), clean), EL_STR("\"}"));
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int _argc, char** _argv) {
|
||||
el_runtime_init_args(_argc, _argv);
|
||||
bind_str = env(EL_STR("ENGRAM_BIND"));
|
||||
if (str_eq(bind_str, EL_STR(""))) {
|
||||
bind_str = EL_STR(":8742");
|
||||
}
|
||||
bind_raw = env(EL_STR("ENGRAM_BIND"));
|
||||
bind_str = ({ el_val_t _if_result_25 = 0; if (str_eq(bind_raw, EL_STR(""))) { _if_result_25 = (EL_STR(":8742")); } else { _if_result_25 = (bind_raw); } _if_result_25; });
|
||||
port = parse_port(bind_str);
|
||||
data_dir = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
if (str_eq(data_dir, EL_STR(""))) {
|
||||
data_dir = EL_STR("/tmp/engram");
|
||||
}
|
||||
data_dir_raw = env(EL_STR("ENGRAM_DATA_DIR"));
|
||||
data_dir = ({ el_val_t _if_result_26 = 0; if (str_eq(data_dir_raw, EL_STR(""))) { _if_result_26 = (EL_STR("/tmp/engram")); } else { _if_result_26 = (data_dir_raw); } _if_result_26; });
|
||||
snapshot_path = el_str_concat(data_dir, EL_STR("/snapshot.json"));
|
||||
engram_load(snapshot_path);
|
||||
boot_snap = fs_read(snapshot_path);
|
||||
if (!str_eq(boot_snap, EL_STR(""))) {
|
||||
if (engram_node_count() == 0) {
|
||||
println(EL_STR("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes \xe2\x80\x94 preserving copy at snapshot.failed-load.json"));
|
||||
fs_write(el_str_concat(data_dir, EL_STR("/snapshot.failed-load.json")), boot_snap);
|
||||
} else {
|
||||
fs_write(el_str_concat(data_dir, EL_STR("/snapshot.boot-backup.json")), boot_snap);
|
||||
}
|
||||
}
|
||||
println(EL_STR("[engram] runtime-native graph engine"));
|
||||
println(el_str_concat(EL_STR("[engram] data_dir="), data_dir));
|
||||
println(el_str_concat(EL_STR("[engram] node_count="), int_to_str(engram_node_count())));
|
||||
|
||||
+248
-49
@@ -76,13 +76,43 @@ fn route_stats(method: String, path: String, body: String) -> String {
|
||||
engram_stats_json()
|
||||
}
|
||||
|
||||
// (2026-07-18 self-review) Scoping sweep: `let` inside an if-block creates an
|
||||
// inner scope only — it does NOT mutate the outer binding (documented with
|
||||
// evidence in awareness.el, 2026-05-25). Every default/reassignment below used
|
||||
// that broken pattern, so defaults never applied: nodes were created with
|
||||
// node_type="" and salience=0.0, /api/search and /api/activate ALWAYS ran with
|
||||
// q="" regardless of input, edges defaulted to relation=""/weight=0.0, and
|
||||
// save/load with no "path" hit engram_save(""). Rewritten to the
|
||||
// `let x = if cond { a } else { b }` expression form (the pattern the newer
|
||||
// routes route_emit_ise/route_capture_knowledge already use correctly).
|
||||
// persist_canonical — save the canonical snapshot after a durable write.
|
||||
//
|
||||
// WHY (2026-07-22 self-review): the 2026-07-21 fix correctly stopped READ
|
||||
// routes from writing the canonical snapshot.json — but nothing was left
|
||||
// that saved it on WRITE. Every mutation (node create, edge create,
|
||||
// knowledge capture, forget, merge) lived only in RAM until someone POSTed
|
||||
// /api/save manually; a process restart silently discarded everything since
|
||||
// the last manual save. Observed live: two engram restarts during the
|
||||
// 2026-07-22 review reverted the store to a ~17h-old snapshot, destroying
|
||||
// same-day writes. Reads must never write the canonical; writes must always
|
||||
// persist it. ISE telemetry is deliberately excluded (48h-pruned, loss-
|
||||
// tolerant, ~2/min — snapshotting the whole store per heartbeat is waste;
|
||||
// any durable write that follows persists the pruning too).
|
||||
fn persist_canonical() -> Int {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
engram_save(dir + "/snapshot.json")
|
||||
return 1
|
||||
}
|
||||
|
||||
fn route_create_node(method: String, path: String, body: String) -> String {
|
||||
let content: String = json_get_string(body, "content")
|
||||
let node_type: String = json_get_string(body, "node_type")
|
||||
if str_eq(node_type, "") { let node_type = "Memory" }
|
||||
let salience: Float = json_get_float(body, "salience")
|
||||
if salience == 0.0 { let salience = 0.5 }
|
||||
let nt_raw: String = json_get_string(body, "node_type")
|
||||
let node_type: String = if str_eq(nt_raw, "") { "Memory" } else { nt_raw }
|
||||
let sal_raw: Float = json_get_float(body, "salience")
|
||||
let salience: Float = if sal_raw == 0.0 { 0.5 } else { sal_raw }
|
||||
let id: String = engram_node(content, node_type, salience)
|
||||
let saved: Int = persist_canonical()
|
||||
"{\"id\":\"" + id + "\",\"content\":\"" + content + "\",\"node_type\":\"" + node_type + "\"}"
|
||||
}
|
||||
|
||||
@@ -103,13 +133,14 @@ fn route_scan_nodes(method: String, path: String, body: String) -> String {
|
||||
}
|
||||
|
||||
// route_scan_edges — bulk export of all edges as a JSON array. Implemented
|
||||
// via engram_save → fs_read of the canonical on-disk snapshot, which the
|
||||
// runtime keeps in lockstep with the in-memory graph. Live against the
|
||||
// running graph, not a stale export.
|
||||
// via engram_save → fs_read of a SCRATCH export path. (2026-07-21 self-review:
|
||||
// previously this saved over the canonical snapshot.json on every GET — if the
|
||||
// process ever booted with a partial/empty store, the first read request
|
||||
// clobbered the good snapshot. Read routes must never write the canonical path.)
|
||||
fn route_scan_edges(method: String, path: String, body: String) -> String {
|
||||
let dir: String = env("ENGRAM_DATA_DIR")
|
||||
if str_eq(dir, "") { let dir = "/tmp/engram" }
|
||||
let snap_path: String = dir + "/snapshot.json"
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let snap_path: String = dir + "/.scan-export.json"
|
||||
engram_save(snap_path)
|
||||
let snap: String = fs_read(snap_path)
|
||||
if str_eq(snap, "") { return "[]" }
|
||||
@@ -122,40 +153,34 @@ fn route_scan_edges(method: String, path: String, body: String) -> String {
|
||||
}
|
||||
|
||||
fn route_search(method: String, path: String, body: String) -> String {
|
||||
let q: String = ""
|
||||
if str_eq(method, "GET") {
|
||||
let q = query_param(path, "q")
|
||||
} else {
|
||||
let q = json_get_string(body, "query")
|
||||
}
|
||||
let limit: Int = query_int(path, "limit", 20)
|
||||
if limit == 0 { let limit = json_get_int(body, "limit") }
|
||||
if limit == 0 { let limit = 20 }
|
||||
let q: String = if str_eq(method, "GET") { query_param(path, "q") } else { json_get_string(body, "query") }
|
||||
let lim_url: Int = query_int(path, "limit", 0)
|
||||
let lim_body: Int = json_get_int(body, "limit")
|
||||
let lim_either: Int = if lim_url > 0 { lim_url } else { lim_body }
|
||||
let limit: Int = if lim_either > 0 { lim_either } else { 20 }
|
||||
return engram_search_json(q, limit)
|
||||
}
|
||||
|
||||
fn route_activate(method: String, path: String, body: String) -> String {
|
||||
let q: String = ""
|
||||
let depth: Int = 3
|
||||
if str_eq(method, "GET") {
|
||||
let q = query_param(path, "q")
|
||||
let depth = query_int(path, "depth", 3)
|
||||
} else {
|
||||
let q = json_get_string(body, "query")
|
||||
let bd: Int = json_get_int(body, "depth")
|
||||
if bd > 0 { let depth = bd }
|
||||
}
|
||||
let q: String = if str_eq(method, "GET") { query_param(path, "q") } else { json_get_string(body, "query") }
|
||||
// Guard: engram_activate with an empty query matches zero seeds, which
|
||||
// zeroes ALL carried working-memory weights (documented in awareness.el
|
||||
// perceive()). Never let an empty activation through to wipe WM.
|
||||
if str_eq(q, "") { return err_json("missing query") }
|
||||
let d_raw: Int = if str_eq(method, "GET") { query_int(path, "depth", 3) } else { json_get_int(body, "depth") }
|
||||
let depth: Int = if d_raw > 0 { d_raw } else { 3 }
|
||||
return "{\"results\":" + engram_activate_json(q, depth) + "}"
|
||||
}
|
||||
|
||||
fn route_create_edge(method: String, path: String, body: String) -> String {
|
||||
let from_id: String = json_get_string(body, "from_id")
|
||||
let to_id: String = json_get_string(body, "to_id")
|
||||
let relation: String = json_get_string(body, "relation")
|
||||
if str_eq(relation, "") { let relation = "associates" }
|
||||
let weight: Float = json_get_float(body, "weight")
|
||||
if weight == 0.0 { let weight = 0.5 }
|
||||
let rel_raw: String = json_get_string(body, "relation")
|
||||
let relation: String = if str_eq(rel_raw, "") { "associates" } else { rel_raw }
|
||||
let w_raw: Float = json_get_float(body, "weight")
|
||||
let weight: Float = if w_raw == 0.0 { 0.5 } else { w_raw }
|
||||
engram_connect(from_id, to_id, weight, relation)
|
||||
let saved: Int = persist_canonical()
|
||||
"{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + relation + "\"}"
|
||||
}
|
||||
|
||||
@@ -170,6 +195,7 @@ fn route_strengthen(method: String, path: String, body: String) -> String {
|
||||
let id: String = json_get_string(body, "node_id")
|
||||
if str_eq(id, "") { return err_json("missing node_id") }
|
||||
engram_strengthen(id)
|
||||
let saved: Int = persist_canonical()
|
||||
ok_json()
|
||||
}
|
||||
|
||||
@@ -177,27 +203,24 @@ fn route_forget(method: String, path: String, body: String) -> String {
|
||||
let id: String = extract_id(path, "/api/nodes/")
|
||||
if str_eq(id, "") { return err_json("missing id") }
|
||||
engram_forget(id)
|
||||
let saved: Int = persist_canonical()
|
||||
ok_json()
|
||||
}
|
||||
|
||||
fn route_save(method: String, path: String, body: String) -> String {
|
||||
let p: String = json_get_string(body, "path")
|
||||
if str_eq(p, "") {
|
||||
let dir: String = env("ENGRAM_DATA_DIR")
|
||||
if str_eq(dir, "") { let dir = "/tmp/engram" }
|
||||
let p = dir + "/snapshot.json"
|
||||
}
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
engram_save(p)
|
||||
"{\"ok\":true,\"path\":\"" + p + "\"}"
|
||||
}
|
||||
|
||||
fn route_load(method: String, path: String, body: String) -> String {
|
||||
let p: String = json_get_string(body, "path")
|
||||
if str_eq(p, "") {
|
||||
let dir: String = env("ENGRAM_DATA_DIR")
|
||||
if str_eq(dir, "") { let dir = "/tmp/engram" }
|
||||
let p = dir + "/snapshot.json"
|
||||
}
|
||||
let p_raw: String = json_get_string(body, "path")
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
let p: String = if str_eq(p_raw, "") { dir + "/snapshot.json" } else { p_raw }
|
||||
engram_load(p)
|
||||
ok_json()
|
||||
}
|
||||
@@ -206,6 +229,148 @@ fn route_health(method: String, path: String, body: String) -> String {
|
||||
"{\"status\":\"ok\",\"engine\":\"engram-runtime-native\"}"
|
||||
}
|
||||
|
||||
// route_sync — return a snapshot of non-ISE/non-Working nodes for the soul daemon
|
||||
// to merge into its in-process graph via engram_load_merge.
|
||||
//
|
||||
// The soul calls GET /api/sync every SOUL_REFRESH_MS (default 10 min) to pull
|
||||
// new Knowledge/Memory/BacklogItem nodes from the authoritative HTTP Engram into
|
||||
// its in-process working store. Previously this returned 404 "not found", causing
|
||||
// the soul to write the error JSON to a temp file and attempt an empty merge.
|
||||
//
|
||||
// Strategy: save the current snapshot to disk, read it back, return the full
|
||||
// snapshot JSON. The soul's engram_load_merge handles large files gracefully
|
||||
// (it skips nodes already present by ID). Auth-exempt: same-host internal call.
|
||||
// (2026-06-27 self-review: added this route to fix silent 10-min sync failures)
|
||||
fn route_sync(method: String, path: String, body: String) -> String {
|
||||
let dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let dir: String = if str_eq(dir_raw, "") { "/tmp/engram" } else { dir_raw }
|
||||
// 2026-07-21 self-review: export to a scratch path, never the canonical
|
||||
// snapshot.json — read routes must not be able to clobber the good snapshot.
|
||||
let snap_path: String = dir + "/.sync-export.json"
|
||||
engram_save(snap_path)
|
||||
let snap: String = fs_read(snap_path)
|
||||
if str_eq(snap, "") { return "{\"nodes\":[],\"edges\":[]}" }
|
||||
return snap
|
||||
}
|
||||
|
||||
// route_load_merge — POST /api/load-merge {"path": "..."} — merge a snapshot
|
||||
// file into the live store WITHOUT resetting it (engram_load_merge skips nodes
|
||||
// already present by id). Added 2026-07-21 self-review to restore the 244 kn-
|
||||
// identity Knowledge nodes lost from the snapshot lineage between 05-13 and
|
||||
// 07-13. Requires an explicit path: refuses to run without one so it can never
|
||||
// be triggered accidentally against a default.
|
||||
fn route_load_merge(method: String, path: String, body: String) -> String {
|
||||
let p: String = json_get_string(body, "path")
|
||||
if str_eq(p, "") { return err_json("path is required") }
|
||||
if str_eq(fs_read(p), "") { return err_json("file missing or empty") }
|
||||
let before_n: Int = engram_node_count()
|
||||
let before_e: Int = engram_edge_count()
|
||||
engram_load_merge(p)
|
||||
let added_n: Int = engram_node_count() - before_n
|
||||
let added_e: Int = engram_edge_count() - before_e
|
||||
let saved: Int = persist_canonical()
|
||||
"{\"ok\":true,\"nodes_added\":" + int_to_str(added_n) + ",\"edges_added\":" + int_to_str(added_e) + ",\"node_count\":" + int_to_str(engram_node_count()) + "}"
|
||||
}
|
||||
|
||||
// route_emit_ise — write an InternalStateEvent node from the soul daemon.
|
||||
//
|
||||
// Endpoint: POST /api/neuron/state-events
|
||||
// Body: {"content": "<json-string>"}
|
||||
//
|
||||
// Auth: exempt (internal endpoint, soul daemon on same host, no _auth needed).
|
||||
// The soul's ise_post() sends {"content":"..."} without _auth; enforcing auth
|
||||
// here would silently drop all heartbeat/curiosity ISEs. Unauthenticated POST
|
||||
// to this endpoint is acceptable: ISE writes are observability-only, append-only,
|
||||
// and come from a trusted process on localhost.
|
||||
//
|
||||
// Salience/importance set to match engram_node_full ISE defaults used by the
|
||||
// in-process fallback path in awareness.el (salience=0.3, importance=0.3,
|
||||
// confidence=0.8, tier=Episodic).
|
||||
// (2026-06-26 self-review: added this route after discovering ise_post was
|
||||
// silently failing — the soul posts here but the endpoint didn't exist.)
|
||||
//
|
||||
// Retention (2026-07-16 self-review): an earlier comment here claimed ISEs
|
||||
// got temporal_decay_rate=1.617 — that was never implemented (engram_node_full
|
||||
// hardcodes 0.0), and per-node decay only dampens activation anyway; it never
|
||||
// removes nodes. By 2026-07-16 ISEs were 75% of the store (10,175 of 13,522
|
||||
// nodes, ~4,300/day, unbounded). ISEs are already WM-excluded in
|
||||
// engram_activate, so the fix is retention, not decay: every insert calls
|
||||
// engram_prune_telemetry(), a single O(nodes+edges) compaction pass that
|
||||
// removes ISEs older than ENGRAM_ISE_RETENTION_MS (default 48h), protecting
|
||||
// "session-start" labels and self_review events as durable history. At
|
||||
// ~3 ISEs/min this bounds telemetry at ~8.6k nodes instead of growing forever.
|
||||
fn route_emit_ise(method: String, path: String, body: String) -> String {
|
||||
let content: String = json_get_string(body, "content")
|
||||
if str_eq(content, "") { return err_json("missing content") }
|
||||
let sal: Float = 0.3
|
||||
let imp: Float = 0.3
|
||||
let conf: Float = 0.8
|
||||
let id: String = engram_node_full(
|
||||
content, "InternalStateEvent", "state-event",
|
||||
sal, imp, conf,
|
||||
"Episodic", "[\"internal-state\",\"InternalStateEvent\"]"
|
||||
)
|
||||
let ret_raw: String = env("ENGRAM_ISE_RETENTION_MS")
|
||||
let ret_ms: Int = if str_eq(ret_raw, "") { 172800000 } else { str_to_int(ret_raw) }
|
||||
let pruned: Int = engram_prune_telemetry(ret_ms)
|
||||
"{\"ok\":true,\"id\":\"" + id + "\",\"pruned\":" + int_to_str(pruned) + "}"
|
||||
}
|
||||
|
||||
// ── Knowledge capture ─────────────────────────────────────────────────────────
|
||||
//
|
||||
// route_capture_knowledge — direct Knowledge-node capture over HTTP.
|
||||
//
|
||||
// Endpoint: POST /api/neuron/knowledge/capture (auth required: "_auth" in body)
|
||||
// Body: {"content": "...", "title": "...", "category": "...",
|
||||
// "tier": "note|lesson|canonical", "tags": [...], "project": "...",
|
||||
// "_auth": "<key>"}
|
||||
//
|
||||
// WHY (2026-07-15 self-review): the world-ingestor integrator was designed
|
||||
// against this endpoint (its MCP-unavailable fallback), but the route never
|
||||
// existed — every direct push 404'd, and because the auth gate ran before
|
||||
// routing, the failure surfaced as {"error":"unauthorized"} and was
|
||||
// misdiagnosed for two weeks while world knowledge silently dropped.
|
||||
// POST /api/nodes was no substitute: it discards label/tags/tier, which
|
||||
// makes captured knowledge invisible to tag-scoped search and curiosity.
|
||||
//
|
||||
// The incoming knowledge tier (note/lesson/canonical) is preserved as a
|
||||
// "tier:<x>" tag rather than mapped onto Engram's cognitive tiers — Knowledge
|
||||
// nodes land in Semantic (stable reference), and the epistemic tier stays
|
||||
// queryable without inventing a lossy mapping.
|
||||
fn route_capture_knowledge(method: String, path: String, body: String) -> String {
|
||||
let content: String = json_get_string(body, "content")
|
||||
if str_eq(content, "") { return err_json("missing content") }
|
||||
let title: String = json_get_string(body, "title")
|
||||
let label: String = if str_eq(title, "") { str_slice(content, 0, 60) } else { title }
|
||||
let category_raw: String = json_get_string(body, "category")
|
||||
let category: String = if str_eq(category_raw, "") { "other" } else { category_raw }
|
||||
let ktier_raw: String = json_get_string(body, "tier")
|
||||
let ktier: String = if str_eq(ktier_raw, "") { "note" } else { ktier_raw }
|
||||
let project: String = json_get_string(body, "project")
|
||||
let tags_raw: String = json_get_raw(body, "tags")
|
||||
let tags_base: String = if str_eq(tags_raw, "") { "[]" } else { tags_raw }
|
||||
// Merge category/tier/project markers into the tag array. Search matches
|
||||
// against the tags string, so these make captures findable by facet.
|
||||
let base_len: Int = str_len(tags_base)
|
||||
let head: String = str_slice(tags_base, 0, base_len - 1)
|
||||
let sep: String = if str_eq(head, "[") { "" } else { "," }
|
||||
let safe_cat: String = str_replace(category, "\"", "'")
|
||||
let safe_tier: String = str_replace(ktier, "\"", "'")
|
||||
let safe_proj: String = str_replace(project, "\"", "'")
|
||||
let proj_tag: String = if str_eq(safe_proj, "") { "" } else { ",\"project:" + safe_proj + "\"" }
|
||||
let tags: String = head + sep + "\"category:" + safe_cat + "\",\"tier:" + safe_tier + "\"" + proj_tag + "]"
|
||||
let sal: Float = 0.5
|
||||
let imp: Float = 0.5
|
||||
let conf: Float = 0.9
|
||||
let id: String = engram_node_full(
|
||||
content, "Knowledge", label,
|
||||
sal, imp, conf,
|
||||
"Semantic", tags
|
||||
)
|
||||
let saved: Int = persist_canonical()
|
||||
"{\"ok\":true,\"id\":\"" + id + "\"}"
|
||||
}
|
||||
|
||||
// ── Auth ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
fn check_auth_ok(method: String, body: String) -> Bool {
|
||||
@@ -232,11 +397,22 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
}
|
||||
}
|
||||
|
||||
// ISE posting is auth-exempt (internal soul daemon, same host, no _auth key)
|
||||
if str_eq(method, "POST") && str_eq(clean, "/api/neuron/state-events") {
|
||||
return route_emit_ise(method, path, body)
|
||||
}
|
||||
|
||||
// Auth (when ENGRAM_API_KEY is set)
|
||||
if !check_auth_ok(method, body) {
|
||||
return err_json("unauthorized")
|
||||
}
|
||||
|
||||
// Knowledge capture (auth enforced above; the world-ingestor integrator
|
||||
// and any headless session without MCP push knowledge through this)
|
||||
if str_eq(method, "POST") && str_eq(clean, "/api/neuron/knowledge/capture") {
|
||||
return route_capture_knowledge(method, path, body)
|
||||
}
|
||||
|
||||
// Stats
|
||||
if str_eq(method, "GET") && (str_eq(clean, "/api/stats") || str_eq(clean, "/stats")) {
|
||||
return route_stats(method, path, body)
|
||||
@@ -293,22 +469,45 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
if str_eq(method, "POST") && (str_eq(clean, "/api/load") || str_eq(clean, "/load")) {
|
||||
return route_load(method, path, body)
|
||||
}
|
||||
if str_eq(method, "POST") && (str_eq(clean, "/api/load-merge") || str_eq(clean, "/load-merge")) {
|
||||
return route_load_merge(method, path, body)
|
||||
}
|
||||
|
||||
// Sync — soul daemon periodic pull of non-ISE knowledge into in-process graph
|
||||
if str_eq(method, "GET") && str_eq(clean, "/api/sync") {
|
||||
return route_sync(method, path, body)
|
||||
}
|
||||
|
||||
"{\"error\":\"not found\",\"path\":\"" + clean + "\"}"
|
||||
}
|
||||
|
||||
// ── Entry ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
let bind_str: String = env("ENGRAM_BIND")
|
||||
if str_eq(bind_str, "") { let bind_str = ":8742" }
|
||||
let bind_raw: String = env("ENGRAM_BIND")
|
||||
let bind_str: String = if str_eq(bind_raw, "") { ":8742" } else { bind_raw }
|
||||
let port: Int = parse_port(bind_str)
|
||||
|
||||
// On startup, try to load any existing snapshot (best effort).
|
||||
let data_dir: String = env("ENGRAM_DATA_DIR")
|
||||
if str_eq(data_dir, "") { let data_dir = "/tmp/engram" }
|
||||
let data_dir_raw: String = env("ENGRAM_DATA_DIR")
|
||||
let data_dir: String = if str_eq(data_dir_raw, "") { "/tmp/engram" } else { data_dir_raw }
|
||||
let snapshot_path: String = data_dir + "/snapshot.json"
|
||||
engram_load(snapshot_path)
|
||||
|
||||
// 2026-07-21 self-review boot guard: if the snapshot file has content but the
|
||||
// load produced 0 nodes, something is wrong (corrupt file / parse failure).
|
||||
// Preserve the evidence and warn loudly — and since read routes no longer write
|
||||
// the canonical path, a bad boot can no longer clobber the good snapshot.
|
||||
let boot_snap: String = fs_read(snapshot_path)
|
||||
if !str_eq(boot_snap, "") {
|
||||
if engram_node_count() == 0 {
|
||||
println("[engram] WARNING: snapshot.json is non-empty but load produced 0 nodes — preserving copy at snapshot.failed-load.json")
|
||||
fs_write(data_dir + "/snapshot.failed-load.json", boot_snap)
|
||||
} else {
|
||||
// Good load: keep a boot-time backup of the snapshot as loaded.
|
||||
fs_write(data_dir + "/snapshot.boot-backup.json", boot_snap)
|
||||
}
|
||||
}
|
||||
|
||||
println("[engram] runtime-native graph engine")
|
||||
println("[engram] data_dir=" + data_dir)
|
||||
println("[engram] node_count=" + int_to_str(engram_node_count()))
|
||||
|
||||
@@ -17,6 +17,16 @@
|
||||
// 4. Append dep to order after all its transitive deps
|
||||
// 5. Deduplicate: skip already-ordered vessels
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated install.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn manifest_name(src: String) -> String // manifest.el
|
||||
extern fn manifest_deps(src: String) -> String // manifest.el
|
||||
extern fn registry_token() -> String // registry.el
|
||||
extern fn registry_find(name: String, version: String) -> String // registry.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Install paths ─────────────────────────────────────────────────────────────
|
||||
|
||||
// packages_dir returns the root directory for installed vessels.
|
||||
|
||||
@@ -14,6 +14,15 @@
|
||||
// EPM_REGISTRY_ORG — org name that hosts vessel repos (default: neuron-technologies)
|
||||
// EPM_TOKEN — Gitea personal access token (required for publish)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// These symbols are defined in sibling epm modules or the El runtime and are
|
||||
// resolved at link time. The `extern fn` decls give elc the C prototype so the
|
||||
// generated registry.c compiles cleanly under strict compilers (gcc>=14 / clang)
|
||||
// that reject implicit function declarations. Signature arity must match the
|
||||
// definition; return/param types are informational (all lower to el_val_t).
|
||||
extern fn config(key: String) -> String // El runtime builtin
|
||||
extern fn read_installed() -> String // install.el
|
||||
|
||||
// ── Config helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// registry_api_url returns the Gitea API base URL with no trailing slash.
|
||||
|
||||
@@ -6,6 +6,15 @@
|
||||
// Depends on: registry.el (registry_latest_version, registry_find),
|
||||
// install.el (read_installed, install_vessel, installed_version)
|
||||
|
||||
// ── Cross-module forward declarations ─────────────────────────────────────────
|
||||
// Defined in sibling epm modules; resolved at link time. The `extern fn` decls
|
||||
// give elc the C prototypes so generated update.c compiles cleanly under strict
|
||||
// compilers (gcc>=14 / clang) that reject implicit function declarations.
|
||||
extern fn read_installed() -> String // install.el
|
||||
extern fn installed_version(name: String) -> String // install.el
|
||||
extern fn install_vessel(name: String, version: String) -> Bool // install.el
|
||||
extern fn registry_latest_version(name: String) -> String // registry.el
|
||||
|
||||
// ── Semver helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
// semver_part extracts the Nth dot-separated component from a semver string.
|
||||
|
||||
Vendored
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user