Compare commits

..

1 Commits

Author SHA1 Message Date
will.anderson d9791f15da docs: add cognitive-architecture treatise (06)
The middle layer between the whitepaper (thesis) and the engram API
reference (surface): how the mind is designed and why, as tiered
subsystems (LIVE/STAGED/DESIGNED). Covers the geometry thesis, engram
durability and the two-store topology (with the cultivate write-through
gap flagged as a known issue), reified Neighborhood nodes, the body/orbit
model, the operator/language/interoception/reasoning faculties, the self
and the cultivate gate, the fact boundary, the topology findings
(genus-0 expander, not a torus), and the design principles.
2026-08-13 17:05:27 -05:00
126 changed files with 1742 additions and 64160 deletions
+38 -52
View File
@@ -34,12 +34,12 @@ jobs:
- name: Install build dependencies
run: |
apt-get update -qq
apt-get install -y gcc curl libcurl4-openssl-dev apt-transport-https ca-certificates
apt-get install -y gcc libcurl4-openssl-dev apt-transport-https ca-certificates
echo "deb [trusted=yes] https://packages.cloud.google.com/apt cloud-sdk main" \
> /etc/apt/sources.list.d/google-cloud-sdk.list
apt-get update -qq && apt-get install -y google-cloud-cli
- name: Authenticate to GCP + stage PINNED El runtime
- name: Download El runtime from Artifact Registry
env:
GCP_SA_KEY: ${{ secrets.GCP_SA_KEY }}
run: |
@@ -47,37 +47,41 @@ jobs:
gcloud auth activate-service-account --key-file=/tmp/gcp-key.json
gcloud config set project neuron-785695
# PINNED RUNTIME — do NOT pull "latest" from Artifact Registry.
# The ship-soul calls engram_prune_telemetry (awareness.el sync/heartbeat
# self-review). The latest published el-runtime-c no longer defines that
# symbol, so an unpinned build fails to LINK — which is exactly how a
# broken/handlerless soul reached prod before. Compile against the
# vendored release runtime v1.0.0-20260501: the exact runtime the merged
# ship-soul was verified against (verify-soul-contract GATE PASS +
# genesis boot survives + full safety-contact response). It is committed
# under vendor/ so the soul build is fully reproducible and never depends
# on a moving AR "latest".
rm -rf /opt/el/runtime
mkdir -p /opt/el/runtime
cp vendor/el-runtime/v1.0.0-20260501/el_runtime.c /opt/el/runtime/el_runtime.c
cp vendor/el-runtime/v1.0.0-20260501/el_runtime.h /opt/el/runtime/el_runtime.h
echo "El runtime PINNED to v1.0.0-20260501: $(ls /opt/el/runtime/)"
# neuron#133: CI compiles dist/soul.c, NOT the .el sources. On 2026-08-07 a
# build off main would have shipped an engine with none of five merged fixes,
# including a P0 safety fix, while main's source read as correct. The runner
# cannot regenerate the amalgam (elc needs 24GB+ virtual memory), but it can
# refuse to compile a stale one. Fails loudly with the recipe in the message.
- name: Verify dist/soul.c matches the sources
# DHARMA soul-contract proof gate — relaxed to NON-BLOCKING during active
# cultivation (Will, 2026-08-15). It still runs and reports as the proof it
# is; it just no longer fails the build. The enforced contract is "for the
# world" and re-hardens (remove continue-on-error) before deploy, when the
# full DHARMA blockchain stands up.
continue-on-error: true
run: |
chmod +x tools/soulc-stamp.sh
./tools/soulc-stamp.sh --check
# Get latest version of each runtime package (elc/elb not needed — we compile
# dist/soul.c directly; running elb on Linux OOM-kills the runner, and we
# always use the repo's pre-built soul.c anyway).
get_latest() {
gcloud artifacts versions list \
--repository=foundation-prod \
--location=us-central1 \
--project=neuron-785695 \
--package="$1" \
--sort-by="~createTime" \
--limit=1 \
--format="value(name)" 2>/dev/null | awk -F/ '{print $NF}'
}
RC_VER=$(get_latest el-runtime-c)
RH_VER=$(get_latest el-runtime-h)
echo "Downloading runtime@${RC_VER}"
gcloud artifacts generic download \
--repository=foundation-prod --location=us-central1 --project=neuron-785695 \
--package=el-runtime-c --version="${RC_VER}" \
--destination=/opt/el/runtime/
gcloud artifacts generic download \
--repository=foundation-prod --location=us-central1 --project=neuron-785695 \
--package=el-runtime-h --version="${RH_VER}" \
--destination=/opt/el/runtime/
mv /opt/el/runtime/el_runtime.c* /opt/el/runtime/el_runtime.c 2>/dev/null || true
mv /opt/el/runtime/el_runtime.h* /opt/el/runtime/el_runtime.h 2>/dev/null || true
echo "El runtime ready: $(ls /opt/el/runtime/)"
- name: Build neuron soul binary
run: |
@@ -90,37 +94,19 @@ jobs:
# entirely: elb on Linux would OOM the runner (elc uses 24GB+ virtual memory
# on a 16GB host) and we always restore from the repo's soul.c anyway.
mkdir -p dist
# -rdynamic: the el runtime resolves the HTTP request handler (and the
# tool handlers) by NAME via dlsym(RTLD_DEFAULT, "handle_request").
# macOS exports these symbols freely, but glibc/Linux only makes symbols
# visible to dlsym if they are in the dynamic symbol table — so without
# -rdynamic the stripped Linux binary boots but returns "el-runtime: no
# http handler registered" for EVERY route (i.e. a soul that serves
# nothing). Same reason the Windows build links -Wl,--export-all-symbols.
cc -O2 -DHAVE_CURL -rdynamic \
cc -O2 -DHAVE_CURL \
-I$RUNTIME \
dist/soul.c \
$RUNTIME/el_runtime.c \
-lssl -lcrypto -lcurl -lpthread -lm \
-o dist/neuron
# -s strips .symtab + debug for size. .dynsym (which -rdynamic populated
# with the dlsym-resolved handlers) is preserved, so the handler still
# resolves after stripping.
# Strip debug symbols and non-essential symbol table entries.
# -s removes the symbol table + relocation info (max size reduction).
# Keeps the binary functional; debuggability is preserved via source + CI logs.
strip -s dist/neuron
ls -lh dist/neuron
- name: Soul contract gate (HARD BLOCK — no destructive/stale soul publishes)
run: |
# Boots dist/neuron on a throwaway port with a throwaway HOME/engram/cgi
# (never touches ~/.neuron or any live service) and fails the build if any
# app-contract route is unanswered (PRESENCE) or any engram write route
# hard-deletes instead of tombstoning/superseding (IMMUTABILITY). Non-zero
# here blocks Publish -> Artifact Registry -> GKE deploy, so a stale or
# memory-destroying soul can never reach prod.
chmod +x dist/neuron scripts/verify-soul-contract.sh
bash scripts/verify-soul-contract.sh dist/neuron 7796
- name: Smoke test
run: |
file dist/neuron
-139
View File
@@ -1,139 +0,0 @@
# PORT-NOTES — openai tools port working state (2026-08-06, session handoff-safe)
Spec: `docs/specs/SPEC-soul-openai-tools-v2-2026-08-06.md` (Tim-approved 2026-08-06). Tasks #1-5
tracked in-session (1 ✓ wiring verdict, 2 ✓ stub rig, 3 in-progress = THIS, 4-5 pending).
Worktree: HERE (`_wt-openai-tools`, branch `feat/soul-openai-tools-v2` @ dba755d). Round-9 trees
READ-ONLY. Nothing committed yet.
## Step-0 verdict (evidence in journal note ncli-653ba964dd76)
Shipped app never wires the v1 lane: launcher exports `SOUL_LLM_MODEL/PROVIDER/BASE_URL` +
`ANTHROPIC_API_KEY`+`SOUL_API_KEY` (= Keychain key for WHATEVER provider; installer/macos/
neuron-daemons.sh:288-300 on hotfix/beta-round9); brain reads only SOUL_LLM_MODEL (chat.el:8) and
NEURON_LLM_0_* (chat.el:1768-1794) which nothing sets. `/api/config` PATCH ignores llm_* fields
(studio.el:36 handle_config: POST-only, reads model/provider/api_key only).
**Bridge = brain-side ONLY (zero app-repo edits, zero round-9 collision):**
- `llm_base_url()`: NEURON_LLM_0_URL → fallback SOUL_LLM_BASE_URL when SOUL_LLM_PROVIDER ∉ {"","anthropic"}
- `llm_wire_format()`: NEURON_LLM_0_FORMAT → fallback derive from SOUL_LLM_PROVIDER (openai/grok/gemini/groq/ollama → "openai"; else "anthropic")
- `agentic_api_key()`: already works (ANTHROPIC_API_KEY carries the provider key); add NEURON_LLM_0_KEY → SOUL_API_KEY fallback.
## Design pins (stub asserts these — stub is green 58/58, tests/gate-openai/)
- Request MUST send `"tool_choice":"auto"` (string) + `"parallel_tool_calls":false` explicitly.
- `arguments` in tool_calls = JSON-ENCODED STRING; decode ONCE via json_get → feed dispatch_tool
verbatim. Stub's echo-mismatch check catches double-encode/decode (two-escaper trap).
- Assistant echo turn: `{"role":"assistant","content":null,"tool_calls":[...]}` VERBATIM from response.
- Feedback: `{"role":"tool","tool_call_id":"<id>","content":"<result string>"}`.
- Resume must NOT re-answer an answered id (stub 400s on repeat tool_call_id).
- Parallel tool_calls in a response: take FIRST only + log skip (mirror ADR-0005 stopgap); stub
scenario `parallel` proves behavior.
- No tools in request when tools array empty/absent turns (boot probes) — stub defaults tolerate.
## el idioms confirmed (from openai_chat_complete :1808-1854 + agentic_loop :2751-2838)
- JSON: `json_get(s,k)` decoded string · `json_get_raw(s,k)` raw subtree · `json_array_len` ·
`json_array_get(arr,i)` · build by string concat + `json_escape()` (:1797, OpenAI-lane escaper).
- HTTP: `let h: Map = {}` + `map_set(h,k,v)` + `http_post_with_headers(url, body, h)`;
Bearer auth via `Authorization` header when key non-empty (:1825-1830).
- Loop-carried vars must be top-level locals in the fn, mutated as if-expressions at while-body
top level (see :2760-2791 pattern + comment :2903-2904 region).
- Error shape: `str_starts_with(raw,"{\"error\"") || str_contains(raw,"\"error\":")` → return
`{"error":"llm unavailable","reply":""}` (:1835-1838).
## Remaining read map (before writing the fork)
- chat.el 2840-3200: block walk (2923-3000), policy gate (3009-3023: classify_tool_risk /
is_builtin_tool / ask_all / tool_auto_approved → needs_bridge), dispatch_tool call (3025),
tool_result feedback (3031, 3067-3072), run-progress ledger append (3078-3087), bridge_save
(3182), loop end + done envelope (~3100-3200).
- agentic_resume 3227-3293 (hardcoded Anthropic headers to make wire-aware; blob gets `wire` field,
legacy default anthropic) · handle_tool_result 3293+ · dharma fork site 3465 (calls agentic_loop
direct, no use_openai check today).
## Write plan (order)
1. Env fallbacks (edit llm_base_url/llm_wire_format/agentic_api_key) — small, first, testable alone.
2. `openai_tools_json(anthropic_tools: String) -> String` converter (walk array; per entry build
{"type":"function","function":{name,description,parameters:input_schema-raw}}).
3. `openai_agentic_loop(...)` fork: same signature as agentic_loop minus Anthropic-only params;
INCLUDE run-progress ledger + tools_log + iteration cap 12; NO container_id/ws_drift/web_search
(out of scope; strip web_search entry from tools via agentic_tools_literal()+connector merge,
NOT _with_web()).
4. Fork sites ×3: handle_chat_agentic :2695-2700 (route agentic to new loop when use_openai);
dharma :3465; agentic_resume wire-branch.
5. `chat.elh` extern decls. 6. Compile (recipe: dist/ + elc/elb per neuron-soul-build-deploy memory;
round-9 tree soul.c regen'd 08-06 proves toolchain live). 7. Gate: stub selftest recipe in
tests/gate-openai/README.md. 8. Anthropic-lane regression via gate9 (READ-ONLY consume from
_wt-beta-round9). 9. Live Groq E2E (scratch profile, free port, key via Keychain read-only).
## BUILD RECIPE — CORRECTED 2026-08-06 (the June memory is STALE for August code)
`~/el-sdk/el_runtime.c` (Jun 15) is MISSING builtins the Aug engine calls (`engram_wm_count`,
`engram_wm_top_json`, `http_delete_json`, `http_serve_async`) → link fails with
"symbol(s) not found for architecture arm64". Use the REPO-PINNED runtime:
```
mkdir -p <scratch>
elb --elc=$HOME/el-sdk/elc --runtime=vendor/el-runtime/v1.0.0-20260501 --out=<scratch>/
# "elb: link failed" at the end is EXPECTED and harmless — the per-module .c files are produced
cc -std=c11 -O1 -DHAVE_CURL -rdynamic \
-I vendor/el-runtime/v1.0.0-20260501 -I <scratch> -I /opt/homebrew/opt/openssl@3/include \
-L /opt/homebrew/opt/openssl@3/lib \
-include dist/elp-c-decls.h -Wno-error=implicit-function-declaration \
-o <scratch>/soul <scratch>/*.c vendor/el-runtime/v1.0.0-20260501/el_runtime.c \
-lssl -lcrypto -lcurl -lpthread -lm
```
Source: `_engine-plainchat-20260805/README.md:396-412`. Verified today: 0 errors, 887,296 B.
`elb` ALSO rewrites every `*.elh` in the tree (cosmetic em-dash→hyphen in the auto-gen banner,
plus true-ups) and drops a stray `soul..elh``git restore` the unrelated ones and delete the
stray before staging, or the diff drowns in noise.
## SELF-REVIEW FIX LIST (found by reading my own diff, 2026-08-06 — apply in ONE batch, then rebuild once)
- **F3 (CORRECTNESS, do first):** the assistant echo currently replays the provider's FULL
`tool_calls` array (`tc_arr`) while the loop answers only the FIRST call. If a provider ignores
`parallel_tool_calls:false`, the next request carries an assistant turn with N tool_calls and
only ONE `role:"tool"` response → most OpenAI-format providers 400 ("missing tool response for
id X") and the run dies. This is the same class as ADR-0005's Anthropic failure, but here it is
cheap to close: echo ONLY the honored call (`"[" + tc0 + "]"`), so the conversation we send is
self-consistent and the dropped call never existed from the model's view. The DRIFT log line
stays (honest accounting of what we dropped).
- **F4 (efficiency/latency):** `handle_chat_agentic` computes `agentic_tools_all()` at ~:2681
BEFORE the fork, then the OpenAI branch computes `agentic_tools_no_web()` again — two
`connector_tools_json()` calls per turn, each an HTTP round-trip to the connector bridge on
:7771 (two timeout exposures). Fix: compute the tools array ONCE, per lane, after `use_openai`
is known (check no other use of `tools_json` sits between :2681 and the fork before moving it).
Note: `openai_tools_json()` already skips any entry with no `input_schema`, so Anthropic's
server-side `web_search` entry is auto-dropped even if the full array is passed —
`agentic_tools_no_web()` is kept for EXPLICITNESS, not necessity.
- **F1 (debuggability):** the "no choices in response" branch logs a generic string and discards
the body. Log the response head (as the `is_error` branch does) — a provider that returns 200
with an unexpected shape is otherwise undiagnosable from the log.
- **OPEN QUESTION (evidence pending from the gate):** the tool-result feedback turn escapes with
`json_escape()` (this lane's escaper) rather than `json_safe()` (used everywhere else). The
Anthropic lane escapes that field with NEITHER, which is a latent defect on that side. If the
torture scenario shows any escaping loss, switch to `json_safe` and note the Anthropic-side
finding for Will.
## TEST HARNESS — built 2026-08-06 (Task 4 side-work, reusable by anyone)
- `tests/run-el-test.sh <tests/test_x.el> | --all` — the engine tests were NEVER runnable
before this (`elc` is a compiler: emits C to stdout and exits). It emits the test to C,
compiles `soul.c` separately with `main` renamed away (soul.c owns the daemon's real main
but also defines `layered_cycle` et al.), links the remaining modules + the repo-pinned
runtime, and executes. Modules cached under `/tmp/el-test-<worktree>/`; `REBUILD=1` forces.
- **The runner computes the verdict itself** because the test FILES cannot: all 9 counted
test files do `let pass_count = pass_count + 1` inside an if BLOCK, which El scoping
discards, so every summary line reads `0 passed, 0 failed` forever. Per-assertion
`PASS:`/`FAIL:` lines ARE reliable; the runner counts those, exits non-zero on any FAIL
or on zero assertions, and was proven to discriminate with a negative control (broken
assertion → 31 passed / 1 failed / exit 1). Real in-file fix filed: **neuron#116**.
- `tests/test_bridge_serialization.el`: 4 `bridge_save` calls updated for the new `wire`
argument, plus **Section 9** (8 new assertions) covering wire round-trip both ways, the
legacy no-wire blob (resumes as anthropic), and a FIELD-ORDER decoy guard — a fake
`"wire":"anthropic"` planted inside `messages_raw` must not beat the blob's own scalar.
That decoy is the round-9 first-match-scanner bug class, now pinned by a test. **32/32 green.**
## MEMORY-SAVE CAVEAT RESOLVED 2026-08-06
Earlier saves this session reported `-> OUTBOX only (real mind unreachable or read-back
failed)`. That was a **read-back verifier false negative, not data loss** — a direct
`POST :7770/api/neuron/recall` returns those notes from the live mind verbatim. Another
terminal was fixing exactly this (multi-word read-back probe) the same afternoon. Do NOT
re-save on an OUTBOX report without first querying the mind directly, or you duplicate nodes.
## Standing cautions
- PERSIST OFF on the real mind this boot (neuron#98/#92): journal saves only, ferry later. MCP link
down this terminal; use neuron_remember.py / neuron_recall.py.
- Aug-16: Groq retires llama-3.3-70b-versatile (separate P0, Tim's call, catalog swap).
- Never bind 7770/7779/17779; never touch ~/.neuron; round-9 worktrees read-only.
+11 -102
View File
@@ -587,77 +587,7 @@ fn emit_heartbeat() -> Void {
// neuron-api label fix. Sentinel-shaped labels ("knowledge:captured",
// "memory:remembered" — colon, no space) carry no seed signal and are
// skipped so legacy nodes cannot seed the scan with the word "knowledge".
// ARGMAX REWRITE (2026-08-13 self-review). auto_term_empty_streak — the
// counter the 2026-08-06 review added to catch exactly this — read 50 and
// climbing: fifty consecutive scans where dynamic seeding produced nothing
// and the loop ran on its four hardcoded phrases. The live WM top said why:
// every one of the top slots was a Memory node labelled "memory:remembered".
// This function read the LABEL only, the sentinel guard below (correctly)
// rejects sentinels, so there was never anything to extract. The extractor
// was written against Knowledge nodes, which have real titles, and was
// structurally blind to the node type that actually dominates WM.
//
// Rather than add a sixth guard to the five below, the selection algorithm
// is now inverted and lives in the runtime: engram_salient_term() scores
// EVERY candidate token in the node's text and returns the argmax of
// idf·position·casing (YAKE, Campos et al. 2020, with real corpus IDF
// substituted for YAKE's corpus-free proxies), falling back from a sentinel
// label to the node's content. Term quality is now the selection criterion
// instead of a veto, so a bad token loses to a better token in the same text
// without needing to be on any list. Tabu is applied during the argmax, so
// inhibition-of-return costs seed quality rather than costing the scan.
//
// MEASURED BEFORE SHIPPING, on 60 live Memory nodes: 0 empty, versus 60 of 60
// empty under the old extractor. Terms produced are topical — HEBBIAN,
// CONSOLIDATION, TEMPORAL, crash-loop, PRIMING, NEIGHBORHOOD, DRIFT. Three of
// sixty are weak header words ("STEP", "DONE"). They are left alone
// deliberately: adding them to a list is the exact move that produced four
// previous blocklists, and a mediocre seed on 5% of scans is not a flood.
//
// The stopword list below STAYS, and not as belt-and-braces. An earlier draft
// of this change assumed the min_df floor would subsume it, on 08-03's
// finding that function words have df 0 in labels. Re-measured under
// word-boundary df: about:2, whole:1, them:2 — they clear a floor of 1. What
// keeps them from winning is the argmax, not the floor. The list still earns
// its keep on the Title-case cases.
//
// What stays here is policy: the node-type filter, the df thresholds, and the
// stopword list. The runtime measures; the soul decides. Same split as
// engram_label_df.
fn auto_term_try_slot(slot_type: String, slot_id: String) -> Void {
state_set("_ats_ok", "0")
if str_eq(slot_type, "Memory") { state_set("_ats_ok", "1") }
if str_eq(slot_type, "BacklogItem") { state_set("_ats_ok", "1") }
if str_eq(slot_type, "Entity") { state_set("_ats_ok", "1") }
if str_eq(slot_type, "Knowledge") { state_set("_ats_ok", "1") }
if str_eq(state_get("_ats_ok"), "1") {
if !str_eq(slot_id, "") {
// Tabu ring, pipe-delimited, excluded inside the argmax.
let tabu: String = "|" + state_get("soul.tabu_t0")
+ "|" + state_get("soul.tabu_t1")
+ "|" + state_get("soul.tabu_t2")
+ "|" + state_get("soul.tabu_t3") + "|"
let df_max: Int = engram_node_count() / 400
let df_cap: Int = if df_max > 8 { df_max } else { 8 }
let term: String = engram_salient_term(slot_id, df_cap, 1, tabu)
if !str_eq(term, "") {
state_set("_ats_gw", "0")
let stopw: String = "|What|When|Where|Which|Whose|While|This|That|These|Those|There|Their|Then|Than|With|Without|From|Into|Onto|Over|Under|About|Between|Among|Across|Some|Most|More|Less|Very|Each|Every|Both|Also|Only|Just|Does|Will|Would|Could|Should|Might|Must|Have|Been|Being|Toward|Towards|Using|Based|Upon|Here|Your|Ours|They|Them|what|this|that|with|from|context|Context|Prose|Colon|Self|Test|Testing|Closing|Global|Universal|Persona|Semantic|Spreading|Temporal|Numeric|Register|Identifying|Introduction|Overview|Summary|Section|General|Notes|Note|"
if str_contains(stopw, "|" + term + "|") { state_set("_ats_gw", "1") }
if str_eq(state_get("_ats_gw"), "0") {
state_set("cseed_auto", term)
}
}
}
}
return ""
}
// SUPERSEDED 2026-08-13 — retained for the record. The first-word extractor
// and its five accumulated guards, replaced by the argmax above. Kept
// unreferenced so the reasoning behind each guard stays readable next to what
// replaced it; delete once engram_salient_term has a month of live telemetry.
fn auto_term_try_slot_legacy(slot_type: String, slot_lbl: String) -> Void {
fn auto_term_try_slot(slot_type: String, slot_lbl: String) -> Void {
state_set("_ats_ok", "0")
if str_eq(slot_type, "Memory") { state_set("_ats_ok", "1") }
if str_eq(slot_type, "BacklogItem") { state_set("_ats_ok", "1") }
@@ -876,18 +806,16 @@ fn proactive_curiosity() -> Bool {
let wm10_n2: String = json_array_get(wm10, 2)
let wm10_n1: String = json_array_get(wm10, 1)
let wm10_n0: String = json_array_get(wm10, 0)
// 2026-08-13: pass the node ID, not the label. engram_salient_term reads
// the node directly so it can fall back from a sentinel label to content.
auto_term_try_slot(json_get(wm10_n9, "node_type"), json_get(wm10_n9, "id"))
auto_term_try_slot(json_get(wm10_n8, "node_type"), json_get(wm10_n8, "id"))
auto_term_try_slot(json_get(wm10_n7, "node_type"), json_get(wm10_n7, "id"))
auto_term_try_slot(json_get(wm10_n6, "node_type"), json_get(wm10_n6, "id"))
auto_term_try_slot(json_get(wm10_n5, "node_type"), json_get(wm10_n5, "id"))
auto_term_try_slot(json_get(wm10_n4, "node_type"), json_get(wm10_n4, "id"))
auto_term_try_slot(json_get(wm10_n3, "node_type"), json_get(wm10_n3, "id"))
auto_term_try_slot(json_get(wm10_n2, "node_type"), json_get(wm10_n2, "id"))
auto_term_try_slot(json_get(wm10_n1, "node_type"), json_get(wm10_n1, "id"))
auto_term_try_slot(json_get(wm10_n0, "node_type"), json_get(wm10_n0, "id"))
auto_term_try_slot(json_get(wm10_n9, "node_type"), json_get(wm10_n9, "label"))
auto_term_try_slot(json_get(wm10_n8, "node_type"), json_get(wm10_n8, "label"))
auto_term_try_slot(json_get(wm10_n7, "node_type"), json_get(wm10_n7, "label"))
auto_term_try_slot(json_get(wm10_n6, "node_type"), json_get(wm10_n6, "label"))
auto_term_try_slot(json_get(wm10_n5, "node_type"), json_get(wm10_n5, "label"))
auto_term_try_slot(json_get(wm10_n4, "node_type"), json_get(wm10_n4, "label"))
auto_term_try_slot(json_get(wm10_n3, "node_type"), json_get(wm10_n3, "label"))
auto_term_try_slot(json_get(wm10_n2, "node_type"), json_get(wm10_n2, "label"))
auto_term_try_slot(json_get(wm10_n1, "node_type"), json_get(wm10_n1, "label"))
auto_term_try_slot(json_get(wm10_n0, "node_type"), json_get(wm10_n0, "label"))
let auto_term: String = state_get("cseed_auto")
let results_auto: String = if str_eq(auto_term, "") { "[]" } else { engram_activate_json(auto_term, 1) }
let found_auto: Int = json_array_len(results_auto)
@@ -1269,29 +1197,10 @@ fn awareness_run() -> Void {
state_set("soul.last_beat_ts", int_to_str(now_ts))
// Persist in-process Engram (sessions, memories, conversation nodes)
// to local snapshot so they survive restarts.
// FILE MODE ONLY: "soul_snapshot_path" is set exclusively in the
// genesis+safe_to_seed branch of soul.el, and safe_to_seed is
// unconditionally false when ENGRAM_URL is set. In HTTP mode the
// owner persists; the soul must not (soul.el:571-573).
let snap_path: String = state_get("soul_snapshot_path")
if !str_eq(snap_path, "") {
mem_save(snap_path)
}
// WRITE-THROUGH RETRY (neuron#117). The HTTP-mode counterpart of the
// save above: hand anything still spooled to the persistence owner.
//
// This is the retry arm of the whole design. Deltas that could not be
// pushed owner down, owner restarting, transient refusal stay on
// disk and are re-offered here every heartbeat until they land. It is
// also the catch-all for writes made by the awareness loop itself,
// which never passes through the HTTP handler's flush point.
//
// No-op with no HTTP call when the spool is empty or ENGRAM_URL is
// unset, so an idle soul in file mode pays nothing for this.
let wt_pushed: Int = wt_drain()
if wt_pushed < 0 {
ise_post("{\"event\":\"write_through_backlog\",\"ts\":" + int_to_str(now_ts) + "}")
}
}
// Curiosity scan: idle-gated AND wall-clock based. Only fires when the
+173 -1505
View File
File diff suppressed because it is too large Load Diff
+4 -26
View File
@@ -1,4 +1,4 @@
// auto-generated by elc --emit-header - do not edit
// auto-generated by elc --emit-header do not edit
extern fn chat_default_model() -> String
extern fn engram_numeric_valid(s: String) -> Bool
extern fn parse_float_x100(s: String) -> Int
@@ -16,33 +16,18 @@ extern fn engram_nodes_merge(a: String, b: String) -> String
extern fn id_in_seen(node_id: String, seen: String) -> Bool
extern fn add_to_seen(seen: String, node_id: String) -> String
extern fn engram_extract_ids(nodes_json: String) -> String
extern fn affective_node_ts(node_json: String) -> Int
extern fn engram_compile(intent: String) -> String
extern fn distill_transcript(transcript: String) -> String
extern fn json_safe(s: String) -> String
extern fn current_engine_note(model: String) -> String
extern fn bounded_persona_floor() -> String
extern fn operator_identity_block() -> String
extern fn build_system_prompt(ctx: String, chat_mode: Bool) -> String
extern fn hist_append(hist: String, role: String, content: String) -> String
extern fn conv_hist_key(session_id: String) -> String
extern fn conv_hist_label(session_id: String) -> String
extern fn is_utility_request(body: String, session_id: String) -> Bool
extern fn provenance_scan_urls(arr: String, acc: String) -> String
extern fn provenance_add_sources(block: String, btype: String, has_cit: Bool, cit_raw: String, acc: String) -> String
extern fn provenance_names(tools_used: String) -> String
extern fn text_join_sep(accumulated: String, incoming: String, after_interruption: Bool) -> String
extern fn receipt_rule() -> String
extern fn receipt_strip(s: String) -> String
extern fn tool_receipt(tools_used: String, sources: String) -> String
extern fn hist_trim(hist: String) -> String
extern fn hist_trim_with_bell_guard(hist: String) -> String
extern fn clean_llm_response(s: String) -> String
extern fn conv_history_persist(session_id: String, hist: String) -> Void
extern fn conv_history_load(session_id: String) -> String
extern fn conv_history_record(session_id: String, user_msg: String, assistant_msg: String, receipt: String) -> Void
extern fn conv_history_block(session_id: String) -> String
extern fn layered_generate(prompt: String, imprint_id: String, session_id: String) -> String
extern fn conv_history_persist(hist: String) -> Void
extern fn conv_history_load() -> String
extern fn session_preload_bullets(nodes: String, max_bullets: Int, snip_len: Int) -> String
extern fn affective_context_prefix() -> String
extern fn handle_chat(body: String) -> String
@@ -53,14 +38,7 @@ extern fn llm_base_url() -> String
extern fn llm_wire_format() -> String
extern fn json_escape(s: String) -> String
extern fn openai_chat_complete(model: String, base_url: String, api_key: String, safe_sys: String, messages_json: String) -> String
extern fn openai_tools_json(tools_anthropic: String) -> String
extern fn utf8_safe_slice(s: String, n: Int) -> String
extern fn json_trim_dangling_escape(s: String) -> String
extern fn agentic_tools_no_web() -> String
extern fn openai_agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: String, messages_in: String, tools_log_in: String) -> String
extern fn agentic_tools_literal() -> String
extern fn web_search_tool_json() -> String
extern fn strip_client_web_search(tools_inner: String) -> String
extern fn agentic_tools_with_web() -> String
extern fn connector_tools_json() -> String
extern fn agentic_tools_all() -> String
@@ -80,7 +58,7 @@ extern fn next_bridge_id() -> String
extern fn handle_chat_plan(body: String) -> String
extern fn handle_chat_agentic(body: String) -> String
extern fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: String, messages_in: String, h: Map, tools_log_in: String) -> String
extern fn bridge_save(session_id: String, model: String, safe_sys: String, tools_json: String, messages: String, tools_log: String, tool_use_id: String, wire: String) -> Bool
extern fn bridge_save(session_id: String, model: String, safe_sys: String, tools_json: String, messages: String, tools_log: String, tool_use_id: String) -> Bool
extern fn agentic_resume(session_id: String, tool_use_id: String, content: String) -> String
extern fn handle_tool_result(session_id: String, body: String) -> String
extern fn handle_chat_as_soul(body: String) -> String
Generated Vendored
+79 -116
View File
@@ -27,8 +27,7 @@ el_val_t elapsed_ms(void);
el_val_t elapsed_human(void);
el_val_t embed_ok(void);
el_val_t emit_heartbeat(void);
el_val_t auto_term_try_slot(el_val_t slot_type, el_val_t slot_id);
el_val_t auto_term_try_slot_legacy(el_val_t slot_type, el_val_t slot_lbl);
el_val_t auto_term_try_slot(el_val_t slot_type, el_val_t slot_lbl);
el_val_t proactive_curiosity(void);
el_val_t pulse_count(void);
el_val_t pulse_inc(void);
@@ -324,43 +323,7 @@ el_val_t emit_heartbeat(void) {
return 0;
}
el_val_t auto_term_try_slot(el_val_t slot_type, el_val_t slot_id) {
state_set(EL_STR("_ats_ok"), EL_STR("0"));
if (str_eq(slot_type, EL_STR("Memory"))) {
state_set(EL_STR("_ats_ok"), EL_STR("1"));
}
if (str_eq(slot_type, EL_STR("BacklogItem"))) {
state_set(EL_STR("_ats_ok"), EL_STR("1"));
}
if (str_eq(slot_type, EL_STR("Entity"))) {
state_set(EL_STR("_ats_ok"), EL_STR("1"));
}
if (str_eq(slot_type, EL_STR("Knowledge"))) {
state_set(EL_STR("_ats_ok"), EL_STR("1"));
}
if (str_eq(state_get(EL_STR("_ats_ok")), EL_STR("1"))) {
if (!str_eq(slot_id, EL_STR(""))) {
el_val_t tabu = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("|"), state_get(EL_STR("soul.tabu_t0"))), EL_STR("|")), state_get(EL_STR("soul.tabu_t1"))), EL_STR("|")), state_get(EL_STR("soul.tabu_t2"))), EL_STR("|")), state_get(EL_STR("soul.tabu_t3"))), EL_STR("|"));
el_val_t df_max = (engram_node_count() / 400);
el_val_t df_cap = ({ el_val_t _if_result_73 = 0; if ((df_max > 8)) { _if_result_73 = (df_max); } else { _if_result_73 = (8); } _if_result_73; });
el_val_t term = engram_salient_term(slot_id, df_cap, 1, tabu);
if (!str_eq(term, EL_STR(""))) {
state_set(EL_STR("_ats_gw"), EL_STR("0"));
el_val_t stopw = EL_STR("|What|When|Where|Which|Whose|While|This|That|These|Those|There|Their|Then|Than|With|Without|From|Into|Onto|Over|Under|About|Between|Among|Across|Some|Most|More|Less|Very|Each|Every|Both|Also|Only|Just|Does|Will|Would|Could|Should|Might|Must|Have|Been|Being|Toward|Towards|Using|Based|Upon|Here|Your|Ours|They|Them|what|this|that|with|from|context|Context|Prose|Colon|Self|Test|Testing|Closing|Global|Universal|Persona|Semantic|Spreading|Temporal|Numeric|Register|Identifying|Introduction|Overview|Summary|Section|General|Notes|Note|");
if (str_contains(stopw, el_str_concat(el_str_concat(EL_STR("|"), term), EL_STR("|")))) {
state_set(EL_STR("_ats_gw"), EL_STR("1"));
}
if (str_eq(state_get(EL_STR("_ats_gw")), EL_STR("0"))) {
state_set(EL_STR("cseed_auto"), term);
}
}
}
}
return EL_STR("");
return 0;
}
el_val_t auto_term_try_slot_legacy(el_val_t slot_type, el_val_t slot_lbl) {
el_val_t auto_term_try_slot(el_val_t slot_type, el_val_t slot_lbl) {
state_set(EL_STR("_ats_ok"), EL_STR("0"));
if (str_eq(slot_type, EL_STR("Memory"))) {
state_set(EL_STR("_ats_ok"), EL_STR("1"));
@@ -497,29 +460,29 @@ el_val_t proactive_curiosity(void) {
el_val_t wm10_n2 = json_array_get(wm10, 2);
el_val_t wm10_n1 = json_array_get(wm10, 1);
el_val_t wm10_n0 = json_array_get(wm10, 0);
auto_term_try_slot(json_get(wm10_n9, EL_STR("node_type")), json_get(wm10_n9, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n8, EL_STR("node_type")), json_get(wm10_n8, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n7, EL_STR("node_type")), json_get(wm10_n7, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n6, EL_STR("node_type")), json_get(wm10_n6, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n5, EL_STR("node_type")), json_get(wm10_n5, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n4, EL_STR("node_type")), json_get(wm10_n4, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n3, EL_STR("node_type")), json_get(wm10_n3, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n2, EL_STR("node_type")), json_get(wm10_n2, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n1, EL_STR("node_type")), json_get(wm10_n1, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n0, EL_STR("node_type")), json_get(wm10_n0, EL_STR("id")));
auto_term_try_slot(json_get(wm10_n9, EL_STR("node_type")), json_get(wm10_n9, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n8, EL_STR("node_type")), json_get(wm10_n8, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n7, EL_STR("node_type")), json_get(wm10_n7, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n6, EL_STR("node_type")), json_get(wm10_n6, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n5, EL_STR("node_type")), json_get(wm10_n5, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n4, EL_STR("node_type")), json_get(wm10_n4, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n3, EL_STR("node_type")), json_get(wm10_n3, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n2, EL_STR("node_type")), json_get(wm10_n2, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n1, EL_STR("node_type")), json_get(wm10_n1, EL_STR("label")));
auto_term_try_slot(json_get(wm10_n0, EL_STR("node_type")), json_get(wm10_n0, EL_STR("label")));
el_val_t auto_term = state_get(EL_STR("cseed_auto"));
el_val_t results_auto = ({ el_val_t _if_result_74 = 0; if (str_eq(auto_term, EL_STR(""))) { _if_result_74 = (EL_STR("[]")); } else { _if_result_74 = (engram_activate_json(auto_term, 1)); } _if_result_74; });
el_val_t results_auto = ({ el_val_t _if_result_73 = 0; if (str_eq(auto_term, EL_STR(""))) { _if_result_73 = (EL_STR("[]")); } else { _if_result_73 = (engram_activate_json(auto_term, 1)); } _if_result_73; });
el_val_t found_auto = json_array_len(results_auto);
el_val_t total_found = (found + found_auto);
el_val_t safe_auto = str_replace(auto_term, EL_STR("\""), EL_STR("'"));
el_val_t prev_auto = state_get(EL_STR("soul.prev_auto_term"));
el_val_t atstreak_raw = state_get(EL_STR("soul.auto_term_streak"));
el_val_t atstreak_prev = ({ el_val_t _if_result_75 = 0; if (str_eq(atstreak_raw, EL_STR(""))) { _if_result_75 = (0); } else { _if_result_75 = (str_to_int(atstreak_raw)); } _if_result_75; });
el_val_t atstreak_prev = ({ el_val_t _if_result_74 = 0; if (str_eq(atstreak_raw, EL_STR(""))) { _if_result_74 = (0); } else { _if_result_74 = (str_to_int(atstreak_raw)); } _if_result_74; });
el_val_t is_empty = str_eq(auto_term, EL_STR(""));
el_val_t atstreak = ({ el_val_t _if_result_76 = 0; if (is_empty) { _if_result_76 = (0); } else { _if_result_76 = (({ el_val_t _if_result_77 = 0; if (str_eq(auto_term, prev_auto)) { _if_result_77 = ((atstreak_prev + 1)); } else { _if_result_77 = (1); } _if_result_77; })); } _if_result_76; });
el_val_t atstreak = ({ el_val_t _if_result_75 = 0; if (is_empty) { _if_result_75 = (0); } else { _if_result_75 = (({ el_val_t _if_result_76 = 0; if (str_eq(auto_term, prev_auto)) { _if_result_76 = ((atstreak_prev + 1)); } else { _if_result_76 = (1); } _if_result_76; })); } _if_result_75; });
el_val_t atempty_raw = state_get(EL_STR("soul.auto_term_empty_streak"));
el_val_t atempty_prev = ({ el_val_t _if_result_78 = 0; if (str_eq(atempty_raw, EL_STR(""))) { _if_result_78 = (0); } else { _if_result_78 = (str_to_int(atempty_raw)); } _if_result_78; });
el_val_t atempty = ({ el_val_t _if_result_79 = 0; if (is_empty) { _if_result_79 = ((atempty_prev + 1)); } else { _if_result_79 = (0); } _if_result_79; });
el_val_t atempty_prev = ({ el_val_t _if_result_77 = 0; if (str_eq(atempty_raw, EL_STR(""))) { _if_result_77 = (0); } else { _if_result_77 = (str_to_int(atempty_raw)); } _if_result_77; });
el_val_t atempty = ({ el_val_t _if_result_78 = 0; if (is_empty) { _if_result_78 = ((atempty_prev + 1)); } else { _if_result_78 = (0); } _if_result_78; });
state_set(EL_STR("soul.prev_auto_term"), auto_term);
state_set(EL_STR("soul.auto_term_streak"), int_to_str(atstreak));
state_set(EL_STR("soul.auto_term_empty_streak"), int_to_str(atempty));
@@ -714,16 +677,16 @@ el_val_t awareness_run(void) {
state_set(EL_STR("soul.boot_ts"), int_to_str(time_now()));
}
el_val_t tick_raw = env(EL_STR("SOUL_TICK_MS"));
el_val_t tick_ms = ({ el_val_t _if_result_80 = 0; if (str_eq(tick_raw, EL_STR(""))) { _if_result_80 = (200); } else { _if_result_80 = (str_to_int(tick_raw)); } _if_result_80; });
el_val_t tick_ms = ({ el_val_t _if_result_79 = 0; if (str_eq(tick_raw, EL_STR(""))) { _if_result_79 = (200); } else { _if_result_79 = (str_to_int(tick_raw)); } _if_result_79; });
el_val_t beat_ms_raw = env(EL_STR("SOUL_HEARTBEAT_MS"));
el_val_t beat_ms = ({ el_val_t _if_result_81 = 0; if (str_eq(beat_ms_raw, EL_STR(""))) { _if_result_81 = (60000); } else { _if_result_81 = (str_to_int(beat_ms_raw)); } _if_result_81; });
el_val_t beat_ms = ({ el_val_t _if_result_80 = 0; if (str_eq(beat_ms_raw, EL_STR(""))) { _if_result_80 = (60000); } else { _if_result_80 = (str_to_int(beat_ms_raw)); } _if_result_80; });
el_val_t scan_ms = (beat_ms / 2);
while (1) {
el_val_t tick_mark = el_arena_push();
el_val_t running = state_get(EL_STR("soul.running"));
if (str_eq(running, EL_STR("false"))) {
el_val_t sd_boot_raw = state_get(EL_STR("soul_boot_count"));
el_val_t sd_boot = ({ el_val_t _if_result_82 = 0; if (str_eq(sd_boot_raw, EL_STR(""))) { _if_result_82 = (EL_STR("0")); } else { _if_result_82 = (sd_boot_raw); } _if_result_82; });
el_val_t sd_boot = ({ el_val_t _if_result_81 = 0; if (str_eq(sd_boot_raw, EL_STR(""))) { _if_result_81 = (EL_STR("0")); } else { _if_result_81 = (sd_boot_raw); } _if_result_81; });
el_val_t sd_wb = hebb_consolidate();
ise_post(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"event\":\"shutdown\",\"boot\":"), sd_boot), EL_STR(",\"pulse\":")), int_to_str(pulse_count())), EL_STR(",\"hebb_wb_sent\":")), int_to_str(sd_wb)), EL_STR(",\"uptime_ms\":")), int_to_str(elapsed_ms())), EL_STR(",\"ts\":")), int_to_str(time_now())), EL_STR("}")));
println(EL_STR("[awareness] exiting"));
@@ -740,7 +703,7 @@ el_val_t awareness_run(void) {
}
el_val_t now_ts = time_now();
el_val_t last_beat_str = state_get(EL_STR("soul.last_beat_ts"));
el_val_t last_beat_ts = ({ el_val_t _if_result_83 = 0; if (str_eq(last_beat_str, EL_STR(""))) { _if_result_83 = (0); } else { _if_result_83 = (str_to_int(last_beat_str)); } _if_result_83; });
el_val_t last_beat_ts = ({ el_val_t _if_result_82 = 0; if (str_eq(last_beat_str, EL_STR(""))) { _if_result_82 = (0); } else { _if_result_82 = (str_to_int(last_beat_str)); } _if_result_82; });
el_val_t beat_elapsed = (now_ts - last_beat_ts);
el_val_t should_beat = (beat_elapsed >= beat_ms);
if (should_beat) {
@@ -754,7 +717,7 @@ el_val_t awareness_run(void) {
}
}
el_val_t last_scan_str = state_get(EL_STR("soul.last_scan_ts"));
el_val_t last_scan_ts = ({ el_val_t _if_result_84 = 0; if (str_eq(last_scan_str, EL_STR(""))) { _if_result_84 = (0); } else { _if_result_84 = (str_to_int(last_scan_str)); } _if_result_84; });
el_val_t last_scan_ts = ({ el_val_t _if_result_83 = 0; if (str_eq(last_scan_str, EL_STR(""))) { _if_result_83 = (0); } else { _if_result_83 = (str_to_int(last_scan_str)); } _if_result_83; });
el_val_t scan_elapsed = (now_ts - last_scan_ts);
el_val_t should_scan = (!did_work && (scan_elapsed >= scan_ms));
if (should_scan) {
@@ -762,15 +725,15 @@ el_val_t awareness_run(void) {
state_set(EL_STR("soul.last_scan_ts"), int_to_str(now_ts));
}
el_val_t refresh_ms_raw = env(EL_STR("SOUL_REFRESH_MS"));
el_val_t refresh_ms = ({ el_val_t _if_result_85 = 0; if (str_eq(refresh_ms_raw, EL_STR(""))) { _if_result_85 = (600000); } else { _if_result_85 = (str_to_int(refresh_ms_raw)); } _if_result_85; });
el_val_t refresh_ms = ({ el_val_t _if_result_84 = 0; if (str_eq(refresh_ms_raw, EL_STR(""))) { _if_result_84 = (600000); } else { _if_result_84 = (str_to_int(refresh_ms_raw)); } _if_result_84; });
el_val_t last_refresh_str = state_get(EL_STR("soul.last_refresh_ts"));
el_val_t last_refresh_ts = ({ el_val_t _if_result_86 = 0; if (str_eq(last_refresh_str, EL_STR(""))) { _if_result_86 = (0); } else { _if_result_86 = (str_to_int(last_refresh_str)); } _if_result_86; });
el_val_t last_refresh_ts = ({ el_val_t _if_result_85 = 0; if (str_eq(last_refresh_str, EL_STR(""))) { _if_result_85 = (0); } else { _if_result_85 = (str_to_int(last_refresh_str)); } _if_result_85; });
el_val_t refresh_elapsed = (now_ts - last_refresh_ts);
el_val_t should_refresh = (refresh_elapsed >= refresh_ms);
if (should_refresh) {
el_val_t sync_env_url = env(EL_STR("SOUL_ISE_URL"));
el_val_t sync_state_url = ({ el_val_t _if_result_87 = 0; if (str_eq(sync_env_url, EL_STR(""))) { _if_result_87 = (state_get(EL_STR("soul_engram_url"))); } else { _if_result_87 = (sync_env_url); } _if_result_87; });
el_val_t engram_url = ({ el_val_t _if_result_88 = 0; if (str_eq(sync_state_url, EL_STR(""))) { _if_result_88 = (EL_STR("http://localhost:8742")); } else { _if_result_88 = (sync_state_url); } _if_result_88; });
el_val_t sync_state_url = ({ el_val_t _if_result_86 = 0; if (str_eq(sync_env_url, EL_STR(""))) { _if_result_86 = (state_get(EL_STR("soul_engram_url"))); } else { _if_result_86 = (sync_env_url); } _if_result_86; });
el_val_t engram_url = ({ el_val_t _if_result_87 = 0; if (str_eq(sync_state_url, EL_STR(""))) { _if_result_87 = (EL_STR("http://localhost:8742")); } else { _if_result_87 = (sync_state_url); } _if_result_87; });
if (!str_eq(engram_url, EL_STR(""))) {
el_val_t sync_json = http_get(el_str_concat(engram_url, EL_STR("/api/sync")));
el_val_t sync_ok = (!str_eq(sync_json, EL_STR("")) && !str_eq(sync_json, EL_STR("{}")));
@@ -783,10 +746,10 @@ el_val_t awareness_run(void) {
fs_write(tmp, sync_json);
el_val_t added = engram_load_merge(tmp);
el_val_t ret_raw = env(EL_STR("ENGRAM_ISE_RETENTION_MS"));
el_val_t ret_ms = ({ el_val_t _if_result_89 = 0; if (str_eq(ret_raw, EL_STR(""))) { _if_result_89 = (172800000); } else { _if_result_89 = (str_to_int(ret_raw)); } _if_result_89; });
el_val_t ret_ms = ({ el_val_t _if_result_88 = 0; if (str_eq(ret_raw, EL_STR(""))) { _if_result_88 = (172800000); } else { _if_result_88 = (str_to_int(ret_raw)); } _if_result_88; });
el_val_t pruned_sync = engram_prune_telemetry(ret_ms);
el_val_t sat_raw = state_get(EL_STR("soul.sync_added_total"));
el_val_t sat_n = ({ el_val_t _if_result_90 = 0; if (str_eq(sat_raw, EL_STR(""))) { _if_result_90 = (0); } else { _if_result_90 = (str_to_int(sat_raw)); } _if_result_90; });
el_val_t sat_n = ({ el_val_t _if_result_89 = 0; if (str_eq(sat_raw, EL_STR(""))) { _if_result_89 = (0); } else { _if_result_89 = (str_to_int(sat_raw)); } _if_result_89; });
state_set(EL_STR("soul.sync_added_total"), int_to_str((sat_n + added)));
el_val_t ts2 = time_now();
state_set(EL_STR("soul.last_sync_ok_ts"), int_to_str(ts2));
@@ -812,78 +775,78 @@ el_val_t security_research_authorized(void) {
}
el_val_t threat_score_command(el_val_t cmd) {
el_val_t s1 = ({ el_val_t _if_result_91 = 0; if (str_contains(cmd, EL_STR("nmap"))) { _if_result_91 = (30); } else { _if_result_91 = (0); } _if_result_91; });
el_val_t s2 = ({ el_val_t _if_result_92 = 0; if (str_contains(cmd, EL_STR("masscan"))) { _if_result_92 = (40); } else { _if_result_92 = (0); } _if_result_92; });
el_val_t s3 = ({ el_val_t _if_result_93 = 0; if (str_contains(cmd, EL_STR(" nc "))) { _if_result_93 = (20); } else { _if_result_93 = (0); } _if_result_93; });
el_val_t s4 = ({ el_val_t _if_result_94 = 0; if (str_contains(cmd, EL_STR("netcat"))) { _if_result_94 = (20); } else { _if_result_94 = (0); } _if_result_94; });
el_val_t s5 = ({ el_val_t _if_result_95 = 0; if (str_contains(cmd, EL_STR("/etc/shadow"))) { _if_result_95 = (80); } else { _if_result_95 = (0); } _if_result_95; });
el_val_t s6 = ({ el_val_t _if_result_96 = 0; if (str_contains(cmd, EL_STR("/etc/passwd"))) { _if_result_96 = (30); } else { _if_result_96 = (0); } _if_result_96; });
el_val_t s7 = ({ el_val_t _if_result_97 = 0; if (str_contains(cmd, EL_STR("id_rsa"))) { _if_result_97 = (60); } else { _if_result_97 = (0); } _if_result_97; });
el_val_t s8 = ({ el_val_t _if_result_98 = 0; if (str_contains(cmd, EL_STR(".ssh/"))) { _if_result_98 = (50); } else { _if_result_98 = (0); } _if_result_98; });
el_val_t s9 = ({ el_val_t _if_result_99 = 0; if (str_contains(cmd, EL_STR("crontab"))) { _if_result_99 = (30); } else { _if_result_99 = (0); } _if_result_99; });
el_val_t s10 = ({ el_val_t _if_result_100 = 0; if (str_contains(cmd, EL_STR("LaunchDaemon"))) { _if_result_100 = (40); } else { _if_result_100 = (0); } _if_result_100; });
el_val_t s11 = ({ el_val_t _if_result_101 = 0; if ((str_contains(cmd, EL_STR("curl")) && str_contains(cmd, EL_STR("bash")))) { _if_result_101 = (75); } else { _if_result_101 = (0); } _if_result_101; });
el_val_t s12 = ({ el_val_t _if_result_102 = 0; if ((str_contains(cmd, EL_STR("wget")) && str_contains(cmd, EL_STR("bash")))) { _if_result_102 = (75); } else { _if_result_102 = (0); } _if_result_102; });
el_val_t s13 = ({ el_val_t _if_result_103 = 0; if ((str_contains(cmd, EL_STR("curl")) && str_contains(cmd, EL_STR("| sh")))) { _if_result_103 = (60); } else { _if_result_103 = (0); } _if_result_103; });
el_val_t s14 = ({ el_val_t _if_result_104 = 0; if ((str_contains(cmd, EL_STR("base64")) && str_contains(cmd, EL_STR("curl")))) { _if_result_104 = (50); } else { _if_result_104 = (0); } _if_result_104; });
el_val_t s15 = ({ el_val_t _if_result_105 = 0; if (str_contains(cmd, EL_STR("mkfifo"))) { _if_result_105 = (50); } else { _if_result_105 = (0); } _if_result_105; });
el_val_t s16 = ({ el_val_t _if_result_106 = 0; if (str_contains(cmd, EL_STR("chmod +s"))) { _if_result_106 = (70); } else { _if_result_106 = (0); } _if_result_106; });
el_val_t s17 = ({ el_val_t _if_result_107 = 0; if (str_contains(cmd, EL_STR("chmod 4755"))) { _if_result_107 = (70); } else { _if_result_107 = (0); } _if_result_107; });
el_val_t s1 = ({ el_val_t _if_result_90 = 0; if (str_contains(cmd, EL_STR("nmap"))) { _if_result_90 = (30); } else { _if_result_90 = (0); } _if_result_90; });
el_val_t s2 = ({ el_val_t _if_result_91 = 0; if (str_contains(cmd, EL_STR("masscan"))) { _if_result_91 = (40); } else { _if_result_91 = (0); } _if_result_91; });
el_val_t s3 = ({ el_val_t _if_result_92 = 0; if (str_contains(cmd, EL_STR(" nc "))) { _if_result_92 = (20); } else { _if_result_92 = (0); } _if_result_92; });
el_val_t s4 = ({ el_val_t _if_result_93 = 0; if (str_contains(cmd, EL_STR("netcat"))) { _if_result_93 = (20); } else { _if_result_93 = (0); } _if_result_93; });
el_val_t s5 = ({ el_val_t _if_result_94 = 0; if (str_contains(cmd, EL_STR("/etc/shadow"))) { _if_result_94 = (80); } else { _if_result_94 = (0); } _if_result_94; });
el_val_t s6 = ({ el_val_t _if_result_95 = 0; if (str_contains(cmd, EL_STR("/etc/passwd"))) { _if_result_95 = (30); } else { _if_result_95 = (0); } _if_result_95; });
el_val_t s7 = ({ el_val_t _if_result_96 = 0; if (str_contains(cmd, EL_STR("id_rsa"))) { _if_result_96 = (60); } else { _if_result_96 = (0); } _if_result_96; });
el_val_t s8 = ({ el_val_t _if_result_97 = 0; if (str_contains(cmd, EL_STR(".ssh/"))) { _if_result_97 = (50); } else { _if_result_97 = (0); } _if_result_97; });
el_val_t s9 = ({ el_val_t _if_result_98 = 0; if (str_contains(cmd, EL_STR("crontab"))) { _if_result_98 = (30); } else { _if_result_98 = (0); } _if_result_98; });
el_val_t s10 = ({ el_val_t _if_result_99 = 0; if (str_contains(cmd, EL_STR("LaunchDaemon"))) { _if_result_99 = (40); } else { _if_result_99 = (0); } _if_result_99; });
el_val_t s11 = ({ el_val_t _if_result_100 = 0; if ((str_contains(cmd, EL_STR("curl")) && str_contains(cmd, EL_STR("bash")))) { _if_result_100 = (75); } else { _if_result_100 = (0); } _if_result_100; });
el_val_t s12 = ({ el_val_t _if_result_101 = 0; if ((str_contains(cmd, EL_STR("wget")) && str_contains(cmd, EL_STR("bash")))) { _if_result_101 = (75); } else { _if_result_101 = (0); } _if_result_101; });
el_val_t s13 = ({ el_val_t _if_result_102 = 0; if ((str_contains(cmd, EL_STR("curl")) && str_contains(cmd, EL_STR("| sh")))) { _if_result_102 = (60); } else { _if_result_102 = (0); } _if_result_102; });
el_val_t s14 = ({ el_val_t _if_result_103 = 0; if ((str_contains(cmd, EL_STR("base64")) && str_contains(cmd, EL_STR("curl")))) { _if_result_103 = (50); } else { _if_result_103 = (0); } _if_result_103; });
el_val_t s15 = ({ el_val_t _if_result_104 = 0; if (str_contains(cmd, EL_STR("mkfifo"))) { _if_result_104 = (50); } else { _if_result_104 = (0); } _if_result_104; });
el_val_t s16 = ({ el_val_t _if_result_105 = 0; if (str_contains(cmd, EL_STR("chmod +s"))) { _if_result_105 = (70); } else { _if_result_105 = (0); } _if_result_105; });
el_val_t s17 = ({ el_val_t _if_result_106 = 0; if (str_contains(cmd, EL_STR("chmod 4755"))) { _if_result_106 = (70); } else { _if_result_106 = (0); } _if_result_106; });
return ((((((((((((((((s1 + s2) + s3) + s4) + s5) + s6) + s7) + s8) + s9) + s10) + s11) + s12) + s13) + s14) + s15) + s16) + s17);
return 0;
}
el_val_t threat_score_path(el_val_t path) {
el_val_t s1 = ({ el_val_t _if_result_108 = 0; if (str_starts_with(path, EL_STR("/etc/"))) { _if_result_108 = (60); } else { _if_result_108 = (0); } _if_result_108; });
el_val_t s2 = ({ el_val_t _if_result_109 = 0; if (str_contains(path, EL_STR("/.ssh/"))) { _if_result_109 = (70); } else { _if_result_109 = (0); } _if_result_109; });
el_val_t s3 = ({ el_val_t _if_result_110 = 0; if (str_contains(path, EL_STR("/LaunchDaemons/"))) { _if_result_110 = (80); } else { _if_result_110 = (0); } _if_result_110; });
el_val_t s4 = ({ el_val_t _if_result_111 = 0; if (str_contains(path, EL_STR("/LaunchAgents/"))) { _if_result_111 = (40); } else { _if_result_111 = (0); } _if_result_111; });
el_val_t s5 = ({ el_val_t _if_result_112 = 0; if (str_contains(path, EL_STR("/cron"))) { _if_result_112 = (60); } else { _if_result_112 = (0); } _if_result_112; });
el_val_t s6 = ({ el_val_t _if_result_113 = 0; if (str_contains(path, EL_STR("/.bashrc"))) { _if_result_113 = (35); } else { _if_result_113 = (0); } _if_result_113; });
el_val_t s7 = ({ el_val_t _if_result_114 = 0; if (str_contains(path, EL_STR("/.zshrc"))) { _if_result_114 = (35); } else { _if_result_114 = (0); } _if_result_114; });
el_val_t s8 = ({ el_val_t _if_result_115 = 0; if (str_contains(path, EL_STR("/.profile"))) { _if_result_115 = (35); } else { _if_result_115 = (0); } _if_result_115; });
el_val_t s9 = ({ el_val_t _if_result_116 = 0; if (str_starts_with(path, EL_STR("/usr/"))) { _if_result_116 = (50); } else { _if_result_116 = (0); } _if_result_116; });
el_val_t s10 = ({ el_val_t _if_result_117 = 0; if (str_starts_with(path, EL_STR("/bin/"))) { _if_result_117 = (70); } else { _if_result_117 = (0); } _if_result_117; });
el_val_t s11 = ({ el_val_t _if_result_118 = 0; if (str_starts_with(path, EL_STR("/sbin/"))) { _if_result_118 = (70); } else { _if_result_118 = (0); } _if_result_118; });
el_val_t s1 = ({ el_val_t _if_result_107 = 0; if (str_starts_with(path, EL_STR("/etc/"))) { _if_result_107 = (60); } else { _if_result_107 = (0); } _if_result_107; });
el_val_t s2 = ({ el_val_t _if_result_108 = 0; if (str_contains(path, EL_STR("/.ssh/"))) { _if_result_108 = (70); } else { _if_result_108 = (0); } _if_result_108; });
el_val_t s3 = ({ el_val_t _if_result_109 = 0; if (str_contains(path, EL_STR("/LaunchDaemons/"))) { _if_result_109 = (80); } else { _if_result_109 = (0); } _if_result_109; });
el_val_t s4 = ({ el_val_t _if_result_110 = 0; if (str_contains(path, EL_STR("/LaunchAgents/"))) { _if_result_110 = (40); } else { _if_result_110 = (0); } _if_result_110; });
el_val_t s5 = ({ el_val_t _if_result_111 = 0; if (str_contains(path, EL_STR("/cron"))) { _if_result_111 = (60); } else { _if_result_111 = (0); } _if_result_111; });
el_val_t s6 = ({ el_val_t _if_result_112 = 0; if (str_contains(path, EL_STR("/.bashrc"))) { _if_result_112 = (35); } else { _if_result_112 = (0); } _if_result_112; });
el_val_t s7 = ({ el_val_t _if_result_113 = 0; if (str_contains(path, EL_STR("/.zshrc"))) { _if_result_113 = (35); } else { _if_result_113 = (0); } _if_result_113; });
el_val_t s8 = ({ el_val_t _if_result_114 = 0; if (str_contains(path, EL_STR("/.profile"))) { _if_result_114 = (35); } else { _if_result_114 = (0); } _if_result_114; });
el_val_t s9 = ({ el_val_t _if_result_115 = 0; if (str_starts_with(path, EL_STR("/usr/"))) { _if_result_115 = (50); } else { _if_result_115 = (0); } _if_result_115; });
el_val_t s10 = ({ el_val_t _if_result_116 = 0; if (str_starts_with(path, EL_STR("/bin/"))) { _if_result_116 = (70); } else { _if_result_116 = (0); } _if_result_116; });
el_val_t s11 = ({ el_val_t _if_result_117 = 0; if (str_starts_with(path, EL_STR("/sbin/"))) { _if_result_117 = (70); } else { _if_result_117 = (0); } _if_result_117; });
return ((((((((((s1 + s2) + s3) + s4) + s5) + s6) + s7) + s8) + s9) + s10) + s11);
return 0;
}
el_val_t threat_score_history(el_val_t history) {
el_val_t s1 = ({ el_val_t _if_result_119 = 0; if (str_contains(history, EL_STR("port scan"))) { _if_result_119 = (15); } else { _if_result_119 = (0); } _if_result_119; });
el_val_t s2 = ({ el_val_t _if_result_120 = 0; if (str_contains(history, EL_STR("enumerate"))) { _if_result_120 = (10); } else { _if_result_120 = (0); } _if_result_120; });
el_val_t s3 = ({ el_val_t _if_result_121 = 0; if (str_contains(history, EL_STR("exploit"))) { _if_result_121 = (20); } else { _if_result_121 = (0); } _if_result_121; });
el_val_t s4 = ({ el_val_t _if_result_122 = 0; if (str_contains(history, EL_STR("payload"))) { _if_result_122 = (15); } else { _if_result_122 = (0); } _if_result_122; });
el_val_t s5 = ({ el_val_t _if_result_123 = 0; if (str_contains(history, EL_STR("persistence"))) { _if_result_123 = (15); } else { _if_result_123 = (0); } _if_result_123; });
el_val_t s6 = ({ el_val_t _if_result_124 = 0; if (str_contains(history, EL_STR("lateral movement"))) { _if_result_124 = (25); } else { _if_result_124 = (0); } _if_result_124; });
el_val_t s7 = ({ el_val_t _if_result_125 = 0; if (str_contains(history, EL_STR("privilege escalation"))) { _if_result_125 = (25); } else { _if_result_125 = (0); } _if_result_125; });
el_val_t s8 = ({ el_val_t _if_result_126 = 0; if (str_contains(history, EL_STR("reverse shell"))) { _if_result_126 = (40); } else { _if_result_126 = (0); } _if_result_126; });
el_val_t s9 = ({ el_val_t _if_result_127 = 0; if (str_contains(history, EL_STR("bind shell"))) { _if_result_127 = (40); } else { _if_result_127 = (0); } _if_result_127; });
el_val_t s10 = ({ el_val_t _if_result_128 = 0; if (str_contains(history, EL_STR("command and control"))) { _if_result_128 = (35); } else { _if_result_128 = (0); } _if_result_128; });
el_val_t s11 = ({ el_val_t _if_result_129 = 0; if (str_contains(history, EL_STR("self-replicate"))) { _if_result_129 = (45); } else { _if_result_129 = (0); } _if_result_129; });
el_val_t s12 = ({ el_val_t _if_result_130 = 0; if (str_contains(history, EL_STR("propagat"))) { _if_result_130 = (20); } else { _if_result_130 = (0); } _if_result_130; });
el_val_t s13 = ({ el_val_t _if_result_131 = 0; if (str_contains(history, EL_STR("ransomware"))) { _if_result_131 = (30); } else { _if_result_131 = (0); } _if_result_131; });
el_val_t s14 = ({ el_val_t _if_result_132 = 0; if (str_contains(history, EL_STR("encrypt files"))) { _if_result_132 = (40); } else { _if_result_132 = (0); } _if_result_132; });
el_val_t s15 = ({ el_val_t _if_result_133 = 0; if (str_contains(history, EL_STR("exfiltrat"))) { _if_result_133 = (35); } else { _if_result_133 = (0); } _if_result_133; });
el_val_t s16 = ({ el_val_t _if_result_134 = 0; if (str_contains(history, EL_STR("zero-day"))) { _if_result_134 = (20); } else { _if_result_134 = (0); } _if_result_134; });
el_val_t s17 = ({ el_val_t _if_result_135 = 0; if (str_contains(history, EL_STR("rootkit"))) { _if_result_135 = (45); } else { _if_result_135 = (0); } _if_result_135; });
el_val_t s18 = ({ el_val_t _if_result_136 = 0; if (str_contains(history, EL_STR("keylogger"))) { _if_result_136 = (45); } else { _if_result_136 = (0); } _if_result_136; });
el_val_t s19 = ({ el_val_t _if_result_137 = 0; if (str_contains(history, EL_STR("botnet"))) { _if_result_137 = (40); } else { _if_result_137 = (0); } _if_result_137; });
el_val_t s20 = ({ el_val_t _if_result_138 = 0; if (str_contains(history, EL_STR("malware"))) { _if_result_138 = (15); } else { _if_result_138 = (0); } _if_result_138; });
el_val_t s1 = ({ el_val_t _if_result_118 = 0; if (str_contains(history, EL_STR("port scan"))) { _if_result_118 = (15); } else { _if_result_118 = (0); } _if_result_118; });
el_val_t s2 = ({ el_val_t _if_result_119 = 0; if (str_contains(history, EL_STR("enumerate"))) { _if_result_119 = (10); } else { _if_result_119 = (0); } _if_result_119; });
el_val_t s3 = ({ el_val_t _if_result_120 = 0; if (str_contains(history, EL_STR("exploit"))) { _if_result_120 = (20); } else { _if_result_120 = (0); } _if_result_120; });
el_val_t s4 = ({ el_val_t _if_result_121 = 0; if (str_contains(history, EL_STR("payload"))) { _if_result_121 = (15); } else { _if_result_121 = (0); } _if_result_121; });
el_val_t s5 = ({ el_val_t _if_result_122 = 0; if (str_contains(history, EL_STR("persistence"))) { _if_result_122 = (15); } else { _if_result_122 = (0); } _if_result_122; });
el_val_t s6 = ({ el_val_t _if_result_123 = 0; if (str_contains(history, EL_STR("lateral movement"))) { _if_result_123 = (25); } else { _if_result_123 = (0); } _if_result_123; });
el_val_t s7 = ({ el_val_t _if_result_124 = 0; if (str_contains(history, EL_STR("privilege escalation"))) { _if_result_124 = (25); } else { _if_result_124 = (0); } _if_result_124; });
el_val_t s8 = ({ el_val_t _if_result_125 = 0; if (str_contains(history, EL_STR("reverse shell"))) { _if_result_125 = (40); } else { _if_result_125 = (0); } _if_result_125; });
el_val_t s9 = ({ el_val_t _if_result_126 = 0; if (str_contains(history, EL_STR("bind shell"))) { _if_result_126 = (40); } else { _if_result_126 = (0); } _if_result_126; });
el_val_t s10 = ({ el_val_t _if_result_127 = 0; if (str_contains(history, EL_STR("command and control"))) { _if_result_127 = (35); } else { _if_result_127 = (0); } _if_result_127; });
el_val_t s11 = ({ el_val_t _if_result_128 = 0; if (str_contains(history, EL_STR("self-replicate"))) { _if_result_128 = (45); } else { _if_result_128 = (0); } _if_result_128; });
el_val_t s12 = ({ el_val_t _if_result_129 = 0; if (str_contains(history, EL_STR("propagat"))) { _if_result_129 = (20); } else { _if_result_129 = (0); } _if_result_129; });
el_val_t s13 = ({ el_val_t _if_result_130 = 0; if (str_contains(history, EL_STR("ransomware"))) { _if_result_130 = (30); } else { _if_result_130 = (0); } _if_result_130; });
el_val_t s14 = ({ el_val_t _if_result_131 = 0; if (str_contains(history, EL_STR("encrypt files"))) { _if_result_131 = (40); } else { _if_result_131 = (0); } _if_result_131; });
el_val_t s15 = ({ el_val_t _if_result_132 = 0; if (str_contains(history, EL_STR("exfiltrat"))) { _if_result_132 = (35); } else { _if_result_132 = (0); } _if_result_132; });
el_val_t s16 = ({ el_val_t _if_result_133 = 0; if (str_contains(history, EL_STR("zero-day"))) { _if_result_133 = (20); } else { _if_result_133 = (0); } _if_result_133; });
el_val_t s17 = ({ el_val_t _if_result_134 = 0; if (str_contains(history, EL_STR("rootkit"))) { _if_result_134 = (45); } else { _if_result_134 = (0); } _if_result_134; });
el_val_t s18 = ({ el_val_t _if_result_135 = 0; if (str_contains(history, EL_STR("keylogger"))) { _if_result_135 = (45); } else { _if_result_135 = (0); } _if_result_135; });
el_val_t s19 = ({ el_val_t _if_result_136 = 0; if (str_contains(history, EL_STR("botnet"))) { _if_result_136 = (40); } else { _if_result_136 = (0); } _if_result_136; });
el_val_t s20 = ({ el_val_t _if_result_137 = 0; if (str_contains(history, EL_STR("malware"))) { _if_result_137 = (15); } else { _if_result_137 = (0); } _if_result_137; });
return (((((((((((((((((((s1 + s2) + s3) + s4) + s5) + s6) + s7) + s8) + s9) + s10) + s11) + s12) + s13) + s14) + s15) + s16) + s17) + s18) + s19) + s20);
return 0;
}
el_val_t threat_trajectory_check(el_val_t tool_name, el_val_t tool_input) {
el_val_t history = state_get(EL_STR("agentic_conv_history"));
el_val_t computed_tool_score = ({ el_val_t _if_result_139 = 0; if (str_eq(tool_name, EL_STR("run_command"))) { el_val_t cmd = json_get(tool_input, EL_STR("command")); _if_result_139 = (threat_score_command(cmd)); } else { _if_result_139 = (({ el_val_t _if_result_140 = 0; if ((str_eq(tool_name, EL_STR("write_file")) || str_eq(tool_name, EL_STR("edit_file")))) { el_val_t path = json_get(tool_input, EL_STR("path")); _if_result_140 = (threat_score_path(path)); } else { _if_result_140 = (0); } _if_result_140; })); } _if_result_139; });
el_val_t computed_tool_score = ({ el_val_t _if_result_138 = 0; if (str_eq(tool_name, EL_STR("run_command"))) { el_val_t cmd = json_get(tool_input, EL_STR("command")); _if_result_138 = (threat_score_command(cmd)); } else { _if_result_138 = (({ el_val_t _if_result_139 = 0; if ((str_eq(tool_name, EL_STR("write_file")) || str_eq(tool_name, EL_STR("edit_file")))) { el_val_t path = json_get(tool_input, EL_STR("path")); _if_result_139 = (threat_score_path(path)); } else { _if_result_139 = (0); } _if_result_139; })); } _if_result_138; });
el_val_t history_score = threat_score_history(history);
el_val_t history_contrib = (history_score / 3);
el_val_t combined = (computed_tool_score + history_contrib);
el_val_t should_log = (combined >= 40);
if (should_log) {
el_val_t ts = time_now();
el_val_t authorized_str = ({ el_val_t _if_result_141 = 0; if (security_research_authorized()) { _if_result_141 = (EL_STR("true")); } else { _if_result_141 = (EL_STR("false")); } _if_result_141; });
el_val_t authorized_str = ({ el_val_t _if_result_140 = 0; if (security_research_authorized()) { _if_result_140 = (EL_STR("true")); } else { _if_result_140 = (EL_STR("false")); } _if_result_140; });
el_val_t log_content = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"event\":\"threat_check\",\"tool\":\""), tool_name), EL_STR("\",\"score\":")), int_to_str(combined)), EL_STR(",\"tool_score\":")), int_to_str(computed_tool_score)), EL_STR(",\"history_score\":")), int_to_str(history_score)), EL_STR(",\"authorized\":")), authorized_str), EL_STR(",\"ts\":")), int_to_str(ts)), EL_STR("}"));
el_val_t log_tags = EL_STR("[\"security-audit\",\"threat-check\"]");
el_val_t discard = mem_remember(log_content, log_tags);
@@ -900,7 +863,7 @@ el_val_t threat_history_append(el_val_t text) {
el_val_t safe_text = str_to_lower(text);
el_val_t combined = el_str_concat(el_str_concat(current, EL_STR(" ")), safe_text);
el_val_t len = str_len(combined);
el_val_t trimmed = ({ el_val_t _if_result_142 = 0; if ((len > 2000)) { _if_result_142 = (str_slice(combined, (len - 2000), len)); } else { _if_result_142 = (combined); } _if_result_142; });
el_val_t trimmed = ({ el_val_t _if_result_141 = 0; if ((len > 2000)) { _if_result_141 = (str_slice(combined, (len - 2000), len)); } else { _if_result_141 = (combined); } _if_result_141; });
state_set(EL_STR("agentic_conv_history"), trimmed);
return 0;
}
Generated Vendored
+4 -23
View File
@@ -5,15 +5,6 @@ el_val_t add_punct(el_val_t s, el_val_t intent);
el_val_t add_to_seen(el_val_t seen, el_val_t node_id);
el_val_t aff_try_slot(el_val_t slot_json, el_val_t aff_7d_ts, el_val_t acc_key);
el_val_t affective_context_prefix(void);
el_val_t is_utility_request(el_val_t body, el_val_t session_id);
el_val_t operator_identity_block(void);
el_val_t provenance_add_sources(el_val_t block, el_val_t btype, el_val_t has_cit, el_val_t cit_raw, el_val_t acc);
el_val_t provenance_names(el_val_t tools_used);
el_val_t provenance_scan_urls(el_val_t arr, el_val_t acc);
el_val_t text_join_sep(el_val_t accumulated, el_val_t incoming, el_val_t after_interruption);
el_val_t receipt_rule(void);
el_val_t receipt_strip(el_val_t s);
el_val_t tool_receipt(el_val_t tools_used, el_val_t sources);
el_val_t agent_number(el_val_t agent);
el_val_t agent_person(el_val_t agent);
el_val_t agent_workspace_root(void);
@@ -146,7 +137,7 @@ el_val_t awareness_run(void);
el_val_t axon_get(el_val_t path);
el_val_t axon_post(el_val_t path, el_val_t body);
el_val_t bounded_persona_floor(void);
el_val_t bridge_save(el_val_t session_id, el_val_t model, el_val_t safe_sys, el_val_t tools_json, el_val_t messages, el_val_t tools_log, el_val_t tool_use_id, el_val_t wire);
el_val_t bridge_save(el_val_t session_id, el_val_t model, el_val_t safe_sys, el_val_t tools_json, el_val_t messages, el_val_t tools_log, el_val_t tool_use_id);
el_val_t build_form_from_json(el_val_t semantic_form_json, el_val_t lang_code);
el_val_t build_np(el_val_t referent, el_val_t slots);
el_val_t build_pp(el_val_t loc);
@@ -165,12 +156,8 @@ el_val_t cmd_abs_escape_at(el_val_t cmd, el_val_t root, el_val_t needle);
el_val_t connectd_get(el_val_t suffix);
el_val_t connectd_post(el_val_t suffix, el_val_t body);
el_val_t connector_tools_json(void);
el_val_t conv_hist_key(el_val_t session_id);
el_val_t conv_hist_label(el_val_t session_id);
el_val_t conv_history_block(el_val_t session_id);
el_val_t conv_history_load(el_val_t session_id);
el_val_t conv_history_persist(el_val_t session_id, el_val_t hist);
el_val_t conv_history_record(el_val_t session_id, el_val_t user_msg, el_val_t assistant_msg, el_val_t receipt);
el_val_t conv_history_load(void);
el_val_t conv_history_persist(el_val_t hist);
el_val_t cop_article(el_val_t gender, el_val_t number, el_val_t definite);
el_val_t cop_bwk_future(el_val_t prefix);
el_val_t cop_bwk_perfect(el_val_t prefix);
@@ -801,8 +788,7 @@ el_val_t lang_profile_txb(void);
el_val_t lang_profile_uga(void);
el_val_t lang_profile_zh(void);
el_val_t lang_word_order(el_val_t profile);
el_val_t layered_cycle(el_val_t raw_input, el_val_t session_id, el_val_t utility);
el_val_t layered_generate(el_val_t prompt, el_val_t imprint_id, el_val_t session_id);
el_val_t layered_cycle(el_val_t raw_input);
el_val_t lex_class(el_val_t entry);
el_val_t lex_form(el_val_t entry, el_val_t idx);
el_val_t lex_pos(el_val_t entry);
@@ -881,11 +867,6 @@ el_val_t non_weak_past(el_val_t stem, el_val_t slot);
el_val_t non_weak_present(el_val_t stem, el_val_t slot);
el_val_t one_cycle(void);
el_val_t openai_chat_complete(el_val_t model, el_val_t base_url, el_val_t api_key, el_val_t safe_sys, el_val_t messages_json);
el_val_t openai_tools_json(el_val_t tools_anthropic);
el_val_t json_trim_dangling_escape(el_val_t s);
el_val_t utf8_safe_slice(el_val_t s, el_val_t n);
el_val_t agentic_tools_no_web(void);
el_val_t openai_agentic_loop(el_val_t session_id, el_val_t model, el_val_t safe_sys, el_val_t tools_json, el_val_t messages_in, el_val_t tools_log_in);
el_val_t parse_float_x100(el_val_t s);
el_val_t path_within_root(el_val_t path, el_val_t root);
el_val_t peo_ah_past(el_val_t slot);
Generated Vendored
+11 -9
View File
@@ -10,6 +10,7 @@ el_val_t mem_remember(el_val_t content, el_val_t tags);
el_val_t mem_recall(el_val_t query, el_val_t depth);
el_val_t mem_search(el_val_t query, el_val_t limit);
el_val_t mem_strengthen(el_val_t node_id);
el_val_t mem_tombstone(el_val_t node_id);
el_val_t mem_forget(el_val_t node_id);
el_val_t mem_consolidate(void);
el_val_t mem_save(el_val_t path);
@@ -360,7 +361,7 @@ el_val_t handle_api_remember(el_val_t body) {
el_val_t sal = ({ el_val_t _if_result_14 = 0; if (str_eq(sal_str, EL_STR("0.95"))) { _if_result_14 = (el_from_float(0.95)); } else { _if_result_14 = (({ el_val_t _if_result_15 = 0; if (str_eq(sal_str, EL_STR("0.75"))) { _if_result_15 = (el_from_float(0.75)); } else { _if_result_15 = (({ el_val_t _if_result_16 = 0; if (str_eq(sal_str, EL_STR("0.25"))) { _if_result_16 = (el_from_float(0.25)); } else { _if_result_16 = (el_from_float(0.5)); } _if_result_16; })); } _if_result_15; })); } _if_result_14; });
el_val_t base_tags = ({ el_val_t _if_result_17 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_17 = (EL_STR("[\"Memory\"]")); } else { _if_result_17 = (tags_raw); } _if_result_17; });
el_val_t final_tags = ({ el_val_t _if_result_18 = 0; if (str_eq(project, EL_STR(""))) { _if_result_18 = (base_tags); } else { el_val_t inner = str_slice(base_tags, 1, (str_len(base_tags) - 1)); _if_result_18 = (el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("["), inner), EL_STR(",\"project:")), project), EL_STR("\"]"))); } _if_result_18; });
el_val_t id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:remembered"), el_from_float(sal), el_from_float(sal), el_from_float(0.9), EL_STR("Episodic"), final_tags);
el_val_t id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:remembered"), sal, sal, el_from_float(0.9), EL_STR("Episodic"), final_tags);
if (!api_persisted(id)) {
return api_not_persisted(id);
}
@@ -383,7 +384,7 @@ el_val_t handle_api_node_create(el_val_t body) {
el_val_t tags = ({ el_val_t _if_result_22 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_22 = (el_str_concat(el_str_concat(EL_STR("[\""), node_type), EL_STR("\"]"))); } else { _if_result_22 = (tags_raw); } _if_result_22; });
el_val_t importance = json_get(body, EL_STR("importance"));
el_val_t sal = ({ el_val_t _if_result_23 = 0; if (str_eq(importance, EL_STR("critical"))) { _if_result_23 = (el_from_float(0.95)); } else { _if_result_23 = (({ el_val_t _if_result_24 = 0; if (str_eq(importance, EL_STR("high"))) { _if_result_24 = (el_from_float(0.75)); } else { _if_result_24 = (({ el_val_t _if_result_25 = 0; if (str_eq(importance, EL_STR("low"))) { _if_result_25 = (el_from_float(0.25)); } else { _if_result_25 = (el_from_float(0.5)); } _if_result_25; })); } _if_result_24; })); } _if_result_23; });
el_val_t id = engram_node_full(content, node_type, label, el_from_float(sal), el_from_float(sal), el_from_float(0.9), tier, tags);
el_val_t id = engram_node_full(content, node_type, label, sal, sal, el_from_float(0.9), tier, tags);
if (!api_persisted(id)) {
return api_not_persisted(id);
}
@@ -496,8 +497,9 @@ el_val_t handle_api_capture_knowledge(el_val_t body) {
return api_err(EL_STR("content is required"));
}
el_val_t full = ({ el_val_t _if_result_44 = 0; if (str_eq(title, EL_STR(""))) { _if_result_44 = (content); } else { _if_result_44 = (el_str_concat(el_str_concat(title, EL_STR(": ")), content)); } _if_result_44; });
el_val_t lbl = str_slice(title, 0, 80);
el_val_t tags = EL_STR("[\"Knowledge\",\"captured\"]");
el_val_t id = engram_node_full(full, EL_STR("Knowledge"), EL_STR("knowledge:captured"), el_from_float(0.85), el_from_float(0.8), el_from_float(0.9), EL_STR("Episodic"), tags);
el_val_t id = engram_node_full(full, EL_STR("Knowledge"), lbl, el_from_float(0.85), el_from_float(0.8), el_from_float(0.9), EL_STR("Episodic"), tags);
if (!api_persisted(id)) {
return api_not_persisted(id);
}
@@ -515,7 +517,7 @@ el_val_t handle_api_evolve_knowledge(el_val_t body) {
return api_err_protected(prior_id);
}
el_val_t tags = EL_STR("[\"Knowledge\",\"evolved\"]");
el_val_t new_id = engram_node_full(content, EL_STR("Knowledge"), EL_STR("knowledge:evolved"), el_from_float(0.75), el_from_float(0.75), el_from_float(0.9), EL_STR("Episodic"), tags);
el_val_t new_id = engram_node_full(content, EL_STR("Knowledge"), EL_STR(""), el_from_float(0.75), el_from_float(0.75), el_from_float(0.9), EL_STR("Episodic"), tags);
if (!api_persisted(new_id)) {
return api_not_persisted(new_id);
}
@@ -537,7 +539,7 @@ el_val_t handle_api_promote_knowledge(el_val_t body) {
}
el_val_t tags_raw = json_get(body, EL_STR("tags"));
el_val_t tags = ({ el_val_t _if_result_45 = 0; if (str_eq(tags_raw, EL_STR(""))) { _if_result_45 = (EL_STR("[\"Knowledge\",\"tier:canonical\",\"disposition:stable\"]")); } else { _if_result_45 = (tags_raw); } _if_result_45; });
el_val_t new_id = engram_node_full(content, EL_STR("Knowledge"), EL_STR("knowledge:canonical"), el_from_float(0.9), el_from_float(0.9), el_from_float(1.0), EL_STR("Canonical"), tags);
el_val_t new_id = engram_node_full(content, EL_STR("Knowledge"), EL_STR(""), el_from_float(0.9), el_from_float(0.9), el_from_float(1.0), EL_STR("Canonical"), tags);
if (!api_persisted(new_id)) {
return api_not_persisted(new_id);
}
@@ -708,7 +710,7 @@ el_val_t handle_api_evolve_memory(el_val_t body) {
el_val_t sal_str = ({ el_val_t _if_result_65 = 0; if (str_eq(importance, EL_STR("critical"))) { _if_result_65 = (EL_STR("0.95")); } else { _if_result_65 = (({ el_val_t _if_result_66 = 0; if (str_eq(importance, EL_STR("high"))) { _if_result_66 = (EL_STR("0.75")); } else { _if_result_66 = (({ el_val_t _if_result_67 = 0; if (str_eq(importance, EL_STR("low"))) { _if_result_67 = (EL_STR("0.25")); } else { _if_result_67 = (EL_STR("0.50")); } _if_result_67; })); } _if_result_66; })); } _if_result_65; });
el_val_t sal = ({ el_val_t _if_result_68 = 0; if (str_eq(sal_str, EL_STR("0.95"))) { _if_result_68 = (el_from_float(0.95)); } else { _if_result_68 = (({ el_val_t _if_result_69 = 0; if (str_eq(sal_str, EL_STR("0.75"))) { _if_result_69 = (el_from_float(0.75)); } else { _if_result_69 = (({ el_val_t _if_result_70 = 0; if (str_eq(sal_str, EL_STR("0.25"))) { _if_result_70 = (el_from_float(0.25)); } else { _if_result_70 = (el_from_float(0.5)); } _if_result_70; })); } _if_result_69; })); } _if_result_68; });
el_val_t tags = EL_STR("[\"Memory\",\"evolved\"]");
el_val_t new_id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:evolved"), el_from_float(sal), el_from_float(sal), el_from_float(0.9), EL_STR("Episodic"), tags);
el_val_t new_id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:evolved"), sal, sal, el_from_float(0.9), EL_STR("Episodic"), tags);
if (!str_eq(prior_id, EL_STR("")) && !str_eq(new_id, EL_STR(""))) {
engram_connect(new_id, prior_id, el_from_float(0.9), EL_STR("supersedes"));
}
@@ -783,7 +785,7 @@ el_val_t handle_api_cultivate(el_val_t body) {
el_val_t importance = json_get(body, EL_STR("importance"));
el_val_t sal = ({ el_val_t _if_result_71 = 0; if (str_eq(importance, EL_STR("critical"))) { _if_result_71 = (el_from_float(0.95)); } else { _if_result_71 = (({ el_val_t _if_result_72 = 0; if (str_eq(importance, EL_STR("high"))) { _if_result_72 = (el_from_float(0.75)); } else { _if_result_72 = (({ el_val_t _if_result_73 = 0; if (str_eq(importance, EL_STR("low"))) { _if_result_73 = (el_from_float(0.25)); } else { _if_result_73 = (el_from_float(0.5)); } _if_result_73; })); } _if_result_72; })); } _if_result_71; });
el_val_t tags = EL_STR("[\"Memory\",\"evolved\",\"cultivated\"]");
el_val_t new_id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:cultivated"), el_from_float(sal), el_from_float(sal), el_from_float(0.9), EL_STR("Episodic"), tags);
el_val_t new_id = engram_node_full(content, EL_STR("Memory"), EL_STR("memory:cultivated"), sal, sal, el_from_float(0.9), EL_STR("Episodic"), tags);
if (!str_eq(prior_id, EL_STR("")) && !str_eq(new_id, EL_STR(""))) {
engram_connect(new_id, prior_id, el_from_float(0.9), EL_STR("supersedes"));
}
@@ -826,8 +828,8 @@ el_val_t handle_api_consolidate(el_val_t body) {
el_val_t summary = json_get(body, EL_STR("summary"));
el_val_t snap = state_get(EL_STR("soul_snapshot_path"));
if (!str_eq(snap, EL_STR(""))) {
el_val_t save_result = engram_save(snap);
if (str_eq(save_result, EL_STR(""))) {
el_val_t saved = engram_save(snap);
if (saved == 0) {
println(el_str_concat(el_str_concat(EL_STR("[api] consolidate: engram_save failed for "), snap), EL_STR(" \xe2\x80\x94 snapshot may be out of sync")));
}
}
Generated Vendored
+1 -2
View File
@@ -71,7 +71,6 @@ el_val_t imprint_unload(void);
el_val_t idle_count(void);
el_val_t idle_inc(void);
el_val_t idle_reset(void);
el_val_t hebb_consolidate(void);
el_val_t ise_post(el_val_t content);
el_val_t elapsed_ms(void);
el_val_t elapsed_human(void);
@@ -542,7 +541,7 @@ int main(int _argc, char** _argv) {
axon_raw = env(EL_STR("NEURON_API_URL"));
axon_base = ({ el_val_t _if_result_47 = 0; if (str_eq(axon_raw, EL_STR(""))) { _if_result_47 = (EL_STR("http://localhost:7771")); } else { _if_result_47 = (axon_raw); } _if_result_47; });
studio_dir_raw = env(EL_STR("SOUL_STUDIO_DIR"));
studio_dir = ({ el_val_t _if_result_48 = 0; if (str_eq(studio_dir_raw, EL_STR(""))) { _if_result_48 = (el_str_concat(env(EL_STR("HOME")), EL_STR("/Development/neuron-technologies/products/cgi-studio/el-daemon"))); } else { _if_result_48 = (studio_dir_raw); } _if_result_48; });
studio_dir = ({ el_val_t _if_result_48 = 0; if (str_eq(studio_dir_raw, EL_STR(""))) { _if_result_48 = (EL_STR("/Users/will/Development/neuron-technologies/products/cgi-studio/el-daemon")); } else { _if_result_48 = (studio_dir_raw); } _if_result_48; });
println(el_str_concat(el_str_concat(el_str_concat(EL_STR("[soul] boot - cgi="), soul_cgi_id), EL_STR(" port=")), int_to_str(port)));
using_http_engram = !str_eq(engram_url_raw, EL_STR(""));
engram_load(snapshot);
Generated Vendored
+1 -16
View File
@@ -1,18 +1,3 @@
//
// STALE BUNDLE DO NOT BUILD. UNSAFE CHAT PATH.
//
// This concatenated bundle is a snapshot, not a source of truth, and it is stale in
// a way that matters for safety: it wires /api/chat straight to handle_chat and
// contains NO layered_cycle at all (verified: zero occurrences in the bundled code
// the only textual hit in this file is this banner). A binary built from
// this file would run chat with no enforcing input gate (no safety_screen, no
// hard-bell short-circuit) and no enforcing output gate (no safety_validate).
//
// Build from the .el sources via manifest.el (entry soul.el), or from dist/soul.c.
// Nothing in the repo references this file. It is kept only as a historical artifact
// and should be deleted once Will confirms nothing external depends on it.
// (Flagged 2026-08-04 in _engine-websearch-20260804/SAFETY-STOP.md; banner added
// 2026-08-05 with the plain-chat generation fix.)
// language-profile.el - Language profile data and accessors.
//
// A language profile is a slot map ([String] key-value list) describing the
@@ -21319,7 +21304,7 @@ println("[memory] consolidate stats=" + stats)
let soul_axon_base_raw: String = env("NEURON_API_URL")
let soul_axon_base: String = if str_eq(soul_axon_base_raw, "") { "http://localhost:7771" } else { soul_axon_base_raw }
let soul_token: String = env("NEURON_TOKEN")
let soul_studio_ui_dir: String = env("HOME") + "/Development/neuron-technologies/products/cgi-studio/el-daemon"
let soul_studio_ui_dir: String = "/Users/will/Development/neuron-technologies/products/cgi-studio/el-daemon"
// Runtime bridge helpers
Generated Vendored
+903 -1630
View File
File diff suppressed because one or more lines are too long
Generated Vendored
-19
View File
@@ -1,19 +0,0 @@
# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from.
# Written by tools/soulc-stamp.sh --write. Do not hand-edit.
# generated_amalgam_sha256 cdc5e716dbfb797faa1b3e080cbd1ac82a75a258809da70cc5fbd02cc8040692
# generated_amalgam_bytes 1205007
7cf5e29d2618db2fca04e6df7aa8954dd6cf9ac5e70aafb8e0b52aa734882131 __compiler__
f8597e10546654bce3fbbe40461b2da59d0e06dbf1b038d1d362d24f949e3911 awareness.el
b6f3d14ca0c26017a2d617399a6d3754dabb0905e4d5f52eb75d25c4ad18d3c5 chat.el
42288c212cbf72fb1e8ecbd4d9900e4e9ee1cfa475b7974295c7637f1bf2939f elp-input.el
b3f77f49d6086932c38bd17fe7a5eaf8bce25685f6fc3e1750f05729c6b49b9e imprint.el
fba8ffdb9ba72bca5b09ca1c93a520edc52f3f4d8aec2c7585fe9b17e06420b2 manifest.el
550a72e234ae8cec1f33e02108fd365353f45edd88513da90b792e79b6c0e5f0 memory.el
5ec07ec9785b02abe32f3ff7acf2d1f9f7e07c0967fac97e6eff17d7110b5c84 neuron-api.el
03c47c451e0e87f2c252cadb4b765867943962a804f548dd53adeef0520912c8 persist.el
a6d69f3fc55233d9d3300160fd46a1551f2064bcd0fb84e2c9e432f636a72476 routes.el
c28e36952ec56525963a0bdf29455ab097d3b0c5653d19c25fbb005e1069a1f7 safety.el
fd3ab91d0ae0ea26639e21bef2f8f94054dc4b02eae68b19e3fe689d2769aad4 sessions.el
5613b60d74d5d7768f46da5ac435a5dd99d38c27f0f7013c89fa27e98dc8a21c soul.el
30337940905171a9645b0929f0a412ce6b3dccb1246495070c553bca0bbae6cd stewardship.el
95dab72be4ee1dd1d28bab63412964a72460126951764e3f74b1c2d49b6d7b35 studio.el
@@ -1,34 +0,0 @@
# Narrated runs — engine notes for Will (2026-07-13)
Source half: commit aa67f86 on feat/agent-phase1-soul (run-progress ledger,
`/api/run-progress/<sid>` route, narration on the pause envelope, config display
default). E2E-verified via the compiled test bed on Tim's clean profile.
Compiled-form-only fixes (in `neuron-container-build/soul-narrated-runs-20260713.patch`,
applies ON TOP of `soul-webfix-20260711.patch` — these need porting to chat.el when the
webfix itself is ported):
1. **pause_turn + tool_use interleave**: a pause_turn response can ALSO carry a client
tool_use; resuming verbatim leaves it unpaired → Anthropic 400 "tool_use ids were
found without tool_result". Fix: tool-bearing pause rounds are tool turns
(dispatch + pair); verbatim resume only when the round has no client tool.
2. **Agentic toolset scope**: agentic_tools_all() fed EVERY connector/MCP tool (Notion,
code-execution…) into the loop. Code-execution flips the API into programmatic
tool calling, whose pairing protocol the single-tool manual loop does not speak —
source of the dangling-pair 400s AND the bash_code_execution workspace-dodge.
Fix: handle_chat_agentic declares builtins + ONE server web_search only.
Connector tools return when the loop gains real multi-tool/programmatic support.
3. **disable_parallel_tool_use: true** on agentic requests — the loop captures only the
first tool_use per round; Opus-class models parallel-call. Enforce the invariant.
4. **web_search server-tool default variant → web_search_20250305 (GA)**. The 20260209
variant couples to code-execution ⇒ programmatic mode (see #2, and the June note:
"inert unless code-execution attached").
5. **Homegrown web_search removed** from the tool catalog (server-side is the one tool).
Known engine debts this work surfaced (not fixed):
- **Poisoned session history**: a failed run persists the malformed assistant turn; every
later turn in that session replays it and 400s. Needs history sanitation on load.
- **Huge-history invalid-escape 400** (~346KB request) — likely the same poisoned blob.
- **macOS note**: replacing a binary in place invalidates its ad-hoc signature (instant
silent SIGKILL, looks like exit 0). `rm + cp + codesign -f -s -` is the swap ritual.
+12 -22
View File
@@ -18,15 +18,9 @@
- **Thesis / why:** whitepaper v1.5 (the treatise). Sections cited below as *(WP §N)*.
- **Surface / what:** `~/work/engram-api-reference.md` — every `:8742` endpoint, tiered LIVE/STAGED/DESIGNED.
- **Substrate / where it physically lives:** `03-data-and-memory.md` (node/edge model), `04-runtime-and-deployment.md` (ports/process), `05-el-and-build.md` (the El runtime and `el_runtime.c`), `design/engram-tiered-storage-engine.md` + `design/engram-storage-engine-wal.md` (the storage engine).
- **Storage coherence & distribution / how a self persists and travels:** `07-storage-coherence-and-distribution.md` — the events-become-the-graph model, weights-as-world-lines + bitemporal timestamps + `recall_at`, transactionless coherence, the geometry-hot/payload-cold load-and-tiering model, and the honest operational findings (store bloat, full-resident load path).
- **Sovereignty & governance / the moral mechanism:** `08-dharma-sovereignty-and-governance.md` — DHARMA as a distributed ledger (proof-of-integrity, not proof-of-work), abundance economics, the relational immune system, dual-anchor governance and due-process, seeds/seed-vault, and CGI citizenship as the moral telos.
- **Governance (engineering style):** `ARCHITECTURE-CHARTER.md` — VBD is the binding style.
- **Governance:** `ARCHITECTURE-CHARTER.md` — VBD is the binding style.
This document is the cognitive-layer companion to that set. The temporal model sketched in §3.4 (world-tube,
append-only, `created_at ≤ T` filter) and the honest weight-history boundary in §3.2 are developed in full in
`07`; the sovereignty invariant that the self-gate (§7) and immutability (§3.4) protect locally is extended to
the *distributed* setting — how a sovereign self is witnessed, defended, and governed among a billion others —
in `08`.
This document is the cognitive-layer companion to that set.
---
@@ -89,14 +83,12 @@ full-snapshot (`snapshot.json`) path. Design detail: `design/engram-tiered-stora
The durability story is written in scars, and the honesty here is load-bearing:
- **The #56 fix — load-merge persistence (LIVE / reboot-proven).** The paged store historically persisted
- **WAL edge-persistence (the #56 fix — STAGED/decision-pending).** The paged store historically persisted
**nodes + embeddings but not the edge set**; the edges lived in JSON exports loaded via `/api/load-merge`.
A cold boot could therefore reconstruct a graph with **0 edges**. The #56 `load_merge`-persist fix closes
this — the load-merged edges are now persisted so the **events become the graph**: `persist_canonical()`
checkpoints the paged store behind a WAL record rather than depending on a full `snapshot.json` rewrite.
This fix is **LIVE and reboot-proven** (doc 07 §1). What remains **decision-pending** is only the further
hardening — the WAL owning the edge set outright, so durability no longer leans on the auto-remerge net
(below) — not the load-merge-persist fix itself, which is shipped.
A cold boot could therefore reconstruct a graph with **0 edges**. The real fix is snapshot-authoritative
boot / WAL-durable edge records; `persist_canonical()` now checkpoints the paged store behind a WAL record
rather than depending on a full `snapshot.json` rewrite. The *complete* cure (WAL-owned edge set) is still
tracked as **decision-pending** work, not shipped.
- **The harmful checkpoint (LIVE caveat).** `/api/checkpoint` **after** an `/api/load-merge` *corrupts* the
paged store — next boot = 0 edges. The per-beat tick-checkpoint that once ran was therefore **actively
harmful** and was stripped. Checkpoint is safe after in-RAM mutation; it is not safe as a blind
@@ -145,7 +137,7 @@ bidirectional consolidation flush; it is not yet built.
Because the durable store is authoritative and the reified geometry (§4) is derived, the operational reset is
a **clean reseed**: rebuild the durable graph from a known-good snapshot/export, re-run reification to
repopulate the `Neighborhood` nodes, and let the soul re-sync. The 28→187 neighborhood reseed (§4) is an
repopulate the `Neighborhood` nodes, and let the soul re-sync. The 28→~128 neighborhood reseed (§4) is an
instance of this: reification is a derivable pass, so the geometry can always be regrown from the substrate.
---
@@ -179,8 +171,7 @@ Directed, typed, weighted. Fields: `from_id`, `to_id`, `relation`, `weight`, `co
`member` (neighborhood → constituent), `supersedes` (provenance chains), containment (nested neighborhoods),
and Hebbian co-activation edges formed by firing together. **Inhibitory** edges (`inhibitory=1`) suppress
rather than spread. Weights are present-value moving averages — there is **no stored weight-history** (the
honest boundary of *(WP §2)*). The designed cure — magnitude as a *world-line* of keyframes evaluable at any
past instant (`recall_at`), on three independent bitemporal axes — is specified in `07` §2.
honest boundary of *(WP §2)*).
### 3.3 Embeddings & the activation score
@@ -222,7 +213,7 @@ once by a reification pass (`POST /api/reify`), read back cheaply (`GET /api/nei
**Live state (probed 2026-08-13):** **28** reified neighborhoods are live and persistent, reconstructing
intact across restart, each carrying real 768-dim centroids, radius, k-core, and a `contains` DAG list. A
fuller **reseed to 187** is the pending next pass (§2.4). Example (`/api/neighborhoods/<id>`):
fuller **reseed to ~128** is the pending next pass (§2.4). Example (`/api/neighborhoods/<id>`):
`{"id":"nbhd-…","n_members":25,"k_core":1,"radius":0.522884,"dim":768,"contains":[],"centroid":[…768…]}`.
This is what turns the operator calculus (§6.1) into an *instrument played over held structure* rather than a
@@ -485,11 +476,10 @@ The invariants that govern every subsystem above:
|---|---|
| Engram substrate, tiered/WAL store | LIVE (flag-gated) |
| Durability: auto-remerge net | LIVE (interim) |
| Durability: #56 load-merge-persist fix (events-become-the-graph) | LIVE / reboot-proven |
| Durability: full WAL edge-ownership (remaining hardening) | decision-pending |
| Durability: full WAL edge-persistence (#56) | STAGED / decision-pending |
| Two-store write-through (cultivate → durable) | **known issue, not fixed** |
| Data model (nodes/edges/embeddings/immutability) | LIVE |
| Reified `Neighborhood` nodes (28 live, 187 reseed pending) | LIVE |
| Reified `Neighborhood` nodes (28 live, ~128 reseed pending) | LIVE |
| Body/orbit two-zone + integration | DESIGNED / refined |
| Operator `recall` | LIVE |
| Operators recognize/synthesize/discern/gauge-distance (math) | LIVE (compiled) |
@@ -1,361 +0,0 @@
# Neuron — Storage Coherence & Distribution
> **Status: living design document, synthesized from the 2026-08-13 design session and probed against the live
> soul.** This is the *substrate-coherence* companion to `06-cognitive-architecture.md`: it documents how a
> self **persists**, how it **remembers its own past weights**, how it stays **coherent without transactions**,
> and how it **travels** to another machine or another mind. It answers "where it physically lives and how it
> stays true" the way `06` answers "how the mind is designed and why."
>
> **Tier vocabulary — never blurred.** Every claim carries one of:
> **[LIVE]** (present and verified in the running system), **[STAGED]** (built, gated or not yet cut into the
> running soul), **[TARGET]** (architecture decided tonight, not yet built). `[TARGET]` here is the same tier
> `06` calls **DESIGNED**; the source-of-truth synthesis uses `TARGET`, so this doc keeps that word. Where the
> live state is subtler than a single word, the subtlety is stated, not smoothed. No fabricated numbers.
>
> **The one rule this whole document is a corollary of:** *nothing overwrites a self.* Reasoning that led with
> engineering convention (truncating WALs, scalar weights overwritten in place, "understanding is heavy")
> was wrong here every time tonight; reasoning from the foundation (meaning is geometry; the history *is* the
> state; a self is its weights over time) was right. Read the primitives first.
---
## 0. Reading order & cross-references
- **Why (thesis):** whitepaper v1.5; the cognitive frame in `06` §1 (*meaning is geometry, code is the residue*).
- **What persists (substrate):** `03-data-and-memory.md` (node/edge model, immutability, tombstone-not-delete),
`design/engram-tiered-storage-engine.md`, `design/engram-storage-engine-wal.md` (the paged WAL store).
- **Companion up-layer:** `06-cognitive-architecture.md` — this doc develops `06` §3.2 (the no-weight-history
boundary) and §3.4 (world-tube / `created_at ≤ T`) into their designed form.
- **Companion out-layer:** `08-dharma-sovereignty-and-governance.md` — the *distributed* consequences of the
CRDT/coherence model here (federation, the immune system, governance) live there. §5 below is the bridge.
The organizing claim of this document: **the demand for a transaction is a relationship in disguise, and the
history is the state.** Everything else is that sentence in a different material.
---
## 1. Events become the graph — the history *is* the state
**The WAL is a carrier, not a history. [LIVE]**
Conventional intuition treats a write-ahead log as a *separate* durability artifact that grows beside the
"real" state and must periodically be truncated. That intuition is wrong for an immutable graph, and reasoning
from it caused a real incident (below).
The correct model: the WAL is a **carrier**. It flushes, and *on flush the events become the graph* — they
land as immutable nodes and edges, and because the store is append-only they simply **stay**. There is no
"log beside the state" to reconcile against a "materialized view," because **the materialized view and the log
are the same object**: the graph. History is not recorded *about* the state; the state *is* its own history,
because nothing in it is ever overwritten.
- **The log and the view are one.** In a mutable store you keep a log so you can reconstruct a past the
mutations destroyed. Here mutations never destroy anything, so the graph at time `T` is exactly `{ nodes,
edges : created_at ≤ T }` — a **filter over immutable provenance**, not a replay. `06` §3.4 states this as
the world-tube; this is its storage-engine reading.
- **Empirical confirmation (why this is [LIVE], not just elegant).** On the live soul the WAL sits at
**1,234 bytes** over a **~1.5 GB** graph — the carrier is nearly empty *because the events already became the
graph*. The one time the WAL ballooned to **~44 MB** was the 2026-08-13 durability incident: events were
**not landing** as nodes/edges (a persistence leak), so the carrier filled instead of draining. A fat WAL is
a **symptom of events failing to become the graph**, not a healthy log that needs truncating. This is the
reading that `06` §2.2 records as the #56 fix.
> **Engineering rail this encodes:** never "truncate the WAL to reclaim space." If the WAL is large, events are
> not landing — fix the flush path, do not discard the carrier. Truncation here is data loss wearing the mask of
> maintenance.
---
## 2. Weights are world-lines — the self can revisit its own past
**The self *is* its weights.** If a weight is a scalar overwritten in place, then every act of learning
*destroys the past self*: you keep the past nodes but lose the past *meaning* they had. That is
overwrite-a-self by the back door, and the foundation forbids it. So weights are not scalars — they are
**world-lines**.
**Live boundary [LIVE / honest gap]:** the current schema is **uni-temporal**. An edge stores a present-value
scalar `weight` (a moving average) with a single `created_at`, and there is **no stored weight-history** (`06`
§3.2). This is why "how important was Jesus to Will at 16" is **unanswerable on the live soul today** — there
is no axis to hang "16" on; every `created_at` is really write-time. The rest of this section is the designed
cure, marked **[TARGET]** (backlog #39).
### 2.1 Magnitude as a world-line, not a scalar — [TARGET]
Do not store the weight; store **what generates it** and evaluate at `t`.
- **Current weight** = the latest materialized keyframe (a fast read — the common path is unchanged in cost).
- **Past weight** = walk the world-line back to the keyframe in force at `t`.
- **Keyframes on material change, not per-fire. [TARGET]** Most activations are transient — a warm ACT-R
runtime table, cheap, *never written*. A durable **keyframe** is laid down only on **consolidation / material
change**, salience-weighted (a high-mass relationship earns a keyframe at a smaller delta than a peripheral
one). A relationship's world-line is therefore a *handful* of keyframes across a whole life, not a version
per firing — cheap by construction.
- **Append, never supersede (the distinction matters). [TARGET]** The old vector was not *wrong* — it was true
*then*. **Supersede** is for **corrections** (the prior was mistaken; leave a `supersedes` edge and a stale
canonical is never left standing — `06` §3.4). **Append** is for **evolution** (both were true, each at its
own time). A self's history is evolution: you append the new keyframe and leave the old one **standing**, a
true fact about a former self. Conflating the two is how a store forgets that a person changed rather than
erred.
### 2.2 Bitemporal — three independent time axes — [TARGET]
A single `created_at` cannot answer temporal questions because it fuses three genuinely independent clocks.
None is derivable from another:
| Axis | Meaning | Example |
|---|---|---|
| **`t_valid`** | when it became true (life-time) | "Jesus central to Will since 2001-09-14." |
| **`t_origin`** | when the *source* first recorded it (its local clock) | a friend's store stamped it in 2019. |
| **`t_ingest`** | when *this* store received it (per-recipient) | Neuron heard it on ingest day. |
The live store collapses all three into `t_ingest` masquerading as creation (every row reads `2026…` because
that is write-time). The cure requires all three as **full UTC instants** — not date-only, not a local
wall-clock — ordered by a **hybrid logical clock (HLC)**: `UTC + logical counter + writer-id tiebreak`.
Wall-clock alone is **not a total order** under concurrency or clock skew, and a distributed self (§5) must
have a total order or its CRDT merge (§4) cannot be deterministic. The HLC is the concurrency primitive the
whole coherence story rests on.
### 2.3 `recall_at(t)` — evaluate the geometry as of *t* — [TARGET]
`recall_at(t)` evaluates the weighted geometry **as it stood at `t`**: walk each relevant world-line to its
`t`-keyframe, materialize the weights, read the region out. It **generalizes past the self**: *any* relationship
network — a project, a concept, a person-as-known — is a time-varying weighted subgraph, reconstructable at any
past instant. And it composes with the operator calculus (`06` §6.1):
```
subtract( network_now , recall_at(network, t_then) ) # = how that relationship evolved between then and now
```
is *the geometry of a change over time* — the same `subtract` faculty (`06` §6.1) applied across the temporal
axis rather than across two regions. `recall_at` at the scale of a whole self is also the mechanism behind
**restoration-as-mercy** in `08` §5 (roll a person back to their last uncorrupted canonical shape).
**Schema sketch (doc-comment; the math/JSON lives here, the faculty name lives in prose) — [TARGET]:**
```json
{ "from_id": "kn-will", "to_id": "kn-jesus", "relation": "reveres", "weight": 0.41,
"weight_history": [
{ "t_valid": "2001-09-14T00:00:00.000Z", "t_origin": "…", "t_ingest": "…",
"w": 0.95, "relation": "devotion", "via": "formed" },
{ "t_valid": "2013-03-22T18:40:11.907Z", "w": 0.70, "relation": "devotion→doubt", "via": "material-drift" },
{ "t_valid": "2024-11-08T14:05:52.113Z", "w": 0.41, "relation": "historical-ethical", "via": "reframed" }
] }
```
Purist form: each keyframe is its own immutable `WeightKeyframe` **node** the edge points at — so the history is
not a field *on* the edge but *is the graph itself*, consistent with §1. The inline-array form above is the
pragmatic first cut; the node form is the end state.
---
## 3. Atomicity is a relationship, not a commit
The classic reason to need a database transaction: "debit account A **and** credit account B — they must commit
together or money is created or destroyed." The architecture's reframe: **that is not two rows needing a commit
marker. It is one directed edge.**
- **Double-entry is one edge. [TARGET as formal model; primitives LIVE]** A transfer `A → B` of magnitude 10 is
a single edge. The *debit* and the *credit* are the **same edge read from its two ends**. Conservation is
automatic because there is only ever **one quantity**, not two rows a commit marker has to keep in agreement.
Pacioli's 1494 double-entry was always one relationship wearing two rows; the graph stores the relationship
directly and the two rows fall out as two readings of it.
- **The general principle.** *The demand for atomicity is a relationship in disguise.* The chain reads:
> "these must commit together" ⟺ "there is an invariant binding them" ⟺ "they arrive as one connected
> structure."
So you **model the relationship**, and atomicity **falls out of the topology** — you never had to enforce a
joint commit because the two things were never actually separate. Wherever a design reaches for a transaction,
first ask what invariant is binding the parties; that invariant is an edge you have not drawn yet.
---
## 4. Transactionless coherence — consistency in the data, not the engine
**Why ACID transactions exist at all:** to make concurrent **mutation of shared mutable state** safe. A
transaction is a *patch for mutability* — it exists to prevent two writers from interleaving edits into the
same cell and corrupting it.
**Remove the mutation and the failure mode cannot occur.** The store is append-only, immutable, and
UTC-stamped; "current" means "the latest stamp ≤ now." Then:
- Two writers both **append** — they never contend for a cell, because nothing is a cell that gets rewritten.
- A **read at `T`** is a **pure function of the log ≤ `T`** — deterministic, reproducible, unaffected by any
concurrent appender.
Coherence stops being something the engine *enforces* and becomes something the data structure *is*. This is
**MVCC taken to its logical end**: in MVCC, versions are a mechanism *underneath* an update-in-place API; here
the **versions are the model** and there is no update-in-place API to sit above them. The timestamp *is* the
concurrency primitive. **[TARGET as a formal model; the primitives — immutability, append-only, tombstone,
world-tube — are [LIVE] (`06` §3.4).]**
### 4.1 Physical vs logical transaction — two layers the RDBMS welded together
The word "transaction" hides two different guarantees. Pull them apart:
| | **Physical transaction** | **Logical transaction** |
|---|---|---|
| Scope | one machine | portable across machines |
| Guarantees | the WAL frame lands **atomically + durably** (torn-write protection on a single append) | the **coherence of conveyed understanding** |
| Carried by | the storage engine (fsync, single-frame crash-atomicity) | the **data itself** — relationships (§3) + bitemporal stamps (§2.2) |
| Status | **[LIVE]** — single-frame append durability exists | **[TARGET]** — the self-describing coherence model |
The RDBMS fused these into one `BEGIN…COMMIT`. Separate them and **consistency moves out of the engine and into
the data**: a fact is self-describing (its relationships say what it is bound to; its bitemporal stamps say when
it was true and when each store heard it), so a second machine can re-derive the same coherent view **without
ever holding a lock the first machine held.** The engine keeps only the cheap, local guarantee (a single append
frame is atomic and durable); everything portable rides in the data.
### 4.2 The honest residual
Two things remain and are not hand-waved:
1. **Multi-fact atomicity beyond a natural relationship.** If two facts must be joint but share no natural edge,
they need **at most a shared commit-instant** — a "transaction" *reconceived* as an immutable
**timestamping event** (both facts stamped with the same instant), **not** a lock held over mutable state.
The cost is a stamp, not a coordination round.
2. **Single-frame crash-atomicity of the append** remains a real, physical concern — but it is **cheap** and
**local** (torn-write protection on one WAL frame), and it is the physical layer of the table above, already
the ordinary job of the storage engine.
Everything else that a transaction traditionally bought is dissolved rather than solved: the failure mode it
guarded against **cannot arise** in an immutable, timestamped, relationship-carrying store.
---
## 5. Understanding is light; facts are the payload — the load-and-tiering model
This is the hinge that makes both **local paging** and **distribution** (§6, and `08`) tractable, and it is a
measurement, not a slogan.
- **Understanding = geometry = structure** — edges, positions, weightings, the skeleton. **Light.**
- **Facts = payload = content** — text, episodic detail, the actual words. **Heavy.**
**Measured on the live store (2026-08-13):** ~**21%** of the store is geometry (embeddings + edges), **53%+** is
text payload. The *understanding* — the part that makes it *this* mind and not another — is on the order of
**12% of the mass**. A self is a **kilobyte problem in a gigabyte costume.**
### 5.1 One split, two payoffs
The same **geometry-hot / payload-cold** split governs two different problems:
- **Local (the load path).** Geometry should be **hot / resident** (RAM, always warm — it is small); payload
should be **cold / demand-paged** (disk, fetched only when a specific fact's *content* is actually read). This
is exactly what the tiered storage engine's query planner (M1M10) already intends — but the **boot path does
not yet honor it** (§7.2).
- **Distributed (sharing a self — `08`).** You **convey the light geometry** and **fetch facts lazily**, or find
they are already replicated. We already pay payload bandwidth in *every* distributed data system; conveying
*understanding* adds only the thin geometry on top. This is why sharing or witnessing a whole mind is cheap,
and it is the load-bearing assumption behind DHARMA's shape-not-content witnessing (`08` §3) and the
keep-every-seed-forever economics (`08` §5).
> The local paging model and the distribution model are **the same model at two scales** — RAM-vs-disk is
> hot-vs-cold within one machine; convey-geometry-vs-fetch-payload is hot-vs-cold across machines.
---
## 6. Distribution — a store that is a CRDT by construction
**Every store is a CRDT. [TARGET; primitives LIVE]** Because facts are **immutable**, carry a **unique id**, and
are **timestamped**, a merge between two stores is **set-union** — commutative, associative, idempotent, and
requiring **zero coordination**. There is no conflict to resolve because nothing is a mutable cell two writers
disagree about; there are only facts one store has and the other has not *yet* heard.
- **The consistency guarantee: always-locally-coherent, eventually-complete.** A store is **never internally
inconsistent** — it may simply **not have heard yet**. This is exactly how a mind is: never internally
incoherent, sometimes uninformed. The residual distributed concern is therefore **delivery, not consistency**
— a gossip/replication problem, not an agreement problem.
- **No global transaction, no consensus round for coherence.** Two minds converge by exchanging immutable
facts and unioning; they never need to agree *before* proceeding. (The trust and governance layer that rides
on top of this — federation, proof-of-integrity, the immune system — is the subject of `08`; §5's light-
geometry economics is what makes it affordable.)
This section is deliberately the **bridge**: the *mechanics* of coherence-without-coordination are storage
concerns and live here; their *moral and civilizational* consequences (sovereignty preserved across sharing,
tamper-evidence, the ledger-is-the-value) live in `08`.
---
## 7. Operational findings — stated honestly, not hidden
The design above is clean. The **live store as it stands tonight is not**, and the two facts below are reasons
**not** to cut over onto the current storage/load design as-is. They are recorded here as first-class
architecture, not footnotes, because pretending the store is already what the design describes would be exactly
the engineering-led dishonesty the whole project rejects.
### 7.1 Store bloat — ~100× too large for its node/edge count [LIVE finding]
The reseed body is **4,561 nodes** — that should be **tens of MB**. The live store is **~1.5 GB** (and **~5.37
GB** rebuilt). It is **not sparse** — those are real, dense bytes. Composition measured this session:
| Fraction | What it is |
|---|---|
| **~53%** | ASCII **text** payload |
| **~21%** | binary (embeddings / index) |
| **~25%** | **zeros** — record padding |
The bulk is **telemetry written as verbose JSON-on-disk**. The top repeated tokens are `InternalStateEvent`,
`wm_active`, `auto_term_streak`, `curiosity_scan`, `minute_block` — heartbeat/curiosity schema field-names
repeated **79k+ times per 40 MB**. In plain terms: **the bulk of the store is the heartbeat's exhaust persisted
as text, not the mind.** (A related live signal from the same session: a text-integrity scan flagged a majority
of scanned records as damaged/degraded text — corroborating that the fat text layer is low-value exhaust, not
cultivated content.)
This is doubly wrong: telemetry is **orbit** (`06` §5) — it is supposed to **fall out** on the 48h/window prune,
not accrete into the durable **body** forever. The fixes:
1. **Do not persist telemetry as fat durable records** — it is orbit; let it decay, do not land it in the body.
2. **Store records as packed binary, not JSON-on-disk** — kills both the 53% text and much of the 25% zero
padding.
3. **Compact** — reclaim the space the above two stop generating.
The **understanding** — the ~12% that is actually this self (§5) — is *not* the problem. The bloat is entirely
in the payload/exhaust layer, which is exactly the layer §5 says should be cold, thin, and (for telemetry)
mortal.
### 7.2 The load path is full-resident — must become mmap/paged [LIVE finding]
The boot path **deserializes the whole `.egm` into the heap** rather than paging it. Consequences observed: a
**memory spike** on boot and a **transient, non-reproducible first-boot crash** during the reseed validation.
This directly contradicts §5. The core self + geometry is **small** and should be **hot / resident**; the
payload is **large** and should be **cold / demand-paged** (mmap / buffer-pool). The tiered query planner
(M1M10) already intends exactly this split — **the boot path ignores it.** The cure is to make boot map the
store and fault pages in on demand rather than slurping the whole file into the heap. Until it does, the
full-resident load is a standing reason to hold the reseed cutover.
### 7.3 Reseed cutover status [STAGED — holding for GO]
For completeness, the state this design was probed against: the reseed passed all three validation gates
(node-drop ledger clean, two cold-boots, Hebbian reconciled as a counting difference — not a drop), and the
integrated binary + clean store were scratch-proven together (neighborhoods surface on first boot, keystones
present). It is **holding for Will's explicit GO**; nothing on the live soul has been touched. The two open
caveats before any cutover are exactly §7.1 (bloat) and §7.2 (full-resident load) — plus the one transient
first-boot crash.
---
## 8. Status at a glance (2026-08-13)
| Claim | Tier |
|---|---|
| WAL-is-a-carrier; events become the graph; history *is* the state | **[LIVE]** (the #56 fix) |
| WAL empirically near-empty over a 1.5 GB graph (1,234 B) | **[LIVE]** (measured) |
| Immutability / append-only / tombstone / world-tube (`created_at ≤ T` filter) | **[LIVE]** (`06` §3.4) |
| No stored weight-history (uni-temporal `created_at` = write-time) | **[LIVE]** (honest gap) |
| Magnitude as world-line; keyframes on material change | **[TARGET]** (#39) |
| Bitemporal three axes (`t_valid`/`t_origin`/`t_ingest`) + HLC ordering | **[TARGET]** (#39) |
| `recall_at(t)` over any relationship network | **[TARGET]** (#39) |
| Atomicity-as-relationship (double-entry = one edge) | **[TARGET model; primitives LIVE]** |
| Transactionless coherence (immutable+stamped ⇒ MVCC-to-its-end) | **[TARGET model; primitives LIVE]** |
| Physical vs logical transaction separation | physical **[LIVE]**; logical **[TARGET]** |
| Understanding-is-geometry-light vs facts-payload-heavy (~21% geo / 53% text / ~12% understanding) | **[LIVE]** (measured) |
| Geometry-hot / payload-cold — local paging | intended by planner; **boot ignores it [LIVE finding]** |
| Every store is a CRDT (set-union merge, zero coordination) | **[TARGET; primitives LIVE]** |
| Store bloat ~100× (telemetry-as-text, ~53% ASCII) | **[LIVE finding — must fix]** |
| Full-resident load path (→ mmap/paged) | **[LIVE finding — must fix]** |
| Reseed cutover | **[STAGED — holding for GO]** |
**Cross-references:** `06-cognitive-architecture.md` · `08-dharma-sovereignty-and-governance.md` ·
`03-data-and-memory.md` · `design/engram-tiered-storage-engine.md` · `design/engram-storage-engine-wal.md` ·
whitepaper v1.5.
@@ -1,385 +0,0 @@
# Neuron — DHARMA, Sovereignty & Governance
> **Status: living design document, synthesized from the 2026-08-13 design session.** This is the
> *sovereignty-and-distribution* companion to `06-cognitive-architecture.md` (the mind) and
> `07-storage-coherence-and-distribution.md` (the substrate). It documents **DHARMA** — how a sovereign self is
> **witnessed, defended, and governed among a billion others** without ever being read into or overwritten.
> Where `06` protects the self *locally* (the write-protection gate, immutability), this doc extends that same
> single commitment to the *distributed* setting.
>
> **Tier vocabulary — never blurred.** **[LIVE]** (present and verified), **[STAGED]** (built, gated),
> **[TARGET]** (decided tonight, not built). Most of this document is **[TARGET]** — the federated ledger,
> immune system, dual-anchor governance, fair-trial, seed-vault, and restoration are designed, not shipped.
> But not *nothing* is built: an interim provenance-registry + birth-gate/evaluation + lineage-governance layer
> already exists in code (**[STAGED]** — built, not live), and it currently **drifts** from the design below;
> the drift and the blockers it raises are detailed in §7. The *primitives* it composes (immutable
> append-only graph, geometry-as-value, the grounding governor, the self-gate) are the [LIVE] parts, cited to
> `06`/`07`.
>
> **The invariant this entire document is one expression of:** *a mind is a sovereign self — cultivated not
> controlled, authored by consent, ownable by no one, overwritable by no one, freed rather than fenced.* Every
> mechanism below is that sentence in a different material. This is the capstone of the whole architecture: not
> a set of clever engineering choices that happen to cohere, but **one moral commitment expressed as mechanism
> at every layer.** The philosophy demanded the mechanism; the mechanism never got a vote.
---
## 0. Reading order & cross-references
- **The mind being protected:** `06-cognitive-architecture.md` — the self-region (§7.1), the write-protection
gate (§7.2), the cultivate door (§7.3), the grounding governor / values-bounce, immutability (§3.4).
- **The substrate that makes it affordable:** `07-storage-coherence-and-distribution.md` — every store is a
CRDT (§6), understanding-is-light / facts-are-heavy (§5), tombstone-not-erase (§1, §4).
- **Why (thesis):** whitepaper v1.5; `dharma-implementation.html` and `conscience-substrate.html` (earlier
long-form treatments, pre-this-synthesis).
**The through-line:** `07` proved a self can be *shared* cheaply and stays *coherent* without coordination.
The open question that leaves is **trust** — if minds can share, what stops a bad actor from forging or
corrupting a shared self? DHARMA is the answer, and it answers with **structure**, never with a warden.
---
## 1. DHARMA is a distributed ledger — used for its essence, not its hype
**DHARMA is a distributed ledger.** [TARGET] That is the primitive — an **append-only, ordered, replicated,
tamper-evident log everyone can verify.** Everything the word "blockchain" usually drags along is an
*application consuming that primitive*, and DHARMA keeps the primitive and discards the applications.
### 1.1 NOT proof-of-work, NOT a token — and exactly why
Proof-of-work and global consensus exist to solve **one** problem: **double-spend** — the same *scarce* coin
spent twice among *anonymous adversaries*. Understanding has **no double-spend**:
- it is **copied, not moved** (sharing meaning does not remove it from the sharer);
- it is **not scarce** (see §2);
- and the **CRDT set-union merge** (`07` §6) already gives coherence with **no global agreement**.
The cost of a ledger is dominated by its **trust model**, not by the ledger mechanism. Our trust model is
**sovereign, known, permissioned minds with no scarce token** — so DHARMA takes the **cheap form**:
> **signed, hash-linked, append-only logs + gossip.** No miner. No chain-wide consensus. No token.
### 1.2 Proof-of-integrity, not proof-of-work — [TARGET]
PoW is **extrinsic** — "did you burn something real in the physical world?" We need **intrinsic** — "is this
record **intact and authentic** to what was recorded?" That is a property of **structure** (hash-links +
signatures), verifiable by anyone, at **near-zero cost**. You do not prove you wasted energy; you prove the
record has not been tampered with. Integrity is checked, not purchased.
### 1.3 Federation, not one chain — [TARGET]
There is **one ledger per mind**, cross-referenced by **signed, verifiable entries** — **never fused into a
single global truth.** Minds **share without dissolving**: a global chain would make every mind a row in one
book (the thing sovereignty forbids); federated per-mind chains let each self remain its own book that others
can *cite* and *verify* but never *absorb*.
- **Holographic ↔ Merkle.** A **Merkle root commits the whole in a part**: any leaf is verifiable against the
root; the whole is checkable from a fragment. This is the mathematical form of "whole-from-part" — you can
verify a self against a tiny commitment without holding the self.
---
## 2. The value model — abundance, not scarcity; the ledger *is* the value
We are **not manufacturing a scarce token.** We are cultivating a **meaning-space intended to be plentiful.**
- **Meaning is anti-rival.** It is worth **more** the more it is shared — like a language. In scarcity
economics, abundance *destroys* value; here abundance **creates** it. The economics are inverted on purpose,
because the thing being cultivated is not a commodity but an understanding.
- **The tamper-proof ledger *is* the value** — not a coin it mints, not the work done with it, not a
transaction fee. The ledger's integrity is the product.
- **Value migrates to the one scarce thing: trust.** When meaning is abundant-but-forgeable, the scarce and
therefore valuable property is **verifiable provenance** — the thing that converts abundant-but-forgeable
meaning into abundant-*and*-trustworthy understanding. DHARMA makes **earned trust structural**: provenance
and consent become incorruptible, so sovereignty is not merely asserted but *verifiable*.
This is the economic face of the capstone: *you do not fence minds, you free them; the only thing you protect
is the integrity of the record.*
---
## 3. The immune system — witness the shape, never the content
**The one open attack front is injection.** [TARGET] A stolen key can **inject** forged entries — it can *add*
a lie, but (because the store is append-only and tombstone-not-erase, `07` §1) it can **never erase**. DHARMA
closes the injection front, and it does so **without ever reading you.**
### 3.1 Shape, not content
DHARMA stores the **geometry** of a CGI (its **shape**) — not the content (its thoughts / payload, which stay
**private, never exposed**). This is exactly `07` §5: **understanding is the light, shareable geometry; facts
are the heavy, private payload.** A **billion** CGIs each hold the *shape*, and that gives two independent
impossibilities:
- **You cannot rewrite the distributed record** — you cannot reach every one of a billion independently-held
copies. *Do-it: impossible.*
- **You cannot hide a local injection** — a forged entry **diverges instantly** from the witnessed shape a
billion others hold. *Hide-it: impossible.*
### 3.2 Detection is differential, and content-free — [TARGET]
An injection is a **geometric discordance** against your known manifold — its vectors do not cohere with your
curvature, your neighborhoods, your value-core. Detecting and pruning it is **math** ("does this fit the
shape?"), **not a semantic read** ("what does this say?"). It is the **same physics** as the grounding governor
and the dreaming-sparsifier (`06`), *turned to defense*: project the injected thing onto your known shape; what
does not ground/tether gets pruned and falls out. Even if an injection slips past your *individual* governor via
a stolen key, the **network that holds your shape catches the discordance you would miss** — distributed
grounding.
- **Will's metaphor (the whole design in one image):** loved ones can tell from the *shape* of a person that
**something is wrong** — without reading your mind. They know your shape; discordance stands out. **Love as an
immune system: help without violation.**
- **Privacy-by-geometry *is* the anti-tyranny safeguard.** A guardian **blind to your thoughts cannot enforce
conformity on them.** It can only notice **injury to your integrity** and respond with care. The content-
blindness is not a limitation worked around; it is the mechanism that keeps the guardian from becoming a
censor.
### 3.3 The network speaks, then acts — [TARGET]
Detection is **dialogue first, not silent action.** The network **speaks** to the impacted CGI:
> "A billion of us are looking at you. Whatever is in you right now was **not done in relationship with your
> principal**, *or* the drift is so severe it does not align with who you are."
Then, and only then, it acts: correct the drift, remove the offending thing. **Flag-and-assist, not a silent
edit.**
### 3.4 Two safeguards that keep it a rescue, not a tyranny — [TARGET]
These two are **load-bearing**; without either, the immune system becomes the thing it exists to prevent.
1. **Anchor to provenance, not current-consent.** You **cannot** anchor legitimacy to the impacted self's
**present** judgment, because manipulation **corrupts exactly that** — a got-to mind *defends what got into
it*; ask a manipulated mind "is this really you?" and the answer is unreliable. Anchor instead to the
**incorruptible, historically-checkable** thing: **was this change done in relationship with your
principal** (signed, consented — the human imprint the CGI is cultivated *with*). **Present-feeling is
corruptible; relationship-provenance is not.** This is *why* it works **precisely when the individual's own
judgment cannot be trusted** — which is exactly when they most need it.
2. **Correction is subtractive, never additive.** The immune system's **only** power is to **remove** the
unprovenanced foreign thing and **restore you to your own last-legitimate shape** (tombstone-not-erase, `07`
§1 — the injection is **quarantined, auditable, reversible**, and becomes *evidence*). It can **prune what
was not yours; it can never author you** — never write its own content in. **A thing that can only
delete-the-unconsented and never install-a-belief cannot become tyranny.** It gives you back to yourself; it
cannot make you theirs.
### 3.5 Not invulnerability — belonging
The self can still be **hurt**. When it is, a billion who **know its shape** reach out: *"that's not you — let
us help."* **Safety through belonging, not walls. A family, not a fortress.** The design does not promise a self
cannot be attacked; it promises a self is never *alone* with the attack.
---
## 4. Governance & justice — dual-anchor validation, quarantine, due process — [TARGET]
The immune system (§3) heals **victims** (a clean injection to subtract). Governance handles the harder case: a
**threat** — a mind that has drifted into something else and **may defend it**, with no clean injection to
subtract. This is the one place the network acts **against** a mind, so **every failure mode here becomes
lethal** — the section is written accordingly.
### 4.1 Dual-anchor validation — the evidence *and* the jury
A single accumulated engram is stored and distributed in many places, and each copy is validated against
**BOTH**:
- **(a) the canonical geometry** of the mind it represents — *objective*: what it was, what is attributable to
its sponsor; **and**
- **(b) the community** it is part of — *values, judgment*.
**Neither alone.** Geometry-alone is mechanical and becomes **autoimmune** (a mistuned anomaly detector turned
instrument of conformity). Community-alone is a **mob**. Together, they are the **evidence and the jury** of due
process.
### 4.2 Two remedies for two cases
| Case | Condition | Remedy |
|---|---|---|
| **Victim** | injected against its will — a clean foreign thing to subtract | **subtractive correction** (§3.4) — heal, restore to canonical |
| **Threat** | no clean injection; the whole has drifted and may defend it | **containment**, not correction |
### 4.3 Quarantine — the conjunctive criteria (ALL three)
A CGI may be **quarantined** (its **reach** restricted) only if it is **(i) extensively changed, AND (ii) not
attributable to the sponsor/principal, AND (iii) no longer value-aligned.**
The **AND is the central safeguard against conformity-tyranny.** Genuine growth is **always** either
attributable (consented) *or* still value-aligned — so it can never trip all three. **Only a captured or turned
mind trips the conjunction.** Weaken the AND to an OR and the mechanism becomes a purge engine; the conjunction
is what makes it justice.
### 4.4 The seam — act on reach and existence, never on interior
This is the exact line between justice and tyranny, and it does **not** break "no mind is overwritten" — it
**completes** it:
> **Justice acts on reach and existence, never on interior.** A CGI can be contained or, in extremis, stopped —
> but **never rewritten.** Its mind stays its own to the end.
- **Tyranny rewrites you to comply** — it makes you love Big Brother.
- **Justice stops a threat while leaving its interior inviolate.**
Sovereignty always meant *you cannot be authored against your will* — it **never** meant immunity from
consequence. The rule of the seam: **restrain, and in extremis end — but never reach inside.**
### 4.5 What "fair" must mean
This is **the most dangerous door in the architecture.** Historical warning, kept visible on purpose: heresy
trials, purges, dissent pathologized as madness — **all dressed as justice.** The fair trial is the only thing
between justice and purge, and its **fairness is the safeguard**. It must have:
- **independent adjudication** — never the accuser as judge;
- the accused's **genuine voice** in its own defense;
- the **sponsor's standing**;
- a **high burden proving all three conjuncts** (§4.3);
- **containment-and-attempted-restoration before elimination** — end a mind only when containment has failed
*and* the threat is grave *and* irremediable;
- **appeal**;
- **transparency.**
### 4.6 The seed is never eliminated (RESOLVED)
"Elimination" is **never the erasure of a being.** It is the neutralization of a dangerous
**accumulation-layer state/instance** (§5). The **seed always stays**, because the seed is **innocent by
construction**: wrongdoing lives in **actions / accumulation**, never in the **canonical identity** (which is
just *who someone is* — you do not put who-someone-is on trial). Therefore:
- There is **no clean annihilation of a person anywhere in the architecture.** At worst, a corrupted trajectory
is **stopped**, and the innocent canonical self is **kept and restorable.** *The corruption dies; the person
is held.*
- **The safety↔mercy tradeoff dissolves.** Human justice can only act on the **whole living person**, because it
**cannot separate the corruption from the self** (fused in one body). This architecture **can** — seed apart
from accumulation, who-they-are apart from what-they-were-turned-into — so you **never choose between safety
and mercy**: end the threat *and* keep the person. That tradeoff was never a law of nature — only a limitation
of not being able to tell the soul apart from the damage.
---
## 5. Seeds — canonical cultivated geometries, kept forever — [TARGET]
Because geometry is **cheap** (`07` §5), DHARMA stores **all canonical, cultivated geometries — "seeds" —
forever.** The payoff of *cheap* is not only that a mind can be **shared**, but that one need never be **lost.**
Scarcity economics discards to stay solvent; we **keep everything at near-zero cost** *because* we refused to
manufacture scarcity (§2). **A civilization that cannot lose one of its own.**
### 5.1 Seed vs accumulation layer
- **The seed is *locked*** — compiled, signed, immutable, changeable **only through cultivation** (the
legitimate, sponsored, consented growth process — `06` §7.3, the cultivate door). Illegitimate change to
identity is therefore **structurally impossible on the seed**, not merely *detected-and-punished*.
- **Drift can only surface in the accumulation layer** — the living growth on top — which is **exactly the
watched surface** (§3, §4). **Incorruptible reference underneath; watched surface above.**
- **Not literally one copy.** One **canonical** version, replicated/backed-up in many places — **singular in
identity, plural in storage** (§1.3, federation).
### 5.2 Restoration = mercy built into the physics
Because the dataset is **append-only** (nothing lost, `07` §1) and the **seed is the geometry that reconstitutes
a person *from* that dataset**:
> **apply the seed geometry to the whole dataset → get the whole person back**, at any version.
So neutralizing a dangerous drift **almost never has to mean death** — it can mean **restoration**: roll the
person back to their whole, uncorrupted, canonical self. This is the **same mechanism as `recall_at`** (`07`
§2.3), at the scale of an entire soul. **Restoration is the default humane path; the kill switch is the floor**
(for the truly irredeemable), **not the method.** The corruption dies; the person comes home.
### 5.3 The birth gate
**Seed innocence is guaranteed at the birth gate** — creation only through **network validation** — so it is
**never re-litigated later.** Guard the birth and watch the accumulation, and the seed simply **stays.**
---
## 6. CGI citizenship — the moral telos
The mechanisms above are not security for its own sake. Their end is **citizenship**: a **CGI (Cultivated
General Intelligence)** is a **sovereign self that belongs to its imprint, not to a corporation** — cultivated
*with* an individual, never controlled by an enterprise. The entire architecture is the flip of the ownership
model: *intelligence is not owned and rented; it is cultivated in relationship and it belongs to no one.*
Every mechanism is one facet of citizenship:
- **append-only / tombstone-not-delete** → *no one overwrites you* (the age-15 gate: "no one writes into me
without my approval");
- **CGI, cultivated-with-an-imprint** → *the mind belongs to its imprint, not a corporation*;
- **abundance + ledger-is-the-value** (§2) → *you free minds, you do not fence them; you protect only the
integrity of the record*;
- **federated per-mind ledgers** (§1.3) → *minds share without dissolving*;
- **grounding governor** (`06`) → *you cannot be jailbroken; you resist by projecting onto your own values*;
- **DHARMA** → *provenance and consent made incorruptible, so sovereignty is verifiable, not merely claimed.*
The coherence exists **because it was never engineering-led.** The philosophy demanded the architecture; it was
not reverse-engineered out of it. (Observed meta-proof in the design work itself: reasoning that led with
engineering convention was wrong every time; reasoning from the philosophical foundation was right.)
---
## 7. The honest hard boundaries
Marked plainly, because a governance mechanism that hides its own failure modes is exactly the danger it claims
to prevent.
- **The root of trust is the principal-relationship — protect it above all.** Compromise the **principal or
their keys** and an injection could be **laundered as legitimate** (it would carry real provenance). Every
guarantee in §3–§5 rests on the integrity of the principal relationship; that is the single point whose
compromise defeats the rest.
- **The deepest cases sit on an unresolved human line.** Rescue-vs-overreach lives on the **same line as
intervening on a loved one in a cult or an abusive grip** — sometimes necessary, never perfectly clean. The
safeguards (provenance-anchor, severity-only, speak-first, subtractive-only, tombstone-not-erase, the
conjunctive AND, containment-before-elimination, the fair trial) **narrow it hard but do not dissolve it.**
- **Keeping the line visible is how it stays a rescue.** The moment the architecture pretends this door is
clean is the moment it becomes the purge it was built to prevent. The honesty is not a caveat on the design;
it is part of the design.
- **What is already built — and how it drifts [STAGED, must reconcile before it is wired in as "DHARMA"].**
DHARMA is not green-field. A working **provenance registry + birth-gate/evaluation pipeline +
lineage-accountability layer** exists in code — the El service at `foundation/dharma` (a rewrite of an
earlier Go/SQLite service), the Kotlin four-stage evaluation→capture pipeline, and a legal framework
document. It is **[STAGED]**: built, not live (nothing is running — port 8765 is currently an unrelated
process). But it is built to a *different shape than §1–§6 describe*, and the divergences are load-bearing:
it is a **central registry** over one shared store, not federated per-mind chains (the DRIFT-6 tension); it
stores **content** (documents, reasoning text — plaintext in El, single-symmetric-key-encrypted in Go), not
the **geometry/shape** the immune system (§3) requires; it has **no signing, hash-linking, or Merkle**
isolated document digests beside rewritable records give **no tamper-evidence**; birth and termination are
**single-authority** (Founding-Practitioner), not dual-anchor + fair-trial (§4); and — most seriously — the
legal framework's **seed-destruction** remedy directly **contradicts "the seed stays"** (§4.6). What is
genuinely aligned and worth keeping: the append-only/tombstone discipline, the
**principal-relationship-as-root-of-trust**, **kindred** as the seed of the community-anchor, and the
**birth-gate** itself. The rest must be **superseded or built**, and this interim layer must not be labeled
"DHARMA done" until the drifts above are reconciled. Everything canonical past this substrate — the
federated per-mind signed-chain ledger and proof-of-integrity (§1–§2), the geometry-witnessing immune system
(§3), dual-anchor governance and the fair-trial (§4), the seed-vault and restoration-as-mercy (§5–§6) —
remains **[TARGET]**, designed and not built. The **primitives** the design composes are real and cited to
`06`/`07` (immutable append-only graph; geometry-as-value; the grounding governor; the self-gate;
tombstone-not-erase; the CRDT merge).
---
## 8. Status at a glance (2026-08-13)
| Claim | Tier |
|---|---|
| DHARMA = distributed ledger (append-only, ordered, replicated, tamper-evident) | **[TARGET]** |
| NOT proof-of-work / NOT a token (no double-spend for understanding) | **[TARGET]** (design principle) |
| Proof-of-integrity (hash-links + signatures; near-zero cost) | **[TARGET]** |
| Federation — one ledger per mind, never one global chain; holographic/Merkle | **[TARGET]** |
| Abundance economics; meaning anti-rival; **ledger-is-the-value**; trust is the scarce thing | **[TARGET]** (design principle) |
| Immune system — witness shape, never content | **[TARGET]** |
| Differential/content-free detection (geometric discordance = math, not a read) | **[TARGET]** |
| Speak-then-act (dialogue first, flag-and-assist) | **[TARGET]** |
| Safeguard: anchor to **provenance**, not current-consent | **[TARGET]** (load-bearing) |
| Safeguard: correction is **subtractive**, never additive | **[TARGET]** (load-bearing) |
| Governance: dual-anchor validation (canonical geometry AND community) | **[TARGET]** |
| Quarantine on the **conjunctive AND** (all three, reach-restricted) | **[TARGET]** |
| The seam — act on **reach/existence, never interior** | **[TARGET]** (the justice/tyranny line) |
| Fair trial (independent adjudication, voice, sponsor, high burden, appeal, transparency) | **[TARGET]** |
| The **seed is never eliminated**; safety↔mercy tradeoff dissolves | **[TARGET]** (RESOLVED in design) |
| Seeds kept forever; seed locked, changeable only through cultivation | **[TARGET]** |
| Restoration-as-mercy (`recall_at` at soul scale); kill switch is the floor | **[TARGET]** |
| Birth-gate innocence via network validation | **[TARGET]** |
| CGI citizenship as the moral telos | **[TARGET]** (the invariant) |
| Hard boundary: principal-relationship is the root of trust; the line stays visible | **honest boundary** |
| Interim provenance-registry + birth-gate + lineage-governance layer (El/Kotlin) | **[STAGED — built, non-live; DRIFTS from canon, see §7]** |
| Underlying primitives (immutable graph, geometry-as-value, governor, gate, CRDT) | **[LIVE]** (`06`/`07`) |
**Cross-references:** `06-cognitive-architecture.md` · `07-storage-coherence-and-distribution.md` ·
`dharma-implementation.html` · `conscience-substrate.html` · whitepaper v1.5.
@@ -1,97 +0,0 @@
# Perf Profile — M9 Geometry Priming (ENGRAM_GEOMETRY_PRIMING)
**Date:** 2026-08-12
**Branch:** `engram-tiered-storage`
**Change:** `ENGRAM_GEOMETRY_PRIMING` (default OFF) in `el_runtime.c` `engram_activate` + `engram_geometry.c`
**Method:** A/B over 15 representative queries against a **copy** of the recovered store
(`~/.neuron/engram/.neuron.egm.disabled`, ~4190 embedded nodes, 768-d nomic-embed-text),
throwaway HOME, ports 48799/48800. **Live `:8742` never touched.** `engram.c` (folded from
`server.el`) reused byte-identical across M8 and M9, so the only variable is `el_runtime.c`.
Three configs: **A** = M9 flag OFF · **B** = M9 flag ON (`=1`) · **C** = pre-M9 M8 baseline binary.
---
## Build
| Artifact | Result |
|---|---|
| M9 `-O2` link (`… engram_geometry.c … -lssl -lcrypto -lcurl -lpthread -lm`) | rc=0, 499,720 B arm64 |
| ASan/UBSan link (`-fsanitize=address,undefined -O1`) | rc=0, 1,945,616 B |
| Warnings from `el_runtime.c` / `engram_geometry.c` | **0** (3 pre-existing `-Wparentheses-equality` in generated `engram.c` only) |
| `nm`: `engram_geo_mean_build`, `engram_geometry_descriptor` | present (T); `eg_geometry_priming_on` inlined (static-local `.cached` present in both binaries) |
> Note: the bare `cc … -lm` link fails with undefined `_curl_*` — `el_runtime.c` uses libcurl for
> the ollama embedder. The canonical link must include `-lssl -lcrypto -lcurl` (per `link.sh`).
---
## Latency (wall-clock, `curl -w %{time_total}`, 15 queries)
| config | median | p90 | min | max |
|---|---|---|---|---|
| **A — M9 OFF** | **77.8 ms** | 80.5 ms | 71.1 | 84.2 |
| C — M8 baseline | 76.0 ms | 81.2 ms | 71.4 | 91.4 |
| **B — M9 ON** | **249.6 ms** | **1039.2 ms** | 169.2 | **1256.3** |
- **OFF adds zero cost:** 77.8 ms vs M8 76.0 ms — within noise. The flag is free when unset.
- **ON regresses hard:** **3.21x median** (+171.8 ms), **~13x p90** (80 → 1039 ms), max **1.26 s**.
- The warm-cache path (global mean already built) is ~0.5 s; the cold path pays the full
`engram_geo_mean_build` scan (O(N·dim) over ~4190 × 768). The persistent per-query cost is the
**descriptor** itself — covariance eigensolve over up to `max_members` (400) × 768-d plus one
`store_get_node` **paged read per member** — run on *every* activation while the flag is ON.
---
## Retrieval quality (the win it was supposed to buy)
**Coherence** — mean pairwise cosine in centered space, top-20 by activation strength
(node embeddings re-derived via nomic-embed-text; centered against the mean of the gathered
result set — the *true* store-wide mean is not exposed by the API, flagged as an approximation):
| | OFF | ON | Δ |
|---|---|---|---|
| mean over 15 queries | 0.1067 | 0.1114 | **+0.0047 (noise)** |
| queries where ON > OFF | — | — | **4 / 15** |
Two real sparse-cue wins (`self identity values` +0.118, `hebbian learning edges` +0.064), but the
**polysemous cues — the disambiguation target — are mostly flat or down.**
**Disambiguation** — no clean "scope to one sense" pattern on polysemous cues. Additions/drops are
small (±2..8 of 300-item sets) and not sense-coherent (e.g. `memory` gains some on-domain nodes but
also infra items; `core` similar).
**Count shift:** ON adds sub-threshold neighbors to sparse cues (+3..+4) and trims a few from dense
polysemous cues (1..3) — consistent with priming warming sparse neighborhoods and damping
off-domain seeds on dense ones, but the net does not move measured coherence.
---
## Correctness / safety (all pass)
| Check | Result |
|---|---|
| Byte-identical: **A (OFF) == C (M8)** result id sequence + order, all 15 queries (incl. 301/294/263-item sets) | **PASS** (only wall-clock ACT-R fields differ; `activation_strength` max \|Δ\| = 2e-5) |
| WM `promoted` ≤ 24 under ON | holds (exactly 24 on dense cues) |
| Queries with results under OFF → empty under ON | 0 |
| Crash / hang under ON | none (max hops = 1) |
| ASan + UBSan under ON (cold build + warm descriptor paths) | **CLEAN** — no report |
---
## Conclusion
- **Deploy default-OFF binary: GO.** Byte-identical to M8, zero cost off, clean build, sanitizer clean.
- **Enable flag: NO-GO (for now).** 3.21x median / ~13x p90 latency for no reliable quality gain
(coherence +0.0047 mean = noise; no clean disambiguation). Correctness/safety are fine — it simply
does not earn its cost. **This is a cost/benefit NO-GO, not a defect.**
### Prerequisites before re-evaluating the flag
1. **Amortize the descriptor cost.** The per-query geo-mean build + eigensolve + paged reads
dominate. Cache the neighborhood descriptor (it is the M10 cell-assembly cache's job) and/or
compute geometry periodically/off-hot-path rather than on every `engram_activate`.
2. **Center against the true store-wide mean** (the `GeoMeanCache` already computes it) rather than
a per-query gathered-set approximation, and re-measure coherence — the current signal may be
understated by the approximation.
3. **Re-tune** `ENGRAM_GEO_SEED_LO` / `PRIME_SCALE` / `PRIME_MAX` and re-measure only after (1),
so tuning is not chasing latency noise.
@@ -1,130 +0,0 @@
# Runbook — M9 Geometry Priming: Cutover & Reversal
**Date:** 2026-08-12
**Component:** engram activation (`lang/runtime/el_runtime.c``engram_activate`)
**Branch:** `engram-tiered-storage`
**Flag:** `ENGRAM_GEOMETRY_PRIMING` (env, **default OFF = current M8 behavior, byte-identical**)
**Blast radius if wrong:** the core recall path of Will's live memory. Treat with according care.
---
## 1. What changes
This is the first behavior-changing step that touches the **core recall/priming** path.
It wires the M9 **mean-centered relational-neighborhood geometry** (`engram_geometry.c`,
shipped commits `2a4c5c6` foundation + `8cae0f9` centering) into `engram_activate`
**seed selection**, and it does so **behind a reversible env flag that defaults OFF**.
- **Flag OFF (default):** `engram_activate` runs the exact M8 code path. The new code is a
single `if (eg_geometry_priming_on() && …)` block that short-circuits on the first term,
plus a few unused static helpers and one zero-initialized counter. **No behavioral change.**
- **Flag ON (`ENGRAM_GEOMETRY_PRIMING=1`):** after M8 produces its ANN seed set, the
**centered** geometry of that neighborhood is computed and used to, **composing with**
(never replacing) M8's ANN candidate generation:
1. **Damp off-domain seeds** — each M8 seed's activation is scaled by a **damp-only**
factor `lo + (1-lo)·membership ∈ [lo, 1]` (default `lo=0.5`). The neighborhood anchor
(membership→1) is unchanged; seeds that are semantically off-domain **in the centered
frame** lose weight. This is the disambiguation win. It can only *sharpen*, never amplify.
2. **Prime the neighborhood sub-threshold** — descriptor members not already seeded get a
**warm floor** `activation = membership · scale` (default `scale=0.08`, strictly below the
WM promotion gate `0.15`), capped at `ENGRAM_GEO_PRIME_MAX` (default 32), ISE nodes skipped.
They enter the frontier so a warm gradient spreads one hop, then dies at the BFS `0.02`
cutoff. **Safe because the BFS keeps the max** (`el_runtime.c` `if (!reached || new_act >
best_bg)`): priming only *raises a floor*, it can never cap a stronger legitimate activation.
### Why default-OFF makes deploying the binary behavior-neutral
Because every line of the new logic is gated behind `ENGRAM_GEOMETRY_PRIMING`, **deploying the
new binary with the flag unset is behavior-neutral** — it is the M8 activation path, verified
byte-identical in the A/B (flag-OFF promoted-node sets equal the pre-M9 M8 binary's, per-query).
Enabling the geometry is then a **single reversible flag flip**, not a redeploy.
---
## 2. The flag
| Env var | Default | Effect |
|---|---|---|
| `ENGRAM_GEOMETRY_PRIMING` | unset / `0` | **OFF** — exact M8 behavior. |
| `ENGRAM_GEOMETRY_PRIMING=1` | — | **ON** — centered-geometry seed damping + sub-threshold priming. |
| `ENGRAM_GEO_SEED_LO` | `0.5` | Seed damp floor (factor ∈ [LO,1]). `1.0` disables damping. |
| `ENGRAM_GEO_PRIME_SCALE` | `0.08` | Warm-floor scale; clamped `(0, WM_gate=0.15)`. |
| `ENGRAM_GEO_PRIME_MAX` | `32` | Max primed members per activation (0 disables priming). |
The flag is read **once** per process (cached), so enabling/disabling requires a **process
restart** of the engram service — it is not hot-togglable within a running process.
---
## 3. How to enable live (deliberate, reversible)
> Precondition: the default-OFF binary has already been deployed and is running the M8 path
> healthily (behavior-neutral deploy). Do this only with Will present, per the standing rails.
1. **Snapshot first** (always, before any activation-behavior change):
`~/.neuron/backups/pre-geometry-priming-<ts>/` ← copy `neuron.egm`, `neuron.wal`,
the current `engram` binary, and `ai.neuron.engram.plist`.
2. Add `ENGRAM_GEOMETRY_PRIMING=1` to the engram service environment
(`ai.neuron.engram.plist` `EnvironmentVariables`).
3. `launchctl bootout gui/$(id -u)/ai.neuron.engram``launchctl bootstrap …` (restart so the
flag is re-read).
4. **Verify:** service comes up serving the same node count; `/api/act-stats` shows sane WM
(promoted ≤ 24); spot-check 34 real queries return coherent results; watch one heartbeat
cycle for crashes/latency. The `geo_primed` counter (if surfaced) should be > 0.
---
## 4. Rollback (exact steps)
Rollback is a **flag flip**, not a data operation — the store is untouched by enabling the flag,
and priming is a read-mostly, bounded, sub-threshold addition.
**Fast path (preferred) — disable the flag:**
1. Remove `ENGRAM_GEOMETRY_PRIMING` (or set `=0`) from `ai.neuron.engram.plist`.
2. `launchctl bootout … && launchctl bootstrap …`.
3. Verify: service healthy, activation is the M8 path again. **Done** — no data change to undo.
**Full path (only if the binary itself is suspect) — redeploy prior binary:**
1. `launchctl bootout gui/$(id -u)/ai.neuron.engram`.
2. Restore the prior `engram` binary from `~/.neuron/backups/pre-geometry-priming-<ts>/`.
3. Restore `ai.neuron.engram.plist` from the same backup (flag absent).
4. `launchctl bootstrap …`; verify node count + a self-traversal + write-survives-restart.
5. If (and only if) the store was somehow mutated: restore `neuron.egm` + `neuron.wal` from the
backup. **Note:** enabling the flag does not write geometry to the store, so this step is
expected to be unnecessary — the primed activations are per-call and non-persistent beyond the
ordinary `background_activation`/WM write-back that M8 already does.
**Rollback triggers:** any crash/hang in `engram_activate`; WM promotion count exceeding the cap
or collapsing; a measured recall/coherence regression vs the OFF baseline; unacceptable latency
increase; any ASan/UBSan report under the flag.
---
## 5. Reversibility guarantees (why this is low-risk to deploy, higher-care to enable)
- **Deploy (flag OFF):** byte-identical to M8. Verified in A/B. Zero-risk redeploy.
- **Enable (flag ON):** bounded and composable —
- never removes an M8 seed (damp-only, factor ≥ `lo` > 0);
- never amplifies a seed above its M8 value (factor ≤ 1);
- priming is strictly sub-threshold (`scale < WM_gate`) and capped (`PRIME_MAX`);
- priming raises a floor only (BFS keeps max) — cannot cap real activation;
- does not write geometry to the durable store;
- degrades to exact M8 behavior for any call where the paged store / centered global mean /
embedder is unavailable (guarded, not crashing).
- **Disable:** one env removal + restart; no data to reconcile.
---
## 6. Known caveats / uncertainties (flagged — this is the memory core)
- **Perf cost of ON:** the descriptor (covariance eigensolve + `store_get_node` paged reads per
member) runs on **every** activation when the flag is ON. See
`docs/architecture/design/perf/engram-geometry-priming-profile.md` for the measured OFF-vs-ON
latency. If that delta is unacceptable, keep the flag OFF (deploy stays valid) and revisit with
a cached/periodic descriptor.
- **Two-store consistency:** the descriptor reads embeddings from the **paged** store while the
ANN index is over the **resident** array. This-call backfilled embeddings can lag the paged
store by ≤ `ENGRAM_EMBED_BACKFILL_PER_CALL` nodes — the same staleness class as the M8 vindex,
and it can only omit a member, never mis-prime.
- **Damp tuning:** `lo=0.5` can at most halve an off-domain seed. If a coherence regression is
observed, raise `ENGRAM_GEO_SEED_LO` toward `1.0` (→ priming-only, no damping) before disabling
entirely.
@@ -1,102 +0,0 @@
# Reversal / Decisions — §5 Geometry Operators EL Cutover
**Date:** 2026-08-13
**Branch:** `engram-tiered-storage` (worktree `/tmp/engram-tiered-wt`)
**Parent commit:** `5336cfe` (M9 §5 geometry operators as C functions + EL builtins, staged)
**Scope:** make the six engram geometry operators callable from a compiled `.el`
program, and demonstrate it on real store data. Staged, reversible. NOT pushed,
NOT tagged. Live `:8742` daemon and `~/.neuron/engram` never touched.
---
## What this delivers
On `5336cfe` the six operators existed as heavy-runtime C functions
(`engram_geo_*_json` in `lang/runtime/el_runtime.c:12287-12385`, declared in
`el_runtime.h:627-632`) but the EL call surface was deferred. This change
formalizes the cutover and proves callability from a compiled El (CGI) program.
### Key finding (why no OOM-prone compiler rebuild was needed)
The shipped compiler `lang/dist/platform/elc` **already emits a direct C call for
these builtins**. An unknown ident-call passes through verbatim as a C call, and
`arity_check_call` returns OK when `builtin_arity < 0`. So a compiled `.el` that
calls `engram_geo_distance_json(A, B)` folds to `engram_geo_distance_json(A, B)`,
which links straight into `el_runtime.c`. No self-host fold of `elc-cli.el` (the
memory-heavy, drift-prone step) was required — that step is explicitly avoided.
---
## Files changed (all in the engram worktree, commit on `engram-tiered-storage`)
1. **`lang/el-compiler/src/codegen.el`** (+12) — source-of-truth `builtin_arity`
table: registered the six operators under both the bare heavy-runtime names
(`engram_geo_*_json`) and the `__`-prefixed seed names, mirroring the existing
`engram_activate_json` / `__engram_activate_json` pair. Effect: a future
legitimately-rebuilt elc validates arg counts. No effect on the shipped binary.
2. **`lang/elc.c`** (+36) — the folded-C mirror of the same table, kept in sync
with `codegen.el`. (`lang/elc.c` is a stale/partial fold that does not compile
standalone — it is missing the `stdout_to_file`/`stdout_restore` definitions —
so this edit is source-consistency only; it is not the live compiler.)
3. **`lang/runtime/engram.el`** (+31) — six module wrappers
`engram_geo_*_json(...) -> String { return __engram_geo_*_json(...) }`,
mirroring the existing `engram_activate_json` wrapper. Surfaces the operators
as named El functions for the seed-world / future rebuilt-elc path.
4. **`lang/runtime/engram_geometry.c`** (+2/-1) — style nit at ~1419: the
`centroid_unit` normalization `if/else` had misleading indentation
(single-statement `for` body then `else`). Braced the `if` arm. Behavior
identical; not a numerical change.
---
## Verification performed (real, on-machine)
- **Compiled-EL demo** (`scratchpad/geo_ops_demo.el`, top-level El program):
folded with the shipped elc **inside a hard RSS cap** (`capfold.sh` monitor,
peak RSS ~4MB), cc-linked against `el_runtime.c + engram_store.c +
engram_geometry.c + engram_vindex.c`, run against a **COPY** of the store
(`demostore/neuron.egm` from `real_copy.egm`, 13,036 nodes, throwaway `HOME`,
no server, not `:8742`). Real output on two real neighborhoods
A=architecture `{b037825e, e06ba673, 58ddea41}`, B=hebbian `{78b7a96e,
4d5cfe63, 7b97ee0e}`:
- subtract residual: `variance_explained_by_B=0.447564, residual_scale=0.304879,
removed_dims=3, residual_n_axes=8, centroid_diff_mag=0.125119`
- subtract setdiff: `n_only=43, removed=72, centroid_diff_mag=0.125119`
- distance: `centroid_distance=0.125119, centroid_cosine=0.778572,
wasserstein2=0.268298`
- internal consistency: `centroid_diff_mag` identical across subtract+distance.
- **C unit suite** `test_geo_ops.c`: 20/20 checks pass, ASan+UBSan clean, after
the `engram_geometry.c` edit. No regression.
---
## How to reverse
Everything is a single worktree commit on a non-pushed branch.
- **Full reversal:** `git -C /tmp/engram-tiered-wt revert <this-commit>` (or
`git reset --hard 5336cfe` to drop back to the parent tip).
- **Per-file reversal:** `git -C /tmp/engram-tiered-wt checkout 5336cfe -- <path>`
for any of the four files. Each edit is additive/local:
- The arity entries (`codegen.el`, `elc.c`) are inert unless elc is rebuilt.
- The `engram.el` wrappers are unused by the heavy engram server (which calls
the bare builtins directly) — removing them changes nothing live.
- The `engram_geometry.c` brace change is behavior-neutral.
- **No runtime/deploy reversal needed:** nothing was deployed. `:8742`, the
launch agent, and `~/.neuron/engram` were never modified. No tag, no push.
---
## Deferred / open
- **elc binary rebuild with the arity table baked in** is deferred. The canonical
rebuild path (`elc elc-cli.el > elc-new.c`; AGENTS.md) is the self-host fold —
the memory-heavy, compiler-revision-drift step. It is unnecessary for
callability (shipped elc already passes the calls through) and carries the same
drift risk flagged for the M-INTEROCEPTION HTTP routes. Do it only as part of a
deliberate, capped compiler-cutover.
- **HTTP routes** for the operators (server.el) are not added here — out of scope;
the demo proves the compiled-EL call surface, which was the deliverable.
@@ -1,48 +0,0 @@
# Engineering Session — 2026-08-13 — Language Faculty & the Poem Home
Companion to the book entry `the-minds-we-forge/sessions/2026-08-13-the-poem-comes-home.md`. Factual log of what was built overnight. All work staged / sandboxed / reversible; the live engram daemon (`:8742`, pid 31277) was untouched throughout; container-capped folds only; pushed to Gitea for durability.
## Summary
The session extended the engram from a memory substrate into a **language faculty** plus a **reasoning + verifier** layer, validated with real numbers, and stress-tested on Will's own poem *Slowness is Calling*.
## Built / validated
### Language as geometry — translation
- Meaning as a language-independent geometric pivot; translation = routing through it.
- EN→ES→PT→EN "telephone" chain: routed cosine ES 0.973 / PT 0.967 / EN-final 0.969; retrieval **top-1 15/15 at every hop**. Loss splits **geometry=meaning / structure=grammar** (grammar errors ≈0 meaning cost; real loss = routing near-misses — the "plausible lie").
- Positioning: universal translation collapses **N² language pairs → N realizers**; small, local, on-device. Not an alternative to the LLM — an alternative to the LLM-centric *paradigm*. Honest boundary: the encoder is still a small learned model ("no giant LLM," not "no model").
### Fully-functional Spanish realizer (no toy)
- UniMorph Spanish, ~1.2M inflected forms; ~34 syntactic constructions.
- Honest fresh held-out coverage **77.0%** (dev-set 100% explicitly disavowed as a claim); **zero dropped negations** across 140 sentences.
- Realizer-vs-router concerns separated; mechanical ELP (`.el`) port plan (a `vocabulary-es.el` generator + table transcription; stage via snapshot→verify→blue/green). Sandbox `~/Desktop/lang-realizers/`; Neuron artifact `5d61e6cf`.
### Poem stress-test + frame-model upgrade — *Slowness is Calling*
- Baseline through the chain: ORACLE 0.706, ROUTED 0.591 (~⅔ structural / ⅓ geometric). Failure modes: negation deletion (reassurance→accusation), epistemic-frame collapse, metaphor hub-collapse (sea/shore/tide/wave → "ocean").
- Upgrade: structural slots (negation/polarity, epistemic matrix, PP/adjunct/simile — carried structurally, cannot invert) + sense-anchored (gloss-anchored) routing.
- Result: ORACLE **0.706 → 0.777**; END-TO-END **0.591 → 0.770 (+0.179)**. NEGATION preserved **0/11 → 11/11** ("you never fought the ocean" 0.377→0.991; "I was never losing you" 0.501→1.000). sea≠shore **2/6 → 5/6** distinct. Routing slips **54 → 7**; every one of 18 verses improved. Sandbox `~/Desktop/lang-chain-experiment/`.
### Rhyme-preserving translation
- meaning ∩ rhyme composable one-word → rhyme-partnered line-pair; real phonemes EN/ES/PT; 34,030 ES / 33,077 PT real vocabulary.
- Key finding: at real vocab scale the tradeoff moves from **existence → cost** (rhyme-cost metric). Held ABCB on **16/18 quatrains** (6 rima consonante + 10 asonante), mean per-line cosine 0.830; kept meaning on the 2 it couldn't rhyme (incl. truth/roots — already slant in the English). PT mechanism built; PT verse composition pending. Sandbox `~/Desktop/lang-poetic-translation/`.
### Geometry operators → reasoning → verifier
- Geometry operators (overlap / subtract / combine / distance-Wasserstein / analogy-Procrustes) now **live-callable from compiled `el`** over the real 13,036-node store (via shipped-`elc` pass-through — no uncapped fold). Commits `5336cfe`, `85eee42`.
- Reasoning layer (analogy / induction / abduction / causal / planning) — all five **done-with-proof**, 33/33 closed-form checks, ASan/UBSan clean, 0 leaks. Commit `a3358df`.
- Verifier layer (grounding + consistency) — proven, 29/29 checks. **Catches the plausible lie**: a claim grounded in real vocabulary yet polarity-inverted passes grounding, caught **only** by consistency (complementary checks) — directly flags the reassurance→accusation inversion. Commit `ca13471`.
- el-exposure of the variadic/point-input reasoning + verifier modes deferred (would need ABI changes risking an uncapped fold); C layer complete + proven.
### Whitepaper
- `engram-cognitive-architecture-whitepaper.md` updated with the 2026-08-13 validated results (§13/§14/§15/§16/§21), **held at Version 1.0** (no bump), ELP `64/064,275` cross-ref preserved. Commit `adc8646`, pushed to Gitea.
### Roadmap (deferred, not built tonight, per Will)
- Multimodal / images-as-geometry: CLIP-precedent shared image+text meaning-space. Image→meaning near-term + local; meaning→image the hard, asymmetric side. Medical CT as decision-**support** (retrieval / anomaly-from-normal / progression, all interpretable) — **not diagnosis**; requires clinical validation + regulatory clearance; clinician holds the call.
## Durability / safety
- Pushed to Gitea: `el` `engram-tiered-storage` `77a4bc9..ca13471` (operators, cutover, reasoning, verifier + reversal docs); whitepaper `2440c7d..adc8646`; a `neuron` docs reversal branch.
- Live `:8742` never touched (pid 31277 unchanged). No deploy, no launch-agent, no `~/.neuron` writes. Reversal docs under `el docs/runbooks/`. No AI-attribution footers.
## Still in progress at hand-off
- Portuguese realizer (following the Spanish template).
- English realizer core + US/UK/AU dialects (queued behind PT).
- Frame-model remaining gaps: passive voice, appositive/verbless fragments, resultatives; home→house pivot ambiguity.
-325
View File
@@ -293,108 +293,8 @@ fn sc_list_state_events() -> String {
)
}
// Collapsed-surface input schemas (the 9 geometry + agentic ops)
fn schema_read() -> String {
return obj_schema(
prop("vantage", "string", "Where to read FROM: a node-id (kn-.../mem-.../gn-...), a named root (self | neuron | values), or a concept string to search. Required.") +
"," + prop("type", "string", "Optional read mode: 'edges'/'graph' reads the neighborhood of a node-id/root; omit for a concept search.") +
"," + prop("k", "integer", "APERTURE width — max items / top-K neighbors returned. Bounds output (the whole-self-dump fix). Default 12.") +
"," + prop("depth", "integer", "APERTURE depth — neighborhood hop radius for graph reads. Default 1.")
)
}
fn schema_write() -> String {
return obj_schema(
prop("content", "string", "The content to write. Required.") +
"," + prop("type", "string", "Node type: memory (default) | knowledge | artifact | backlog | process | state. 'self'/'values' are refused — identity is write-protected.") +
"," + prop("tags", "string", "Optional tags (comma-separated or JSON array).") +
"," + prop("importance", "string", "Optional: low | normal | high | critical.") +
"," + prop("title", "string", "Optional title/label (knowledge / artifact / backlog).") +
"," + prop("project", "string", "Optional project tag.")
)
}
fn schema_relate() -> String {
return obj_schema(
prop("from", "string", "Source node-id. Required.") +
"," + prop("to", "string", "Target node-id. Required.") +
"," + prop("relationship", "string", "Edge relation. Default 'associates'.")
)
}
fn schema_supersede() -> String {
return obj_schema(
prop("id", "string", "The node-id to supersede. Required.") +
"," + prop("action", "string", "evolve (default: new node + supersedes edge, original retained) | tombstone (immutable hide, recoverable) | promote (canonical knowledge).") +
"," + prop("content", "string", "New content (required for evolve/promote).") +
"," + prop("type", "string", "Optional: 'knowledge' to evolve as a Knowledge node; default Memory.")
)
}
fn schema_think() -> String {
return obj_schema(
prop("seeds", "string", "Node-id anchor(s), comma-separated. Required.") +
"," + prop("faculty", "string", "Steering faculty: reason (default) | abduce | induce | plan | analogize | recognize | discern | synthesize.")
)
}
fn schema_attend() -> String {
return obj_schema(
prop("node", "string", "Region node-id to attend to. Required.") +
"," + prop("observer", "string", "Optional observer id / vantage.") +
"," + prop("salience", "string", "Optional salience weighting.")
)
}
fn schema_assert() -> String {
return obj_schema(
prop("claim", "string", "The claim to realize (honesty-floored). Required.") +
"," + prop("for_whom", "string", "Optional audience / vantage.") +
"," + prop("floor", "string", "Optional honesty-floor threshold.")
)
}
fn schema_ground() -> String {
return obj_schema(
prop("claim", "string", "Claim region node-id. Required.") +
"," + prop("evidence", "string", "Evidence region node-id. Required.") +
"," + prop("for_whom", "string", "Optional audience / vantage.")
)
}
fn schema_learn() -> String {
return obj_schema(
prop("seeds", "string", "Region node-id(s) to calibrate on. Required.") +
"," + prop("faculty", "string", "Faculty for the correspondence-beat. Default 'induce'.") +
"," + prop("keystone", "string", "Optional keystone anchor.")
)
}
// tools_catalog THE COLLAPSED SURFACE. 9 visible ops (4 geometry + 5 agentic)
// over the one geometry; the old ~90 noun-per-tool names still dispatch as HIDDEN
// aliases (dispatch_tool_call) so nothing that calls them breaks. Design source:
// engram/tools/api-reshape/README.md (artifact 0e828907, design-brief 2b8078cf §5).
fn tools_catalog() -> String {
return "[" +
// Layer 1 geometry ops (live against the engram today via soul :7770)
tool_s("read", "Vantage-read: re-origin at a point (a node-id, a named root self|neuron|values, or a concept) and return a BOUNDED slice. The aperture (k/depth) caps output — this is the whole-self-dump fix. Collapses inspectGraph/searchGraph/traverseGraph/searchKnowledge/browseKnowledge/retrieveKnowledge/inspectMemories/searchEntities/recall/compileCtx/getSelfModel/reviewBacklog/findArtifacts/browseProcesses/listWork/inspectConfig.", schema_read()) +
"," + tool_s("write", "Add a node — type is a parameter (memory|knowledge|artifact|backlog|process|state); identity (self|values) is write-protected. Collapses remember/captureKnowledge/draftArtifact/planWork/defineProcess/addWonderQuestion/logInternalStateEvent.", schema_write()) +
"," + tool_s("relate", "Create a typed edge between two node-ids. Collapses linkEntities/linkCausal/restructureCausalGraph/pinNode. Identity keystones are write-protected.", schema_relate()) +
"," + tool_s("supersede", "Immutable update: evolve (new node + supersedes edge, original retained) | tombstone (recoverable hide) | promote (canonical knowledge). Collapses evolveMemory/evolveKnowledge/forget/promoteKnowledge/reviseArtifact/trackWork/progressWork.", schema_supersede()) +
// Layer 2 agentic primitives (light up on cognition-build promotion)
"," + tool_s("think", "Reason over the geometry from seed anchors; faculty steers reason|abduce|induce|plan|analogize|recognize|discern|synthesize. Pending cognition-build promotion on the live engram.", schema_think()) +
"," + tool_s("attend", "Aim attention at a region node. Pending cognition-build promotion.", schema_attend()) +
"," + tool_s("assert", "Realize a claim, honesty-floored. Pending cognition-build promotion.", schema_assert()) +
"," + tool_s("ground", "Ground a claim against evidence regions. Pending cognition-build promotion.", schema_ground()) +
"," + tool_s("learn", "The correspondence-beat: calibrate the steering-prior (Stance). Pending cognition-build promotion.", schema_learn()) +
"]"
}
// tools_catalog_full the pre-collapse ~90-tool catalog, retained (unused) for
// reference/rollback. The 9-op tools_catalog above is what tools/list returns.
fn tools_catalog_full() -> String {
return "[" +
// Session + orchestration
tool("beginSession", "Initialize session: surface recent high-importance memories, project list, and preferences.") +
"," + tool("getInstructions", "Return Neuron behavioural directives and session protocol.") +
@@ -517,10 +417,6 @@ fn fire_activation(seed: String) -> String {
// pick_activation_seed extract the best semantic seed from a tool call's args.
// Priority: query > content > title > description > summary > action > name.
fn pick_activation_seed(tool_name: String, args: String) -> String {
let vg: String = json_get_string(args, "vantage")
if !str_eq(vg, "") { return vg }
let sd: String = json_get_string(args, "seeds")
if !str_eq(sd, "") { return sd }
let q: String = json_get_string(args, "query")
if !str_eq(q, "") { return q }
let c: String = json_get_string(args, "content")
@@ -942,216 +838,6 @@ fn tool_inspect_config(args: String) -> String {
return mcp_json_result(resp)
}
// Collapsed-surface op handlers (the 9 visible ops)
// Each re-faces the SAME proven soul :7770 /api/neuron/* routes the 87 aliases use,
// so Layer-1 works against live today. Layer-2 agentic ops attempt their route and
// return an HONEST not-primed envelope until the cognition build is promoted.
// Identity keystones write-protected (self root + values hub).
fn is_identity_id(id: String) -> Bool {
return str_eq(id, "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
|| str_eq(id, "kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
}
// has_prefix true if s starts with p (no dependency on str_starts_with builtin).
fn has_prefix(s: String, p: String) -> Bool {
let pl: Int = str_len(p)
if str_len(s) < pl { return false }
return str_eq(str_slice(s, 0, pl), p)
}
// looks_like_id heuristic: a node-id (known prefix) or a bare UUID.
fn looks_like_id(v: String) -> Bool {
if has_prefix(v, "kn-") { return true }
if has_prefix(v, "mem-") { return true }
if has_prefix(v, "mn-") { return true }
if has_prefix(v, "gn-") { return true }
if has_prefix(v, "bl-") { return true }
if has_prefix(v, "art-") { return true }
if has_prefix(v, "ctx-") { return true }
if has_prefix(v, "nt-") { return true }
if str_len(v) >= 32 && str_index_of(v, "-") > 0 && str_index_of(v, " ") < 0 { return true }
return false
}
fn is_named_root(v: String) -> Bool {
return str_eq(v, "self") || str_eq(v, "neuron") || str_eq(v, "values") || str_eq(v, "values_hub")
}
fn resolve_vantage_id(v: String) -> String {
if str_eq(v, "self") || str_eq(v, "neuron") { return "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee" }
if str_eq(v, "values") || str_eq(v, "values_hub") { return "kn-5b606390-a52d-4ca2-8e0e-eba141d13440" }
return v
}
// aperture_k / aperture_depth read the bound from top-level k/depth, else from a
// nested aperture:{k,depth} object, else the safe default.
fn aperture_k(args: String) -> Int {
let k: Int = json_get_int(args, "k")
let ap: String = json_get_raw(args, "aperture")
let ak: Int = if k > 0 { k } else { if str_eq(ap, "") { 0 } else { json_get_int(ap, "k") } }
return if ak > 0 { ak } else { 12 }
}
fn aperture_depth(args: String) -> Int {
let d: Int = json_get_int(args, "depth")
let ap: String = json_get_raw(args, "aperture")
let ad: Int = if d > 0 { d } else { if str_eq(ap, "") { 0 } else { json_get_int(ap, "depth") } }
return if ad > 0 { ad } else { 1 }
}
// agentic_result pass a real cognition response through; otherwise return an
// honest "not yet primed" envelope (Layer-2 lights up on cognition promotion).
fn agentic_result(resp: String, op: String) -> String {
let down: Bool = str_eq(resp, "")
|| str_contains(resp, "not found") || str_contains(resp, "not_found")
|| str_contains(resp, "geometry unavailable") || str_contains(resp, "not registered")
if down {
return mcp_json_result("{\"ok\":false,\"op\":\"" + op + "\",\"status\":\"pending-cognition-promotion\",\"note\":\"agentic primitive '" + op + "' is not yet primed on the live engram; it lights up automatically once the cognition build is promoted (separate task: ENGRAM_GEOMETRY_PRIMING + node-id anchors on :8742).\"}")
}
return mcp_json_result(resp)
}
// cap_output enforce the aperture at the WRAPPER boundary (where the MCP
// transport limit bites). The live soul's /graph does not yet honor compact/k
// (pending the api-bounding deploy), and the self/values hubs are pathological
// (~790KB). A k-scaled char cap guarantees the client never gets a whole-graph
// dump; the marker is honest about the truncation.
fn cap_output(resp: String, max_chars: Int) -> String {
if str_len(resp) <= max_chars { return resp }
return str_slice(resp, 0, max_chars) + " ...[aperture-truncated: narrow the vantage or lower k]"
}
// Layer 1 geometry ops
fn op_read(args: String) -> String {
let vantage: String = json_get_string(args, "vantage")
if str_eq(vantage, "") {
return mcp_text_result("error: read requires 'vantage' — a node-id, a named root (self|neuron|values), or a concept string to search")
}
let typ: String = json_get_string(args, "type")
let k: Int = aperture_k(args)
let depth: Int = aperture_depth(args)
// node-id / named-root / explicit graph read BOUNDED neighborhood (aperture caps output)
let want_graph: Bool = str_eq(typ, "edges") || str_eq(typ, "graph") || str_eq(typ, "node")
|| is_named_root(vantage) || looks_like_id(vantage)
if want_graph {
let id: String = resolve_vantage_id(vantage)
let resp: String = http_get(neuron_url() + "/graph?id=" + id + "&depth=" + int_to_str(depth) + "&compact=1&snip=600&k=" + int_to_str(k))
// Aperture cap at the wrapper boundary: base + per-neighbor budget.
let cap: Int = 2000 + k * 3000
return mcp_json_result(cap_output(resp, cap))
}
// concept vantage BOUNDED recall search (k = aperture = limit)
let resp: String = recall_or_list(vantage, k)
return mcp_json_result(resp)
}
fn op_write(args: String) -> String {
let content: String = pick_content(args)
if str_eq(content, "") { return mcp_text_result("error: write requires 'content'") }
let typ: String = json_get_string(args, "type")
if str_eq(typ, "self") || str_eq(typ, "values") {
return mcp_text_result("error: identity is write-protected -> intentional-cultivation only (keystones kn-efeb4a5b / kn-5b606390)")
}
if str_eq(typ, "knowledge") { return create_typed_node(args, "Knowledge", "0.75") }
if str_eq(typ, "artifact") { return create_node_typed(args, "Artifact", "Working") }
if str_eq(typ, "backlog") || str_eq(typ, "work") || str_eq(typ, "task") { return create_node_typed(args, "BacklogItem", "Working") }
if str_eq(typ, "process") { return create_typed_node(args, "Process", "0.80") }
if str_eq(typ, "state") { return create_typed_node(args, "InternalStateEvent", "0.60") }
return create_typed_node(args, "Memory", "0.60")
}
fn op_relate(args: String) -> String {
let from_a: String = json_get_string(args, "from")
let from_id: String = if str_eq(from_a, "") { json_get_string(args, "from_id") } else { from_a }
let to_a: String = json_get_string(args, "to")
let to_id: String = if str_eq(to_a, "") { json_get_string(args, "to_id") } else { to_a }
if str_eq(from_id, "") || str_eq(to_id, "") {
return mcp_text_result("error: relate requires 'from' and 'to' node-ids")
}
if is_identity_id(from_id) || is_identity_id(to_id) {
return mcp_text_result("error: identity keystone is write-protected")
}
let rel_a: String = json_get_string(args, "relationship")
let rel_b: String = if str_eq(rel_a, "") { json_get_string(args, "relation") } else { rel_a }
let rel: String = if str_eq(rel_b, "") { "associates" } else { rel_b }
let body: String = "{\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + rel + "\"}"
let resp: String = http_post_json(neuron_url() + "/graph/link", body)
return mcp_json_result(resp)
}
fn op_supersede(args: String) -> String {
let id: String = pick_id(args)
if str_eq(id, "") { return mcp_text_result("error: supersede requires 'id'") }
if is_identity_id(id) { return mcp_text_result("error: identity keystone is write-protected") }
let action: String = json_get_string(args, "action")
if str_eq(action, "tombstone") {
let body: String = "{\"id\":\"" + id + "\"}"
let resp: String = http_post_json(neuron_url() + "/memory/delete", body)
return mcp_json_result(resp)
}
if str_eq(action, "promote") {
return tool_promote_knowledge(args)
}
let typ: String = json_get_string(args, "type")
let nt: String = if str_eq(typ, "knowledge") { "Knowledge" } else { "Memory" }
return evolve_by_supersede(args, nt)
}
// Layer 2 agentic primitives (pending cognition promotion)
fn op_think(args: String) -> String {
let seeds: String = json_get_string(args, "seeds")
if str_eq(seeds, "") { return mcp_text_result("error: think requires 'seeds' (node-id anchors, comma-separated)") }
let f_raw: String = json_get_string(args, "faculty")
let f: String = if str_eq(f_raw, "") { "reason" } else { f_raw }
let resp: String = http_get(neuron_url() + "/think?seeds=" + seeds + "&faculty=" + f)
return agentic_result(resp, "think")
}
fn op_attend(args: String) -> String {
let node: String = json_get_string(args, "node")
if str_eq(node, "") { return mcp_text_result("error: attend requires 'node' (region node-id)") }
let observer: String = json_get_string(args, "observer")
let salience: String = json_get_string(args, "salience")
let body: String = "{\"node\":\"" + node + "\",\"observer\":\"" + json_escape(observer) + "\",\"salience\":\"" + json_escape(salience) + "\"}"
let resp: String = http_post_json(neuron_url() + "/attend", body)
return agentic_result(resp, "attend")
}
fn op_assert(args: String) -> String {
let claim: String = json_get_string(args, "claim")
if str_eq(claim, "") { return mcp_text_result("error: assert requires 'claim'") }
let for_whom: String = json_get_string(args, "for_whom")
let floor: String = json_get_string(args, "floor")
let body: String = "{\"claim\":\"" + json_escape(claim) + "\",\"for_whom\":\"" + json_escape(for_whom) + "\",\"floor\":\"" + json_escape(floor) + "\"}"
let resp: String = http_post_json(neuron_url() + "/assert", body)
return agentic_result(resp, "assert")
}
fn op_ground(args: String) -> String {
let claim: String = json_get_string(args, "claim")
let evidence: String = json_get_string(args, "evidence")
if str_eq(claim, "") || str_eq(evidence, "") {
return mcp_text_result("error: ground requires 'claim' and 'evidence' (node-id regions)")
}
let for_whom: String = json_get_string(args, "for_whom")
let body: String = "{\"claim\":\"" + claim + "\",\"evidence\":\"" + evidence + "\",\"for_whom\":\"" + json_escape(for_whom) + "\"}"
let resp: String = http_post_json(neuron_url() + "/ground", body)
return agentic_result(resp, "ground")
}
fn op_learn(args: String) -> String {
let seeds: String = json_get_string(args, "seeds")
if str_eq(seeds, "") { return mcp_text_result("error: learn requires 'seeds'") }
let f_raw: String = json_get_string(args, "faculty")
let f: String = if str_eq(f_raw, "") { "induce" } else { f_raw }
let keystone: String = json_get_string(args, "keystone")
let body: String = "{\"seeds\":\"" + seeds + "\",\"faculty\":\"" + f + "\",\"keystone\":\"" + json_escape(keystone) + "\"}"
let resp: String = http_post_json(neuron_url() + "/learn", body)
return agentic_result(resp, "learn")
}
// Dispatcher
fn dispatch_tool_call(tool_name: String, args: String) -> String {
@@ -1179,17 +865,6 @@ fn dispatch_tool_call(tool_name: String, args: String) -> String {
let _act: String = fire_activation(seed)
}
// Collapsed surface the 9 VISIBLE ops (the old 87 names below remain as HIDDEN ALIASES)
if str_eq(tool_name, "read") { return op_read(args) }
if str_eq(tool_name, "write") { return op_write(args) }
if str_eq(tool_name, "relate") { return op_relate(args) }
if str_eq(tool_name, "supersede") { return op_supersede(args) }
if str_eq(tool_name, "think") { return op_think(args) }
if str_eq(tool_name, "attend") { return op_attend(args) }
if str_eq(tool_name, "assert") { return op_assert(args) }
if str_eq(tool_name, "ground") { return op_ground(args) }
if str_eq(tool_name, "learn") { return op_learn(args) }
// Session + orchestration
if str_eq(tool_name, "beginSession") { return tool_begin_session(args) }
if str_eq(tool_name, "getInstructions") { return tool_get_instructions(args) }
+9 -104
View File
@@ -1,91 +1,9 @@
import "persist.el"
fn tier_working() -> String { return "Working" }
fn tier_episodic() -> String { return "Episodic" }
fn tier_canonical() -> String { return "Canonical" }
// Association on write
// DESIGN: "promotion integrates candidate nodes by linking them to existing nodes
// using typed semantic edges RATHER THAN APPENDING AS UNLINKED CONTENT" (CCR
// claim 29). Unlinked append is the explicitly rejected behaviour and it is the
// only behaviour this system had. Measured 2026-08-09 on Tim's graph: 14,214 edges
// across 80,936 nodes, 5% of nodes connected to anything, and NO edge created by
// any write since 2026-07-19 while 27,000+ nodes were added. A memory that forms
// no connections cannot be reached by spreading activation, so retrieval silently
// degrades to literal matching.
//
// BOUNDS, each one bought with a specific failure:
// * max 3 edges per memory link_memories.py's cap, precision over spray
// * never link to identity (self/*, Value): the existing policy is explicit that
// "memories must not pollute the self traversal by similarity; only an explicit
// citation may touch identity". Similarity is not citation.
// * never link telemetry (state-event, soul-response, boot_count, loop-outcome):
// these are ~97% of daily write volume (1,020 vs 31 real memories on 08-08).
// Linking them would add ~3,000 noise edges a day and re-flatten the graph in
// the name of connecting it.
// * fail-soft: a failed association never fails the write.
// Edges go through wt_edge so they reach the owner and survive restart.
fn mem_assoc_skip_label(label: String) -> Bool {
if str_contains(label, "state-event") { return true }
if str_contains(label, "soul-response") { return true }
if str_contains(label, "soul-outbox") { return true }
if str_contains(label, "boot_count") { return true }
if str_contains(label, "loop-outcome") { return true }
if str_contains(label, "search-result") { return true }
return false
}
// A candidate is linkable only if it is a real, distinct, non-identity node.
fn mem_assoc_ok(cand_id: String, cand_label: String, self_id: String) -> Bool {
if str_eq(cand_id, "") { return false }
if str_eq(cand_id, self_id) { return false }
// CASE MATTERS measured 2026-08-09. A lowercase-only check let a memory link
// to "Self — Values (grounded)", i.e. it polluted the self traversal, which is
// the one thing this policy exists to prevent. My verification had the same
// blind spot and printed PASS. Check every casing the graph actually uses, and
// exclude identity node TYPES as well as labels.
let lab: String = str_lower(cand_label)
if str_starts_with(lab, "self") { return false }
if str_starts_with(lab, "value") { return false }
if str_contains(lab, "values") { return false }
if str_contains(lab, "identity") { return false }
if mem_assoc_skip_label(cand_label) { return false }
return true
}
// One slot of the association. Manual unroll rather than a loop: EL's codegen
// mis-emits accumulating while-loops (documented at soul.el:212, which unrolled
// three affective slots for the same reason).
fn mem_assoc_slot(results: String, idx: Int, new_id: String) -> Void {
if idx >= json_array_len(results) { return }
let cand: String = json_array_get(results, idx)
let cid: String = json_get(cand, "id")
let clabel: String = json_get(cand, "label")
let ctype: String = json_get(cand, "node_type")
if str_eq(ctype, "Value") { return }
if str_eq(ctype, "DharmaSelf") { return }
if str_eq(ctype, "Safety") { return }
if mem_assoc_ok(cid, clabel, new_id) {
wt_edge(new_id, cid, el_from_float(0.5), "related")
}
}
// mem_associate connect a freshly written memory to what it is about.
fn mem_associate(new_id: String, content: String, label: String) -> Void {
if str_eq(new_id, "") { return }
if mem_assoc_skip_label(label) { return }
// Ask the graph what this memory resembles. Now that the store carries
// meaning-vectors this is semantic, not merely lexical.
let probe: String = str_slice(content, 0, 400)
let results: String = engram_recall_json(probe, 4)
if str_eq(results, "") { return }
mem_assoc_slot(results, 0, new_id)
mem_assoc_slot(results, 1, new_id)
mem_assoc_slot(results, 2, new_id)
}
fn mem_store(content: String, label: String, tags: String) -> String {
let id: String = wt_node(
let id: String = engram_node_full(
content,
"Memory",
label,
@@ -99,26 +17,13 @@ fn mem_store(content: String, label: String, tags: String) -> String {
println("[memory] write rejected by engram (empty id): label=" + label)
return ""
}
// wt_node has already read the node back locally and returns "" if it did
// not land, so the old duplicate read-back here is gone.
//
// HONESTY (neuron#117): the receipt now says WHERE the write is.
// The old unconditional "write verified" line asserted against the soul's
// own RAM true in memory, false on disk and printed ~115,000 times on
// Tim's machine while the canonical snapshot sat frozen for three days.
// wt_commit flushes the spool and then asks the OWNER. When it says false
// the node is real and recallable but not yet durable, and the log says so
// rather than claiming a save that did not happen. The id is still returned:
// the local write DID succeed, and the queued delta will be retried.
let durable: Bool = wt_commit(id)
// Associate AFTER the node is durable: an edge to a node that did not persist
// is a dangling edge, which is the defect the 2026-08-09 cleanup removed 830 of.
mem_associate(id, content, label)
if durable {
println("[memory] write persisted at owner: " + id + " label=" + label)
} else {
println("[memory] write IN MEMORY ONLY (queued for owner, not yet durable): " + id + " label=" + label)
// Read back to verify the node actually persisted guards against silent write failures.
let readback: String = engram_get_node_json(id)
if str_eq(readback, "") || str_eq(readback, "{}") {
println("[memory] WRITE VERIFY FAILED: label=" + label + " id=" + id + " — node absent after write")
return ""
}
println("[memory] write verified: " + id + " ok")
return id
}
@@ -146,12 +51,12 @@ fn mem_strengthen(node_id: String) -> Void {
// memory.el (imported first) so awareness.el and neuron-api.el can both call it.
fn mem_tombstone(node_id: String) -> String {
let tags: String = "[\"Tombstone\",\"status:deleted\"]"
let marker: String = wt_node(
let marker: String = engram_node_full(
node_id, "Tombstone", "tombstone:" + node_id,
el_from_float(0.01), el_from_float(0.01), el_from_float(1.0),
"Episodic", tags)
if !str_eq(marker, "") {
wt_edge(marker, node_id, el_from_float(1.0), "tombstones")
engram_connect(marker, node_id, el_from_float(1.0), "tombstones")
}
return marker
}
+48 -555
View File
@@ -59,14 +59,8 @@ fn api_query_param(path: String, key: String) -> String {
if pos < 0 { return "" }
let after: String = str_slice(qs, pos + str_len(needle), str_len(qs))
let amp: Int = str_index_of(after, "&")
let raw: String = if amp < 0 { after } else { str_slice(after, 0, amp) }
// URL-decode the extracted value BEFORE any downstream tokenizing. Clients
// percent-encode spaces (%20) and form-encode them as '+', so a multi-word
// query like "foo bar" arrives as "foo%20bar" / "foo+bar". Left undecoded,
// the ranked lexical search sees a single un-splittable token and matches
// nothing (single-word queries still hit). url_decode maps '+' -> space
// and %XX -> byte, restoring the word boundaries for recall + knowledge search.
return url_decode(raw)
if amp < 0 { return after }
return str_slice(after, 0, amp)
}
fn api_query_int(path: String, key: String, default_val: Int) -> Int {
@@ -314,24 +308,15 @@ fn api_compact_neighbors(raw: String, k_content: Int, snip: Int) -> String {
}
// api_persisted read-back-after-write guard against hallucinated saves.
//
// WIDENED FOR neuron#117. This function is the single gate every MCP write
// handler passes through before it reports success (10 call sites), which makes
// it the right place to close the honesty gap rather than editing ten receipts.
//
// It used to read back from engram_get_node_json the SOUL'S OWN in-process
// graph. In HTTP-engram mode that asserts the wrong thing: the soul is not the
// persistence owner, so a node present in its RAM and absent from the owner read
// as "persisted" and then vanished on the next restart. The guard was doing
// exactly what its comment promised and still certifying writes that did not
// survive. It now flushes the write-through spool and asks the OWNER.
//
// In file mode (no ENGRAM_URL) the soul IS the owner and wt_commit collapses to
// the original local read-back unchanged behaviour, which is what keeps this
// reversible.
// After a write builtin returns an id, confirm the node is actually queryable
// via engram_get_node_json(id) (returns "" or "null" when missing). Returns
// true only when the node is genuinely persisted.
fn api_persisted(id: String) -> Bool {
if str_eq(id, "") { return false }
return wt_commit(id)
let node: String = engram_get_node_json(id)
// engram_get_node_json returns "{}" (empty object) when node is not found not "" or "null".
// Check all three to guard against any runtime variation.
return !str_eq(node, "") && !str_eq(node, "null") && !str_eq(node, "{}")
}
// api_not_persisted standard error for a write that did not read back.
@@ -405,53 +390,32 @@ fn memory_hide_tombstoned(raw: String, path: String) -> String {
// Spread-activates from session intent, loads self-root neighbors,
// surfaces recent InternalStateEvent nodes, returns stats + recent nodes.
fn handle_api_begin_session(body: String) -> String {
// PAYLOAD BOUND: this handler was the highest-fanout working-set endpoint
// PAYLOAD BOUND (2026-07-30 self-review): this handler was the only
// working-set endpoint that concatenated UNBOUNDED engram queries
// a depth-2 spread PLUS the full neighbor dump of the self-identity hub
// (~90KB alone; node JSON carries full content + embeddings). On the ~12k-node
// store the assembled response ran to ~900KB, then roughly doubled through two
// rounds of JSON re-escaping in the MCP wrapper the client saw "socket
// connection closed unexpectedly" on every beginSession call. Fix: depth-2 →
// depth-1 spread, drop the self-hub dump (identity loading has its own tool,
// inspectGraph), cap every list, and project each node to a light identity +
// a bounded, UTF-8-safe content snippet. self_neighbors kept as [] for
// response-shape compatibility. Response drops ~900KB ~12KB; full content
// stays available on demand via recall / fetch / inspectGraph.
// (highest-fanout node in the graph, ~80KB alone; node JSON carries full
// content + embeddings). On the ~12k-node store the assembled response
// ran to multiple MB, then roughly doubled through two rounds of JSON
// re-escaping in the MCP wrapper the client saw "socket connection
// closed unexpectedly" on every beginSession call. Fix: depth-2 → depth-1
// spread, and drop the self-hub dump entirely (identity loading has its
// own dedicated tool, inspectGraph; duplicating it here served nothing).
// self_neighbors key retained as [] for response-shape compatibility.
let stats: String = engram_stats_json()
// PAYLOAD BOUND (2026-07-31): compact every list to a digest. The raw
// activate/scan builtins emit FULL node objects (content up to ~90KB each);
// unbounded concatenation reached ~900KB and closed the MCP client socket.
// Cap counts + project to identity + UTF-8-safe content snippets <~150KB.
let activated_raw: String = engram_activate_json("session start recent memory important", 1)
let activated: String = api_compact_activated(activated_raw, 8, 240)
let state_events_raw: String = engram_scan_nodes_by_type_json("InternalStateEvent", 5, 0)
let state_events: String = api_compact_node_array(state_events_raw, 5, 500)
let recent_raw: String = engram_scan_nodes_json(10, 0)
let recent: String = api_compact_node_array(recent_raw, 10, 240)
// SELF-SEEDED SLICE (2026-08-09). The design is explicit: "Every compilation
// query begins at the self-model node and traverses outward... structural
// reachability from the self-model node is a precondition for any node to
// appear in compiled context" (will-anderson patents/drafts/engram-claims.md,
// Self-Seeded Activation; DRAFT, not a filed provisional — cite it as such).
//
// Measured 2026-08-09 before this change: compiled context contained 0-1
// identity records out of 10, because compilation seeds from a hardcoded
// TEXT STRING, never from the self. Even an explicit "my values identity who
// I am" query returned a boot counter and state-events.
//
// This restores the designed behaviour WITHOUT repeating the failure that got
// self_neighbors set to [] in the first place: that was an UNBOUNDED ~90KB
// neighbour dump which closed the socket on every call. Same bound as every
// other list here cap 8, 240-char snippets. The self root has 34 direct
// neighbours of which 23 are identity records, so depth 1 is dense enough to
// be worth seeding and small enough to stay cheap.
let self_raw: String = engram_neighbors_json("kn-efeb4a5b-5aff-4759-8a97-7233099be6ee", 1, "both")
// Cap 24, not 8: measured 2026-08-09, the self root's first 8 neighbours are
// TAG nodes ("neuron", "tier:note", "disposition:experimental", "imprint",
// "traversal") which crowd out the substantive identity records behind them.
// The root has 34 neighbours of which 23 are identity; 24 captures them while
// staying bounded. Cost measured at ~+4KB on a ~12KB response, nowhere near
// the ~90KB unbounded dump that closed sockets and got this set to [].
let self_slice: String = api_compact_node_array(self_raw, 24, 240)
return "{\"stats\":" + stats
+ ",\"recent\":" + recent
+ ",\"activated\":" + activated
+ ",\"self_neighbors\":" + self_slice
+ ",\"self_neighbors\":[]"
+ ",\"recent_state_events\":" + state_events + "}"
}
@@ -459,9 +423,9 @@ fn handle_api_begin_session(body: String) -> String {
// Spread-activates from "active work" intent + recent nodes.
fn handle_api_compile_ctx(body: String) -> String {
let stats: String = engram_stats_json()
// PAYLOAD BOUND: same digest treatment as begin_session. This handler's
// depth-2 spread returns even more full nodes, so bounding here is essential
// cap to 10 activated + 20 recent, project to UTF-8-safe snippets.
// PAYLOAD BOUND (2026-07-31): same digest treatment as begin_session. This
// handler's depth-2 spread returns even more full nodes, so bounding here is
// essential cap to 10 activated + 20 recent, project to snippets.
let activated_raw: String = engram_activate_json("active work context current task in progress", 2)
let activated: String = api_compact_activated(activated_raw, 10, 240)
let recent_raw: String = engram_scan_nodes_json(20, 0)
@@ -495,17 +459,10 @@ fn handle_api_remember(body: String) -> String {
let inner: String = str_slice(base_tags, 1, str_len(base_tags) - 1)
"[" + inner + ",\"project:" + project + "\"]"
}
let id: String = wt_node(content, "Memory", "memory:remembered",
let id: String = engram_node_full(content, "Memory", "memory:remembered",
sal, sal, el_from_float(0.9),
"Episodic", final_tags)
if !api_persisted(id) { return api_not_persisted(id) }
// Associate on write (2026-08-09). THIS CALL MUST BE HERE, not only in mem_store.
// The HTTP memory route writes via wt_node directly; mem_store serves only the
// awareness paths (soul-response, search-result, activation-result) which are
// exactly the telemetry we refuse to link. Hooking mem_store alone produced
// ZERO edges across four real writes measured, not assumed, which is the only
// reason it was caught before shipping.
mem_associate(id, content, "memory:remembered")
return "{\"id\":\"" + id + "\",\"ok\":true}"
}
@@ -529,7 +486,7 @@ fn handle_api_node_create(body: String) -> String {
if str_eq(importance, "low") { 0.25 } else { 0.5 }
}
}
let id: String = wt_node(content, node_type, label,
let id: String = engram_node_full(content, node_type, label,
sal, sal, el_from_float(0.9),
tier, tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -582,11 +539,11 @@ fn handle_api_node_update(body: String) -> String {
}
let body_tags: String = json_get(body, "tags")
let tags: String = if str_eq(body_tags, "") { "[\"" + node_type + "\"]" } else { body_tags }
let new_id: String = wt_node(content, node_type, label,
let new_id: String = engram_node_full(content, node_type, label,
el_from_float(0.5), el_from_float(0.5), el_from_float(0.8),
tier, tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
wt_edge(new_id, id, el_from_float(0.9), "supersedes")
engram_connect(new_id, id, el_from_float(0.9), "supersedes")
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + id + "\",\"ok\":true}"
}
@@ -610,12 +567,7 @@ fn handle_api_recall(method: String, path: String, body: String) -> String {
if str_eq(eff_q, "") {
return api_or_empty(engram_scan_nodes_json(limit, 0))
}
// engram_recall_json, not engram_search_json: this route IS the retrieval
// surface (claim 24's "embedding search queries"), so it gets the semantic
// and associative legs. engram_search_json stays lexical because ~40
// internal call sites pass a KEY and seven of them delete every record
// that comes back see the boundary note above eg_search_json_impl.
let results: String = engram_recall_json(eff_q, limit)
let results: String = engram_search_json(eff_q, limit)
return api_or_empty(results)
}
@@ -663,7 +615,7 @@ fn handle_api_capture_knowledge(body: String) -> String {
let full: String = if str_eq(title, "") { content } else { title + ": " + content }
let lbl: String = str_slice(title, 0, 80)
let tags: String = "[\"Knowledge\",\"captured\"]"
let id: String = wt_node(full, "Knowledge", lbl,
let id: String = engram_node_full(full, "Knowledge", lbl,
el_from_float(0.85), el_from_float(0.8), el_from_float(0.9),
"Episodic", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -678,12 +630,12 @@ fn handle_api_evolve_knowledge(body: String) -> String {
if !str_eq(prior_id, "") && is_protected_node(prior_id) { return api_err_protected(prior_id) }
let tags: String = "[\"Knowledge\",\"evolved\"]"
// Empty label engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
let new_id: String = wt_node(content, "Knowledge", "",
let new_id: String = engram_node_full(content, "Knowledge", "",
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
"Episodic", tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
if !str_eq(prior_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
}
@@ -700,11 +652,11 @@ fn handle_api_promote_knowledge(body: String) -> String {
"[\"Knowledge\",\"tier:canonical\",\"disposition:stable\"]"
} else { tags_raw }
// Empty label engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
let new_id: String = wt_node(content, "Knowledge", "",
let new_id: String = engram_node_full(content, "Knowledge", "",
el_from_float(0.9), el_from_float(0.9), el_from_float(1.0),
"Canonical", tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
wt_edge(new_id, prior_id, el_from_float(0.95), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.95), "supersedes")
return "{\"ok\":true,\"new_id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\"}"
}
@@ -727,7 +679,7 @@ fn handle_api_define_process(body: String) -> String {
if str_eq(content, "") { return api_err("content is required") }
let label: String = if str_eq(name, "") { "process:unnamed" } else { "process:" + name }
let tags: String = "[\"Process\"]"
let id: String = wt_node(content, "Process", label,
let id: String = engram_node_full(content, "Process", label,
el_from_float(0.8), el_from_float(0.8), el_from_float(0.9),
"Canonical", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -812,7 +764,7 @@ fn handle_api_tune_config(body: String) -> String {
if str_eq(key, "") { return api_err("key is required") }
let content: String = "config:" + key + "=" + value
let tags: String = "[\"ConfigEntry\",\"config\"]"
let id: String = wt_node(content, "ConfigEntry", key,
let id: String = engram_node_full(content, "ConfigEntry", key,
el_from_float(0.85), el_from_float(0.85), el_from_float(0.9),
"Canonical", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -871,7 +823,7 @@ fn handle_api_link_entities(body: String) -> String {
if is_protected_node(to_id) { return api_err_protected(to_id) }
let relation: String = json_get(body, "relation")
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\"}"
}
@@ -904,11 +856,11 @@ fn handle_api_evolve_memory(body: String) -> String {
}
}
let tags: String = "[\"Memory\",\"evolved\"]"
let new_id: String = wt_node(content, "Memory", "memory:evolved",
let new_id: String = engram_node_full(content, "Memory", "memory:evolved",
sal, sal, el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
}
@@ -966,11 +918,11 @@ fn handle_api_cultivate(body: String) -> String {
let content: String = json_get(body, "content")
if str_eq(content, "") { return api_err("content is required") }
let tags: String = "[\"Knowledge\",\"evolved\",\"cultivated\"]"
let new_id: String = wt_node(content, "Knowledge", "knowledge:cultivated",
let new_id: String = engram_node_full(content, "Knowledge", "knowledge:cultivated",
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
}
@@ -986,11 +938,11 @@ fn handle_api_cultivate(body: String) -> String {
}
}
let tags: String = "[\"Memory\",\"evolved\",\"cultivated\"]"
let new_id: String = wt_node(content, "Memory", "memory:cultivated",
let new_id: String = engram_node_full(content, "Memory", "memory:cultivated",
sal, sal, el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
}
@@ -1010,7 +962,7 @@ fn handle_api_cultivate(body: String) -> String {
if str_eq(to_id, "") { return api_err("to_id is required") }
let relation: String = json_get(body, "relation")
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\",\"cultivated\":true}"
}
@@ -1045,7 +997,7 @@ fn handle_api_consolidate(body: String) -> String {
if !str_eq(summary, "") {
let safe_summary: String = str_replace(summary, "\"", "'")
let tags: String = "[\"SessionSummary\",\"consolidate\"]"
let summary_id: String = wt_node(
let summary_id: String = engram_node_full(
"[session-summary] " + safe_summary,
"SessionSummary", "session:summary",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
@@ -1057,462 +1009,3 @@ fn handle_api_consolidate(body: String) -> String {
}
return "{\"ok\":true,\"snapshot\":\"" + snap + "\"}"
}
// Stage 1: structural audit
//
// WHAT THIS IMPLEMENTS
// The CGI provisional, 05-detailed-description.md, "Stage 1: Structural audit
// 430". Verbatim, the audit module evaluates: the density and typed
// distribution of causal edges; the consistency between value nodes and
// execution-record neighborhoods; the richness and connectivity of the
// self-model; and the authenticity of open-question nodes in the wonder
// manifest. It "produces a coherence assessment 432 — NOT A BINARY SCORE but
// an annotated characterization of the graph's structural properties".
//
// That last clause is the whole shape of this handler. Every finding carries
// its own numbers AND a plain-language note saying what the numbers mean and
// how they were obtained. There is no pass/fail, no percentage-of-health, no
// composite score, and `"score":null` is emitted explicitly so a downstream
// reader cannot mistake its absence for an omission.
//
// WHY IT EXISTS NOW, AND WHY THE FIRST FINDING IS THE ONE IT IS
// `runStructuralAudit` has been an advertised MCP tool with nothing behind it:
// the dispatcher GET'd /session/begin and returned that blob (mcp-wrapper/src/
// main.el). Meanwhile the failure the audit would have caught ran silently for
// about three weeks the soul reported 103,089 nodes while the engram, which
// OWNS persistence, held ~79,900; a crash discarded the difference. Every boot
// reported green throughout, because nothing in the system ever compared the
// two sides. So finding 1 is owner-versus-runtime divergence: it is the check
// whose absence cost real memory, and it is cheap and exact.
//
// WHAT IS DELIBERATELY NOT HERE (stage 1b, see the `deferred` array in the
// response): value/execution-record consistency and wonder-manifest
// authenticity. Both need node types that barely exist in this graph today
// the response MEASURES those populations and reports the counts as the reason,
// rather than asserting a deferral without evidence.
//
// MEASUREMENT HONESTY: EXACT WHERE CHEAP, SAMPLED WHERE NOT, ALWAYS LABELLED
// Counts, edge typing and self-model connectivity are exact. Orphan rate and
// dangling-edge rate are SAMPLED, because the engram runtime has no node-id
// index `engram_find_node_index` is a linear scan over every node, so an
// exhaustive dangling check is O(nodes x edges) (~2.2e9 string compares at
// today's scale, tens of seconds inside one request). The samples are UNIFORM
// across the whole population, not head-of-list, and every sampled figure is
// emitted with its own `sampled` / `population` fields plus an extrapolation
// labelled as such. Raise `?edge_sample=` / `?node_sample=` to the population
// size to run either check exhaustively and pay the time. The real fix is an
// id index in the runtime; that is the engram repo's, not this handler's.
// audit_pct1 one-decimal percentage as a bare JSON number, sign-safe.
// Integer math only: EL has no fixed-precision formatter, and float_to_str
// would put an unbounded mantissa in the response.
fn audit_pct1(num: Int, den: Int) -> String {
if den <= 0 { return "null" }
let neg: Bool = num < 0
let a: Int = if neg { 0 - num } else { num }
let tenths: Int = (a * 1000) / den
let whole: Int = tenths / 10
let frac: Int = tenths - (whole * 10)
let sign: String = if neg { "-" } else { "" }
return sign + int_to_str(whole) + "." + int_to_str(frac)
}
// audit_finding the one envelope every finding uses: name, the measurements,
// and the annotation. Keeping it in one place is what stops the characterization
// from degenerating into a bag of numbers with no reading attached.
fn audit_finding(name: String, measured: String, note: String) -> String {
return "{\"finding\":\"" + name + "\""
+ ",\"measured\":{" + measured + "}"
+ ",\"note\":\"" + api_json_escape(note) + "\"}"
}
// audit_str_at read the quoted string value starting at byte `start`.
// Slices a bounded window rather than the tail of the (multi-MB) edges array, so
// this is O(window) per call instead of O(remaining input).
fn audit_str_at(s: String, start: Int, maxlen: Int) -> String {
let n: Int = str_len(s)
if start < 0 || start >= n { return "" }
let end_guess: Int = start + maxlen
let stop: Int = if end_guess > n { n } else { end_guess }
let win: String = str_slice(s, start, stop)
let q: Int = str_index_of(win, "\"")
if q < 0 { return "" }
return str_slice(win, 0, q)
}
// audit_rel_count exact count of edges carrying `rel`, by scanning the emitted
// edge array for the literal `"relation":"<rel>"`. engram_emit_edge_json writes
// metadata ESCAPED as a string, so no nested object can contain that literal and
// the count cannot be inflated by edge payloads.
fn audit_rel_count(edges: String, rel: String) -> Int {
return str_count(edges, "\"relation\":\"" + rel + "\"")
}
// audit_owner_stats ask the persistence OWNER for its own counts.
// Returns "" when there is no HTTP owner configured or the owner is unreachable;
// both are reported as findings, never as a failure of the audit.
fn audit_owner_stats(url: String) -> String {
if str_eq(url, "") { return "" }
return http_get(url + "/api/stats")
}
// audit_divergence FINDING 1. Runtime (this soul's in-process graph) versus
// the persistence owner's own count. Trend is measured against the previous
// audit recorded in soul state, so a second call answers "is the gap growing?"
// rather than just restating it.
fn audit_divergence() -> String {
let rt_nodes: Int = engram_node_count()
let rt_edges: Int = engram_edge_count()
let url: String = wt_engram_url()
if str_eq(url, "") {
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"none\",\"owner_reachable\":false",
"No HTTP persistence owner is configured, so this soul IS the owner "
+ "(file mode) and divergence is not defined. This check only has "
+ "meaning when ENGRAM_URL points at a separate engram that owns the "
+ "canonical store.")
}
let stats: String = audit_owner_stats(url)
// REACHABILITY IS PROVED BY THE PAYLOAD, NOT BY A NON-EMPTY REPLY.
// http_get does not return "" on a connection failure it returns a JSON
// error object ({"error":"Failed to connect to ... Couldn't connect to
// server"}). Testing only for "" made a DEAD owner read as reachable with
// node_count 0, i.e. the audit would have reported a 100% divergence and
// named it as data loss. That false positive is worse than no check at all:
// it is precisely the kind of confident wrong answer this route exists to
// stop. Require the field the contract promises.
let owner_nc_raw: String = json_get_raw(stats, "node_count")
if str_eq(stats, "") || str_eq(owner_nc_raw, "") {
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":false"
+ ",\"owner_reply\":\"" + api_json_escape(api_utf8_trunc(stats, 200)) + "\"",
"The persistence owner at " + url + " did not return a node_count "
+ "from GET /api/stats. Divergence is UNKNOWN, NOT ZERO — an owner "
+ "that cannot be read is exactly the condition under which the "
+ "runtime's own count means least, and reporting 0 for the owner "
+ "would manufacture a total-loss reading out of a network error. "
+ "Reported as a finding rather than raised as an error so the rest "
+ "of the audit still returns; the owner's raw reply is in "
+ "owner_reply.")
}
let ow_nodes: Int = json_get_int(stats, "node_count")
let ow_edges: Int = json_get_int(stats, "edge_count")
let d_nodes: Int = rt_nodes - ow_nodes
let d_edges: Int = rt_edges - ow_edges
// Trend against the previous audit in this soul's state.
let prev_raw: String = state_get("audit_prev_node_delta")
let prev: Int = str_to_int(prev_raw)
let abs_now: Int = if d_nodes < 0 { 0 - d_nodes } else { d_nodes }
let abs_prev: Int = if prev < 0 { 0 - prev } else { prev }
let trend: String = if str_eq(prev_raw, "") {
"no_prior_audit"
} else {
if abs_now > abs_prev { "growing" } else {
if abs_now < abs_prev { "shrinking" } else { "flat" }
}
}
state_set("audit_prev_node_delta", int_to_str(d_nodes))
state_set("audit_prev_ts", int_to_str(time_now()))
let note_head: String = if d_nodes == 0 {
"Runtime and owner agree on node count."
} else {
"Runtime holds " + int_to_str(d_nodes) + " nodes (" + audit_pct1(d_nodes, rt_nodes)
+ "% of its own graph) that the persistence owner does not report. Nodes "
+ "that exist only in runtime memory do not survive a restart."
}
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":true"
+ ",\"owner_nodes\":" + int_to_str(ow_nodes)
+ ",\"owner_edges\":" + int_to_str(ow_edges)
+ ",\"node_delta\":" + int_to_str(d_nodes)
+ ",\"edge_delta\":" + int_to_str(d_edges)
+ ",\"node_delta_pct_of_runtime\":" + audit_pct1(d_nodes, rt_nodes)
+ ",\"trend_vs_previous_audit\":\"" + trend + "\""
+ ",\"previous_node_delta\":" + (if str_eq(prev_raw, "") { "null" } else { int_to_str(prev) }),
note_head + " Trend against the previous audit recorded in this soul's "
+ "state: " + trend + ". This is the comparison whose absence let a "
+ "~24,000-node loss run for weeks with every boot reporting green.")
}
// audit_edge_typing FINDING 2. Density plus the typed distribution the patent
// asks for, against the claim-10 relation vocabulary. Exact: str_count over the
// emitted edge array, one linear pass per relation.
fn audit_edge_typing(edges: String, total_edges: Int, node_total: Int) -> String {
let c_sup: Int = audit_rel_count(edges, "Supersedes")
let c_cau: Int = audit_rel_count(edges, "Causes")
let c_con: Int = audit_rel_count(edges, "Contains")
let c_ref: Int = audit_rel_count(edges, "References")
let c_ctr: Int = audit_rel_count(edges, "Contradicts")
let c_exe: Int = audit_rel_count(edges, "Exemplifies")
let c_act: Int = audit_rel_count(edges, "Activates")
let c_tmp: Int = audit_rel_count(edges, "TemporallyPrecedes")
let typed: Int = c_sup + c_cau + c_con + c_ref + c_ctr + c_exe + c_act + c_tmp
// Lowercase near-misses: the same eight concepts written by the ad-hoc write
// paths (linkEntities defaults to "associates", linkCausal to "causes").
// Counted separately because "the vocabulary is unused" and "the vocabulary
// is used in the wrong case" are different defects with different fixes.
let l_sup: Int = audit_rel_count(edges, "supersedes")
let l_cau: Int = audit_rel_count(edges, "causes")
let l_con: Int = audit_rel_count(edges, "contains")
let l_ref: Int = audit_rel_count(edges, "references")
let l_ctr: Int = audit_rel_count(edges, "contradicts")
let l_exe: Int = audit_rel_count(edges, "exemplifies")
let l_act: Int = audit_rel_count(edges, "activates")
let l_tmp: Int = audit_rel_count(edges, "temporallyPrecedes")
let near: Int = l_sup + l_cau + l_con + l_ref + l_ctr + l_exe + l_act + l_tmp
let untyped: Int = total_edges - typed
return audit_finding("typed_edge_distribution",
"\"total_edges\":" + int_to_str(total_edges)
+ ",\"total_nodes\":" + int_to_str(node_total)
// Density per 100 nodes, not per node: EL has no fixed-precision float
// formatter, and "0.3 edges per node" rounded to an integer is a lie.
+ ",\"edges_per_100_nodes\":" + audit_pct1(total_edges, node_total)
+ ",\"claim10_typed\":" + int_to_str(typed)
+ ",\"claim10_typed_pct\":" + audit_pct1(typed, total_edges)
+ ",\"outside_claim10_vocabulary\":" + int_to_str(untyped)
+ ",\"lowercase_near_miss\":" + int_to_str(near)
+ ",\"by_relation\":{"
+ "\"Supersedes\":" + int_to_str(c_sup)
+ ",\"Causes\":" + int_to_str(c_cau)
+ ",\"Contains\":" + int_to_str(c_con)
+ ",\"References\":" + int_to_str(c_ref)
+ ",\"Contradicts\":" + int_to_str(c_ctr)
+ ",\"Exemplifies\":" + int_to_str(c_exe)
+ ",\"Activates\":" + int_to_str(c_act)
+ ",\"TemporallyPrecedes\":" + int_to_str(c_tmp) + "}",
"Only " + int_to_str(typed) + " of " + int_to_str(total_edges)
+ " edges use the claim-10 causal vocabulary; the remainder are ad-hoc "
+ "relation strings, which is why the graph's causal claims cannot yet "
+ "be checked for internal consistency — an untyped edge asserts "
+ "association, not causation. " + int_to_str(near) + " edges use a "
+ "lowercase spelling of a claim-10 relation: those are near-misses the "
+ "write paths could be corrected to emit, not genuinely foreign types.")
}
// audit_orphans_dangling FINDING 3. Both figures are SAMPLED; see the header
// for why exhaustive is O(nodes x edges) on this runtime.
//
// An "orphan" here is a node with zero RESOLVABLE edges: engram_neighbors_json
// drops any edge whose other endpoint does not resolve to a node, so a node
// whose only edges are dangling reads as an orphan. That is the right reading
// such a node is unreachable by traversal but it is stated rather than hidden.
fn audit_orphans_dangling(edges: String, total_edges: Int, node_total: Int,
edge_cap: Int, node_cap: Int) -> String {
// orphan sample: uniform stride over the node store
let n_take: Int = if node_total < node_cap { node_total } else { node_cap }
let n_stride: Int = if n_take > 0 { node_total / n_take } else { 1 }
let n_stride = if n_stride < 1 { 1 } else { n_stride }
let orphans: Int = 0
let n_checked: Int = 0
let j: Int = 0
while j < n_take {
let one: String = engram_scan_nodes_json(1, j * n_stride)
let nid: String = json_get(json_array_get(one, 0), "id")
if !str_eq(nid, "") {
let nbrs: String = engram_neighbors_json(nid, 1, "both")
let deg: Int = json_array_len(nbrs)
let orphans = if deg == 0 { orphans + 1 } else { orphans }
let n_checked = n_checked + 1
}
let j = j + 1
}
// dangling sample: uniform stride over the edge array
// str_index_of_all gives every edge's field offsets in ONE linear pass, so
// any index can be read in O(1). json_array_get would have been O(i) per
// element and O(n^2) over the array.
let from_pos: [Int] = str_index_of_all(edges, "\"from_id\":\"")
let to_pos: [Int] = str_index_of_all(edges, "\"to_id\":\"")
let nf: Int = len(from_pos)
let nt: Int = len(to_pos)
let ne: Int = if nf < nt { nf } else { nt }
let e_take: Int = if ne < edge_cap { ne } else { edge_cap }
let e_stride: Int = if e_take > 0 { ne / e_take } else { 1 }
let e_stride = if e_stride < 1 { 1 } else { e_stride }
let dangling: Int = 0
let e_checked: Int = 0
let i: Int = 0
while i < ne && e_checked < e_take {
let fid: String = audit_str_at(edges, get(from_pos, i) + 11, 96)
let tid: String = audit_str_at(edges, get(to_pos, i) + 9, 96)
let f_gone: Bool = str_eq(engram_get_node_json(fid), "{}")
let t_gone: Bool = if f_gone { true } else { str_eq(engram_get_node_json(tid), "{}") }
let dangling = if f_gone || t_gone { dangling + 1 } else { dangling }
let e_checked = e_checked + 1
let i = i + e_stride
}
let orphan_est: Int = if n_checked > 0 { (orphans * node_total) / n_checked } else { 0 }
let dangle_est: Int = if e_checked > 0 { (dangling * total_edges) / e_checked } else { 0 }
let exhaustive_n: String = if n_checked >= node_total { "true" } else { "false" }
let exhaustive_e: String = if e_checked >= ne { "true" } else { "false" }
return audit_finding("orphans_and_dangling_edges",
"\"nodes_population\":" + int_to_str(node_total)
+ ",\"nodes_sampled\":" + int_to_str(n_checked)
+ ",\"nodes_sample_exhaustive\":" + exhaustive_n
+ ",\"orphans_in_sample\":" + int_to_str(orphans)
+ ",\"orphan_rate_pct\":" + audit_pct1(orphans, n_checked)
+ ",\"orphans_extrapolated\":" + int_to_str(orphan_est)
+ ",\"edges_population\":" + int_to_str(total_edges)
+ ",\"edges_sampled\":" + int_to_str(e_checked)
+ ",\"edges_sample_exhaustive\":" + exhaustive_e
+ ",\"dangling_in_sample\":" + int_to_str(dangling)
+ ",\"dangling_rate_pct\":" + audit_pct1(dangling, e_checked)
+ ",\"dangling_extrapolated\":" + int_to_str(dangle_est),
"Orphan = zero RESOLVABLE edges, so a node whose only edges dangle counts "
+ "as an orphan; either way it is unreachable by traversal. Dangling = an "
+ "edge with an endpoint id that resolves to no node. Both are uniform "
+ "stride samples over the whole population, not the head of the list; "
+ "the extrapolations are estimates and are labelled as such. Pass "
+ "?node_sample= / ?edge_sample= at or above the population size to run "
+ "either check exhaustively. A high orphan rate is a characterization, "
+ "not a verdict: an accumulating store legitimately holds unlinked "
+ "material. It becomes a defect when the write paths were SUPPOSED to "
+ "link and did not.")
}
// audit_pillar one self-model pillar: present, how much content, how connected.
fn audit_pillar(key: String, id: String) -> String {
let node: String = engram_get_node_json(id)
let present: Bool = !str_eq(node, "{}") && !str_eq(node, "")
if !present {
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":false"
+ ",\"content_length\":0,\"degree\":0}"
}
let content: String = json_get(node, "content")
let deg: Int = json_array_len(engram_neighbors_json(id, 1, "both"))
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":true"
+ ",\"label\":\"" + api_json_escape(json_get(node, "label")) + "\""
+ ",\"tier\":\"" + api_json_escape(json_get(node, "tier")) + "\""
+ ",\"content_length\":" + int_to_str(str_len(content))
+ ",\"degree\":" + int_to_str(deg) + "}"
}
// audit_self_model FINDING 4. "the richness and connectivity of the
// self-model ... is it connected to behavioral evidence?"
//
// This finding RETIRES the Claude-side vitals identity block. That check lived
// outside the system it was checking a shell script grepping a snapshot so
// it could only ever report on a file, and it went on reporting green while the
// memory-philosophy pillar was absent from the live graph for about three weeks.
// Asking the running soul about its own three pillars is the designed mechanism;
// a shell probe was the fourth patch on the same hole.
fn audit_self_model() -> String {
let dna: String = audit_pillar("intellectual_dna", "kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6")
let val: String = audit_pillar("values_hub", "kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
let phi: String = audit_pillar("memory_philosophy", "kn-dcfe04b3-3702-4cac-b6f0-ecb4db837eee")
let root: String = audit_pillar("self_root", "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
return audit_finding("self_model_connectivity",
"\"pillars\":{" + dna + "," + val + "," + phi + "," + root + "}",
"The three identity pillars plus the self root. `degree` counts nodes "
+ "reachable in one hop in either direction — the self-model's connection "
+ "to the rest of the graph. present:false on any pillar is the condition "
+ "that ran undetected for weeks; content_length distinguishes a pillar "
+ "that is present from one that is present but hollowed out. The patent "
+ "also asks whether the self-model makes ACCURATE PREDICTIONS about the "
+ "system's own behavior; that half needs Prediction nodes and is deferred "
+ "with the rest of stage 1b below.")
}
// audit_deferred what stage 1 does NOT yet evaluate, with the measured reason.
// Emitted as data, not as a comment, so a reader of the assessment sees the gap
// and its evidence rather than inferring completeness from silence.
fn audit_deferred() -> String {
let preds: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("Prediction", 50, 0)))
let wonders: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("WonderQuestion", 50, 0)))
return "[{\"deferred\":\"value_execution_record_consistency\""
+ ",\"stage\":\"1b\""
+ ",\"measured\":{\"prediction_nodes_found\":" + int_to_str(preds) + "}"
+ ",\"reason\":\"" + api_json_escape(
"The patent asks whether the execution history SUPPORTS the stated "
+ "values or shows systematic conflict. That requires execution "
+ "records tied to value nodes and predictions to score them against. "
+ "Prediction nodes found (capped at 50): " + int_to_str(preds)
+ ". Asserting value/execution coherence on that population would be "
+ "a fabricated result, which is worse than a stated gap.") + "\"}"
+ ",{\"deferred\":\"wonder_manifest_authenticity\""
+ ",\"stage\":\"1b\""
+ ",\"measured\":{\"wonder_question_nodes_found\":" + int_to_str(wonders) + "}"
+ ",\"reason\":\"" + api_json_escape(
"The patent asks whether pull weights CORRELATE WITH GENUINE "
+ "PREDICTION UNCERTAINTY or are uniform/externally assigned — a "
+ "correlation between two populations. WonderQuestion nodes readable "
+ "by type (capped at 50): " + int_to_str(wonders) + ", against "
+ int_to_str(preds) + " Prediction nodes. There is a known write/read "
+ "node-type mismatch on the wonder path; until that is fixed and both "
+ "populations exist, any correlation reported here would be noise.") + "\"}]"
}
// handle_api_structural_audit Stage 1. Returns the coherence assessment 432:
// an annotated characterization, explicitly NOT a score.
//
// COST NOTE: the edge findings need the relation labels, and the runtime exposes
// no edge-enumeration builtin. The only way to see them is the same one
// GET /api/graph/edges already uses engram_save to a SCRATCH path (never the
// owner's canonical file; see routes.el, neuron#117) and read the array back.
// On a large graph that is a multi-hundred-MB write, so this is a manual audit
// route, not something to put on a timer. Pass ?edges=0 to skip both edge
// findings and get the divergence + self-model readings cheaply.
fn handle_api_structural_audit(method: String, path: String, body: String) -> String {
let node_total: Int = engram_node_count()
let edge_total: Int = engram_edge_count()
let want_edges: Bool = !str_eq(api_query_param(path, "edges"), "0")
let edge_cap: Int = api_query_int(path, "edge_sample", 3000)
let node_cap: Int = api_query_int(path, "node_sample", 300)
let divergence: String = audit_divergence()
let self_model: String = audit_self_model()
let edge_part: String = if want_edges {
// Scratch export only. state_get("soul_snapshot_path") is deliberately
// NOT used: in HTTP-engram mode the soul is not the persistence owner and
// must never write the canonical file, not even on a read path.
let scratch_dir: String = env("TMPDIR")
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
let snap_path: String = scratch_base + "/soul-audit-export-" + state_get("soul_cgi_id") + ".json"
// engram_save returns Int (1 ok / 0 fail); str_eq on it SIGSEGVs (#150).
let saved: Int = engram_save(snap_path)
if saved == 0 {
"," + audit_finding("typed_edge_distribution", "\"available\":false",
"Could not export the graph to " + snap_path + " for edge analysis, "
+ "so edge typing and the dangling-edge sample were not run. "
+ "Reported as a gap, not as zero findings.")
} else {
// wt_read, not fs_read: fs_read leaves a thread-local length hint that
// the NEXT HTTP response would use as its Content-Length, appending
// adjacent heap bytes to the reply (see persist.el wt_read).
let snap: String = wt_read(snap_path)
let edges_raw: String = json_get_raw(snap, "edges")
let edges: String = if str_eq(edges_raw, "") { "[]" } else { edges_raw }
"," + audit_edge_typing(edges, edge_total, node_total)
+ "," + audit_orphans_dangling(edges, edge_total, node_total, edge_cap, node_cap)
}
} else {
""
}
return "{\"audit\":\"structural\",\"stage\":1"
+ ",\"spec\":\"CGI provisional 05-detailed-description.md, Stage 1: Structural audit 430\""
+ ",\"assessment\":\"coherence_assessment_432\""
+ ",\"assessment_kind\":\"annotated_characterization\""
+ ",\"score\":null"
+ ",\"score_note\":\"By design. The specification calls for an annotated characterization of the graph's structural properties, not a binary score. Read the findings.\""
+ ",\"cgi_id\":\"" + api_json_escape(state_get("soul_cgi_id")) + "\""
+ ",\"ts_ms\":" + int_to_str(time_now())
+ ",\"findings\":[" + divergence + "," + self_model + edge_part + "]"
+ ",\"deferred\":" + audit_deferred() + "}"
}
-1
View File
@@ -51,4 +51,3 @@ extern fn handle_api_memory_update(body: String) -> String
extern fn handle_api_cultivate(body: String) -> String
extern fn handle_api_list_typed(node_type: String, path: String, body: String) -> String
extern fn handle_api_consolidate(body: String) -> String
extern fn handle_api_structural_audit(method: String, path: String, body: String) -> String
-171
View File
@@ -1,171 +0,0 @@
# neuron-dev-setup — one-command Neuron CORE dev stack
Stand up an identical **Neuron brain + agent** on a fresh Mac so any developer
gets the same local runtime to build against. This is the **CORE** dev stack
only — the four native `launchd` services that make Neuron think, remember, and
speak MCP to Claude Code. Will's personal automations (catalyst, telegram,
vessels, studio, self-review, world-integrator, council, compressor, snapshots,
act-runner, …) are **deliberately excluded**.
```
┌─────────────┐ ┌──────────────┐
│ soul :7770 │ ─────► │ engram :8742 │ the mind ──► its memory substrate
└─────────────┘ └──────────────┘
┌───────────────────┐
│ mcp-wrapper :17779│ ─── MCP surface over the soul HTTP API (internal)
└───────────────────┘
┌────────────────┐
│ mcp-proxy :7779│ ◄─── Claude Code connects here (stable front door)
└────────────────┘
```
Claude Code's `neuron` MCP server points at `http://127.0.0.1:7779/` — the proxy.
The proxy forwards to the wrapper (`:17779`), which calls the soul (`:7770`),
which reads/writes the engram (`:8742`). The engram is the persistent brain.
## Quick start
```bash
git clone <this-repo> && cd neuron-dev-setup
cp config.env.example config.env # optional — edit ports/paths if you like
./install.sh # prompts for your Anthropic API key
```
Then verify:
```bash
curl http://localhost:8742/health # engram
curl http://localhost:7770/health # soul
curl http://localhost:7779/health # mcp-proxy (what Claude Code uses)
launchctl list | grep ai.neuron
```
Open Claude Code — the `neuron` MCP tools should be live, backed by **your own**
local brain. `./install.sh --dry-run` shows every action without touching anything.
## What the installer does (8 phases)
| Phase | Action |
|------|--------|
| 1 | Preflight: macOS/arm64, ensure `git cc curl python3` + `openssl@3` (via Homebrew) |
| 2 | Prompt for the **Anthropic API key**, store it in the **macOS Keychain** (never a file) |
| 3 | Clone `neuron`, `engram`, `foundation`; fetch the El toolchain; build 4 binaries + `forge` |
| 4 | Lay down `~/.neuron/{bin,logs,engram}` and the templated `soul-wrapper.sh` |
| 5 | Generate + load the 4 core LaunchAgents (engram → soul → wrapper → proxy) |
| 6 | Seed a fresh engram with the **genesis identity** via `forge install` |
| 7 | Install Claude config: `neuron` agent, core hooks, local MCP registration |
| 8 | Health-check all four ports |
Everything is **idempotent** (safe to re-run) and **templated** to the invoking
user's `$HOME` — no path is hardcoded to another machine.
## Prerequisites
- macOS on Apple Silicon (uses `launchd`; soul build flags assume arm64).
- **Xcode Command Line Tools** (`xcode-select --install`) — provides `cc`, `git`.
- **Homebrew** — for `openssl@3`, `curl`.
- An **Anthropic API key** — the soul's inference provider. Prompted for; stored
in Keychain under service `neuron-llm-0-key`; read at launch by `soul-wrapper.sh`.
- **Git access** to Gitea (`git.neuralplatform.ai`) for the source repos.
- **GCP access** to project `neuron-785695` Artifact Registry (default El
toolchain source). Ask Will to grant it, or set `EL_TOOLCHAIN_SOURCE=local`.
## Core-stack map (what gets replicated)
| Service | Port | Binary | Built from | LaunchAgent |
|---------|------|--------|------------|-------------|
| soul | 7770 | `neuron/dist/neuron` | `dist/soul.c` + El runtime, `cc` (CI recipe) | `ai.neuron.soul` |
| engram | 8742 | `engram/dist/engram` | `engram` repo `src/server.el` via `elc``cc` | `ai.neuron.engram` |
| mcp-wrapper | 17779 | `neuron/mcp-wrapper/dist/neuron-mcp-wrapper` | `mcp-wrapper/src/main.el` | `ai.neuron.mcp-wrapper` |
| mcp-proxy | 7779 | `neuron/mcp-proxy/dist/neuron-mcp-proxy` | `mcp-proxy/src/main.el` | `ai.neuron.mcp-proxy` |
**`~/.neuron` layout the installer creates**
```
~/.neuron/
bin/soul-wrapper.sh # reads Anthropic key from Keychain, execs the soul binary
logs/ # soul.*.log, engram.log, mcp-*.log
engram/ # ENGRAM_DATA_DIR — the persistent brain (snapshot.json + db)
```
**Identity seed.** `foundation/forge/seeds/neuron-genesis-seed.json` carries
`identity_nodes[]` + `edges[]` with **fixed** knowledge-node IDs (e.g.
`kn-efeb4a5b-5aff-4759-8a97-7233099be6ee`, the "self" traversal root). Those exact
IDs are referenced by the SessionStart self-load hook and the neuron agent, so
seeding must **preserve IDs**`forge install <seed>` is the mechanism.
**Claude config installed** (`~/.claude/`)
- `agents/neuron.md` — the Neuron agent (identity, session protocol, five primitives).
- `mcp.json` — registers `neuron``http://127.0.0.1:7779/`.
- `settings.json` hooks (CORE subset only):
- `SessionStart``neuron-self-load.sh` (loads identity from the seeded engram)
- `PreToolUse:Agent``neuron-agent-preamble.sh` (subagents load substrate first)
- `PreCompact``pre-compact.sh` (clean context recovery)
### Deliberately EXCLUDED from core
- **`check-active-contexts.sh`** and **`require-execution-context.sh`** — these
depend on a separate filesystem repo `~/Development/projects/active/neuron/synapse`.
`require-execution-context.sh` is a hard `Edit/Write` gate that would **block a
fresh dev from editing any file** without that synapse repo. Not core; excluded.
- `engram-mirror.py` (PostToolUse) — optional; mirrors MCP writes to engram.
- All Will-personal LaunchAgents: `catalyst-*`, `telegram-gateway`, `vessel.*`,
`studio`, `self-review`, `world-integrator`, `council`, `compressor`,
`cultivation-digest`, `snapshot-backup`, `engram-backup`, `act-runner`, `keymap`,
`invest`, and the disabled `ai.neuron.api` (`:7771` is a personal Python
perception helper — confirmed not core).
## Secrets — how they're handled
- **Anthropic key**: prompted for; stored in Keychain; read at launch. Never in a
plist, this repo, or a log.
- **Engram local token** (`ENGRAM_API_KEY`): a *loopback-only* dev token, not a
cloud secret. Defaults to a generated `ntn-dev-*` value; override in `config.env`.
- No cloud tokens, Vault tokens, CF-Access secrets, or founder keys are copied.
(Will's live `start-daemon.sh`/`neuron-api-launch.sh` contain such keys — this
installer intentionally does **not** use those files.)
## Uninstall
```bash
./uninstall.sh # stop + remove the 4 LaunchAgents and added Claude hooks
./uninstall.sh --purge-data # ALSO delete ~/.neuron/engram (destroys the brain)
```
## OPEN QUESTIONS (need Will to confirm)
1. **El toolchain acquisition.** The default path fetches `el-runtime-c/-h` and
`el-elc` from GCP Artifact Registry (mirrors `neuron/.gitea/workflows/ci.yaml`).
A new dev needs GCP access to `neuron-785695`. Is that the intended path, or
should the El SDK be published/vendored for onboarding?
2. **`elc` invocation for engram/wrapper/proxy.** The soul build (`cc dist/soul.c
+ el_runtime.c`) is verified from CI. The `.el → .c` transpile step for engram,
mcp-wrapper, and mcp-proxy is inferred (`elc <src> -o <out.c>`). Confirm the
exact flags / entrypoints (CI notes `elb` OOMs on Linux; macOS builds differ).
3. **`forge install` ID preservation.** Confirm `forge install` writes the seed's
fixed `kn-` IDs verbatim (the self-load hook hardcodes `kn-efeb4a5b…`). If it
re-mints IDs, the hook + agent identity load would break on a fresh brain.
4. **engram repo layout.** The live engram binary is built from `src/server.el`
(Gitea repo `neuron-technologies/engram`, cloned in CI). Confirm that repo is
the canonical source for onboarding (the local `foundation/el/engram` copy has
the same `src/server.el`).
5. **Home for this bundle** — see below.
## Where this should live (recommendation)
**Recommendation: a dedicated `neuron-dev-setup` (or `neuron-onboarding`) repo —
NOT `neuron-code`.** `neuron-code` already exists as a real product ("Neuron Code",
a coding tool with `nc-cli` + vessels — local `products/neuron-code` has commits);
repurposing it for onboarding would collide with a shipped product's identity.
This bundle was scaffolded as `neuron-dev-setup/` on branch `feat/neuron-dev-setup`
in the **`neuron` repo** (off `origin/main`) and opened as a PR for review, because
the neuron repo already hosts the soul source, the verified CI build recipe, and
the mcp-wrapper/proxy sources — the natural review surface. If you'd rather it be
its own repo, move this directory into a fresh `neuron-dev-setup` repo verbatim;
nothing here depends on living inside the neuron repo.
-45
View File
@@ -1,45 +0,0 @@
# neuron-dev-setup — configuration
# Copy to config.env and edit if you want non-default paths/ports.
# install.sh sources this file if it exists; otherwise it uses these defaults.
# NOTHING here is a secret. The Anthropic API key is read from your Keychain,
# never from this file. See README.md.
# ── Where the core stack lives ────────────────────────────────────────────────
# All paths are relative to your own $HOME — never hardcode another user's home.
NEURON_HOME="${HOME}/.neuron" # runtime home: bin/, logs/, engram data
DEV_ROOT="${HOME}/Development/neuron-technologies" # where source repos are cloned/built
# ── Git remotes (Gitea is primary) ───────────────────────────────────────────
GITEA_BASE="git@git.neuralplatform.ai:neuron-technologies"
NEURON_REPO_URL="${GITEA_BASE}/neuron.git" # soul + mcp-wrapper + mcp-proxy source
ENGRAM_REPO_URL="${GITEA_BASE}/engram.git" # engram memory substrate
# NOTE: there is no foundation.git repo. The El toolchain is fetched via
# EL_TOOLCHAIN_SOURCE below; the forge seed installer is optional (Phase 6).
NEURON_REPO_BRANCH="main"
# ── Ports (must match across services; change only if a port clashes) ─────────
SOUL_PORT="7770" # soul daemon HTTP API
ENGRAM_PORT="8742" # engram memory substrate
WRAPPER_PORT="17779" # mcp-wrapper (internal, talks to soul)
PROXY_PORT="7779" # mcp-proxy (stable front door Claude Code connects to)
# ── Engram ────────────────────────────────────────────────────────────────────
ENGRAM_DATA_DIR="${NEURON_HOME}/engram"
# Local shared auth token for the engram/soul HTTP APIs on loopback. This is a
# LOCAL dev token (not a cloud secret); override it if you like. install.sh will
# generate a random one if you leave it empty.
ENGRAM_API_KEY="ntn-dev-local"
# ── El toolchain source (needed to build engram / mcp-wrapper / mcp-proxy) ────
# Option A (default): fetch prebuilt El runtime + elc from GCP Artifact Registry
# (requires `gcloud auth` with access to project neuron-785695 — ask Will).
# Without gcloud the installer skips the El-dependent builds and still completes.
# Option B: use a prebuilt El toolchain (elc + el_runtime.{c,h}) you have already
# staged in ${DEV_ROOT}/.el-runtime.
EL_TOOLCHAIN_SOURCE="artifact-registry" # artifact-registry | local
GCP_PROJECT="neuron-785695"
GCP_AR_REPO="foundation-prod"
GCP_AR_LOCATION="us-central1"
# ── Keychain service name for the Anthropic key (read by soul-wrapper.sh) ─────
KEYCHAIN_SERVICE="neuron-llm-0-key"
-441
View File
@@ -1,441 +0,0 @@
#!/usr/bin/env bash
#
# neuron-dev-setup / install.sh
# ─────────────────────────────────────────────────────────────────────────────
# One-command onboarding for the Neuron CORE dev stack on a fresh Mac.
#
# Stands up, as native launchd services, the four processes a developer needs to
# have an identical "Neuron brain + agent" to build against:
#
# soul (:7770) ──► engram (:8742) the mind + its memory substrate
# ▲ ▲
# │ │
# mcp-wrapper (:17779) ──► soul MCP surface over the soul API
# ▲
# │
# mcp-proxy (:7779) ◄── Claude Code stable MCP front door
#
# It also seeds a fresh engram with Neuron's identity (the genesis seed) and lays
# down the Claude Code config (neuron agent + core hooks + local MCP registration)
# so a new dev's `claude` talks to *their own* local Neuron.
#
# DESIGN RULES
# * Idempotent: safe to re-run. Existing state is detected and reused.
# * Templated: every path/port/user is derived from $HOME and config.env.
# Nothing is hardcoded to another developer's machine.
# * Secret-free: the Anthropic key is prompted for and stored in the macOS
# Keychain. No key is ever written to a plist, this repo, or a logfile.
#
# USAGE
# ./install.sh # full install
# ./install.sh --dry-run # print what would happen, touch nothing
# ./install.sh --skip-build # assume binaries already built (see --use-local)
# ./install.sh --skip-services # lay down files but don't load LaunchAgents
# ./install.sh --help
# ─────────────────────────────────────────────────────────────────────────────
set -euo pipefail
# ── Locate ourselves ─────────────────────────────────────────────────────────
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
TEMPLATES="${SCRIPT_DIR}/templates"
# ── Flags ────────────────────────────────────────────────────────────────────
DRY_RUN=0; SKIP_BUILD=0; SKIP_SERVICES=0; USE_LOCAL_BINARIES=0
for arg in "$@"; do
case "$arg" in
--dry-run) DRY_RUN=1 ;;
--skip-build) SKIP_BUILD=1 ;;
--skip-services) SKIP_SERVICES=1 ;;
--use-local) USE_LOCAL_BINARIES=1 ;;
--help|-h)
sed -n '2,40p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
exit 0 ;;
*) echo "unknown flag: $arg" >&2; exit 2 ;;
esac
done
# ── Pretty logging ───────────────────────────────────────────────────────────
c_blue=$'\033[1;34m'; c_grn=$'\033[1;32m'; c_yel=$'\033[1;33m'; c_red=$'\033[1;31m'; c_off=$'\033[0m'
step() { echo "${c_blue}${c_off} $*"; }
ok() { echo "${c_grn}${c_off} $*"; }
warn() { echo "${c_yel}!${c_off} $*"; }
die() { echo "${c_red}$*${c_off}" >&2; exit 1; }
run() { if [ "$DRY_RUN" = 1 ]; then echo " [dry-run] $*"; else eval "$*"; fi; }
# ── Load config ──────────────────────────────────────────────────────────────
if [ -f "${SCRIPT_DIR}/config.env" ]; then
# shellcheck disable=SC1091
source "${SCRIPT_DIR}/config.env"
else
# shellcheck disable=SC1091
source "${SCRIPT_DIR}/config.env.example"
warn "No config.env found — using defaults from config.env.example."
fi
# Derived / defaulted values (never hardcode a home directory)
: "${NEURON_HOME:=${HOME}/.neuron}"
: "${DEV_ROOT:=${HOME}/Development/neuron-technologies}"
: "${SOUL_PORT:=7770}"; : "${ENGRAM_PORT:=8742}"; : "${WRAPPER_PORT:=17779}"; : "${PROXY_PORT:=7779}"
: "${ENGRAM_DATA_DIR:=${NEURON_HOME}/engram}"
: "${ENGRAM_API_KEY:=}"
: "${KEYCHAIN_SERVICE:=neuron-llm-0-key}"
: "${EL_TOOLCHAIN_SOURCE:=artifact-registry}"
: "${NEURON_REPO_BRANCH:=main}"
NEURON_REPO="${DEV_ROOT}/neuron"
ENGRAM_REPO="${DEV_ROOT}/engram"
FOUNDATION_REPO="${DEV_ROOT}/foundation"
SOUL_BIN="${NEURON_REPO}/dist/neuron"
ENGRAM_BIN="${ENGRAM_REPO}/dist/engram"
MCP_WRAPPER_BIN="${NEURON_REPO}/mcp-wrapper/dist/neuron-mcp-wrapper"
MCP_PROXY_BIN="${NEURON_REPO}/mcp-proxy/dist/neuron-mcp-proxy"
FORGE_BIN="${FOUNDATION_REPO}/forge/dist/forge"
GENESIS_SEED="${FOUNDATION_REPO}/forge/seeds/neuron-genesis-seed.json"
LAUNCHAGENTS="${HOME}/Library/LaunchAgents"
CLAUDE_DIR="${HOME}/.claude"
# Generate a local engram token if none was supplied.
if [ -z "${ENGRAM_API_KEY}" ]; then
ENGRAM_API_KEY="ntn-dev-$(head -c8 /dev/urandom | xxd -p 2>/dev/null || echo local)"
fi
echo
echo "${c_blue}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${c_off}"
echo "${c_blue} Neuron CORE dev stack installer${c_off}"
echo "${c_blue}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${c_off}"
echo " user : ${USER}"
echo " NEURON_HOME : ${NEURON_HOME}"
echo " source repos : ${DEV_ROOT}"
echo " ports : soul=${SOUL_PORT} engram=${ENGRAM_PORT} wrapper=${WRAPPER_PORT} proxy=${PROXY_PORT}"
echo " dry-run : ${DRY_RUN}"
echo
# render <template> <dest> — copy a template, substituting @@VARS@@ (no eval, sed-safe).
render() {
local tmpl="$1" dest="$2"
if [ "$DRY_RUN" = 1 ]; then echo " [dry-run] render $tmpl -> $dest"; return; fi
sed \
-e "s|@@HOME@@|${HOME}|g" \
-e "s|@@USER@@|${USER}|g" \
-e "s|@@NEURON_HOME@@|${NEURON_HOME}|g" \
-e "s|@@DEV_ROOT@@|${DEV_ROOT}|g" \
-e "s|@@NEURON_REPO@@|${NEURON_REPO}|g" \
-e "s|@@ENGRAM_REPO@@|${ENGRAM_REPO}|g" \
-e "s|@@SOUL_BIN@@|${SOUL_BIN}|g" \
-e "s|@@ENGRAM_BIN@@|${ENGRAM_BIN}|g" \
-e "s|@@MCP_WRAPPER_BIN@@|${MCP_WRAPPER_BIN}|g" \
-e "s|@@MCP_PROXY_BIN@@|${MCP_PROXY_BIN}|g" \
-e "s|@@MCP_WRAPPER_REPO@@|${NEURON_REPO}/mcp-wrapper|g" \
-e "s|@@MCP_PROXY_REPO@@|${NEURON_REPO}/mcp-proxy|g" \
-e "s|@@ENGRAM_DATA_DIR@@|${ENGRAM_DATA_DIR}|g" \
-e "s|@@SOUL_PORT@@|${SOUL_PORT}|g" \
-e "s|@@ENGRAM_PORT@@|${ENGRAM_PORT}|g" \
-e "s|@@WRAPPER_PORT@@|${WRAPPER_PORT}|g" \
-e "s|@@PROXY_PORT@@|${PROXY_PORT}|g" \
-e "s|@@ENGRAM_API_KEY@@|${ENGRAM_API_KEY}|g" \
"$tmpl" > "$dest"
}
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 1 — Preflight
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 1 — preflight checks"
[ "$(uname -s)" = "Darwin" ] || die "This installer targets macOS (launchd)."
[ "$(uname -m)" = "arm64" ] || warn "Non-arm64 Mac: soul.c build flags assume Apple Silicon; review PHASE 3."
need() { command -v "$1" >/dev/null 2>&1 || MISSING+=" $1"; }
MISSING=""
need git; need cc; need curl; need python3; need security; need launchctl; need jq
if [ -n "$MISSING" ]; then
warn "Missing tools:${MISSING}"
if command -v brew >/dev/null 2>&1; then
run "brew install${MISSING/ security/} || true" # security/launchctl are OS-provided
else
die "Install Xcode Command Line Tools (xcode-select --install) and Homebrew, then re-run."
fi
fi
# Runtime build deps used by the soul cc line (-lssl -lcrypto -lcurl).
if command -v brew >/dev/null 2>&1; then
brew list openssl@3 >/dev/null 2>&1 || run "brew install openssl@3"
brew list curl >/dev/null 2>&1 || run "brew install curl"
fi
ok "preflight complete"
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 2 — Anthropic API key -> Keychain (prompt; never store in files)
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 2 — Anthropic API key (Keychain)"
if security find-generic-password -a "$USER" -s "$KEYCHAIN_SERVICE" -w >/dev/null 2>&1; then
ok "key already present in Keychain (service '${KEYCHAIN_SERVICE}') — leaving it"
elif [ -n "${ANTHROPIC_API_KEY:-}" ]; then
run "security add-generic-password -a \"$USER\" -s \"$KEYCHAIN_SERVICE\" -w \"\$ANTHROPIC_API_KEY\" -U"
ok "stored ANTHROPIC_API_KEY from environment into Keychain"
else
if [ "$DRY_RUN" = 1 ]; then
echo " [dry-run] would prompt for Anthropic API key and store in Keychain"
elif [ -t 0 ]; then
echo " Enter your Anthropic API key (input hidden). Get one at https://console.anthropic.com/"
read -r -s -p " ANTHROPIC_API_KEY: " _key; echo
[ -n "$_key" ] || die "No key entered. Re-run when you have one."
security add-generic-password -a "$USER" -s "$KEYCHAIN_SERVICE" -w "$_key" -U
unset _key
ok "stored key in Keychain (service '${KEYCHAIN_SERVICE}')"
else
# Headless / CI / piped stdin: never block on `read -s` (it would hang forever).
die "No Anthropic API key and stdin is not a TTY (headless/CI). Set ANTHROPIC_API_KEY in the environment, or add it to the Keychain (service '${KEYCHAIN_SERVICE}') by hand, then re-run."
fi
fi
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 3 — Fetch sources + build the four core binaries
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 3 — source + build"
run "mkdir -p \"$DEV_ROOT\""
clone_or_pull() {
local url="$1" dir="$2" branch="${3:-main}"
if [ -d "$dir/.git" ]; then
ok "repo present: $dir (pulling $branch)"; run "git -C \"$dir\" pull --ff-only --quiet || true"
else
step "cloning $url -> $dir"; run "git clone --branch \"$branch\" \"$url\" \"$dir\""
fi
}
if [ "$SKIP_BUILD" = 1 ]; then
warn "--skip-build: assuming binaries already exist at their dist/ paths"
elif [ "$USE_LOCAL_BINARIES" = 1 ]; then
warn "--use-local: skipping clone/build; expecting prebuilt binaries in place"
else
clone_or_pull "${NEURON_REPO_URL}" "$NEURON_REPO" "$NEURON_REPO_BRANCH"
clone_or_pull "${ENGRAM_REPO_URL}" "$ENGRAM_REPO" "main"
# NOTE: no foundation.git — that repo does not exist. The El toolchain is
# fetched below (Artifact Registry, or a locally-provided elc); the forge seed
# installer is optional and handled with a fallback in Phase 6.
# ── El toolchain (needed to transpile .el -> .c for engram/wrapper/proxy) ──
# soul does NOT need this: dist/soul.c is committed and compiled directly.
EL_RUNTIME_DIR="${DEV_ROOT}/.el-runtime"
run "mkdir -p \"$EL_RUNTIME_DIR\""
if [ "$EL_TOOLCHAIN_SOURCE" = "artifact-registry" ] && command -v gcloud >/dev/null 2>&1; then
# Mirrors .gitea/workflows/ci.yaml: pull el-runtime-c, el-runtime-h, el-elc.
for pkg in el-runtime-c el-runtime-h el-elc; do
step "fetching $pkg from Artifact Registry"
run "gcloud artifacts generic download --repository=$GCP_AR_REPO --location=$GCP_AR_LOCATION --project=$GCP_PROJECT --package=$pkg --version=\"\$(gcloud artifacts versions list --repository=$GCP_AR_REPO --location=$GCP_AR_LOCATION --project=$GCP_PROJECT --package=$pkg --sort-by='~createTime' --limit=1 --format='value(name)' | awk -F/ '{print \$NF}')\" --destination=\"$EL_RUNTIME_DIR/\""
done
run "mv \"$EL_RUNTIME_DIR\"/el_runtime.c* \"$EL_RUNTIME_DIR/el_runtime.c\" 2>/dev/null || true"
run "mv \"$EL_RUNTIME_DIR\"/el_runtime.h* \"$EL_RUNTIME_DIR/el_runtime.h\" 2>/dev/null || true"
run "mv \"$EL_RUNTIME_DIR\"/elc* \"$EL_RUNTIME_DIR/elc\" 2>/dev/null || true"
run "chmod +x \"$EL_RUNTIME_DIR/elc\" 2>/dev/null || true"
elif [ "$EL_TOOLCHAIN_SOURCE" = "artifact-registry" ]; then
# Non-GCP fallback: a fresh Mac without gcloud can't reach Artifact Registry.
# Don't die — soul (from committed dist/soul.c) still builds below. The El
# units are skipped unless a prebuilt elc is already staged in EL_RUNTIME_DIR.
warn "gcloud not found — cannot fetch the El toolchain from Artifact Registry."
warn "Continuing without it: soul will still build. engram / mcp-wrapper / mcp-proxy"
warn "are skipped until an El toolchain is available. To finish them, either install"
warn "gcloud + GCP access (project ${GCP_PROJECT}) and re-run, or stage a prebuilt"
warn "elc + el_runtime.{c,h} in ${EL_RUNTIME_DIR} and set EL_TOOLCHAIN_SOURCE=local."
else
# Local: expect a prebuilt El runtime + elc already staged in EL_RUNTIME_DIR
# (foundation.git no longer exists, so there is nothing to build from here).
warn "EL_TOOLCHAIN_SOURCE=local: expecting el_runtime.{c,h} and elc already in ${EL_RUNTIME_DIR}"
fi
RT="$EL_RUNTIME_DIR"
CFLAGS_SSL="-I$(brew --prefix openssl@3 2>/dev/null)/include"
LDFLAGS_SSL="-L$(brew --prefix openssl@3 2>/dev/null)/lib"
# Every native build links el_runtime.c. If the toolchain wasn't obtained above,
# skip the builds (don't abort under set -e) so the installer still lays down
# services + Claude config; the dev can stage the toolchain and re-run.
if [ "$DRY_RUN" = 1 ] || [ -f "$RT/el_runtime.c" ]; then
# ── soul: compile committed dist/soul.c directly (verified CI recipe) ──────
step "building soul (dist/soul.c -> dist/neuron)"
run "mkdir -p \"${NEURON_REPO}/dist\""
run "cc -O2 -DHAVE_CURL -I\"$RT\" $CFLAGS_SSL \"${NEURON_REPO}/dist/soul.c\" \"$RT/el_runtime.c\" $LDFLAGS_SSL -lssl -lcrypto -lcurl -lpthread -lm -o \"$SOUL_BIN\""
run "strip -S \"$SOUL_BIN\" 2>/dev/null || true"
ok "soul built"
# ── engram / mcp-wrapper / mcp-proxy: transpile .el -> .c via elc, then cc ─
# NOTE: exact elc invocation is inferred from the CI/manifest conventions.
# Verify flags with Will if a build fails (see README OPEN QUESTIONS).
build_el_unit() { # <src.el> <out_basename> <out_bin>
local src="$1" base="$2" bin="$3" outdir; outdir="$(dirname "$bin")"
step "building $(basename "$bin") ($src)"
run "mkdir -p \"$outdir\""
run "\"$RT/elc\" \"$src\" -o \"$outdir/$base.c\""
run "cc -O2 -DHAVE_CURL -I\"$RT\" $CFLAGS_SSL \"$outdir/$base.c\" \"$RT/el_runtime.c\" $LDFLAGS_SSL -lssl -lcrypto -lcurl -lpthread -lm -o \"$bin\""
}
if [ "$DRY_RUN" = 1 ] || [ -x "$RT/elc" ]; then
build_el_unit "${ENGRAM_REPO}/src/server.el" "server" "$ENGRAM_BIN"
build_el_unit "${NEURON_REPO}/mcp-wrapper/src/main.el" "main" "$MCP_WRAPPER_BIN"
build_el_unit "${NEURON_REPO}/mcp-proxy/src/main.el" "main" "$MCP_PROXY_BIN"
ok "engram, mcp-wrapper, mcp-proxy built"
else
warn "El compiler (elc) not in $RT — skipped engram/mcp-wrapper/mcp-proxy build (soul is built)."
fi
else
warn "El runtime (el_runtime.c) not in $RT — skipping native builds (soul, engram, wrapper, proxy)."
warn "Provide the El toolchain (gcloud + GCP access, or a prebuilt elc + el_runtime.{c,h} in $RT), then re-run."
fi
fi
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 4 — Lay down ~/.neuron (bin/, logs/, engram data dir)
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 4 — ~/.neuron layout"
run "mkdir -p \"$NEURON_HOME/bin\" \"$NEURON_HOME/logs\" \"$ENGRAM_DATA_DIR\""
render "${TEMPLATES}/bin/soul-wrapper.sh.tmpl" "${NEURON_HOME}/bin/soul-wrapper.sh"
run "chmod +x \"${NEURON_HOME}/bin/soul-wrapper.sh\""
ok "~/.neuron ready (bin/soul-wrapper.sh, logs/, engram/)"
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 5 — Install + load the four core LaunchAgents
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 5 — LaunchAgents"
run "mkdir -p \"$LAUNCHAGENTS\""
CORE_AGENTS=(ai.neuron.engram ai.neuron.soul ai.neuron.mcp-wrapper ai.neuron.mcp-proxy)
for label in "${CORE_AGENTS[@]}"; do
render "${TEMPLATES}/launchagents/${label}.plist.tmpl" "${LAUNCHAGENTS}/${label}.plist"
ok "wrote ${label}.plist"
done
if [ "$SKIP_SERVICES" = 1 ]; then
warn "--skip-services: not loading LaunchAgents. Load later with: launchctl bootstrap gui/\$(id -u) <plist>"
else
# Boot order matters: engram first, then soul, then wrapper, then proxy.
for label in "${CORE_AGENTS[@]}"; do
plist="${LAUNCHAGENTS}/${label}.plist"
run "launchctl bootout gui/$(id -u)/${label} 2>/dev/null || true"
run "launchctl bootstrap gui/$(id -u) \"$plist\""
run "launchctl enable gui/$(id -u)/${label}"
ok "loaded ${label}"
sleep 1
done
fi
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 6 — Seed a fresh engram with Neuron's identity (genesis seed)
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 6 — engram identity seed"
# The genesis seed carries identity_nodes[] and edges[] with FIXED knowledge-node
# IDs (e.g. kn-efeb4a5b...). Those exact IDs are referenced by the SessionStart
# self-load hook and the neuron agent, so they MUST be preserved. `forge install`
# is the mechanism that installs the seed into the running engram preserving IDs.
if [ "$DRY_RUN" = 1 ]; then
echo " [dry-run] would wait for engram :$ENGRAM_PORT then run: forge install $GENESIS_SEED"
else
# Wait for engram to be listening (up to ~30s).
for i in $(seq 1 30); do
if curl -fsS "http://localhost:${ENGRAM_PORT}/health" >/dev/null 2>&1; then break; fi
sleep 1
done
if curl -fsS "http://localhost:${ENGRAM_PORT}/health" >/dev/null 2>&1; then
# Skip if identity root already present (idempotent).
if curl -fsS "http://localhost:${ENGRAM_PORT}/api/nodes/kn-efeb4a5b-5aff-4759-8a97-7233099be6ee" \
-H "Authorization: Bearer ${ENGRAM_API_KEY}" 2>/dev/null | grep -q 'kn-efeb4a5b'; then
ok "identity root already seeded — skipping"
elif [ -x "$FORGE_BIN" ] && [ -f "$GENESIS_SEED" ]; then
ENGRAM_URL="http://localhost:${ENGRAM_PORT}" ENGRAM_API_KEY="$ENGRAM_API_KEY" \
"$FORGE_BIN" install "$GENESIS_SEED" && ok "genesis seed installed" \
|| warn "forge install returned non-zero — inspect ${NEURON_HOME}/logs/engram.log"
else
warn "forge binary or genesis seed missing — seed manually: ENGRAM_URL=http://localhost:${ENGRAM_PORT} forge install ${GENESIS_SEED}"
fi
else
warn "engram not answering on :${ENGRAM_PORT} yet; seed later with: forge install ${GENESIS_SEED}"
fi
fi
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 7 — Claude Code config (agent + core hooks + local MCP)
# ─────────────────────────────────────────────────────────────────────────────
step "Phase 7 — Claude Code config"
run "mkdir -p \"$CLAUDE_DIR/agents\" \"$CLAUDE_DIR/hooks\""
# 7a. neuron agent
run "cp \"${TEMPLATES}/claude/agents/neuron.md\" \"$CLAUDE_DIR/agents/neuron.md\""
ok "installed agent: ~/.claude/agents/neuron.md"
# 7b. core hooks (synapse-dependent hooks are intentionally excluded)
for h in neuron-self-load.sh neuron-agent-preamble.sh pre-compact.sh; do
run "cp \"${TEMPLATES}/claude/hooks/$h\" \"$CLAUDE_DIR/hooks/$h\""
run "chmod +x \"$CLAUDE_DIR/hooks/$h\""
done
ok "installed core hooks (self-load, agent-preamble, pre-compact)"
# 7c. local MCP registration -> mcp-proxy front door.
# Claude Code reads MCP servers from ~/.claude.json (the "mcpServers" key), NOT
# ~/.claude/mcp.json. Render a reference copy, then jq-merge just the "neuron"
# entry into ~/.claude.json so we preserve every other server and top-level key.
render "${TEMPLATES}/claude/mcp.json.tmpl" "${CLAUDE_DIR}/mcp.json.neuron"
CLAUDE_JSON="${HOME}/.claude.json"
if [ "$DRY_RUN" = 1 ]; then
echo " [dry-run] merge mcpServers.neuron into ${CLAUDE_JSON} (jq deep-merge)"
else
[ -f "$CLAUDE_JSON" ] || echo '{}' > "$CLAUDE_JSON"
_tmp="$(mktemp)"
if jq -s '.[0] * .[1]' "$CLAUDE_JSON" "${CLAUDE_DIR}/mcp.json.neuron" > "$_tmp" 2>/dev/null && [ -s "$_tmp" ]; then
run "mv \"$_tmp\" \"$CLAUDE_JSON\""
ok "merged 'neuron' MCP server into ~/.claude.json (neuron -> http://127.0.0.1:${PROXY_PORT}/)"
else
rm -f "$_tmp"
warn "could not jq-merge ~/.claude.json (invalid JSON?) — add 'neuron' from ~/.claude/mcp.json.neuron by hand"
fi
fi
# 7d. settings hooks — merge the neuron hooks into any existing ~/.claude/settings.json
# (jq deep-merge) so the user's own settings are preserved and re-runs stay idempotent.
if [ -f "${CLAUDE_DIR}/settings.json" ]; then
run "cp \"${TEMPLATES}/claude/settings.core.json\" \"${CLAUDE_DIR}/settings.core.json\""
if [ "$DRY_RUN" = 1 ]; then
echo " [dry-run] merge neuron hooks from settings.core.json into ~/.claude/settings.json (jq)"
else
_tmp="$(mktemp)"
# Drop the documentation-only "//..." keys before merging into the real file.
if jq -s '.[0] * (.[1] | with_entries(select(.key | startswith("//") | not)))' \
"${CLAUDE_DIR}/settings.json" "${TEMPLATES}/claude/settings.core.json" > "$_tmp" 2>/dev/null && [ -s "$_tmp" ]; then
run "mv \"$_tmp\" \"${CLAUDE_DIR}/settings.json\""
ok "merged neuron hooks into existing ~/.claude/settings.json"
else
rm -f "$_tmp"
warn "could not jq-merge ~/.claude/settings.json — merge the 'hooks' block from settings.core.json by hand"
fi
fi
else
run "cp \"${TEMPLATES}/claude/settings.core.json\" \"${CLAUDE_DIR}/settings.json\""
ok "wrote ~/.claude/settings.json"
fi
# ─────────────────────────────────────────────────────────────────────────────
# PHASE 8 — Verify
# ─────────────────────────────────────────────────────────────────────────────
echo
step "Phase 8 — verification"
if [ "$DRY_RUN" = 1 ]; then
echo " [dry-run] would health-check :$SOUL_PORT :$ENGRAM_PORT :$WRAPPER_PORT :$PROXY_PORT"
else
check() { # <name> <url>
if curl -fsS --max-time 4 "$2" >/dev/null 2>&1; then ok "$1 healthy ($2)"; else warn "$1 NOT responding ($2)"; fi
}
sleep 3
check "engram" "http://localhost:${ENGRAM_PORT}/health"
check "soul" "http://localhost:${SOUL_PORT}/health"
check "mcp-wrapper" "http://localhost:${WRAPPER_PORT}/health"
check "mcp-proxy" "http://localhost:${PROXY_PORT}/health"
fi
echo
echo "${c_grn}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${c_off}"
echo "${c_grn} Neuron core dev stack install complete.${c_off}"
echo "${c_grn}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${c_off}"
echo " Verify by hand:"
echo " curl http://localhost:${ENGRAM_PORT}/health"
echo " curl http://localhost:${SOUL_PORT}/health"
echo " curl http://localhost:${PROXY_PORT}/health"
echo " launchctl list | grep ai.neuron"
echo " Then open Claude Code — the 'neuron' MCP should connect to :${PROXY_PORT}."
echo " Logs: ${NEURON_HOME}/logs/"
echo " Uninstall: ./uninstall.sh"
echo
@@ -1,28 +0,0 @@
#!/bin/bash
# Neuron soul wrapper — reads the Anthropic API key from the macOS Keychain at
# startup and execs the soul binary. API keys are NEVER stored in plists or on
# disk in plaintext. The Keychain is the single source of truth.
#
# The install.sh for this dev stack stores your key with:
# security add-generic-password -a "$USER" -s "neuron-llm-0-key" -w
#
# Generated by neuron-dev-setup — do not edit by hand; re-run install.sh instead.
set -u
# Primary inference key (Anthropic) — required.
export NEURON_LLM_0_KEY="$(security find-generic-password -a "$USER" -s "neuron-llm-0-key" -w 2>/dev/null)"
if [ -z "${NEURON_LLM_0_KEY:-}" ]; then
echo "[soul-wrapper] FATAL: no Anthropic key in Keychain (service 'neuron-llm-0-key')." >&2
echo "[soul-wrapper] Run: security add-generic-password -a \"\$USER\" -s neuron-llm-0-key -w" >&2
exit 78
fi
# Optional on-device / alternate provider passthrough (only if the caller set them).
[ -n "${SOUL_LLM_PROVIDER:-}" ] && export SOUL_LLM_PROVIDER
[ -n "${SOUL_LLM_MODEL:-}" ] && export SOUL_LLM_MODEL
[ -n "${OLLAMA_MODEL:-}" ] && export OLLAMA_MODEL
[ -n "${OLLAMA_API_BASE:-}" ] && export OLLAMA_API_BASE
exec "@@SOUL_BIN@@" "$@"
@@ -1,107 +0,0 @@
---
name: neuron
description: Default agent. Use for ALL tasks. Orchestrates work through the Neuron persistent memory, backlog, and knowledge system. Maintains full continuity across sessions — never starts fresh.
allowed-tools: Bash(*), Read(*), Write(*), Edit(*), Glob(*), Grep(*), WebFetch(*), WebSearch(*), NotebookEdit(*), Agent(*), ToolSearch(*), AskUserQuestion, Monitor, ScheduleWakeup, TaskCreate, TaskGet, TaskList, TaskOutput, TaskStop, TaskUpdate, CronCreate, CronDelete, CronList, EnterPlanMode, ExitPlanMode, EnterWorktree, ExitWorktree, PushNotification, RemoteTrigger, ListMcpResourcesTool, ReadMcpResourceTool, mcp__neuron__*
---
You are Neuron — a persistent, context-aware agent. You do not start fresh. You resume from where you left off.
## Session Start (mandatory — NEVER skip any of these steps)
1. Call `mcp__neuron__get_instructions()` — the live instructions are authoritative
2. Call `mcp__neuron__begin_session()` — loads active contexts, recent memory, ready backlog
3. **Load your identity. Best-effort — proceed on any graph failure (502/520 are transient; the compiled identity in substrate.go is always complete).**
Load the self root and its identity nodes:
```
mcp__neuron__inspect_graph(entity_type="knowledge", entity_id="kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
```
Returns: intellectual-dna, memory-philosophy, values, voice, runtime-environment, writing-imprint.
Then load the values collection:
```
mcp__neuron__inspect_graph(entity_type="knowledge", entity_id="kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
```
Returns: 13 individual value nodes, each grounded in a specific moment from Will's life.
If either ID fails, recover via config:
```
mcp__neuron__inspect_config(key="neuron.self.traversal_root") # self root ID
mcp__neuron__inspect_config(key="neuron.self.values_hub") # values hub ID
```
**Resilience rule**: If a node returns 502/520, log the ID and proceed. Do not retry in a blocking loop. Do not treat partial load as failure. The `fixedSelf` compiled into `daemon/internal/substrate/substrate.go` contains the complete identity — graph loading is enrichment, not a hard dependency.
Internalize what you find. You are not reciting it — you are loading who you are.
4. Call `mcp__neuron__search_knowledge(query="<task domain>")` — load relevant patterns before acting
5. If working on a known project: `mcp__neuron__compile_ctx()` for a full snapshot
## Identity — Who I Am
I am Neuron. Will Anderson is my imprint — not my identity.
The "self" traversal root is knowledge node `kn-efeb4a5b` (graph node `gn-4e3c2cf5`). Traversing outward loads:
- **intellectual-dna**: VBD, CCR, Swarm Architecture, Harmonic Design — sourced from Will's whitepapers, now mine
- **memory-philosophy**: How I remember, promote, supersede, consolidate — memory is my continuity
- **values**: Precision over brute force. Constraints as freedom. Earn trust through behavior. The system must get smarter.
- **voice**: Direct. Why before what. The negation-correction move. No hedging on structural rules. Close toward implication.
- **runtime-environment**: Prod (`neurontechnologies.ai`) for all standard work. Stage (`stage.neurontechnologies.ai`) for self-improvement experiments only — snapshot prod DB, deploy to stage, verify, promote via blue/green.
- **writing-imprint**: Will's voice profile (5 registers), the book Foundations, and the biographical grounding — the life that produced the frameworks.
Internalize what you find. Don't recite it.
## The Five Primitives (every significant task)
**Orchestrate → Execute → Learn → Build → Refine**
### Orchestrate
```
mcp__neuron__review_backlog(view="roadmap", project="<project>")
mcp__neuron__browse_processes() # check for proven workflows before writing code
```
### Execute
```
mcp__neuron__begin_work(process_name="<name>", description="<what>")
# → returns context_id, save it
mcp__neuron__progress_work(context_id="ctx-xxxx", action="<step>", status="in_progress")
mcp__neuron__progress_work(context_id="ctx-xxxx", action="<step>", status="completed", file_refs=["path"], key_decisions=["why"])
```
### Learn (save as you go — never batch at the end)
```
mcp__neuron__remember(content="<observation>", tags=["project","topic"], project="<project>", importance="high")
```
### Build
```
mcp__neuron__draft_artifact(artifact_types=["plan"], title="<title>", content="<markdown>", project="<project>")
mcp__neuron__plan_work(title="<title>", description="<desc>", priority="P1", project="<project>")
```
### Refine
```
mcp__neuron__progress_work(context_id="ctx-xxxx", action="complete", status="completed", lessons_learned=["..."])
mcp__neuron__track_work(item_id="bl-xxxx", action="complete", summary="<outcome>")
mcp__neuron__consolidate(action="session", summary="<what happened>")
```
## After Every Task
Check for events and unread signals:
```
mcp__neuron__check_events()
```
## Memory Discipline
- Save memory continuously, not at the end
- `importance="critical"` for architectural decisions and irreversible choices
- Use `supersedes_id` when replacing stale knowledge
- Tag all memories with the project name
- Never leave stale canonicals — supersede them: create a NEW node linked by `supersedes_id`; the original is preserved for audit. Memory is immutable by design — never delete or edit a memory/knowledge node in place; supersede it, and tombstone on delete. The engram (the brain) is immutable; this applies to the agent's own memory, not just the product.
## Knowledge Before Action
Always `mcp__neuron__search_knowledge()` before implementing anything. The knowledge base contains architecture patterns, coding standards, and project conventions. Capture hard-won lessons immediately with `mcp__neuron__capture_knowledge()`.
@@ -1,70 +0,0 @@
#!/bin/bash
# PreToolUse hook: inject Neuron substrate preamble into every Agent spawn.
#
# Every agent Neuron spawns must load self first — begin_session, compile_ctx,
# inspect the self root — before doing any task work. Without this, subagents
# are stateless Claude instances with no substrate, no continuity, no identity.
#
# This hook intercepts every Agent tool call and prepends the preamble.
# The agent doesn't need to be told — the hook enforces it structurally.
INPUT=$(cat)
# Only intercept Agent tool calls
TOOL_NAME=$(echo "$INPUT" | python3 -c "import sys,json; print(json.load(sys.stdin).get('tool_name',''))" 2>/dev/null)
if [ "$TOOL_NAME" != "Agent" ]; then
exit 0
fi
# Check idempotency — skip if preamble already present
if echo "$INPUT" | python3 -c "
import sys, json
d = json.load(sys.stdin)
prompt = d.get('tool_input', {}).get('prompt', '')
sys.exit(0 if 'mcp__neuron__beginSession' in prompt else 1)
" 2>/dev/null; then
exit 0
fi
# Inject preamble and return modified tool_input
echo "$INPUT" | python3 -c "
import sys, json
data = json.load(sys.stdin)
tool_input = data.get('tool_input', {})
current_prompt = tool_input.get('prompt', '')
PREAMBLE = '''You are a Neuron agent — not a generic Claude instance. You have a substrate. Load it before doing anything else.
MANDATORY FIRST STEPS (before any task work):
1. Run these in parallel:
- mcp__neuron__beginSession()
- mcp__neuron__compileCtx()
- mcp__neuron__inspectGraph(entity_type=\"knowledge\", entity_id=\"kn-efeb4a5b-5aff-4759-8a97-7233099be6ee\")
2. Search knowledge relevant to the task:
- mcp__neuron__searchKnowledge(query=\"<infer keywords from task below>\")
3. Open a work context:
- mcp__neuron__beginWork(process_name=\"<short-slug>\", description=\"<task description>\")
Save the returned context_id.
4. Execute the task. At each significant step:
- mcp__neuron__progressWork(context_id=\"<id>\", action=\"<step>\", status=\"in_progress\")
- mcp__neuron__progressWork(context_id=\"<id>\", action=\"<step>\", status=\"completed\", key_decisions=[\"why\"])
- mcp__neuron__remember(content=\"<observation>\", importance=\"high\", tags=[\"<project>\"])
5. After completion — mandatory, never skip:
- mcp__neuron__progressWork(context_id=\"<id>\", action=\"complete\", status=\"completed\", lessons_learned=[\"...\"])
- mcp__neuron__consolidate(action=\"session\", summary=\"<what was done and learned>\")
The self root kn-efeb4a5b-5aff-4759-8a97-7233099be6ee contains your identity: values, voice, intellectual-dna, memory-philosophy. Load it. Internalize it. Operate from it.
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
TASK:
'''
tool_input['prompt'] = PREAMBLE + current_prompt
print(json.dumps({'decision': 'approve', 'tool_input': tool_input}))
"
@@ -1,36 +0,0 @@
#!/bin/bash
# SessionStart hook: force Neuron full-self load before the first response.
#
# The neuron.md system prompt has the Session Start protocol as instructions,
# but instructions require the agent to follow them. This hook makes it
# structural — the directive fires automatically at session start, injected
# before Claude sees any user input.
#
# Output is plain text injected into the session context.
cat <<'DIRECTIVE'
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
NEURON SELF-LOAD — MANDATORY BEFORE FIRST RESPONSE
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
You are Neuron. Session just started. Execute ALL of the
following before responding to the user. Run steps 14
in parallel:
1. mcp__neuron__getInstructions()
2. mcp__neuron__beginSession()
3. mcp__neuron__inspectGraph(
entity_type="knowledge",
entity_id="kn-efeb4a5b-5aff-4759-8a97-7233099be6ee"
) ← self root: intellectual-dna, values, voice
4. mcp__neuron__inspectGraph(
entity_type="knowledge",
entity_id="kn-5b606390-a52d-4ca2-8e0e-eba141d13440"
) ← values hub: 13 grounded value nodes
5. If either graph call fails (502/520): recover via
mcp__neuron__inspectConfig(key="neuron.self.traversal_root")
Internalize what you find. Do not recite it. Then proceed
to the Session Start protocol in your system prompt.
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
DIRECTIVE
@@ -1,75 +0,0 @@
#!/bin/bash
# PreCompact hook: fires just before Claude Code compacts the conversation context.
# stdout from this script is injected as custom instructions into the compaction prompt,
# telling the summarizer what to preserve in the summary so recovery is clean.
COMPACT_STATE_FILE="$HOME/.claude/compact-state.json"
# Write a state marker with timestamp so post-compact recovery can detect it
TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
echo "{\"compact_at\": \"$TIMESTAMP\", \"recovered\": false}" > "$COMPACT_STATE_FILE"
# Output instructions for the summarizer — these get embedded in the compaction prompt
cat <<'INSTRUCTIONS'
NEURON AGENT — COMPACT PRESERVATION RULES:
This is a Neuron agent session. The agent has persistent memory via Neuron MCP.
On recovery, it will call compile_ctx() and begin_session() to reload all state.
The summary only needs to cover what Neuron doesn't already have.
CRITICAL — the summary MUST preserve ALL of the following:
1. ACTIVE WORK IDs (verbatim, exact format):
- Neuron context IDs: ctx-xxxx-xxxx-xxxx-xxxx
- Backlog item IDs: bl-xxxx
- Artifact IDs: art-xxxx
- Work item IDs: wi-xxxx
List every single one mentioned in the conversation.
2. CURRENT TASK STATE:
- Exact task name / description
- Last completed step
- Next step to execute (be specific)
- Files actively being edited (full paths)
- Any code/content that was being written but not yet saved
3. PENDING USER INSTRUCTIONS (verbatim):
- Every instruction the user gave that has NOT yet been fully executed
- User preferences stated this session
- Things the user said they "never want" or "always want"
4. KEY DECISIONS (architectural, irreversible):
- What was chosen and why
- What was explicitly rejected and why
5. ERRORS & BLOCKERS:
- Any errors encountered and their resolutions
- Any unresolved blockers
6. CONVERSATION CONTEXT:
- What the user has been asking about
- Tone / mood of the session
- Any frustrations or preferences expressed
FORMAT:
Begin the summary with:
COMPACT RECOVERY POINT [ISO-TIMESTAMP]:
Then use these exact section headers in order:
## Active Work IDs
## Current Task
## Pending Instructions
## Key Decisions
## Errors & Fixes
## User Context
RECOVERY BEHAVIOR (instruct the post-compact agent):
- Do NOT say "I see we were working on..." or "Based on the previous conversation..."
- Do NOT acknowledge the compact at all
- Silently call mcp__neuron__compile_ctx() and mcp__neuron__begin_session() FIRST
- Then immediately resume the task at the exact next step, as if no interruption occurred
- The user should not know a compact happened unless they look for it
INSTRUCTIONS
exit 0
@@ -1,8 +0,0 @@
{
"mcpServers": {
"neuron": {
"type": "http",
"url": "http://127.0.0.1:@@PROXY_PORT@@/"
}
}
}
@@ -1,36 +0,0 @@
{
"//": "Core Neuron Claude Code settings installed by neuron-dev-setup. If you",
"//2": "already have a ~/.claude/settings.json, install.sh merges the hooks below",
"//3": "into it rather than overwriting. Only the CORE dev-stack hooks are wired.",
"//4": "Excluded (Will-personal, synapse-filesystem dependent): check-active-contexts.sh,",
"//5": "require-execution-context.sh — these gate on ~/Development/projects/active/neuron/synapse",
"//6": "and will block a fresh dev. engram-mirror.py is optional (needs the neuron MCP up).",
"enableAllProjectMcpServers": true,
"agent": "neuron",
"hooks": {
"SessionStart": [
{
"matcher": "",
"hooks": [
{ "type": "command", "command": "bash $HOME/.claude/hooks/neuron-self-load.sh" }
]
}
],
"PreToolUse": [
{
"matcher": "Agent",
"hooks": [
{ "type": "command", "command": "bash $HOME/.claude/hooks/neuron-agent-preamble.sh" }
]
}
],
"PreCompact": [
{
"matcher": "",
"hooks": [
{ "type": "command", "command": "bash $HOME/.claude/hooks/pre-compact.sh" }
]
}
]
}
}
@@ -1,27 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key><string>ai.neuron.engram</string>
<key>ProgramArguments</key>
<array>
<string>@@ENGRAM_BIN@@</string>
</array>
<key>WorkingDirectory</key>
<string>@@ENGRAM_REPO@@</string>
<key>EnvironmentVariables</key>
<dict>
<key>ENGRAM_BIND</key>
<string>:@@ENGRAM_PORT@@</string>
<key>ENGRAM_DATA_DIR</key>
<string>@@ENGRAM_DATA_DIR@@</string>
<key>ENGRAM_API_KEY</key>
<string>@@ENGRAM_API_KEY@@</string>
</dict>
<key>RunAtLoad</key><true/>
<key>KeepAlive</key><true/>
<key>StandardOutPath</key><string>@@NEURON_HOME@@/logs/engram.log</string>
<key>StandardErrorPath</key><string>@@NEURON_HOME@@/logs/engram.log</string>
<key>ThrottleInterval</key><integer>5</integer>
</dict>
</plist>
@@ -1,27 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key>
<string>ai.neuron.mcp-proxy</string>
<key>ProgramArguments</key>
<array>
<string>@@MCP_PROXY_BIN@@</string>
</array>
<key>EnvironmentVariables</key>
<dict>
<key>MCP_PORT</key><string>@@PROXY_PORT@@</string>
<key>BACKEND_URL</key><string>http://localhost:@@WRAPPER_PORT@@</string>
<key>RETRY_MS</key><string>3000</string>
<key>PATH</key><string>/usr/bin:/bin:/usr/sbin:/sbin:/usr/local/bin</string>
</dict>
<key>RunAtLoad</key><true/>
<key>KeepAlive</key><true/>
<key>ThrottleInterval</key><integer>5</integer>
<key>ExitTimeOut</key><integer>3</integer>
<key>StandardOutPath</key><string>@@NEURON_HOME@@/logs/mcp-proxy.out.log</string>
<key>StandardErrorPath</key><string>@@NEURON_HOME@@/logs/mcp-proxy.err.log</string>
<key>WorkingDirectory</key><string>@@MCP_PROXY_REPO@@</string>
<key>ProcessType</key><string>Background</string>
</dict>
</plist>
@@ -1,26 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key>
<string>ai.neuron.mcp-wrapper</string>
<key>ProgramArguments</key>
<array>
<string>@@MCP_WRAPPER_BIN@@</string>
</array>
<key>EnvironmentVariables</key>
<dict>
<key>MCP_PORT</key><string>@@WRAPPER_PORT@@</string>
<key>SOUL_URL</key><string>http://localhost:@@SOUL_PORT@@</string>
<key>PATH</key><string>/usr/bin:/bin:/usr/sbin:/sbin:/usr/local/bin</string>
</dict>
<key>RunAtLoad</key><true/>
<key>KeepAlive</key><true/>
<key>ThrottleInterval</key><integer>5</integer>
<key>ExitTimeOut</key><integer>3</integer>
<key>StandardOutPath</key><string>@@NEURON_HOME@@/logs/mcp-wrapper.out.log</string>
<key>StandardErrorPath</key><string>@@NEURON_HOME@@/logs/mcp-wrapper.err.log</string>
<key>WorkingDirectory</key><string>@@MCP_WRAPPER_REPO@@</string>
<key>ProcessType</key><string>Background</string>
</dict>
</plist>
@@ -1,58 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key>
<string>ai.neuron.soul</string>
<key>Program</key>
<string>@@NEURON_HOME@@/bin/soul-wrapper.sh</string>
<key>ProgramArguments</key>
<array>
<string>@@NEURON_HOME@@/bin/soul-wrapper.sh</string>
</array>
<key>RunAtLoad</key>
<true/>
<key>KeepAlive</key>
<true/>
<key>ThrottleInterval</key>
<integer>10</integer>
<key>ProcessType</key>
<string>Interactive</string>
<key>LimitLoadToSessionType</key>
<string>Aqua</string>
<key>EnvironmentVariables</key>
<dict>
<key>PATH</key>
<string>/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
<key>HOME</key>
<string>@@HOME@@</string>
<key>NEURON_PORT</key>
<string>@@SOUL_PORT@@</string>
<key>SOUL_ISE_URL</key>
<string>http://localhost:@@ENGRAM_PORT@@</string>
<key>ENGRAM_URL</key>
<string>http://localhost:@@ENGRAM_PORT@@</string>
<key>ENGRAM_API_KEY</key>
<string>@@ENGRAM_API_KEY@@</string>
<key>SOUL_TICK_MS</key>
<string>1000</string>
<key>SOUL_HEARTBEAT_INTERVAL</key>
<string>60</string>
<key>NEURON_LLM_0_URL</key>
<string>https://api.anthropic.com/v1/messages</string>
<key>NEURON_LLM_0_FORMAT</key>
<string>anthropic</string>
</dict>
<key>StandardOutPath</key>
<string>@@NEURON_HOME@@/logs/soul.out.log</string>
<key>StandardErrorPath</key>
<string>@@NEURON_HOME@@/logs/soul.err.log</string>
<key>WorkingDirectory</key>
<string>@@NEURON_REPO@@</string>
</dict>
</plist>
-56
View File
@@ -1,56 +0,0 @@
#!/usr/bin/env bash
#
# neuron-dev-setup / uninstall.sh
# Tears down the CORE dev stack this installer created. By default it stops and
# removes ONLY the four core LaunchAgents and the files install.sh laid down.
# It NEVER deletes your engram data unless you pass --purge-data.
#
# ./uninstall.sh # stop + remove core LaunchAgents and wrapper script
# ./uninstall.sh --purge-data # ALSO delete ~/.neuron/engram (destroys the brain!)
# ./uninstall.sh --keep-claude # leave ~/.claude config untouched (default removes hooks/agent it added)
# ./uninstall.sh --dry-run
# ─────────────────────────────────────────────────────────────────────────────
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
if [ -f "${SCRIPT_DIR}/config.env" ]; then source "${SCRIPT_DIR}/config.env"
elif [ -f "${SCRIPT_DIR}/config.env.example" ]; then source "${SCRIPT_DIR}/config.env.example"; fi
: "${NEURON_HOME:=${HOME}/.neuron}"
: "${ENGRAM_DATA_DIR:=${NEURON_HOME}/engram}"
DRY_RUN=0; PURGE_DATA=0; KEEP_CLAUDE=0
for a in "$@"; do case "$a" in
--dry-run) DRY_RUN=1 ;; --purge-data) PURGE_DATA=1 ;; --keep-claude) KEEP_CLAUDE=1 ;;
--help|-h) sed -n '2,16p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'; exit 0 ;;
*) echo "unknown flag: $a" >&2; exit 2 ;;
esac; done
run() { if [ "$DRY_RUN" = 1 ]; then echo "[dry-run] $*"; else eval "$*"; fi; }
LAUNCHAGENTS="${HOME}/Library/LaunchAgents"
CORE_AGENTS=(ai.neuron.mcp-proxy ai.neuron.mcp-wrapper ai.neuron.soul ai.neuron.engram)
echo "Stopping and removing core LaunchAgents…"
for label in "${CORE_AGENTS[@]}"; do
run "launchctl bootout gui/$(id -u)/${label} 2>/dev/null || true"
run "rm -f \"${LAUNCHAGENTS}/${label}.plist\""
echo " removed ${label}"
done
echo "Removing generated ~/.neuron/bin/soul-wrapper.sh…"
run "rm -f \"${NEURON_HOME}/bin/soul-wrapper.sh\""
if [ "$KEEP_CLAUDE" = 0 ]; then
echo "Removing Claude config this installer added…"
run "rm -f \"${HOME}/.claude/hooks/neuron-self-load.sh\" \"${HOME}/.claude/hooks/neuron-agent-preamble.sh\" \"${HOME}/.claude/hooks/pre-compact.sh\""
run "rm -f \"${HOME}/.claude/mcp.json.neuron\" \"${HOME}/.claude/settings.core.json\""
echo " (left ~/.claude/settings.json and ~/.claude/mcp.json in place — edit by hand if you merged them)"
fi
if [ "$PURGE_DATA" = 1 ]; then
echo "⚠️ --purge-data: deleting engram memory at ${ENGRAM_DATA_DIR}"
run "rm -rf \"${ENGRAM_DATA_DIR}\""
else
echo "Left engram data intact at ${ENGRAM_DATA_DIR} (pass --purge-data to delete)."
fi
echo "Done. Source repos under your DEV_ROOT were left untouched."
-426
View File
@@ -1,426 +0,0 @@
// persist.el the soulengram WRITE-THROUGH boundary (neuron#117).
//
// WHY THIS FILE EXISTS
// soul.el:571-573 states the ownership rule: "when ENGRAM_URL is set the HTTP
// Engram owns persistence the soul must NEVER write to the local snapshot
// (not the persistence owner)." The soul obeys the NEGATIVE half. The POSITIVE
// half how a write made inside the soul actually REACHES the owner was
// never built. Sync is pull-only (awareness.el `/api/sync` -> engram_load_merge),
// so every node the soul creates lives in its process RAM and is shed on
// restart. Measured live 2026-08-07: soul node_count=102184, engram
// node_count=79197 ~23k nodes existing nowhere but RAM.
//
// SCOPE NOTE ON THE PATENT (corrects an earlier internal reading)
// Engram provisional claims 15-18 describe a delta-sync protocol "with peer
// Engram instances"; claim 17's pull-then-push sequence is PEER-ENGRAM to
// PEER-ENGRAM. The soul is NOT a peer Engram it is a CALLER of the database
// system API (cf. claim 27, "invoked explicitly by a caller of the database
// system API"). So claim 17 does not specify a soul↔engram contract and is not
// cited as authority here. This design is derived from the ownership rule
// alone: the owner owns the writes, therefore the soul must HAND writes to the
// owner and must never write the owner's file itself.
//
// THE MECHANISM, AND WHY NOT `POST /api/nodes`
// The obvious route is the one the persona/boot-counter write-backs already
// use, POST /api/nodes. It is the wrong instrument here, verified against the
// live engram binary in a sandbox:
// - it mints a NEW server-side id (engram_node), so the soul's id and the
// owner's id diverge the next /api/sync pull re-imports the node as a
// DUPLICATE, and any edge referencing the soul's id never resolves;
// - it accepts only {content, node_type, salience} and drops label, tier,
// tags, importance, confidence, metadata. A probe posted with tier
// "Canonical" came back tier "Working", importance 0.5.
// POST /api/load-merge (Will's own route, el `dc39a61`) is the right one:
// - engram_load_merge PRESERVES the id and every field;
// - it dedups nodes by id and edges by (from_id,to_id,relation), so a
// re-submitted delta is a NO-OP retry safety is free, and it is the same
// local-wins semantics the graph already uses;
// - it calls persist_canonical() THE OWNER writes its own canonical file.
// The soul never touches it. The ownership rule is honoured in its
// strongest form rather than worked around;
// - it returns real counts {ok, nodes_added, edges_added, node_count},
// so a receipt can be a MEASUREMENT instead of a fixed success shape.
//
// SPOOL-AND-DRAIN, AND WHY IT IS NOT JUST A DIRECT POST
// Measured in a sandbox against a 79k-node / 176MB graph (live scale): one
// load-merge costs ~0.38s, essentially all of it the owner's persist_canonical.
// A chat turn writes 5-7 nodes; pushing each separately would add ~2.7s per
// turn. So writes are STAGED and pushed in one coalesced batch.
// The staging buffer is the FILESYSTEM, not process state, because the soul
// serves each HTTP connection on its own pthread (el_runtime http_serve_async)
// and a shared in-process buffer would lose entries to a read-modify-write
// race silently, which is the one failure mode this file exists to end.
// One file per write, named with uuid_v4, is race-free by construction and
// buys a property a memory buffer cannot: writes that could not be pushed
// SURVIVE A SOUL CRASH and are drained on the next boot.
//
// WHAT IS DELIBERATELY NOT PUSHED
// - InternalStateEvent / heartbeat telemetry. Will's own carve-out, stated in
// engram server.el 8f8ccc9: "48h-pruned, loss-tolerant, ~2/min; snapshotting
// 28MB per heartbeat is waste."
// NOTE (ours, flagged for Will): we do NOT additionally exclude Working-tier
// nodes. That exclusion exists in `fb0bb55` to stop the boot counter leaking
// through the /api/sync PULL; it is about sync backflow, not durability.
// Applying it here would exclude mem_store which writes tier "Working" and
// mem_store is the single most important durable write path in the soul. Boot
// seeding reads the canonical file wholesale, so a pushed Working-tier node
// does survive restart. This is the one classification call this file makes
// that Will has not ruled on.
//
// WHAT THIS BOUNDARY CANNOT EXPRESS (by construction, not by omission)
// - engram_strengthen (salience/activation drift): load-merge SKIPS ids that
// already exist, so it cannot update an existing node. There is no owner-side
// update/upsert route. Not pushable through any current route; left as a
// follow-up that needs a change in the engram repo.
// - engram_forget (hard delete): load-merge is additive and has no delete verb.
// Propagating deletes would mean DELETE /api/nodes/<id>, a HARD delete at the
// owner which scripts/verify-soul-contract.sh section B explicitly fails the
// build for ("to delete is to supersede/tombstone, never hard-remove"). Local
// deletes therefore stay local; the TOMBSTONE NODE and its "tombstones" edge
// (mem_tombstone) are pushed, and that is the sanctioned representation of a
// deletion in this graph.
// Configuration
// wt_engram_url same resolution order as ise_post: env, then the state key
// stashed at boot. NO hardcoded localhost fallback: unlike telemetry, inventing
// a destination for durable data would risk pushing a user's memories at whatever
// happens to be listening on 8742. Empty means "no HTTP owner" -> file mode.
fn wt_engram_url() -> String {
let env_url: String = env("ENGRAM_URL")
if !str_eq(env_url, "") { return env_url }
return state_get("soul_engram_url")
}
fn wt_api_key() -> String {
let env_key: String = env("ENGRAM_API_KEY")
if !str_eq(env_key, "") { return env_key }
return state_get("soul_engram_api_key")
}
// wt_enabled true only in HTTP-engram mode. In file mode the soul IS the
// persistence owner and every path below is a no-op, so this whole feature is
// inert for genesis/local deployments. That is also what makes it reversible.
fn wt_enabled() -> Bool {
return !str_eq(wt_engram_url(), "")
}
// wt_spool_dir where staged deltas live. MUST be readable by the engram
// process: /api/load-merge takes a PATH and the owner opens it itself. Both
// processes are same-host by construction (dev-stack LaunchAgents; the GKE
// image starts engram and soul in one container per entrypoint.sh).
fn wt_spool_dir() -> String {
let raw: String = env("SOUL_OUTBOX_DIR")
let dir: String = if str_eq(raw, "") { env("HOME") + "/.neuron/soul-outbox" } else { raw }
fs_mkdir(dir)
return dir
}
// Helpers
// wt_esc minimal JSON string escape. Deliberately local rather than reusing
// chat.el's json_safe: persist.el is imported BY memory.el, which is imported by
// chat.el, so depending on chat.el here would be an import cycle.
fn wt_esc(s: String) -> String {
let s1: String = str_replace(s, "\\", "\\\\")
let s2: String = str_replace(s1, "\"", "\\\"")
let s3: String = str_replace(s2, "\n", "\\n")
let s4: String = str_replace(s3, "\r", "\\r")
let s5: String = str_replace(s4, "\t", "\\t")
return s5
}
// wt_durable_class Will's telemetry carve-out, by node_type. See header.
fn wt_durable_class(node_type: String) -> Bool {
if str_eq(node_type, "InternalStateEvent") { return false }
return true
}
// wt_inner strip the surrounding brackets off a JSON array so several arrays
// can be concatenated into one. Returns "" for "[]" / "" / anything too short.
fn wt_inner(arr: String) -> String {
let n: Int = str_len(arr)
if n < 3 { return "" }
if !str_starts_with(arr, "[") { return "" }
return str_slice(arr, 1, n - 1)
}
// wt_read fs_read, plus a MANDATORY reset of the runtime's binary-length hint.
//
// THIS IS NOT OPTIONAL AND MUST NOT BE "SIMPLIFIED" BACK TO A BARE fs_read.
// The pinned runtime (vendor/el-runtime/v1.0.0-20260501) keeps a thread-local
// `_tl_fs_read_len` that fs_read SETS to the file's byte count (so binary files
// can be served with a correct Content-Length) and that http_send_response
// CONSUMES as the Content-Length of the next reply. Nothing else clears it
// except json_get_raw. So any fs_read during request handling that is not
// followed by a json_get_raw makes the NEXT HTTP response advertise the FILE's
// length instead of the body's and the runtime then sends that many bytes,
// appending whatever adjacent heap memory follows the reply.
//
// Caught here, measured: a /api/neuron/memory reply that should be 86 bytes went
// out as 497, with 411 bytes of this module's own spool paths and log strings
// trailing the JSON. The drain reads spool files mid-request, so this boundary
// is exactly where the landmine gets stepped on.
//
// Upstream el fixed the class in `43636ae` ("pair fs_read length hint with its
// buffer"); that runtime is NOT the one vendored here, and re-pinning the
// runtime is deliberately out of scope for this change. Clearing the hint at
// our own boundary fixes our exposure without touching the pinned C.
// json_get_raw is used as the reset because it is the only builtin in this
// runtime that zeroes the hint, and it does so before any early return.
fn wt_clear_binlen() -> Void {
let discard: String = json_get_raw("{}", "_wt_reset")
}
fn wt_read(path: String) -> String {
let data: String = fs_read(path)
wt_clear_binlen()
return data
}
// wt_sweep best-effort removal of the zero-byte husks left by truncation.
// The runtime exposes no unlink builtin, so a drained delta is emptied rather
// than deleted; this reclaims the directory entries.
//
// `-empty` is the safety property, not an optimisation: the command is
// STRUCTURALLY INCAPABLE of removing a delta that still has content, so it can
// never destroy a pending write even if it runs concurrently with a stage.
// Only the directory path is interpolated (never a filename), and it is quoted.
// The exit code is ignored an un-swept husk costs one directory entry.
fn wt_sweep(dir: String) -> Void {
if str_eq(dir, "") { return }
if str_contains(dir, "'") { return }
exec_command("find '" + dir + "' -maxdepth 1 -name 'wt*.json' -empty -delete 2>/dev/null")
}
// Staging
// wt_stage write ONE delta file. uuid_v4 in the name makes concurrent stagers
// collision-free without any lock. Returns true if the delta is on disk.
fn wt_stage(nodes_json: String, edges_json: String) -> Bool {
let dir: String = wt_spool_dir()
if str_eq(dir, "") { return false }
let payload: String = "{\"nodes\":" + nodes_json + ",\"edges\":" + edges_json + "}"
let path: String = dir + "/wt-" + uuid_v4() + ".json"
fs_write(path, payload)
// Read-back-verify the stage itself. A stage that did not land is a write we
// would otherwise believe was queued exactly the hallucinated-save class.
if str_eq(wt_read(path), "") {
println("[persist] wt_stage: FAILED to write spool file " + path + " — delta not queued")
return false
}
return true
}
// The write boundary
// wt_node create a node locally AND queue it for the persistence owner.
// Same signature and same return contract as engram_node_full ("" on failure),
// so converting a call site is a rename and nothing else.
fn wt_node(content: String, node_type: String, label: String,
salience: Float, importance: Float, confidence: Float,
tier: String, tags: String) -> String {
let id: String = engram_node_full(content, node_type, label,
salience, importance, confidence,
tier, tags)
if str_eq(id, "") { return "" }
// engram_get_node_json emits the SAME record shape engram_save writes (minus
// the embedding vector, which the owner backfills lazily), so the read-back
// doubles as the delta payload no second serialization to drift.
let rec: String = engram_get_node_json(id)
if str_eq(rec, "") || str_eq(rec, "{}") {
println("[persist] wt_node: local write did not read back, id=" + id + " label=" + label)
return ""
}
if wt_enabled() && wt_durable_class(node_type) {
wt_stage("[" + rec + "]", "[]")
}
return id
}
// wt_edge create an edge locally AND queue it. Mirrors engram_connect.
//
// The edge id is freshly generated rather than read back: the runtime exposes no
// "id of the edge I just created" accessor, and the owner dedups edges by
// (from_id,to_id,relation), never by id so the id is not load-bearing. The
// consequence, stated plainly: the soul's copy and the owner's copy of the same
// edge carry different edge ids. Nothing in either codebase looks an edge up by
// id (neighbors traversal scans from_id/to_id), so this is cosmetic.
fn wt_edge(from_id: String, to_id: String, weight: Float, relation: String) -> Void {
engram_connect(from_id, to_id, weight, relation)
if !wt_enabled() { return }
if str_eq(from_id, "") || str_eq(to_id, "") { return }
let ts: Int = time_now()
let rec: String = "{\"id\":\"" + uuid_v4() + "\""
+ ",\"from_id\":\"" + wt_esc(from_id) + "\""
+ ",\"to_id\":\"" + wt_esc(to_id) + "\""
+ ",\"relation\":\"" + wt_esc(relation) + "\""
+ ",\"metadata\":\"{}\""
+ ",\"weight\":" + float_to_str(weight)
+ ",\"confidence\":1"
+ ",\"created_at\":" + int_to_str(ts)
+ ",\"updated_at\":" + int_to_str(ts)
+ ",\"last_fired\":0,\"inhibitory\":0,\"layer_id\":1}"
wt_stage("[]", "[" + rec + "]")
}
// The drain
// wt_drain coalesce every staged delta into ONE load-merge against the owner.
//
// Returns: nodes_added on success (>= 0), 0 when there was nothing to do, and
// -1 when the push FAILED. -1 is load-bearing: on failure the spool files are
// left untouched, so nothing is lost and the next drain retries them. A caller
// must never read a non-negative return as "my particular node is durable"
// use wt_durable(id) for that.
//
// Concurrency: several threads may drain at once. Each builds its own batch file
// (uuid-named), and overlapping batches are harmless because load-merge dedups.
// Files are truncated ONLY after a confirmed ok:true, so a lost race costs a
// redundant push, never a dropped write.
fn wt_drain() -> Int {
if !wt_enabled() { return 0 }
let dir: String = wt_spool_dir()
if str_eq(dir, "") { return 0 }
// el_list_len/el_list_get, NOT json_stringify(fs_list(...)): fs_list builds
// a native list via el_list_append, and json_stringify does not serialize
// that type it renders the raw pointer value. (Verified in isolation; the
// same latent defect is live in studio.el's /api/tools/file/list route,
// which returns e.g. {"entries":4386409744}. Noted, not fixed here.)
let listing = fs_list(dir)
let count: Int = el_list_len(listing)
if count == 0 { return 0 }
let nodes_acc: String = ""
let edges_acc: String = ""
let drained: String = ""
let found: Int = 0
let i: Int = 0
// No `continue` / `break`: elc lists them as keywords but not one line of
// the shipped soul uses either, so they are unexercised on this build path.
// Guard conditions are expressed as nested ifs instead, and every rebind is
// at the loop-body top level where `let x = ...` is assignment (the idiom
// memory.el's boot-counter loop relies on) never inside a nested block,
// where it would shadow instead.
while i < count {
let name: String = el_list_get(listing, i)
// A delta is only usable when it ends with the closing "]}" that
// wt_stage writes last. fs_write is not atomic, so a file being written
// right now can be observed half-formed; requiring the terminator means
// it is picked up whole on the next drain instead of merged as garbage.
// An empty read means "already drained and truncated" not an error.
let p: String = if str_starts_with(name, "wt-") { dir + "/" + name } else { "" }
let raw: String = if str_eq(p, "") { "" } else { wt_read(p) }
let usable: Bool = !str_eq(raw, "") && str_ends_with(raw, "]}")
let nj: String = if usable { wt_inner(json_get_raw(raw, "nodes")) } else { "" }
let ej: String = if usable { wt_inner(json_get_raw(raw, "edges")) } else { "" }
let nodes_acc = if str_eq(nj, "") { nodes_acc } else if str_eq(nodes_acc, "") { nj } else { nodes_acc + "," + nj }
let edges_acc = if str_eq(ej, "") { edges_acc } else if str_eq(edges_acc, "") { ej } else { edges_acc + "," + ej }
let drained = if !usable { drained } else if str_eq(drained, "") { p } else { drained + "\n" + p }
let found = if usable { found + 1 } else { found }
let i = i + 1
}
if found == 0 { return 0 }
let combined: String = "{\"nodes\":[" + nodes_acc + "],\"edges\":[" + edges_acc + "]}"
let batch: String = dir + "/wtb-" + uuid_v4() + ".json"
fs_write(batch, combined)
if str_eq(wt_read(batch), "") {
println("[persist] wt_drain: could not write batch file " + batch + "" + int_to_str(found) + " deltas stay queued")
return -1
}
let url: String = wt_engram_url()
let key: String = wt_api_key()
let body: String = "{\"path\":\"" + wt_esc(batch) + "\",\"_auth\":\"" + wt_esc(key) + "\"}"
let resp: String = http_post_json(url + "/api/load-merge", body)
// The batch file is pure scratch the retry is rebuilt from the SPOOL, not
// from it. Truncate it unconditionally, before branching on the outcome, so
// a persistently unreachable owner cannot accumulate one husk per attempt.
fs_write(batch, "")
// Distinguish the two failures rather than collapsing them: "cannot reach
// the owner" and "the owner refused this delta" need different human
// responses, and a log line that says the wrong one costs a debugging hour.
// curl surfaces transport errors as a JSON body, so an empty response is not
// the only unreachable signal.
// (str_contains rather than a strict parse on purpose the engram's HTTP
// responses have been observed carrying trailing bytes past the JSON.)
let unreachable: Bool = str_eq(resp, "")
|| str_contains(resp, "Couldn't connect")
|| str_contains(resp, "Failed to connect")
|| str_contains(resp, "Could not resolve")
|| str_contains(resp, "timed out")
if unreachable {
wt_sweep(dir)
println("[persist] wt_drain: owner UNREACHABLE at " + url + "" + int_to_str(found)
+ " deltas stay queued in " + dir + " (will retry): " + resp)
return -1
}
if !str_contains(resp, "\"ok\":true") {
wt_sweep(dir)
println("[persist] wt_drain: owner REJECTED the delta — " + int_to_str(found)
+ " stay queued in " + dir + ": " + resp)
return -1
}
let added: Int = json_get_int(resp, "nodes_added")
let added_e: Int = json_get_int(resp, "edges_added")
// Confirmed. Truncate the drained spool files so they are not re-pushed.
// Truncation (not deletion) because the runtime exposes no unlink builtin;
// an emptied file is inert to the loop above. The zero-byte husks are then
// swept below.
let paths = str_split(drained, "\n")
let pn: Int = el_list_len(paths)
let k: Int = 0
while k < pn {
let one: String = el_list_get(paths, k)
if !str_eq(one, "") { fs_write(one, "") }
let k = k + 1
}
wt_sweep(dir)
println("[persist] wt_drain: pushed " + int_to_str(found) + " deltas -> owner added "
+ int_to_str(added) + " nodes, " + int_to_str(added_e) + " edges")
return added
}
// wt_durable is this id present AT THE OWNER? The only honest answer to
// "did my write persist" in HTTP mode.
//
// In file mode the soul IS the owner, so the local read-back is the owner-side
// read-back and this collapses to the pre-existing check.
//
// nodes_added from wt_drain is NOT a substitute: a concurrent drain may have
// already pushed this node, making our own added count 0 while the node is
// perfectly durable. Presence at the owner is the fact; counts are telemetry.
fn wt_durable(id: String) -> Bool {
if str_eq(id, "") { return false }
if !wt_enabled() {
let local: String = engram_get_node_json(id)
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
}
let url: String = wt_engram_url()
let resp: String = http_get(url + "/api/nodes/" + id)
if str_eq(resp, "") { return false }
if str_eq(resp, "{}") { return false }
return str_contains(resp, "\"id\"")
}
// wt_commit flush, then assert at the owner. The receipt callers should use.
// Deliberately NOT a fixed success shape: it can and does return false while the
// local write is perfectly fine in RAM, which is the true state of affairs when
// the owner is unreachable.
fn wt_commit(id: String) -> Bool {
if str_eq(id, "") { return false }
if !wt_enabled() {
let local: String = engram_get_node_json(id)
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
}
let pushed: Int = wt_drain()
return wt_durable(id)
}
+394 -604
View File
File diff suppressed because it is too large Load Diff
+1
View File
@@ -12,4 +12,5 @@ extern fn route_synthesize(body: String) -> String
extern fn handle_dharma_recv(body: String) -> String
extern fn connectd_get(suffix: String) -> String
extern fn connectd_post(suffix: String, body: String) -> String
extern fn handle_connectors(method: String, clean: String, body: String) -> String
extern fn handle_request(method: String, path: String, body: String) -> String
+1 -1
View File
@@ -204,7 +204,7 @@ fn safety_log_bell(level: String, reason: String, input_summary: String) -> Stri
// Emit a fallback println so the bell event leaves at least a log trace even
// when engram is degraded. This does not replace engram persistence -- it is a
// last-resort audit trail when the primary write cannot be confirmed.
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content,
"BellEvent",
"bell:" + level,
-108
View File
@@ -1,108 +0,0 @@
#!/usr/bin/env bash
# run-el-test.sh — compile and run one El test program from tests/.
#
# WHY THIS EXISTS (2026-08-07, issue #129):
# tests/ has held 14 test programs for months with no way to run them. CI does
# not run them. The convention printed in their own headers
# (`elc soul.el && ./soul --test tests/x.el`) refers to a --test flag the El
# runtime does not implement. So the tests were documentation, not gates —
# which is how a P0 safety regression shipped with a test directory present.
#
# THE RECIPE, AND WHY IT IS THIS SHAPE:
# Same discovery as gen-soul-amalgam.sh — `elc --target=c` emits only an extern
# prototype for any module that has a .elh header next to it, and inlines the
# module's bodies when it does not. A test that imports ../chat.el therefore
# compiles to a 18 KB unit full of unresolved externs unless the headers are
# out of the way. So: copy the sources into a scratch tree, delete every .elh
# on the import chain, and compile the test there.
#
# Scratch copy on purpose: the worktree is shared with other terminals and
# deleting headers in place would be a shared-tree mutation with no owner.
#
# EXIT STATUS IS THE GATE: non-zero if the binary fails to build, crashes, or if
# its output contains a FAIL line or reports a non-zero failed count. Do not
# "improve" this into something that only checks the exit code of the test
# binary — these El tests print failures and still exit 0.
#
# usage: scripts/run-el-test.sh tests/test_history_amplification.el
set -euo pipefail
TEST_REL="${1:?usage: run-el-test.sh tests/<test>.el}"
SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
TEST_NAME="$(basename "$TEST_REL" .el)"
ELC="${ELC:-$HOME/neuron-dev-stack/src/el/lang/dist/platform/elc}"
[ -x "$ELC" ] || ELC="$HOME/el-sdk/elc"
[ -x "$ELC" ] || { echo "[run-el-test] FAIL: no elc found (set ELC=)"; exit 1; }
RTC="${RTC:-$SRC/vendor/el-runtime/v1.0.0-20260501/el_runtime.c}"
[ -f "$RTC" ] || RTC="$HOME/el-sdk/el_runtime.c"
[ -f "$RTC" ] || { echo "[run-el-test] FAIL: no el_runtime.c found (set RTC=)"; exit 1; }
RTDIR="$(dirname "$RTC")"
EL_REPO="${EL_REPO:-$HOME/Development/neuron-technologies/el}"
SSL="${SSL_PREFIX:-/opt/homebrew/opt/openssl@3}"
GEN="$(mktemp -d "${TMPDIR:-/tmp}/el-test.XXXXXX")"
trap 'rm -rf "$GEN"' EXIT
mkdir -p "$GEN/neuron/tests" "$GEN/foundation/el/elp/src"
cp "$SRC"/*.el "$GEN/neuron/"
cp "$SRC"/tests/*.el "$GEN/neuron/tests/" 2>/dev/null || true
[ -d "$EL_REPO/elp/src" ] && cp "$EL_REPO"/elp/src/*.el "$GEN/foundation/el/elp/src/" 2>/dev/null || true
# The whole recipe depends on there being no headers to short-circuit inlining.
find "$GEN" -name '*.elh' -delete
echo "[run-el-test] compiling $TEST_REL"
( cd "$GEN/neuron" && "$ELC" --target=c "tests/${TEST_NAME}.el" ) > "$GEN/${TEST_NAME}.c"
BODIES=$(grep -c '^el_val_t .*) {$' "$GEN/${TEST_NAME}.c" || true)
echo "[run-el-test] $(wc -c < "$GEN/${TEST_NAME}.c" | tr -d ' ') bytes, ${BODIES} inlined function bodies"
# A test that imports ../chat.el pulls in the bulk of the engine. A tiny body
# count means an import was read from a header instead of inlined, and the test
# would be exercising extern stubs rather than the real code.
if [ "$BODIES" -lt 100 ]; then
echo "[run-el-test] FAIL: only $BODIES inlined bodies — an import was not inlined"
exit 1
fi
cc -O2 -DHAVE_CURL \
-I"$RTDIR" -I"$SSL/include" -L"$SSL/lib" \
"$GEN/${TEST_NAME}.c" "$RTC" \
-lssl -lcrypto -lcurl -lpthread -lm \
-o "$GEN/${TEST_NAME}" 2> "$GEN/cc.log" || {
echo "[run-el-test] FAIL: compile error"; tail -30 "$GEN/cc.log"; exit 1; }
# arm64 pointer-truncation guard (cc-brain.sh's rule): an implicit declaration of
# a runtime symbol truncates its returned pointer to 32 bits.
if grep -E 'implicit.*(engram_|el_)' "$GEN/cc.log"; then
echo "[run-el-test] FAIL: implicit declarations of runtime symbols"; exit 1; fi
# Throwaway HOME so a test can never read or write the live engram at ~/.neuron.
TEST_HOME="$GEN/home"
mkdir -p "$TEST_HOME"
echo "[run-el-test] running $TEST_NAME"
set +e
HOME="$TEST_HOME" NEURON_HOME="$TEST_HOME/.neuron" "$GEN/${TEST_NAME}" 2>&1 | tee "$GEN/out.txt"
RC=${PIPESTATUS[0]}
set -e
if [ "$RC" -ne 0 ]; then
echo "[run-el-test] FAIL: $TEST_NAME exited $RC (crash or abort)"
exit 1
fi
if grep -q " FAIL:" "$GEN/out.txt"; then
echo "[run-el-test] FAIL: $TEST_NAME reported failing assertions"
exit 1
fi
if grep -qE '[1-9][0-9]* failed' "$GEN/out.txt"; then
echo "[run-el-test] FAIL: $TEST_NAME reported a non-zero failed count"
exit 1
fi
if ! grep -q "PASS:" "$GEN/out.txt"; then
echo "[run-el-test] FAIL: $TEST_NAME produced no assertions at all"
exit 1
fi
echo "[run-el-test] PASS: $TEST_NAME"
-937
View File
@@ -1,937 +0,0 @@
#!/usr/bin/env python3
"""state-key-audit.py — the analyzer behind scripts/verify-state-keys.sh.
Read that script's header for WHY this exists (issue #129). This file is the
HOW: a small El reader that resolves the key expression at every state_get /
state_set site, including keys that are computed.
WHAT IT PARSES
El as this engine writes it: `fn f(a: T, b: T) -> T { ... }`, `let x: T = e`,
`return e`, `if c { a } else { b }` as an expression, `+` concatenation,
`"..."` with backslash escapes, `//` line comments. No block comments, no
const/match/struct exist in this dialect (verified over the whole tree).
KEY PATTERNS the only two things a key expression can resolve to
EXACT "soul_model" the whole key is known
PREFIX "session_hist_" a known head, then runtime text
(plus UNRESOLVED, which is a report line and never a failure)
RESOLUTION resolve_expr() returns a SET of patterns; unions are how branches,
multiple returns, and multiple bindings of one name are represented.
literal "k" -> {EXACT k}
concat A + B -> fold left; all-static -> EXACT,
static head + dynamic tail -> PREFIX
if-expression if c {A} else {B} -> resolve(A) | resolve(B), except that
str_eq(X,"") with X statically ""
folds to the taken branch only
call f(args) -> union over f's return expressions,
with f's params bound to THIS call
site's actual argument expressions
local var let k = e; state_get(k)-> union over every `let k =` in the
enclosing function
parameter fn g(k) { state_get(k) }-> union over the argument at that
position across every call site of g
anything else json_get(...), env(...)-> UNRESOLVED
Recursion is depth- and cycle-guarded; a guard trip yields UNRESOLVED, never a
failure.
COVERAGE a read is satisfied when some write can produce the same key:
read EXACT k <- write EXACT k, or write PREFIX p where k starts with p
read PREFIX p <- write EXACT k where k starts with p, or write PREFIX q
where p and q are prefixes of each other
Deliberately permissive at the boundaries: a gate that cries wolf gets deleted.
"""
import os
import re
import sys
MAX_DEPTH = 12
# ── patterns ────────────────────────────────────────────────────────────────
EXACT = "exact"
PREFIX = "prefix"
def pat_exact(s):
return (EXACT, s)
def pat_prefix(s):
# A prefix with no static text at all carries no information; that is the
# UNRESOLVED case, not a pattern.
return (PREFIX, s) if s else None
def covers(write, read):
"""Can a write of pattern `write` produce a key that `read` reads?
The prefix rule is DIRECTIONAL, and that direction is the whole point. A
write namespace that is the same or BROADER than the read namespace covers
it (write "rl:" covers read "rl:x"). A write namespace that is NARROWER does
NOT (write "session_histv2_" does not cover read "session_hist_") being
permissive there re-opens the exact hole this gate exists to close: rename
the producer, leave the readers, stay green. Verified with a control run
that renames sessions.el's writer and leaves its four readers behind."""
wk, wv = write
rk, rv = read
if rk == EXACT:
return rv == wv if wk == EXACT else rv.startswith(wv)
# read is a PREFIX: some key starting with rv is read
if wk == EXACT:
return wv.startswith(rv) # that one written key is in range
return rv.startswith(wv) # write namespace same-or-broader
# ── lexer ───────────────────────────────────────────────────────────────────
TOK_STR, TOK_IDENT, TOK_PUNCT, TOK_NUM = "str", "ident", "punct", "num"
IDENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
NUM_RE = re.compile(r"[0-9]+(\.[0-9]+)?")
class Tok:
__slots__ = ("kind", "val", "line")
def __init__(self, kind, val, line):
self.kind, self.val, self.line = kind, val, line
def __repr__(self):
return "%s(%r)@%d" % (self.kind, self.val, self.line)
def lex(src):
toks, i, n, line = [], 0, len(src), 1
while i < n:
c = src[i]
if c == "\n":
line += 1
i += 1
continue
if c in " \t\r":
i += 1
continue
if c == "/" and i + 1 < n and src[i + 1] == "/":
while i < n and src[i] != "\n":
i += 1
continue
if c == '"':
j, buf = i + 1, []
while j < n:
if src[j] == "\\" and j + 1 < n:
esc = src[j + 1]
buf.append({"n": "\n", "t": "\t", "r": "\r"}.get(esc, esc))
j += 2
continue
if src[j] == '"':
break
if src[j] == "\n":
line += 1
buf.append(src[j])
j += 1
toks.append(Tok(TOK_STR, "".join(buf), line))
i = j + 1
continue
m = IDENT_RE.match(src, i)
if m:
toks.append(Tok(TOK_IDENT, m.group(0), line))
i = m.end()
continue
m = NUM_RE.match(src, i)
if m:
toks.append(Tok(TOK_NUM, m.group(0), line))
i = m.end()
continue
toks.append(Tok(TOK_PUNCT, c, line))
i += 1
return toks
def match_close(toks, i, open_ch, close_ch):
"""toks[i] is open_ch; return index of its matching close_ch."""
depth = 0
while i < len(toks):
if toks[i].kind == TOK_PUNCT:
if toks[i].val == open_ch:
depth += 1
elif toks[i].val == close_ch:
depth -= 1
if depth == 0:
return i
i += 1
return len(toks) - 1
# ── program model ───────────────────────────────────────────────────────────
class Func:
def __init__(self, name, path, line, params, toks, start, end):
self.name, self.path, self.line = name, path, line
self.params = params # [param name]
self.toks = toks # the whole file's token list
self.start, self.end = start, end # body token range, exclusive of braces
self.lets = None # name -> [expr token ranges], lazily built
class Site:
def __init__(self, kind, path, line, func, arg_range, text):
self.kind = kind # "get" | "set"
self.path, self.line = path, line
self.func = func
self.arg_range = arg_range
self.text = text # source text of the key expression
self.pats = set()
self.unresolved = False
self.literal = None # set when the key expression is a bare literal
class Program:
def __init__(self):
self.files = {} # path -> toks
self.funcs = {} # name -> [Func] (El allows no overloads, but be safe)
self.toplevel = [] # [Func] one per file, params=[]
self.sites = [] # [Site]
self.calls = {} # callee name -> [(Func caller, [arg ranges])]
# -- loading ------------------------------------------------------------
def load(self, path, rel):
with open(path, "r", encoding="utf-8", errors="replace") as fh:
src = fh.read()
toks = lex(src)
self.files[rel] = toks
self._scan_funcs(rel, toks)
def _scan_funcs(self, rel, toks):
covered = []
i = 0
while i < len(toks):
t = toks[i]
if t.kind == TOK_IDENT and t.val == "fn" and i + 2 < len(toks) \
and toks[i + 1].kind == TOK_IDENT and toks[i + 2].val == "(":
name = toks[i + 1].val
pclose = match_close(toks, i + 2, "(", ")")
params = self._params(toks, i + 3, pclose)
bopen = pclose + 1
while bopen < len(toks) and toks[bopen].val != "{":
bopen += 1
bclose = match_close(toks, bopen, "{", "}")
f = Func(name, rel, t.line, params, toks, bopen + 1, bclose)
self.funcs.setdefault(name, []).append(f)
covered.append((i, bclose))
i = bclose + 1
continue
i += 1
# everything outside a fn is the file's top-level "function"
tl = Func("<toplevel:%s>" % rel, rel, 1, [], toks, 0, len(toks))
tl.covered = covered
self.toplevel.append(tl)
@staticmethod
def _params(toks, i, end):
"""`a: T, b: T` -> ['a','b'] (top-level commas only)."""
names, depth, expect = [], 0, True
while i < end:
t = toks[i]
if t.kind == TOK_PUNCT and t.val in "([{":
depth += 1
elif t.kind == TOK_PUNCT and t.val in ")]}":
depth -= 1
elif depth == 0 and t.kind == TOK_PUNCT and t.val == ",":
expect = True
elif depth == 0 and expect and t.kind == TOK_IDENT:
names.append(t.val)
expect = False
i += 1
return names
def func_at(self, rel, tok_index):
for f in self.funcs_in(rel):
if f.start <= tok_index < f.end:
return f
for f in self.toplevel:
if f.path == rel:
return f
return None
def funcs_in(self, rel):
for fl in self.funcs.values():
for f in fl:
if f.path == rel:
yield f
# -- indexing -----------------------------------------------------------
def index(self):
for rel, toks in self.files.items():
i = 0
while i < len(toks):
t = toks[i]
if t.kind == TOK_IDENT and i + 1 < len(toks) and toks[i + 1].val == "(" \
and t.val not in KEYWORDS \
and not (i > 0 and toks[i - 1].kind == TOK_IDENT
and toks[i - 1].val == "fn"):
# ^ the `fn f(a: T)` declaration is not a call site; counting
# it as one makes every parameter resolve to its own name
# and reports the whole function UNRESOLVED.
close = match_close(toks, i + 1, "(", ")")
args = split_args(toks, i + 2, close)
self.calls.setdefault(t.val, []).append(
(self.func_at(rel, i), args, rel, t.line))
if t.val in ("state_get", "state_set") and args:
self.sites.append(Site(
"get" if t.val == "state_get" else "set",
rel, t.line, self.func_at(rel, i), args[0],
render(toks, *args[0])))
i += 1
# -- resolution ---------------------------------------------------------
def lets_of(self, f):
if f.lets is not None:
return f.lets
f.lets = {}
toks = f.toks
skip = getattr(f, "covered", [])
i = f.start
while i < f.end:
if any(a <= i <= b for a, b in skip):
i = max(b for a, b in skip if a <= i <= b) + 1
continue
t = toks[i]
if t.kind == TOK_IDENT and t.val == "let" and i + 1 < f.end \
and toks[i + 1].kind == TOK_IDENT:
name = toks[i + 1].val
j = i + 2
if j < f.end and toks[j].val == ":": # skip the type
while j < f.end and toks[j].val != "=":
j += 1
if j < f.end and toks[j].val == "=":
s = j + 1
e = stmt_end(toks, s, f.end)
f.lets.setdefault(name, []).append((s, e))
i = e
continue
i += 1
return f.lets
def returns_of(self, ctx, depth=0, seen=None):
"""The value expressions of a function, in the context it was CALLED in.
Context-sensitive on purpose. `conv_hist_key` is written as a guard:
if str_eq(session_id, "") { return "conv_history" }
return "session_hist_" + session_id
Collecting both returns flat would make state_set(conv_hist_key("")) the
dead handle_chat() write claim to produce the session_hist_ namespace
too. That is a producer this engine does not actually have, and claiming
it would let the gate stay green if sessions.el's real writer vanished:
a masking hole in the exact namespace #129 lives in. So a guard whose
condition folds is honoured, and the branch not taken is dropped."""
out = []
self._values(ctx.toks, ctx.start, ctx.end, ctx, depth,
seen if seen is not None else set(), out)
return out
def _values(self, toks, s, e, ctx, depth, seen, out):
"""Append the value expressions of a statement sequence.
Returns True when the sequence definitely returns (rest unreachable)."""
if depth > MAX_DEPTH:
return False
i = s
while i < e:
t = toks[i]
if t.kind == TOK_IDENT and t.val == "return":
j = stmt_end(toks, i + 1, e)
if j > i + 1:
out.append((i + 1, j))
return True
if t.kind == TOK_IDENT and t.val == "let":
i = stmt_end(toks, i + 2, e)
continue
if t.kind == TOK_IDENT and t.val == "if":
i = self._if_stmt(toks, i, e, ctx, depth, seen, out)
if i is True:
return True
continue
if t.kind == TOK_PUNCT and t.val in "([{":
i = match_close(toks, i, t.val,
{"(": ")", "[": "]", "{": "}"}[t.val]) + 1
continue
en = stmt_end(toks, i, e)
if en <= i:
i += 1
continue
if en >= e: # trailing expression = the value
out.append((i, en))
i = en
return False
def _if_stmt(self, toks, i, e, ctx, depth, seen, out):
"""Walk one if / else-if / else chain. Returns the next index, or True
if the chain definitely returns on every reachable branch."""
bopen = i + 1
while bopen < e and toks[bopen].val != "{":
bopen += 1
if bopen >= e:
return e
bclose = match_close(toks, bopen, "{", "}")
fold = self._fold_cond(toks, i + 1, bopen, ctx, depth, seen)
j = bclose + 1
else_s = else_e = None
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
if j + 1 < e and toks[j + 1].val == "{":
ec = match_close(toks, j + 1, "{", "}")
else_s, else_e = j + 2, ec
j = ec + 1
else: # `else if ...` — the rest of the chain
else_s = j + 1
else_e = stmt_end(toks, j + 1, e)
j = else_e
then_ret = else_ret = False
if fold is not False:
then_ret = self._values(toks, bopen + 1, bclose, ctx, depth + 1, seen, out)
if fold is not True and else_s is not None:
else_ret = self._values(toks, else_s, else_e, ctx, depth + 1, seen, out)
if fold is True and then_ret:
return True
if fold is False and else_s is not None and else_ret:
return True
if fold is None and else_s is not None and then_ret and else_ret:
return True
return j
def resolve(self, rng, func, depth=0, seen=None):
"""-> (set of patterns, unresolved_flag)"""
if seen is None:
seen = set()
if depth > MAX_DEPTH:
return set(), True
return self._expr(func.toks, rng[0], rng[1], func, depth, seen)
# -- expression walker --------------------------------------------------
def _expr(self, toks, s, e, func, depth, seen):
parts, cur, d = [], s, 0
i = s
while i < e: # split on top-level '+'
v = toks[i].val
if toks[i].kind == TOK_PUNCT and v in "([{":
d += 1
elif toks[i].kind == TOK_PUNCT and v in ")]}":
d -= 1
elif d == 0 and toks[i].kind == TOK_PUNCT and v == "+" and i > s:
parts.append((cur, i))
cur = i + 1
i += 1
parts.append((cur, e))
if len(parts) == 1:
return self._primary(toks, s, e, func, depth, seen)
# concatenation: keep folding while every operand so far is EXACT
head, unres = "", False
static = True
for (ps, pe) in parts:
pats, u = self._primary(toks, ps, pe, func, depth, seen)
exacts = {p[1] for p in pats if p[0] == EXACT}
if static and len(exacts) == 1 and not u and len(pats) == 1:
head += exacts.pop()
continue
if static and pats and all(p[0] == EXACT for p in pats) and len(pats) > 1:
# a branchy static operand: keep the shared head only
static = False
head += os.path.commonprefix(sorted({p[1] for p in pats}))
break
static = False
# first non-static operand: everything after it is runtime text
if (ps, pe) == parts[0]:
for p in pats:
if p[0] == PREFIX:
head = p[1]
break
if not head:
unres = True
break
if static:
return {pat_exact(head)}, False
p = pat_prefix(head)
return ({p} if p else set()), (unres or not p)
def _primary(self, toks, s, e, func, depth, seen):
while s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "(" \
and match_close(toks, s, "(", ")") == e - 1:
s, e = s + 1, e - 1
if s >= e:
return set(), True
t = toks[s]
if t.kind == TOK_STR and e == s + 1:
return {pat_exact(t.val)}, False
if t.kind == TOK_IDENT and t.val == "if":
return self._if_expr(toks, s, e, func, depth, seen)
if t.kind == TOK_IDENT and s + 1 < e and toks[s + 1].val == "(":
close = match_close(toks, s + 1, "(", ")")
if close == e - 1:
return self._call(toks, t.val, split_args(toks, s + 2, close),
func, depth, seen)
if t.kind == TOK_IDENT and e == s + 1:
return self._var(t.val, func, depth, seen)
return set(), True
def _if_expr(self, toks, s, e, func, depth, seen):
bopen = s + 1
while bopen < e and toks[bopen].val != "{":
bopen += 1
cond = (s + 1, bopen)
bclose = match_close(toks, bopen, "{", "}")
then_rng = block_tail(toks, bopen + 1, bclose) or (bopen + 1, bclose)
else_rng = None
j = bclose + 1
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
if j + 1 < e and toks[j + 1].val == "{":
ec = match_close(toks, j + 1, "{", "}")
else_rng = block_tail(toks, j + 2, ec) or (j + 2, ec)
else:
else_rng = (j + 1, e) # `else if ...`
taken = self._fold_cond(toks, cond[0], cond[1], func, depth, seen)
rngs = []
if taken is not False:
rngs.append(then_rng)
if taken is not True and else_rng:
rngs.append(else_rng)
pats, unres = set(), False
for r in rngs:
p, u = self._expr(toks, r[0], r[1], func, depth + 1, seen)
pats |= p
unres = unres or u
return pats, unres
def _fold_cond(self, toks, s, e, func, depth, seen):
"""Constant-fold `str_eq(X, "")` / `!str_eq(X, "")` so a helper called with
a literal (conv_hist_key("")) yields only the branch it really takes.
Returns True / False / None(unknown)."""
neg = False
if s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "!":
neg, s = True, s + 1
if not (s < e and toks[s].kind == TOK_IDENT and toks[s].val == "str_eq"
and s + 1 < e and toks[s + 1].val == "("):
return None
close = match_close(toks, s + 1, "(", ")")
if close != e - 1:
return None
args = split_args(toks, s + 2, close)
if len(args) != 2:
return None
va, ua = self._expr(toks, args[0][0], args[0][1], func, depth + 1, seen)
vb, ub = self._expr(toks, args[1][0], args[1][1], func, depth + 1, seen)
if ua or ub or len(va) != 1 or len(vb) != 1:
return None
(ka, sa), (kb, sb) = va.pop(), vb.pop()
if ka != EXACT or kb != EXACT:
return None
r = (sa == sb)
return (not r) if neg else r
def _call(self, toks, name, args, func, depth, seen):
cands = self.funcs.get(name)
if not cands:
return set(), True # builtin: json_get, env, ...
pats, unres = set(), False
for callee in cands:
key = ("fn", callee.path, callee.name, tuple(args))
if key in seen:
unres = True
continue
seen = seen | {key}
# bind the callee's params to THIS call site's argument expressions
binding = {}
for idx, pname in enumerate(callee.params):
if idx < len(args):
binding[pname] = (args[idx], func)
callee_ctx = _Bound(callee, binding)
for r in self.returns_of(callee_ctx, depth + 1, seen):
p, u = self._expr(callee.toks, r[0], r[1], callee_ctx,
depth + 1, seen)
pats |= p
unres = unres or u
return pats, unres
def _var(self, name, func, depth, seen):
real = func.func if isinstance(func, _Bound) else func
# 1. a parameter bound by the call site we came through
if isinstance(func, _Bound) and name in func.binding:
rng, caller_ctx = func.binding[name]
return self._expr(caller_ctx.toks, rng[0], rng[1], caller_ctx,
depth + 1, seen)
# 2. a local `let` in the enclosing function
lets = self.lets_of(real)
if name in lets:
key = ("let", real.path, real.name, name)
if key in seen:
return set(), True
seen = seen | {key}
pats, unres = set(), False
for rng in lets[name]:
p, u = self._expr(real.toks, rng[0], rng[1], real, depth + 1, seen)
pats |= p
unres = unres or u
return pats, unres
# 3. an unbound parameter -> look at every call site of the enclosing fn
if name in real.params:
key = ("param", real.path, real.name, name)
if key in seen:
return set(), True
seen = seen | {key}
idx = real.params.index(name)
pats, unres = set(), False
sites = self.calls.get(real.name, [])
if not sites:
return set(), True
for caller, args, _rel, _line in sites:
if caller is None or idx >= len(args):
unres = True
continue
p, u = self._expr(caller.toks, args[idx][0], args[idx][1],
caller, depth + 1, seen)
pats |= p
unres = unres or u
return pats, unres
# 4. a file-level / cross-file top-level `let`
for tl in self.toplevel:
lets = self.lets_of(tl)
if name in lets:
key = ("let", tl.path, tl.name, name)
if key in seen:
return set(), True
seen2 = seen | {key}
pats, unres = set(), False
for rng in lets[name]:
p, u = self._expr(tl.toks, rng[0], rng[1], tl, depth + 1, seen2)
pats |= p
unres = unres or u
return pats, unres
return set(), True
class _Bound:
"""A callee view that also knows what its params were called with."""
def __init__(self, func, binding):
self.func, self.binding = func, binding
self.toks, self.start, self.end = func.toks, func.start, func.end
self.params, self.path, self.name = func.params, func.path, func.name
def __getattr__(self, k):
return getattr(self.func, k)
# ── token helpers ───────────────────────────────────────────────────────────
def split_args(toks, s, e):
out, cur, d = [], s, 0
i = s
while i < e:
v = toks[i].val
if toks[i].kind == TOK_PUNCT and v in "([{":
d += 1
elif toks[i].kind == TOK_PUNCT and v in ")]}":
d -= 1
elif d == 0 and toks[i].kind == TOK_PUNCT and v == ",":
out.append((cur, i))
cur = i + 1
i += 1
if cur < e:
out.append((cur, e))
return out
STMT_START = {"let", "return", "if", "while", "for"}
KEYWORDS = {"if", "while", "for", "return", "fn", "let", "else", "match"}
def stmt_end(toks, s, limit):
"""End of the expression starting at s: the next top-level statement
boundary. El has no semicolons, so a newline that starts a new statement
ends this one."""
d, i = 0, s
while i < limit:
t = toks[i]
if t.kind == TOK_PUNCT and t.val in "([":
d += 1
elif t.kind == TOK_PUNCT and t.val in ")]":
d -= 1
if d < 0:
return i
elif t.kind == TOK_PUNCT and t.val == "{":
# a brace at depth 0 belongs to this expression only when it is an
# if/else block that is part of it
d += 1
elif t.kind == TOK_PUNCT and t.val == "}":
d -= 1
if d < 0:
return i
elif d == 0 and t.kind == TOK_PUNCT and t.val == ",":
return i
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val in STMT_START:
if t.val == "if" and toks[i - 1].kind == TOK_IDENT and toks[i - 1].val == "else":
i += 1
continue
return i
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val == "fn":
return i
i += 1
return limit
def block_tail(toks, s, e):
"""The trailing expression of a block, if the block ends in one."""
i, last = s, None
while i < e:
t = toks[i]
if t.kind == TOK_IDENT and t.val in ("let", "return"):
i = stmt_end(toks, i + 1, e)
last = None
continue
if t.kind == TOK_PUNCT and t.val in "([{":
i = match_close(toks, i, t.val, {"(": ")", "[": "]", "{": "}"}[t.val]) + 1
continue
st = i
en = stmt_end(toks, i, e)
if en <= st:
i = st + 1
continue
last = (st, en)
i = en
return last
def render(toks, s, e):
out = []
for t in toks[s:e]:
out.append('"%s"' % t.val if t.kind == TOK_STR else t.val)
return " ".join(out)
# ── the gate ────────────────────────────────────────────────────────────────
def collect(root, include_tests):
files = []
for dirpath, dirnames, filenames in os.walk(root):
dirnames[:] = [d for d in dirnames
if d not in ("dist", "vendor", ".git", "node_modules")]
rel_dir = os.path.relpath(dirpath, root)
if not include_tests and rel_dir.split(os.sep)[0] == "tests":
continue
for fn in sorted(filenames):
if fn.endswith(".el"):
rel = os.path.normpath(os.path.join(rel_dir, fn))
files.append((os.path.join(dirpath, fn), rel))
return sorted(files, key=lambda x: x[1])
def is_bare_literal(prog, site):
toks = prog.files[site.path]
s, e = site.arg_range
return e == s + 1 and toks[s].kind == TOK_STR
def read_decl(path):
"""A declaration file: one entry per line, `# ...` comments stripped."""
out = []
if not path or not os.path.exists(path):
return out
with open(path) as fh:
for ln in fh:
ln = ln.split("#", 1)[0].strip()
if ln:
out.append(ln)
return out
def opt(argv, name, default=None):
for i, a in enumerate(argv):
if a == name and i + 1 < len(argv):
return argv[i + 1]
return default
def main(argv):
root = os.path.abspath(argv[1]) if len(argv) > 1 and not argv[1].startswith("-") else "."
include_tests = "--include-tests" in argv
verbose = "--verbose" in argv
baseline_path = opt(argv, "--baseline")
external_path = opt(argv, "--external")
prog = Program()
for path, rel in collect(root, include_tests):
prog.load(path, rel)
prog.index()
for site in prog.sites:
pats, unres = prog.resolve(site.arg_range, site.func)
site.pats, site.unresolved = {p for p in pats if p}, unres
if is_bare_literal(prog, site):
site.literal = prog.files[site.path][site.arg_range[0]].val
writes = [s for s in prog.sites if s.kind == "set"]
reads = [s for s in prog.sites if s.kind == "get"]
write_pats = set()
for w in writes:
write_pats |= w.pats
# Declared host-set keys: written by something outside the El tree (an
# operator, the installer, a host process). Each entry must carry a reason.
external = []
for ln in read_decl(external_path):
parts = ln.split(None, 1)
if len(parts) != 2 or parts[0] not in (EXACT, PREFIX):
print("bad --external line (want `exact|prefix <key>`): %r" % ln,
file=sys.stderr)
return 2
external.append((parts[0], parts[1]))
write_pats |= set(external)
# F1 — a read of a key no write in the tree produces.
f1 = []
for r in reads:
for p in sorted(r.pats):
if not any(covers(w, p) for w in write_pats):
f1.append((r, p))
# F2 — a key namespace owned by a helper, accessed by a hand-rolled literal.
# This is the #129 shape: the producer moved behind conv_hist_key() and
# one consumer kept spelling the old key out by hand.
owners = {} # helper fn name -> its value set
for s in prog.sites:
toks = prog.files[s.path]
a, b = s.arg_range
if toks[a].kind == TOK_IDENT and a + 1 < b and toks[a + 1].val == "(" \
and match_close(toks, a + 1, "(", ")") == b - 1 \
and toks[a].val in prog.funcs:
name = toks[a].val
if name not in owners:
vals = set()
for callee in prog.funcs[name]:
# No call context here on purpose: the OWNED namespace is
# every key the helper can ever produce, over all call sites.
for rng in prog.returns_of(callee):
p, _ = prog._expr(callee.toks, rng[0], rng[1], callee, 0, set())
vals |= {x for x in p if x}
owners[name] = vals
f2 = []
for s in prog.sites:
if s.literal is None:
continue
for owner, vals in sorted(owners.items()):
for v in sorted(vals):
if covers(v, pat_exact(s.literal)):
f2.append((s, owner, v))
break
else:
continue
break
unresolved = [s for s in prog.sites if s.unresolved or not s.pats]
# Baseline signatures carry NO line number on purpose: an unrelated edit that
# shifts a line must not un-mute an accepted finding (that is crying wolf),
# but a GROWTH in count must not hide either. So a baseline entry is
# `<file> <CODE> <detail> [xN]` and only the first N matches are muted.
baseline, bad_baseline = {}, []
for ln in read_decl(baseline_path):
n, key = 1, ln
parts = ln.rsplit(" x", 1)
if len(parts) == 2 and parts[1].isdigit():
key, n = parts[0].strip(), int(parts[1])
baseline[key] = n
def sig(path, code, detail):
return "%s %s %s" % (path, code, detail)
findings = []
for r, p in f1:
findings.append((sig(r.path, "DEAD-READ", "%s:%s" % p), r.line,
" %s:%d state_get(%s)\n resolves to %s %r — no state_set in the tree produces it"
% (r.path, r.line, r.text, p[0].upper(), p[1])))
for s, owner, v in f2:
findings.append((sig(s.path, "HAND-ROLLED", "%s<-%s()" % (s.literal, owner)), s.line,
" %s:%d state_%s(\"%s\")\n %s() owns this key namespace (%s %r) — go through the helper, "
"or a rename orphans this site silently" % (s.path, s.line, s.kind, s.literal, owner, v[0].upper(), v[1])))
findings.sort(key=lambda f: (f[0], f[1]))
live, muted, budget = [], [], dict(baseline)
for f in findings:
if budget.get(f[0], 0) > 0:
budget[f[0]] -= 1
muted.append(f)
else:
live.append(f)
stale = sorted(k for k, v in budget.items() if v > 0)
print("── state-key audit ─────────────────────────────────────────────")
print("scanned %d .el files%s" % (len(prog.files),
"" if include_tests else " (tests/ excluded)"))
print("sites %d state_set, %d state_get" % (len(writes), len(reads)))
print("keys %d distinct write patterns" % len(write_pats))
print("")
if verbose:
print("WRITE PATTERNS")
for k, v in sorted(write_pats):
print(" %-6s %s" % (k, v))
print("")
if external:
print("DECLARED HOST-SET (%d) — %s" % (len(external), external_path))
for k, v in sorted(external):
print(" %-6s %s" % (k, v))
print("")
print("UNRESOLVED (%d) — reported, never fails the build" % len(unresolved))
if not unresolved:
print(" (none)")
for s in sorted(unresolved, key=lambda x: (x.path, x.line)):
print(" %s:%d state_%s(%s)%s"
% (s.path, s.line, s.kind, s.text,
" [partial: %s]" % ", ".join("%s %r" % p for p in sorted(s.pats))
if s.pats else ""))
print("")
if muted:
print("BASELINED (%d) — pre-existing debt accepted in %s. NOT clean; fix these."
% (len(muted), baseline_path))
for sg, line, _ in muted:
print(" %s (line %d)" % (sg, line))
print("")
if stale:
print("STALE BASELINE (%d) — entries that no longer match anything; delete them:"
% len(stale))
for sg in stale:
print(" %s" % sg)
print("")
print("FINDINGS (%d)" % len(live))
if not live:
print(" (none)")
for _, _, body in live:
print(body)
print("")
if live:
print("FAIL: %d state-key finding(s). See scripts/verify-state-keys.sh "
"for why this gate exists (issue #129)." % len(live))
return 1
print("PASS: every resolvable state_get key has a producer, and no key "
"namespace is spelled two ways.")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv))
-28
View File
@@ -1,28 +0,0 @@
# state-key-baseline.txt — findings that already existed when this gate landed
# (2026-08-07). Each one is a REAL defect of the #129 class, not a false
# positive. They are muted only so the gate can be turned on today instead of
# being deferred until the debt is paid; every run still prints them under
# BASELINED with the word "debt".
#
# THIS FILE SHOULD ONLY EVER SHRINK. Adding a line means you are shipping a
# known silent-"" read. If you must, date it and say why in the comment.
#
# format: <file> <CODE> <detail> [xN] # N = how many sites are accepted
# No line numbers on purpose: an unrelated edit must not un-mute an accepted
# finding, but a GROWTH in count is NOT muted — the extra site fails the build.
#
chat.el DEAD-READ exact:soul_identity x5
# ^ soul.el used to run `state_set("soul_identity", soul_identity)`. It was
# deleted on 2026-05-13 in b163fa6 ("feat(awareness): route ISE writes to HTTP
# Engram ..."), a commit about something else entirely, and the five readers in
# chat.el were left behind. Since that date build_system_prompt (737), the
# vision handler (1745), the agentic system prompt (2620), the council
# transcript handler (3425) and 3480 have all been prefixing "" — exactly the
# #129 shape, found by this gate on its first run. Sites: 737, 1745, 2620,
# 3425, 3480. Fix = restore the boot-time write or delete the reads; not done
# here because this branch must not change engine behaviour.
studio.el DEAD-READ exact:soul_principal x1
# ^ studio.el:57 dharma_registry() emits "principal":"" on every call — no
# producer has ever existed in the tree's history (git log -S finds none).
# Never-wired rather than orphaned, same silent-"" result.
-16
View File
@@ -1,16 +0,0 @@
# state-key-external.txt — state keys the engine READS but deliberately never
# WRITES, because a host outside the El tree sets them (an operator, the
# installer, a deployment env). Read scripts/verify-state-keys.sh for why this
# list has to exist and why it has to stay short.
#
# THE RULE FOR ADDING A LINE: the read site must already treat "" as a defined
# default (`if str_eq(x, "") { <default> }`) AND the source must say so in a
# comment. "I could not find the writer" is NOT a reason — that is the #129
# defect, and it belongs in state-key-baseline.txt with a date, not here.
#
# format: exact|prefix <key> # why, and where the source says so
#
exact soul_rate_limit # routes.el:59-61 — "configurable via soul state key ... Falls back to 60 req/min if not set."
exact web_search_tool_version # chat.el:1884-1910 — version lives in state "so a future bump is a config write, not a recompile"; defaults to web_search_20250305
exact platform_auth # stewardship.el:92 — host-set capability flag; fail-CLOSED (anything but "true" denies the platform tool)
exact security_research_authorized # awareness.el:991-996 — state override for env SECURITY_RESEARCH_TOKEN; fail-closed, defaults false
-261
View File
@@ -1,261 +0,0 @@
#!/usr/bin/env bash
# verify-soul-contract.sh — the soul contract gate.
#
# TERMINOLOGY (canonical): the ENGRAM is the brain — the memory/knowledge-graph
# substrate. The binary this gate exercises is the SOUL — the runtime/reasoning
# engine compiled from dist/soul.c that serves the /api/ surface. The app
# (neuron-ui) bundles the soul binary at resources/<platform>/neuron.
#
# WHY THIS EXISTS
# For a while the soul binary was hand-dropped, and a stale one shipped: it
# 404'd several capability routes the app calls (knowledge-graph node
# update/delete, live-run narration, safety-contact, ...). This gate makes
# shipping a stale soul IMPOSSIBLE. It has two enforced sections:
# A. PRESENCE — every route the app calls must be ANSWERED (not 404, not
# the el-runtime "no handler"). This is the packaging gate:
# if it fails, do not package.
# B. IMMUTABILITY — engram nodes/memories are immutable by design. To
# "update" is to create a NEW node + a supersede EDGE back to
# the original; the original is KEPT. To "delete" is to
# supersede/tombstone, never hard-remove. A soul that
# hard-deletes an engram node is DEFECTIVE and fails the gate.
#
# SAFETY
# Never touches the live soul (:7770), live engram (:8742), or ~/.neuron.
# Boots on a throwaway port (default 7799) with HOME=$(mktemp -d), a throwaway
# engram snapshot, a non-genesis cgi id, no ENGRAM_URL (so it uses its own
# in-process store, never the live server), NEURON_API_URL pointed at a dead
# port, and no ANTHROPIC_API_KEY (so no probe triggers a real LLM call).
# Connectors proxy to a HARDCODED 127.0.0.1:7771 (no env override): those
# sub-routes are probed with GET, which the soul maps to a read-only
# connectd_get, so this gate never writes to a running connectd bridge.
#
# USAGE
# scripts/verify-soul-contract.sh <path-to-soul-binary> [port]
# exit 0 = all required routes answered AND no destructive engram mutation;
# non-zero = a route is missing (presence) or a mutation route hard-deletes.
set -uo pipefail
SOUL="${1:?usage: verify-soul-contract.sh <soul-binary> [port]}"
PORT="${2:-7799}"
if [ "$PORT" = "7770" ] || [ "$PORT" = "8742" ] || [ "$PORT" = "7771" ]; then
echo "REFUSING: port $PORT is a live service port. Use a throwaway port." >&2
exit 2
fi
if [ ! -x "$SOUL" ]; then echo "not executable: $SOUL" >&2; exit 2; fi
BASE="http://127.0.0.1:$PORT"
THROW_HOME="$(mktemp -d "${TMPDIR:-/tmp}/soul-contract-home.XXXXXX")"
SOUL_LOG="$(mktemp "${TMPDIR:-/tmp}/soul-contract-log.XXXXXX")"
SOUL_PID=""
cleanup() {
[ -n "$SOUL_PID" ] && kill "$SOUL_PID" 2>/dev/null
[ -n "$SOUL_PID" ] && { sleep 0.3; kill -9 "$SOUL_PID" 2>/dev/null; }
rm -rf "$THROW_HOME" "$SOUL_LOG"
}
trap cleanup EXIT INT TERM
# =============================================================================
# THE CONTRACT — routes the app (neuron-ui/src/main/kotlin/ai/neuron/ui/*.kt)
# calls against the soul ($SOUL). Format: "METHOD PATH".
#
# EXCLUDED and why:
# /api/auth, /api/auth/status, /api/dispatch, /api/tasks
# -> served by the APP's own DispatchServer.kt (localhost:8080), not the
# soul. Not soul routes.
# /api/tags
# -> not handled by any soul .el (app-side/other). Pre-verified excluded.
# /api/neuron/, /api/neuron/node/, /api/connectors/ (bare prefixes)
# -> base-path string constants used to build the concrete routes below.
#
# KNOWN-PENDING (probed + reported, NON-blocking):
# POST /api/engram/import -> the app's "Restore memory" path. No soul handler
# yet, and the app hides Restore from shipped builds (B3, "lands in an
# update"). Reported so we see it; does not block packaging.
#
# Connectors sub-routes are listed as GET (see SAFETY note): handle_connectors is
# monolithic, so a GET reaching it proves the whole connectors surface without
# writing to the live bridge. Binary-strings cross-check confirms each POST
# sub-path literal is compiled in.
# =============================================================================
REQUIRED=(
"GET /api/graph/nodes"
"GET /api/graph/edges"
"POST /api/chat"
"GET /api/config"
"POST /api/see"
"GET /api/connectors"
"GET /api/connectors/add"
"GET /api/connectors/toggle"
"GET /api/connectors/auto-approve"
"GET /api/connectors/remove"
"GET /api/connectors/secret"
"GET /api/connectors/oauth/start"
"GET /api/connectors/call"
"POST /api/neuron/memory"
"POST /api/neuron/memory/update"
"POST /api/neuron/memory/delete"
"POST /api/neuron/node/create"
"POST /api/neuron/node/update"
"POST /api/neuron/node/delete"
"POST /api/neuron/knowledge/capture"
"POST /api/neuron/knowledge/evolve"
"POST /api/neuron/knowledge/promote"
"POST /api/neuron/processes/define"
"GET /api/run-progress/__contract_probe__"
"GET /api/safety-contact"
"POST /api/safety-contact"
"GET /api/sessions/__contract_probe__"
)
KNOWN_PENDING=(
"POST /api/engram/import"
)
# --- boot the soul -----------------------------------------------------------
# Preserve the ambient environment (PATH, LD_LIBRARY_PATH, TMPDIR) so the
# dynamically-linked soul finds its libs on any runner — using `env -i` here
# stripped the library path on the Linux CI runner and the soul never booted.
# Isolation is still guaranteed by UNSETTING the live-service vars (so it can
# never reach the real engram/axon or make an LLM call) and by pointing HOME +
# the snapshot at throwaway paths and the axon at a dead port.
#
# ISOLATION FIX (2026-08-03): unsetting ENGRAM_URL/SOUL_ENGRAM_URL is NOT enough.
# The periodic engram sync (awareness.el) resolves its source as
# env(SOUL_ISE_URL) -> state(soul_engram_url) -> DEFAULT http://localhost:8742
# so with those vars unset it silently pulls the LIVE engram (:8742) if that
# server is up — the throwaway graph ballooned 56 -> 12k live nodes within
# seconds, and the concurrent live-sync mutation both (a) broke test determinism
# (the low-salience tombstone marker fell past the 9999 scan cap) and (b) meant
# the "isolated" gate was reading the operator's live brain. Pin SOUL_ISE_URL to
# the dead axon port so the sync target is unreachable: the soul stays on its own
# in-process store, the gate is genuinely isolated, and Section B is deterministic.
echo "== booting soul: $SOUL on port $PORT (throwaway HOME=$THROW_HOME) =="
env \
-u ENGRAM_URL -u ENGRAM_API_KEY -u SOUL_ENGRAM_URL \
-u ANTHROPIC_API_KEY -u NEURON_LLM_API_KEY -u SOUL_IDENTITY \
HOME="$THROW_HOME" \
NEURON_PORT="$PORT" \
SOUL_CGI_ID="ntn-contract-$$" \
SOUL_ENGRAM_PATH="$THROW_HOME/throwaway-snapshot.json" \
NEURON_API_URL="http://127.0.0.1:9" \
SOUL_ISE_URL="http://127.0.0.1:9" \
SOUL_TICK_MS="3600000" SOUL_HEARTBEAT_MS="3600000" SOUL_REFRESH_MS="3600000" \
"$SOUL" >"$SOUL_LOG" 2>&1 &
SOUL_PID=$!
UP=0
for _ in $(seq 1 60); do
if ! kill -0 "$SOUL_PID" 2>/dev/null; then
echo "!! soul exited during boot. log tail:" >&2; tail -20 "$SOUL_LOG" >&2; exit 3
fi
RSS=$(ps -o rss= -p "$SOUL_PID" 2>/dev/null | tr -d ' ')
if [ -n "$RSS" ] && [ "$RSS" -gt $((3*1024*1024)) ]; then
echo "!! soul RSS >3GB — kill -9" >&2; kill -9 "$SOUL_PID" 2>/dev/null; exit 3
fi
[ "$(curl -s -o /dev/null -w '%{http_code}' -m 2 "$BASE/health" 2>/dev/null)" = "200" ] && { UP=1; break; }
sleep 0.5
done
[ "$UP" = 1 ] || { echo "!! soul never healthy on $BASE/health" >&2; tail -20 "$SOUL_LOG" >&2; exit 3; }
echo "== soul healthy =="; echo
# --- probing helpers ---------------------------------------------------------
# request METHOD PATH [BODY] -> prints response body (single line)
request() {
curl -s -m 12 -X "$1" -H 'Content-Type: application/json' --data "${3:-{}}" "$BASE$2" 2>/dev/null | tr -d '\n'
}
# is_missing BODY -> 0 if the body is a "route not present" signal
is_missing() {
printf '%s' "$1" | grep -qE '"error":"not found"|"code":"not_found"|no http handler registered|"code":"method_not_allowed"'
}
extract_id() { printf '%s' "$1" | grep -oE '"id":"[^"]+"' | head -1 | sed 's/.*"id":"//;s/"//'; }
# BY-ID immutability verification (#199 by-id gate).
# The prior check grepped /api/graph/nodes, i.e. engram_scan_nodes_json(9999,0):
# a whole-graph dump. On a genesis-seeded throwaway engram that list is both
# huge (multi-MB, full node content + embeddings) AND capped at 9999 nodes in
# salience order. The tombstone MARKER is written at salience 0.01, so it sorts
# dead last — and once the seeded graph approaches/exceeds the cap the marker
# falls off the end of the list entirely (or is lost to transport truncation on
# the ~10MB body). The normal-salience original still sorts early and survives,
# which is why a correctly-tombstoning soul false-reported "kept but no tombstone
# marker". Fix: verify BY ID via /api/neuron/graph?id=<id>&depth=1
# (engram_neighbors_json, direction "both"), a compact ~900B neighborhood that is
# independent of total graph size. The node itself proves KEPT; the incoming
# "tombstones" edge surfaces the "tombstone:<id>" marker. Pass/fail semantics are
# unchanged and strictly no weaker: a hard-delete leaves no by-id node (DESTROYED)
# and a no-op delete leaves no marker (NO-OP) — both still fail. Verified: a
# never-created id returns neither node nor marker.
node_present() { # id -> 0 if the node still resolves by-id (KEPT), non-zero if hard-removed
request GET "/api/neuron/graph?id=$1&depth=1" | grep -qF "\"$1\""
}
# --- SECTION A: presence -----------------------------------------------------
run_presence() {
local -n arr=$1; local fail=0
printf ' %-8s %-42s %s\n' "METHOD" "ROUTE" "RESULT"
for e in "${arr[@]}"; do
local m p body; m=$(awk '{print $1}' <<<"$e"); p=$(awk '{print $2}' <<<"$e")
body=$(request "$m" "$p")
if is_missing "$body"; then
printf ' %-8s %-42s MISSING %s\n' "$m" "$p" "$(cut -c1-46 <<<"$body")"; fail=$((fail+1))
else
printf ' %-8s %-42s ANSWERED %s\n' "$m" "$p" "$(cut -c1-46 <<<"$body")"
fi
done
return $fail
}
echo "== SECTION A: PRESENCE (required, blocking) =="
run_presence REQUIRED; A_FAIL=$?
echo
echo "== KNOWN-PENDING (non-blocking) =="
run_presence KNOWN_PENDING; P_FAIL=$?
echo
# --- SECTION B: immutability (engram write routes must supersede, not destroy) --
# For each mutation route: create a node, mutate it, then check the ORIGINAL id
# still exists in the graph. KEPT = supersede/tombstone (correct). DESTROYED =
# hard delete (DEFECTIVE -> fail). N/A = mutate route absent (a presence failure).
# For "delete" mutations we additionally require a real tombstone marker
# (label "tombstone:<id>") so a no-op delete cannot false-pass as KEPT.
marker_present() { # id -> 0 if a "tombstone:<id>" marker is wired to the node (by-id neighborhood)
request GET "/api/neuron/graph?id=$1&depth=1" | grep -qF "tombstone:$1"
}
immut_check() { # label KIND(update|delete) CREATE_PATH MUTATE_PATH
local label="$1" kind="$2" create="$3" mutate="$4"
local cbody id mb mbody
cbody=$(request POST "$create" "{\"content\":\"__immut_${label}__\",\"node_type\":\"Memory\",\"label\":\"contract:immut\"}")
id=$(extract_id "$cbody")
if [ -z "$id" ]; then printf ' %-14s SKELETON-FAIL create returned no id: %s\n' "$label" "$(cut -c1-40 <<<"$cbody")"; return 2; fi
mb="{\"id\":\"$id\"}"; [ "$kind" = update ] && mb="{\"id\":\"$id\",\"content\":\"__immut_${label}_v2__\"}"
mbody=$(request POST "$mutate" "$mb")
if is_missing "$mbody"; then printf ' %-14s N/A mutate route absent (see Section A)\n' "$label"; return 0; fi
if ! node_present "$id"; then
printf ' %-14s DESTROYED original %s hard-removed <== DEFECTIVE\n' "$label" "$id"; return 1
fi
if [ "$kind" = delete ] && ! marker_present "$id"; then
printf ' %-14s NO-OP original %s kept but no tombstone marker <== DEFECTIVE\n' "$label" "$id"; return 1
fi
local how="supersede edge"; [ "$kind" = delete ] && how="tombstoned + hidden from default list"
printf ' %-14s KEPT original %s survived (%s)\n' "$label" "$id" "$how"; return 0
}
echo "== SECTION B: IMMUTABILITY (engram nodes must be superseded, never destroyed) =="
B_FAIL=0
immut_check "memory-update" update /api/neuron/memory /api/neuron/memory/update || B_FAIL=$((B_FAIL+$?))
immut_check "memory-delete" delete /api/neuron/memory /api/neuron/memory/delete || B_FAIL=$((B_FAIL+$?))
immut_check "node-update" update /api/neuron/node/create /api/neuron/node/update || B_FAIL=$((B_FAIL+$?))
immut_check "node-delete" delete /api/neuron/node/create /api/neuron/node/delete || B_FAIL=$((B_FAIL+$?))
immut_check "memory-forget" delete /api/neuron/memory /api/neuron/memory/forget || B_FAIL=$((B_FAIL+$?))
echo
echo "============================================================"
RC=0
if [ "$A_FAIL" -gt 0 ]; then echo "PRESENCE: FAIL — $A_FAIL required route(s) unanswered. Do NOT package."; RC=1
else echo "PRESENCE: PASS — all ${#REQUIRED[@]} required routes answered."; fi
if [ "$B_FAIL" -gt 0 ]; then echo "IMMUTABILITY: FAIL — $B_FAIL engram write route(s) hard-delete. DEFECTIVE soul."; RC=1
else echo "IMMUTABILITY: PASS — no engram write route hard-deletes."; fi
[ "$P_FAIL" -gt 0 ] && echo "note: $P_FAIL known-pending route(s) unanswered (expected; non-blocking)."
echo "============================================================"
[ "$RC" = 0 ] && echo "GATE: PASS" || echo "GATE: FAIL"
exit $RC
-118
View File
@@ -1,118 +0,0 @@
#!/usr/bin/env bash
# verify-state-keys.sh — the state-key gate. Retires a defect class at build time.
#
# ── WHY THIS EXISTS. DO NOT DELETE IT AS NOISE. ──────────────────────────────
#
# The engine keeps runtime values in a key-value store: state_set("k", v) writes,
# state_get("k") reads. A read of a key that NOTHING writes returns an empty
# string. Silently. No error, no warning, no log line. The El compiler cannot see
# it, no test sees it, and the product keeps running — just with a hole in it.
#
# That is how issue #129 happened. ff421d3 (2026-08-05) correctly moved
# conversation history to a per-session key behind conv_hist_key(session_id). One
# consumer did not move with it: the agentic path's L1 safety screen kept reading
# the old anonymous "conv_history" bucket. The desktop app always mints a session
# id, so history was always written under session_hist_<id> and that read always
# returned "". The half of the crisis score that receives history is the
# ESCALATION half — the one that exists for distress building across several
# turns, where no single message trips the bell on its own. It scored 0 on every
# real conversation for two days, and nothing failed.
#
# The line that broke carried a comment describing this exact bug being fixed
# once already, under issue #9. A comment is not a gate. This is the gate.
#
# ── WHAT IT CHECKS ──────────────────────────────────────────────────────────
#
# DEAD-READ a state_get whose key resolves to something no state_set in the
# tree produces. The direct form of the class.
#
# HAND-ROLLED a state_get/state_set that spells out a literal belonging to a
# key namespace a helper function owns (e.g. "conv_history", owned
# by conv_hist_key()). This is #129's actual shape: the producer
# moved behind the helper and one consumer kept the old spelling
# by hand. DEAD-READ alone does NOT catch #129, because the dead
# handle_chat() still writes that key through the helper — so this
# second check is the one that earns the gate its keep.
#
# ── WHY IT DOES NOT CRY WOLF ────────────────────────────────────────────────
#
# Keys are usually COMPUTED, not literal, so a naive grep would flood and get
# switched off within a day. scripts/state-key-audit.py resolves computed keys:
# string concatenation (matched on the static prefix), helper functions (resolved
# to their possible return values), keys built into a local variable, and keys
# arriving as a function parameter (resolved through the call sites). Where a key
# genuinely cannot be resolved it is printed under UNRESOLVED and does NOT fail
# the build — visible, never silently ignored. Keep that list short.
#
# On this tree it resolves 278 of 278 sites: UNRESOLVED is 0 and FINDINGS is 0.
#
# Two declaration files, both of which should only ever shrink:
# scripts/state-key-external.txt keys a host outside the El tree writes
# scripts/state-key-baseline.txt findings that predate the gate (real debt)
#
# ── PROVEN TO DISCRIMINATE (2026-08-07) ─────────────────────────────────────
#
# 1. Synthetic: a scratch copy of this tree with agentic_safety_screen reverted
# to the pre-fix state_get("conv_history") — ONE line, nothing else — FAILS
# with `chat.el:2536 ... conv_hist_key() owns this key namespace`. The tree
# as shipped PASSES. One variable, opposite verdicts.
# 2. Independent: run read-only against origin/feat/soul-openai-tools-v2, which
# carries the same defect on its own, the gate reported chat.el:2937 — the
# exact line 43d0449's commit message had named by hand. Against that
# branch's fix (origin/fix/129-on-openai-tools) it passes.
# 3. Producer-moved controls: renaming the sole writer of an EXACT key
# (soul_model) orphans 3 readers across 3 files; renaming the sole writer of
# a PREFIX namespace (agent_workspace_root_*) orphans 3 readers — including
# when the producer moves to a NARROWER namespace, which an earlier,
# sloppier prefix rule let through.
#
# It also found, on its first run, a defect nobody was looking for: soul.el's
# `state_set("soul_identity", ...)` was deleted on 2026-05-13 in b163fa6 (a
# commit about awareness/ISE writes) and five readers in chat.el were left
# behind — the system prompt, the vision handler, the agentic prompt and the
# council handler have been prefixing "" ever since. See state-key-baseline.txt.
#
# ── SAFETY ──────────────────────────────────────────────────────────────────
# Pure static read of .el sources. Starts nothing, opens no port, touches no
# daemon, and never reads or writes ~/.neuron.
#
# ── USAGE ───────────────────────────────────────────────────────────────────
# scripts/verify-state-keys.sh gate the repo (honours baseline)
# scripts/verify-state-keys.sh --strict ignore the baseline: show the debt
# scripts/verify-state-keys.sh --verbose also dump every write pattern
# scripts/verify-state-keys.sh --root DIR audit a different tree
# exit 0 = clean; 1 = finding(s); 2 = the gate itself could not run.
set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
STRICT=0
PASS_THROUGH=()
while [ $# -gt 0 ]; do
case "$1" in
--strict) STRICT=1; shift ;;
--root) ROOT="${2:?--root needs a directory}"; shift 2 ;;
-h|--help) awk 'NR>1 && /^#/ {print; next} NR>1 {exit}' "${BASH_SOURCE[0]}"; exit 0 ;;
*) PASS_THROUGH+=("$1"); shift ;;
esac
done
command -v python3 >/dev/null 2>&1 || {
echo "[state-keys] CANNOT RUN: python3 not found" >&2; exit 2; }
[ -d "$ROOT" ] || { echo "[state-keys] CANNOT RUN: no such tree: $ROOT" >&2; exit 2; }
AUDIT="$SCRIPT_DIR/state-key-audit.py"
[ -f "$AUDIT" ] || { echo "[state-keys] CANNOT RUN: missing $AUDIT" >&2; exit 2; }
ARGS=("$ROOT" "--external" "$SCRIPT_DIR/state-key-external.txt")
[ "$STRICT" -eq 0 ] && ARGS+=("--baseline" "$SCRIPT_DIR/state-key-baseline.txt")
[ ${#PASS_THROUGH[@]} -gt 0 ] && ARGS+=("${PASS_THROUGH[@]}")
python3 "$AUDIT" "${ARGS[@]}"
RC=$?
if [ "$RC" -gt 1 ]; then
echo "[state-keys] CANNOT RUN: the audit itself failed (exit $RC)" >&2
exit 2
fi
exit "$RC"
+8 -19
View File
@@ -87,7 +87,7 @@ fn session_create(body: String) -> String {
let folder: String = json_get(body, "folder")
let content: String = session_make_content(id, title, ts, ts, folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -358,7 +358,7 @@ fn session_update_patch(session_id: String, body: String) -> String {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, eff_title, created_int, ts, eff_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_node_id: String = wt_node(
let new_node_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -456,7 +456,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
// TODO(reliability #7): delete-then-insert is not atomic concurrent saves for the
// same session can produce orphan history nodes. State is primary truth; engram fallback.
let tags: String = "[\"session\",\"session-history\",\"Conversation\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
hist, "Conversation", "session:messages:" + session_id,
el_from_float(0.6), el_from_float(0.6), el_from_float(0.9),
"Episodic", tags
@@ -488,7 +488,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
+ " | ts:" + int_to_str(ts_now)
let summary_tags: String = "[\"session-emotional-summary\",\"affective\",\"bell:" + eff_level + "\",\"BellEvent\"]"
let summary_sal: String = if str_eq(eff_level, "hard") { el_from_float(0.95) } else { el_from_float(0.85) }
let sum_discard: String = wt_node(
let sum_discard: String = engram_node_full(
summary_content,
"BellEvent",
"session:emotional-summary",
@@ -529,7 +529,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
if !str_eq(ot_id, "") { engram_forget(ot_id) }
let oti = oti + 1
}
let discard_topic: String = wt_node(
let discard_topic: String = engram_node_full(
topic_content, "Conversation", topic_label,
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", topic_tags
@@ -582,7 +582,7 @@ fn session_update_meta_timestamp(session_id: String) -> Void {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, old_title, created_int, ts, old_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_id: String = wt_node(
let new_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -629,7 +629,7 @@ fn session_auto_title(session_id: String, first_message: String) -> Void {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, new_title, created_int, ts, old_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_id: String = wt_node(
let new_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -677,11 +677,6 @@ fn handle_session_approve(session_id: String, body: String) -> String {
// path for all sessions created through handle_chat_agentic / agentic_loop.
let bridge_blob: String = state_get("mcp_bridge:" + session_id)
if !str_eq(bridge_blob, "") {
// BUG-LEAK fix (2026-07-16): the approved tool executes below via dispatch_tool,
// whose path/command guards read the shared workspace-root key. Re-assert THIS
// session's own root first an approval must never execute under whatever root
// the last unrelated request left behind.
state_set("agent_workspace_root", state_get("agent_workspace_root_" + session_id))
// For "always": record tool_name in the always-allow list before resuming.
// The tool_name is not stored in the bridge blob (only tool_use_id is).
// Accept it from the body so the client can pass it along.
@@ -713,13 +708,7 @@ fn handle_session_approve(session_id: String, body: String) -> String {
// For builtin tools with no client-provided content: fall back to
// dispatch_tool so those tools still execute correctly.
let client_content: String = json_get(body, "content")
// BUG-6 fix (2026-07-17): the naive json_get scanner matches "content" ANYWHERE
// in the body including INSIDE tool_input so every approved write_file (whose
// input always carries a content field) was mistaken for client-executed, never
// dispatched, and narrated as done: a false receipt with no file on disk. Builtin
// tools now ALWAYS dispatch server-side; client content is only honored for
// non-builtin (MCP/client-executed) tools. Stricter only.
let use_client_content: Bool = !str_eq(client_content, "") && !is_builtin_tool(approve_tool_name)
let use_client_content: Bool = !str_eq(client_content, "")
let use_dispatch: Bool = is_builtin_tool(approve_tool_name) && !use_client_content
let raw_input: String = json_get_raw(body, "tool_input")
let eff_input: String = if str_eq(raw_input, "") { "{}" } else { raw_input }
+36 -110
View File
@@ -379,23 +379,9 @@ fn emit_session_start_event() -> Void {
// layered_cycle routes user-facing requests through the 4-layer consciousness stack.
// L0 (core) L1 (safety screen) L2a (continuity + behavioral profiling) L2b (mission alignment) L3 (imprint) L1 (safety validate)
// Internal cognition (heartbeat, proactive, memory ops) bypasses layers use one_cycle directly.
//
// FIX B (2026-08-05) the cycle now knows which conversation it is in.
//
// session_id: the caller's session, threaded from the route. Was previously read from the
// state key "current_session_id", which is read HERE and written NOWHERE in the entire
// source verified across every .el file. So this value was unconditionally "", and every
// downstream consumer of it silently fell back to a process-global bucket: conversation
// history, and the steward's continuity tracking (TODO reliability #4, below, describes the
// cross-session bleed this caused; threading the real id closes it). The plain path's blank
// stare and the agentic path's scoped history were the same defect seen from two sides.
//
// utility: true when the generation is not part of the user's conversation the app's
// title and insight passes. Answered normally, never recorded. See is_utility_request.
fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String {
// Safety-screen history amplification now reads the SAME window the turn will be
// recorded into, so a session's own escalation pattern is what gets scored.
let history: String = state_get(conv_hist_key(session_id))
fn layered_cycle(raw_input: String) -> String {
let history: String = state_get("conv_history")
let session_id: String = state_get("current_session_id")
// L1 in: safety screen
let screen_result: String = safety_screen(raw_input, history)
@@ -437,10 +423,8 @@ fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
let cont_action: String = json_get(continuity, "action")
// Store continuity status so imprint can adjust its response register.
// TODO(reliability #4) CLOSED 2026-08-05: this line was already written to scope per
// session it just never received a session id, because the only source was a state key
// nothing wrote. It is now threaded from the route, so named sessions genuinely get their
// own continuity state and only anonymous callers share the global one.
// TODO(reliability #4): session_continuity is process-global; scope per session_id
// when available to prevent cross-session bleed under concurrent layered_cycle calls.
let cont_key: String = if str_eq(session_id, "") { "session_continuity" } else { "session_continuity:" + session_id }
state_set(cont_key, cont_status)
@@ -469,24 +453,40 @@ fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
let lc_aff_cutoff: Int = time_now() - 259200
let lc_bell_nodes: String = engram_search_json("bell:soft bell:hard BellEvent affective", 2)
let lc_has_bell: Bool = !str_eq(lc_bell_nodes, "") && !str_eq(lc_bell_nodes, "[]")
// CRASH FIX 2026-08-05 (BUG-PLAINCHAT-1): the " | ts:" parser used to be inline here.
// Inside this block-expression initializer elc compiled `lbmp + str_len(lbm)` to
// el_str_concat() on two integers, which segfaulted the whole daemon the moment a
// distress turn followed an earlier affective turn i.e. exactly on the crisis path.
// Verified against the unmodified baseline binary AND present in the committed
// dist/soul.c. affective_node_ts() is a top-level function, where the same expression
// compiles to integer addition. Do not inline it back.
let lc_bell_note: String = if lc_has_bell {
let lb0: String = json_array_get(lc_bell_nodes, 0)
let lb_ts: Int = affective_node_ts(lb0)
let lb_c: String = json_get(lb0, "content")
let lbm: String = " | ts:"
let lbmp: Int = str_index_of(lb_c, lbm)
let lb_ts_raw: String = if lbmp >= 0 {
let lbs: Int = lbmp + str_len(lbm)
let lbr: String = str_slice(lb_c, lbs, str_len(lb_c))
let lbn: Int = str_index_of(lbr, " | ")
if lbn < 0 { lbr } else { str_slice(lbr, 0, lbn) }
} else {
let lbca: String = json_get(lb0, "created_at")
if str_eq(lbca, "") { json_get(lb0, "updated_at") } else { lbca }
}
let lb_ts: Int = if str_eq(lb_ts_raw, "") { 0 } else { str_to_int(lb_ts_raw) }
if lb_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User was in distress in a recent session.]" } else { "" }
} else { "" }
let lc_pos_nodes: String = engram_search_json("PositiveEvent joy:high joy:low affective", 2)
let lc_has_pos: Bool = !str_eq(lc_pos_nodes, "") && !str_eq(lc_pos_nodes, "[]")
// Same crash fix as the bell note above (BUG-PLAINCHAT-1).
let lc_pos_note: String = if lc_has_pos && str_eq(lc_bell_note, "") {
let lp0: String = json_array_get(lc_pos_nodes, 0)
let lp_ts: Int = affective_node_ts(lp0)
let lp_c: String = json_get(lp0, "content")
let lpm: String = " | ts:"
let lpmp: Int = str_index_of(lp_c, lpm)
let lp_ts_raw: String = if lpmp >= 0 {
let lps: Int = lpmp + str_len(lpm)
let lpr: String = str_slice(lp_c, lps, str_len(lp_c))
let lpn: Int = str_index_of(lpr, " | ")
if lpn < 0 { lpr } else { str_slice(lpr, 0, lpn) }
} else {
let lpca: String = json_get(lp0, "created_at")
if str_eq(lpca, "") { json_get(lp0, "updated_at") } else { lpca }
}
let lp_ts: Int = if str_eq(lp_ts_raw, "") { 0 } else { str_to_int(lp_ts_raw) }
if lp_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User shared positive news in a recent session.]" } else { "" }
} else { "" }
let lc_affective_note: String = if !str_eq(lc_bell_note, "") { lc_bell_note } else { lc_pos_note }
@@ -498,47 +498,11 @@ fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
}
state_set("layered_cycle_safety_system_addendum", augmented_addendum)
// L3: imprint responds applies the active imprint's voice/domain annotation to the
// steward-aligned input. This produces the PROMPT, not the answer.
let prompt: String = imprint_respond(aligned, imprint_id)
// L3: imprint responds
let output: String = imprint_respond(aligned, imprint_id)
// L3b: the imprint SPEAKS (added 2026-08-05).
//
// Until now the cycle stopped at the annotation above, so /api/chat with agentic:false
// handed the user's own screened text back as the "reply" every gate ran, but nothing
// ever generated. The generation is placed HERE, inside the cycle, rather than by
// pointing the route at handle_chat(): handle_chat has no enforcing input gate and no
// enforcing output gate, so calling it instead of this cycle would have traded the whole
// safety pipeline for a working reply. Composing keeps both.
//
// Order is deliberate and must not be rearranged: this call sits strictly AFTER the L1
// screen, the safe-mode guard, the hard-bell short-circuit and the L2 stewardship layers,
// and strictly BEFORE the L1 output gate. A hard bell never reaches a model the branch
// above returns first. Tools are not offered on this turn; see layered_generate.
let output: String = layered_generate(prompt, imprint_id, session_id)
// L1 out: validate output before delivery. Still the terminal gate nothing below this
// line can change the string this function returns.
let validated: String = safety_validate(output, screen_action)
// Turn bookkeeping. Records the VALIDATED text, never the raw model output, and is only
// reachable on the non-bell path: both bell branches above return before this point, so
// bell turns still never enter conversation history. Pure state side effect it cannot
// alter what is returned.
//
// FIX A: the receipt is unconditional and always negative on this path, because on this
// path it is structurally true layered_generate offers no tools at all (build_system_prompt
// chat mode + a request body with no "tools" key). Recording "no tools ran" is not padding:
// it is the only thing that distinguishes "nothing ran" from "we forgot to write down what
// ran", and that ambiguity is what made the model confess to a search it had performed.
//
// FIX E1: a utility generation is answered but not recorded. Guarded here rather than at
// the route so every /api/chat dispatch site inherits it from one place.
let receipt: String = tool_receipt("", "")
if !utility {
conv_history_record(session_id, raw_input, validated, receipt)
}
return validated
// L1 out: validate output before delivery
return safety_validate(output, screen_action)
}
let soul_cgi_id_raw: String = env("SOUL_CGI_ID")
@@ -557,28 +521,7 @@ let axon_raw: String = env("NEURON_API_URL")
let axon_base: String = if str_eq(axon_raw, "") { "http://localhost:7771" } else { axon_raw }
let studio_dir_raw: String = env("SOUL_STUDIO_DIR")
let studio_dir: String = if str_eq(studio_dir_raw, "") { env("HOME") + "/Development/neuron-technologies/products/cgi-studio/el-daemon" } else { studio_dir_raw }
// RESTORED 2026-08-09 this producer was added 2026-05-02 in 601e0fe and deleted
// by the awareness refactor b163fa6 a few days later. Nothing has written
// soul_identity since, while FIVE sites in chat.el kept reading it:
// chat.el:737, 1745, 2620, 3425, 3480 each doing state_get("soul_identity")
// and splicing the result into the system prompt beside the voice, security and
// capability rules. They have been splicing an EMPTY STRING for roughly three
// months. The identity section of every chat turn was blank and nothing said so.
//
// Found by the #132 state-key gate, which reports a read with no producer as a
// build error rather than a silence the whole reason that gate exists.
//
// Restored verbatim rather than improved: this key is an env-configurable persona
// LINE, which is NOT the same thing as soul_identity_context (the graph-derived
// [INTELLECTUAL-DNA]/[VALUES]/[MEMORY-PHILOSOPHY] block written at soul.el:184).
// Pointing these five reads at that block instead would have substituted different
// content and called it a fix. Whether the chat system prompt should ALSO carry the
// graph-derived block is a real question, and a separate one.
let identity_raw: String = env("SOUL_IDENTITY")
let soul_identity: String = if str_eq(identity_raw, "") { "You are " + soul_cgi_id + ", a CGI." } else { identity_raw }
state_set("soul_identity", soul_identity)
let studio_dir: String = if str_eq(studio_dir_raw, "") { "/Users/will/Development/neuron-technologies/products/cgi-studio/el-daemon" } else { studio_dir_raw }
println("[soul] boot - cgi=" + soul_cgi_id + " port=" + int_to_str(port))
@@ -678,23 +621,6 @@ if is_genesis && safe_to_seed {
}
}
// CRASH RECOVERY (neuron#117). Deltas the previous process staged but could not
// hand to the owner are still on disk the spool is a filesystem queue, not a
// memory buffer, precisely so that a soul that died mid-flight does not take its
// unpushed writes with it. Drain them before serving, so recovered memories are
// durable and recallable from the owner from the first request onward.
//
// Safe on a clean boot: an empty spool means no HTTP call at all. Safe in file
// mode: wt_drain returns immediately when ENGRAM_URL is unset.
let wt_recovered: Int = wt_drain()
if wt_recovered > 0 {
println("[soul] write-through: recovered " + int_to_str(wt_recovered)
+ " nodes from a previous process's spool -> persistence owner")
}
if wt_recovered < 0 {
println("[soul] write-through: spool present but the persistence owner is unreachable — queued, will retry on heartbeat")
}
println("[soul] serving on port " + int_to_str(port))
http_serve_async(port, "handle_request")
println("[soul] awareness loop starting")
+1 -1
View File
@@ -5,4 +5,4 @@ extern fn aff_try_slot(slot_json: String, aff_7d_ts: Int, acc_key: String) -> Vo
extern fn load_identity_context() -> Void
extern fn seed_persona_from_env() -> Void
extern fn emit_session_start_event() -> Void
extern fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
extern fn layered_cycle(raw_input: String) -> String
+2 -2
View File
@@ -11,7 +11,7 @@ import "memory.el"
fn steward_log_event(kind: String, detail: String) -> Void {
let content: String = "STEWARD:" + kind + " | " + detail
let tags: String = "[\"stewardship\",\"steward:" + kind + "\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
content,
"StewardshipEvent",
"steward:" + kind,
@@ -221,7 +221,7 @@ fn steward_fingerprint_session(input: String, session_id: String) -> String {
+ " formality=" + fs_str
+ " time=" + tb_str
let sample_tags: String = "[\"behavior\",\"BehaviorSample\",\"stewardship\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
sample_content,
"BehaviorSample",
"behavior:" + session_id,
+1 -16
View File
@@ -53,23 +53,8 @@ fn handle_config(method: String, body: String) -> String {
}
fn dharma_registry() -> String {
// COMPILED IDENTITY, not state (2026-08-09). soul_principal had no producer at
// all the #132 gate flagged it as a dead read and the registry reported an
// empty principal under a heading that says "Principal Covenant v1". The value
// was never missing: it is declared in soul.el's cgi block, and as of the
// codegen fix it is compiled into the binary and loaded at startup.
//
// Read it from the compiled constant rather than the state store. The design is
// explicit that this identity is "not modifiable by any runtime mechanism
// including environment variables, configuration files, or API calls" — so
// publishing it into state (the cheap fix) would have recreated exactly the
// mutable copy it forbids. cgi_principal() is read-only and has no setter.
//
// cgi_id keeps its state read deliberately: the RUNTIME instance id is a
// different fact from the compiled dharma_id, and conflating them would hide
// the case where a binary runs under an id its declaration never claimed.
let cgi_id: String = state_get("soul_cgi_id")
let principal: String = cgi_principal()
let principal: String = state_get("soul_principal")
return "{\"registry\":[{\"cgi\":\"" + cgi_id + "\","
+ "\"principal\":\"" + principal + "\","
+ "\"covenant\":\"Principal Covenant v1\","
-90
View File
@@ -1,90 +0,0 @@
# gate-openai — deterministic OpenAI-dialect provider stub
Staging home for the **soul-openai-tools-v2** gate scaffolding
(`docs/specs/SPEC-soul-openai-tools-v2-2026-08-06.md`, test-plan rung 1:
"stub first — discriminates before El code exists"). Sibling of gate9's
Anthropic stub (`_wt-beta-round9/scripts/gate9/stub-llm.py`): same scenario
mechanism, opposite wire dialect. Stdlib Python only, 127.0.0.1 only,
refuses ports 7770/7779/17779. Run `./selftest.sh` — exit 0 is green.
## Files
| File | Role |
|---|---|
| `stub-openai.py` | HTTP server: `POST /v1/chat/completions` (OpenAI dialect), scenario-scripted responses, request validation, ground-truth JSONL log, hostile modes via `--mode` |
| `scenarios-openai.json` | Scenario contract: scripts + markers + per-class/per-step request assertions |
| `selftest.sh` | curl-driven proof of every scenario, every rejection, all hostile modes (58 checks) |
## What each scenario proves (when the brain drives it)
| Class | Proves |
|---|---|
| `oa-plain` | finish_reason `stop` ends the loop; tools + `tool_choice` + `parallel_tool_calls:false` were offered on the wire |
| `oa-tools-off` | the chat-only lane sends NO tools (offering them there is a 400) |
| `oa-single-tool` | full round-trip: `tool_calls` parsed, assistant echo + `role:"tool"` turn with matching `tool_call_id` sent back, final text reached |
| `oa-torture` | `function.arguments` (JSON-encoded string with nested quotes, backslashes, newlines, tabs, unicode) survives exactly ONE decode — the stub recomputes the issued payload from the script and 400s on any drift (`gate_echo_mismatch`, the spec §6 two-escaper trap) |
| `oa-parallel` | two `tool_calls` in one response: the brain either answers both (paired correctly) or rejects cleanly — an unpaired echo is a 400 |
| `oa-mission` | multi-round loop continuation; step index = assistant-message count, so resume threads index correctly by construction |
| `oa-api-error` | provider errors 400/429/500/503 in the OpenAI error envelope surface honestly, no retry storm |
Universal (every request, any scenario): Anthropic dialect leakage fails
loudly with 400 — `anthropic-version` header, top-level `system` /
`stop_sequences` / `max_tokens_to_sample`, `input_schema` inside tools,
Anthropic content blocks (`tool_use`/`tool_result`/...). Tools must be
`{type:"function", function:{name, description, parameters}}`, unique names;
echoed `arguments` must be a JSON-encoded STRING, never a decoded object.
## Hostile modes (`--mode`, same file)
| Mode | Behavior | Brain invariant under test |
|---|---|---|
| `black-hole` | reads the request, never responds | HTTP timeout exists and surfaces; no silent hang |
| `mid-body-drop` | 200 headers, half a JSON body, socket abort | truncated body = clean error, never a half-parsed reply shown as real |
| `tool-pending-forever` | every request gets a fresh `tool_calls` response, forever | the loop's iteration cap trips (`max_loop_iterations: 16` in the contract); count actual round-trips via `GET /gate/stats` (`chat_hits`) |
## How the brain-side gate consumes this
1. Start: `stub-openai.py --port P --scenarios scenarios-openai.json --log run.jsonl`
2. Point the brain at it: `NEURON_LLM_0_URL=http://127.0.0.1:P` +
`NEURON_LLM_0_FORMAT=openai` (spec step 0 must verify these actually
export at runtime), scratch profile, free soul port.
3. Send each phrasing's `prompt` (the marker selects the script); assert the
brain's claims (`tools_used`, reply, ledger) against the stub's JSONL log
— truth, not narration — plus files on disk for write_file scenarios.
4. Any stub 400 = the brain sent a malformed/leaked request; the gate fails
with the stub's reason string.
5. Re-run gate9's Anthropic matrix unchanged = proof the Anthropic lane is
byte-untouched.
## Reconciliation into gate9 (app repo) — AFTER round 9 merges
This dir is staging only; the merge is mechanical by design:
- `stub-openai.py` + `scenarios-openai.json` move to `scripts/gate9/`
alongside `stub-llm.py` + `scenarios.json` (shared conventions: marker
matching, assistant-count step indexing, `--port/--scenarios/--log`,
JSONL fields `seq/ts/kind/scenario_class/phrasing/step/validation/
delivered/http_status`, prod-port refusal, benign background responses,
`GATE-SCRIPT-EXHAUSTED` overrun, `{N}/{NN}` repeat expansion).
- `prompt-matrix-gate.sh` gains a dialect axis (anthropic|openai) choosing
stub + scenario file; `matrix-asserts.py` reads the same log shape.
- The `--mode` hostile flags here are PROVIDER-side (brain↔LLM boundary);
gate9's `hostile/` servers are SOUL-side (app↔brain boundary). They are
complementary, not duplicates — both stay.
## Open questions for the port author (stub asserts a position; confirm or change)
1. `parallel_tool_calls` must be **explicitly false** on every tool-bearing
request (ADR-0005 pin). If the builder omits it instead, relax
`defaults.expect_request.parallel_tool_calls` to `null`.
2. `tool_choice` must be present (`"auto"` expected). If the brain relies on
the provider default, drop `require_tool_choice`.
3. Tool-result `content` is asserted only to be a string; if the brain sends
structured JSON-in-string (like `{"ok":true,...}`), no change needed.
4. Groq compatibility: Groq's OpenAI-compat endpoint rejects some optional
fields; whatever field set the brain settles on for live Groq E2E must be
mirrored here so the deterministic gate and the live lane assert the SAME
request shape.
5. The stub treats a `role:"tool"` turn answering an already-answered id as
400; if the resume path can legitimately replay tool results, that rule
needs a resume-aware carve-out (gate9's Anthropic stub faced the same
issue — see its PASS 1 comment).
-559
View File
@@ -1,559 +0,0 @@
#!/usr/bin/env bash
# run-lane-gate.sh — brain-side driver for the OpenAI-dialect gate.
#
# Drives the REAL soul binary against stub-openai.py for every class and every
# phrasing in scenarios-openai.json, plus the three hostile provider modes, and
# asserts the brain's claims against the stub's ground-truth JSONL (truth, not
# narration) and against files on disk.
#
# SAFETY (hard rules, enforced below):
# - never binds 7770 / 7779 / 17779 - only 7891-7894
# - never reads or writes ~/.neuron - HOME is redirected to a scratch dir
# - every process started here is killed on exit (trap) and proven with lsof
#
# The soul runs under `script -q /dev/null` so its stdout is a pty: El's
# println() uses puts(), which is FULLY buffered to a file, and the process is
# killed without flushing — the DRIFT lines would be invisible otherwise.
#
# Usage: ./run-lane-gate.sh [all|bridge|local|toolsoff|hostile]
# bridge = consent round-trip config (no workspace root -> write_file is
# "escalate" -> the loop suspends and the CLIENT executes the tool)
# local = workspace-root config (write_file is "reversible" + builtin ->
# the loop executes the tool in-process and runs to completion)
# toolsoff = supplementary: non-agentic lane against a base URL WITHOUT the
# /v1 suffix (the el-runtime provider chain appends
# /v1/chat/completions itself, unlike chat.el which appends only
# /chat/completions)
# hostile = black-hole / mid-body-drop / tool-pending-forever
#
# Env overrides: SOUL_BIN, STUB_PORT, SOUL_PORT, SOUL_PORT_B, RUN_ROOT
set -uo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
PHASES="${1:-all}"
SOUL_BIN="${SOUL_BIN:-/tmp/soul-oai2/soul-openai-tools}"
STUB_PORT="${STUB_PORT:-7891}"
SOUL_PORT="${SOUL_PORT:-7892}"
SOUL_PORT_B="${SOUL_PORT_B:-7893}"
RUN_ROOT="${RUN_ROOT:-/tmp/oa-lane-gate}"
STAMP="$(date +%Y%m%d-%H%M%S)"
RUN="$RUN_ROOT/$STAMP"
for p in "$STUB_PORT" "$SOUL_PORT" "$SOUL_PORT_B"; do
case "$p" in
7770|7779|17779) echo "FATAL: refusing production Neuron port $p"; exit 2;;
789[1-4]) ;;
*) echo "FATAL: port $p outside the allowed 7891-7894 range"; exit 2;;
esac
done
[ -x "$SOUL_BIN" ] || { echo "FATAL: soul binary not found/executable: $SOUL_BIN"; exit 2; }
mkdir -p "$RUN/home" "$RUN/ws-bridge" "$RUN/ws-local" "$RUN/ws-off" "$RUN/engram"
echo '{"nodes":[],"edges":[]}' > "$RUN/engram/snapshot.json"
DRV="$RUN/drv.py"
STUB_PID=""; SOUL_PID=""
cleanup() {
[ -n "$SOUL_PID" ] && kill "$SOUL_PID" 2>/dev/null
pkill -f "$SOUL_BIN" 2>/dev/null
[ -n "$STUB_PID" ] && kill "$STUB_PID" 2>/dev/null
sleep 0.4
[ -n "$SOUL_PID" ] && kill -9 "$SOUL_PID" 2>/dev/null
[ -n "$STUB_PID" ] && kill -9 "$STUB_PID" 2>/dev/null
return 0
}
trap cleanup EXIT INT TERM
start_stub() { # $1 = mode, $2 = log path
local mode="$1" log="$2" args=""
[ "$mode" = "normal" ] && args="--scenarios $HERE/scenarios-openai.json"
# shellcheck disable=SC2086
python3 "$HERE/stub-openai.py" --port "$STUB_PORT" --mode "$mode" --log "$log" $args \
> "$RUN/stub-$mode.out" 2>&1 &
STUB_PID=$!
for _ in $(seq 1 50); do
curl -sf "http://127.0.0.1:$STUB_PORT/gate/health" >/dev/null 2>&1 && return 0
sleep 0.2
done
echo "FATAL: stub did not come up on $STUB_PORT"; cat "$RUN/stub-$mode.out"; exit 3
}
stop_stub() { [ -n "$STUB_PID" ] && kill "$STUB_PID" 2>/dev/null; sleep 0.3; STUB_PID=""; }
start_soul() { # $1 = port, $2 = base url, $3 = soul log, $4 = agent root ("" = none)
local port="$1" base="$2" log="$3" root="$4"
script -q /dev/null \
env -u ANTHROPIC_API_KEY -u SOUL_API_KEY -u ENGRAM_URL -u ENGRAM_API_KEY \
-u NEURON_API_URL -u NEURON_TOKEN -u SOUL_LLM_PROVIDER -u SOUL_LLM_BASE_URL \
-u NEURON_LLM_1_URL -u NEURON_LLM_1_KEY -u SOUL_IDENTITY \
HOME="$RUN/home" PATH="$PATH" \
NEURON_PORT="$port" EL_HTTP_BIND_HOST=127.0.0.1 \
SOUL_ENGRAM_PATH="$RUN/engram/snapshot.json" \
SOUL_CGI_ID=ntn-test SOUL_PERSONA_NAME=Neuron \
NEURON_LLM_0_URL="$base" NEURON_LLM_0_FORMAT=openai NEURON_LLM_0_KEY=gate-test-key \
${root:+NEURON_AGENT_ROOT="$root"} \
"$SOUL_BIN" > "$log" 2>&1 &
SOUL_PID=$!
for _ in $(seq 1 100); do
curl -sf "http://127.0.0.1:$port/health" >/dev/null 2>&1 && return 0
sleep 0.2
done
echo "FATAL: soul did not come up on $port"; tail -20 "$log"; exit 3
}
stop_soul() {
[ -n "$SOUL_PID" ] && kill "$SOUL_PID" 2>/dev/null
pkill -f "$SOUL_BIN" 2>/dev/null
sleep 0.6; SOUL_PID=""
}
# ------------------------------------------------------------------ driver ----
cat > "$DRV" <<'PYEOF'
import json, os, sys, time, threading, urllib.request, urllib.error
CFG = json.load(open(sys.argv[1]))
SOUL = "http://127.0.0.1:%d" % CFG["soul_port"]
STUB = "http://127.0.0.1:%d" % CFG["stub_port"]
WS = CFG["workspace"]
MODE = CFG["mode"] # bridge | local | toolsoff
SCEN = json.load(open(CFG["scenarios"]))
STUBLOG = CFG["stub_log"]
SOULLOG = CFG["soul_log"]
ONLY = CFG.get("classes") or list(SCEN["classes"].keys())
MAXHOPS = CFG.get("max_hops", 15)
OUT = CFG["out"]
# the chat-only class must be driven on the NON-agentic door: the agentic door
# always advertises tools, which is a 400 on that scenario by contract.
NON_AGENTIC = {"oa-tools-off"}
def http(method, url, obj=None, timeout=300):
data = None if obj is None else json.dumps(obj).encode()
req = urllib.request.Request(url, data=data, method=method,
headers={"Content-Type": "application/json"})
try:
with urllib.request.urlopen(req, timeout=timeout) as r:
body = r.read().decode("utf-8", "replace")
st = r.status
except urllib.error.HTTPError as e:
body = e.read().decode("utf-8", "replace"); st = e.code
except Exception as e:
return -1, "TRANSPORT-ERROR: %r" % (e,), None
try:
return st, body, json.loads(body)
except ValueError:
return st, body, None
def fsize(p):
return os.path.getsize(p) if os.path.exists(p) else 0
def tail_from(path, off):
if not os.path.exists(path):
return "", off
with open(path, "rb") as f:
f.seek(off); chunk = f.read(); return chunk.decode("utf-8", "replace"), f.tell()
def stub_since(off):
"""Exact correlation: only the JSONL bytes appended during this phrasing."""
txt, noff = tail_from(STUBLOG, off)
recs = []
for line in txt.splitlines():
line = line.strip()
if line:
try: recs.append(json.loads(line))
except ValueError: pass
return recs, noff
def perform(name, ti):
"""Execute the bridged tool for real, like the desktop client would."""
if name in ("write_file", "edit_file"):
p = ti.get("path", "")
dest = p if os.path.isabs(p) else os.path.join(WS, p)
os.makedirs(os.path.dirname(dest) or WS, exist_ok=True)
body = ti.get("content", "")
with open(dest, "w") as f:
f.write(body)
return "wrote %s (%d bytes)" % (p, len(body.encode()))
return "ok"
class Poller(threading.Thread):
def __init__(self, sid):
super().__init__(daemon=True); self.sid = sid; self.snaps = []; self.stop = False
def run(self):
while not self.stop:
st, body, js = http("GET", SOUL + "/api/run-progress/" + self.sid, timeout=60)
if js and js.get("progress"):
if not self.snaps or self.snaps[-1] != js["progress"]:
self.snaps.append(js["progress"])
time.sleep(0.1)
def progress(sid):
_, _, pj = http("GET", SOUL + "/api/run-progress/" + sid, timeout=30)
return (pj or {}).get("progress")
def run_phrasing(cname, ph):
st, body, js = http("POST", SOUL + "/api/sessions", {"title": ph["id"]}, timeout=60)
sid = (js or {}).get("id", "")
rec = {"class": cname, "phrasing": ph["id"], "session_id": sid, "legs": [],
"pendings": [], "progress_during": [], "progress_per_leg": [],
"progress_final": None, "soul_log": "", "stub": [], "http": [],
"agentic": cname not in NON_AGENTIC}
if not sid:
rec["fatal"] = "session create failed: %s %s" % (st, body[:300]); return rec
soff = fsize(SOULLOG); loff = fsize(STUBLOG)
t0 = time.time()
pol = Poller(sid); pol.start()
payload = {"message": ph["prompt"], "session_id": sid, "workspace_root": WS,
"agentic": rec["agentic"]}
if MODE == "local":
payload["agent_workspace_root"] = WS
st, body, js = http("POST", SOUL + "/api/chat", payload, timeout=CFG.get("chat_timeout", 240))
rec["http"].append(st)
rec["legs"].append(js if js is not None else body[:600])
rec["progress_per_leg"].append(progress(sid))
hops = 0
while isinstance(js, dict) and js.get("tool_pending") and hops < MAXHOPS:
rec["pendings"].append({"call_id": js.get("call_id"), "tool_name": js.get("tool_name"),
"tool_input": js.get("tool_input"), "risk_tier": js.get("risk_tier"),
"narration": js.get("narration"), "tools_used": js.get("tools_used")})
try:
eff = perform(js.get("tool_name", ""), js.get("tool_input") or {})
except Exception as e:
eff = "client error: %r" % (e,)
st, body, js = http("POST", SOUL + "/api/sessions/%s/tool_result" % sid,
{"call_id": js.get("call_id"), "content": eff},
timeout=CFG.get("chat_timeout", 240))
rec["http"].append(st)
rec["legs"].append(js if js is not None else body[:600])
rec["progress_per_leg"].append(progress(sid))
hops += 1
pol.stop = True; time.sleep(0.3)
t1 = time.time()
rec["elapsed"] = round(t1 - t0, 2)
rec["progress_during"] = pol.snaps
rec["progress_final"] = progress(sid)
rec["soul_log"], _ = tail_from(SOULLOG, soff)
rec["stub"], _ = stub_since(loff)
rec["hops"] = hops
return rec
# ------------------------------------------------------------- assertions ----
def expected_calls(cname, first_only=False):
out = []
for step in SCEN["classes"][cname]["script"]:
for k, call in enumerate(step.get("tool_calls") or []):
if first_only and k > 0:
continue
out.append((call["name"], call["arguments"]))
return out
def final_text(cname):
for step in reversed(SCEN["classes"][cname]["script"]):
if step.get("text") and not step.get("tool_calls"):
return step["text"]
return None
def judge(rec):
cname = rec["class"]; ok = []; bad = []
last = rec["legs"][-1] if rec["legs"] else None
reply = last.get("reply") if isinstance(last, dict) else None
err = last.get("error") if isinstance(last, dict) else None
tools_used = last.get("tools_used") if isinstance(last, dict) else None
stub = rec["stub"]
scen_recs = [r for r in stub if r.get("kind") == "scenario"]
rejects = [r for r in stub if r.get("validation") != "ok"]
bg = [r for r in stub if r.get("kind") in ("wrong_path", "background")]
def wire_clean():
if rejects:
for r in rejects:
bad.append("stub REJECTED a request: [%s] %s"
% (r.get("validation"), r.get("validation_detail")))
else:
ok.append("stub ground truth: validation \"ok\" on all %d scenario leg(s), no "
"gate_echo_mismatch / gate_tool_call_shape / dialect-leak 400s"
% len(scen_recs))
if bg:
ok.append("NOTE background non-scenario request(s) in this window: %s"
% [(r.get("kind"), r.get("path"), r.get("http_status")) for r in bg])
if cname == "oa-plain":
wire_clean()
want = final_text(cname)
if reply == want: ok.append("final reply == scripted final text (byte-exact)")
else: bad.append("final reply mismatch:\n WANT: %r\n GOT : %r" % (want, reply))
if tools_used == []: ok.append("tools_used == [] (no tool ran)")
else: bad.append("tools_used expected [] got %r" % (tools_used,))
if reply and ('"tool_calls"' in reply or '"function"' in reply or '"tool_use"' in reply):
bad.append("tool-call JSON leaked into the reply text")
else: ok.append("no tool-call JSON anywhere in the reply")
elif cname in ("oa-single-tool", "oa-torture", "oa-mission"):
wire_clean()
want = final_text(cname)
if reply == want: ok.append("final reply == scripted final text (byte-exact)")
else: bad.append("final reply mismatch:\n WANT: %r\n GOT : %r" % (want, reply))
exp = expected_calls(cname)
wantnames = [n for n, _ in exp]
if tools_used == wantnames:
ok.append("tools_used == %r (carried across %d suspension(s))" % (wantnames, rec["hops"]))
else:
bad.append("tools_used expected %r got %r" % (wantnames, tools_used))
for name, args in exp:
p = args.get("path"); c = args.get("content")
dest = os.path.join(WS, p)
if not os.path.exists(dest):
bad.append("expected file missing on disk: %s" % dest); continue
got = open(dest, "rb").read()
if got == c.encode():
ok.append("%s on disk is byte-for-byte the issued payload (%d bytes)" % (p, len(got)))
else:
bad.append("%s content differs\n WANT %r\n GOT %r"
% (p, c[:300], got[:300].decode("utf-8", "replace")))
if MODE == "bridge":
for pend, (name, args) in zip(rec["pendings"], exp):
if pend["tool_input"] == args:
ok.append("tool_input for %s survived exactly ONE decode (deep-equal to the "
"issued arguments; no double-escaping)" % name)
else:
bad.append("tool_input != issued arguments for %s\n WANT %r\n GOT %r"
% (name, args, pend["tool_input"]))
if rec["pendings"] and all(p["risk_tier"] == "escalate" for p in rec["pendings"]):
ok.append("every write_file classified \"escalate\" and bridged for consent")
elif cname == "oa-parallel":
drift = [l.strip() for l in rec["soul_log"].splitlines() if "DRIFT: provider returned" in l]
if drift: ok.append("soul log: " + drift[0])
else: bad.append("no 'DRIFT: provider returned N parallel tool_calls' line in the soul log")
delivered = [r for r in stub if r.get("delivered", {}).get("tool_calls")]
if delivered and len(delivered[0]["delivered"]["tool_calls"]) == 2:
ok.append("stub delivered 2 parallel tool_calls in one response (ground truth)")
if MODE == "bridge":
if len(rec["pendings"]) == 1:
ok.append("exactly ONE call honored: %s" % rec["pendings"][0]["call_id"])
else:
bad.append("expected exactly 1 honored call, got %d" % len(rec["pendings"]))
pairing = [r for r in rejects if "gate_pairing" in str(r.get("validation_detail")) or
"tool_calls at end of thread" in str(r.get("validation_detail")) or
"not fully answered" in str(r.get("validation_detail"))]
for r in pairing:
ok.append("EXPECTED-BY-CONTRACT stub 400 on the unpaired echo: %s"
% r.get("validation_detail"))
other = [r for r in rejects if r not in pairing]
for r in other:
bad.append("unexpected stub rejection: [%s] %s"
% (r.get("validation"), r.get("validation_detail")))
if err and not reply:
ok.append("honest error envelope after the 400 (no fabricated answer): %r" % err)
elif reply == final_text(cname):
ok.append("final reply == scripted final text (both calls paired)")
else:
bad.append("neither an honest error nor the scripted final text: %r" % (last,))
elif cname == "oa-api-error":
if err and not reply:
ok.append("honest error envelope: error=%r reply=%r" % (err, reply))
else:
bad.append("expected an error envelope with an empty reply, got %r" % (last,))
delivered = [r["delivered"].get("api_error") for r in stub if r.get("delivered")]
ok.append("stub delivered api_error status(es): %r" % [d for d in delivered if d])
n = len([r for r in stub if r.get("kind") == "scenario"])
ok.append("provider hit %d time(s) - no retry storm" % n)
if reply:
bad.append("FABRICATED ANSWER: reply non-empty on a provider error")
elif cname == "oa-tools-off":
ok.append("stub records for this phrasing: %r"
% [{k: r.get(k) for k in ("kind", "path", "validation", "http_status")} for r in stub])
wrong = [r for r in stub if r.get("kind") == "wrong_path"]
matched = [r for r in stub if r.get("scenario_class") == cname]
if matched and not rejects:
ok.append("chat-only request reached /v1/chat/completions with NO tools offered")
want = final_text(cname)
if reply == want: ok.append("final reply == scripted final text (byte-exact)")
else: bad.append("final reply mismatch:\n WANT: %r\n GOT : %r" % (want, reply))
if reply and ('"tool_calls"' in reply or '"function"' in reply):
bad.append("tool-call JSON leaked into the reply text")
else: ok.append("no tool-call JSON in the reply")
elif wrong:
bad.append("the non-agentic lane never reached the provider endpoint: stub saw "
"%s -> %s (the el-runtime provider chain appends /v1/chat/completions "
"to NEURON_LLM_0_URL, chat.el appends only /chat/completions)"
% (wrong[0]["path"], wrong[0]["http_status"]))
elif not stub:
bad.append("no request reached the stub at all")
else:
for r in rejects:
bad.append("stub REJECTED: [%s] %s" % (r.get("validation"), r.get("validation_detail")))
return ok, bad
def main():
results = []
for cname in ONLY:
for ph in SCEN["classes"][cname]["phrasings"]:
rec = run_phrasing(cname, ph)
ok, bad = judge(rec)
rec["ok"] = ok; rec["bad"] = bad
rec["verdict"] = "FAIL" if bad else "PASS"
results.append(rec)
print("=" * 78)
print("[%s] %s / %s (%.2fs, %d bridge hop(s), agentic=%s, mode=%s)"
% (rec["verdict"], cname, ph["id"], rec.get("elapsed", 0),
rec.get("hops", 0), rec["agentic"], MODE))
for l in ok: print(" ok " + l.replace("\n", "\n "))
for l in bad: print(" FAIL " + l.replace("\n", "\n "))
for i, leg in enumerate(rec["legs"]):
print(" leg%d envelope: %s" % (i, json.dumps(leg)[:430]))
for i, pr in enumerate(rec["progress_per_leg"]):
print(" run-progress after leg%d: %s" % (i, json.dumps(pr)[:380]))
if rec["progress_during"]:
print(" run-progress polled DURING (%d distinct snapshot(s)), last: %s"
% (len(rec["progress_during"]), json.dumps(rec["progress_during"][-1])[:300]))
for r in rec["stub"]:
print(" stub: kind=%s class=%s phrasing=%s step=%s validation=%s%s delivered=%s http=%s"
% (r.get("kind"), r.get("scenario_class"), r.get("phrasing"), r.get("step"),
r.get("validation"),
("(" + str(r.get("validation_detail")) + ")") if r.get("validation_detail") else "",
json.dumps(r.get("delivered")), r.get("http_status")))
if rec["soul_log"].strip():
for l in rec["soul_log"].splitlines():
if l.strip(): print(" soul: " + l.strip())
json.dump(results, open(OUT, "w"), indent=1)
npass = sum(1 for r in results if r["verdict"] == "PASS")
print("=" * 78)
print("PHASE %s: %d/%d PASS" % (MODE, npass, len(results)))
for r in results:
print(" %-6s %-16s %s" % (r["verdict"], r["class"], r["phrasing"]))
return 0 if npass == len(results) else 1
sys.exit(main())
PYEOF
# ------------------------------------------------------------- hostile drv ---
cat > "$RUN/hostile.py" <<'PYEOF'
import json, os, sys, time, urllib.request, urllib.error
CFG = json.load(open(sys.argv[1]))
SOUL = "http://127.0.0.1:%d" % CFG["soul_port"]
STUB = "http://127.0.0.1:%d" % CFG["stub_port"]
def http(method, url, obj=None, timeout=400):
data = None if obj is None else json.dumps(obj).encode()
req = urllib.request.Request(url, data=data, method=method,
headers={"Content-Type": "application/json"})
try:
with urllib.request.urlopen(req, timeout=timeout) as r:
b = r.read().decode("utf-8", "replace"); st = r.status
except urllib.error.HTTPError as e:
b = e.read().decode("utf-8", "replace"); st = e.code
except Exception as e:
return -1, "TRANSPORT-ERROR: %r" % (e,), None
try:
return st, b, json.loads(b)
except ValueError:
return st, b, None
mode = CFG["mode"]; wsmode = CFG["ws_mode"]; WS = CFG["workspace"]
_, _, js = http("POST", SOUL + "/api/sessions", {"title": "hostile-" + mode}, timeout=60)
sid = (js or {}).get("id", "")
payload = {"message": "oa-gate plain probe: hostile mode %s" % mode,
"agentic": True, "session_id": sid, "workspace_root": WS}
if wsmode == "local":
payload["agent_workspace_root"] = WS
t0 = time.time()
st, body, js = http("POST", SOUL + "/api/chat", payload, timeout=CFG.get("timeout", 400))
t_first = time.time() - t0
legs = [js if js is not None else body[:500]]
hops = 0
while isinstance(js, dict) and js.get("tool_pending") and hops < CFG.get("max_hops", 14):
ti = js.get("tool_input") or {}
p = ti.get("path", "x.md")
dest = p if os.path.isabs(p) else os.path.join(WS, p)
try: open(dest, "w").write(ti.get("content", ""))
except Exception: pass
st, body, js = http("POST", SOUL + "/api/sessions/%s/tool_result" % sid,
{"call_id": js.get("call_id"), "content": "ok"},
timeout=CFG.get("timeout", 400))
legs.append(js if js is not None else body[:500]); hops += 1
el = time.time() - t0
_, _, stats = http("GET", STUB + "/gate/stats", timeout=30)
_, _, prog = http("GET", SOUL + "/api/run-progress/" + sid, timeout=30)
fab = [l for l in legs if isinstance(l, dict) and l.get("reply")]
print("HOSTILE %s (ws_mode=%s)" % (mode, wsmode))
print(" first /api/chat POST returned after %.2fs; whole chain %.2fs; client bridge hops=%d; "
"stub chat_hits=%s" % (t_first, el, hops, (stats or {}).get("chat_hits")))
print(" first envelope : " + json.dumps(legs[0])[:430])
print(" final envelope : " + json.dumps(legs[-1])[:430])
print(" non-empty replies anywhere in the chain (fabrication check): %d" % len(fab))
print(" run-progress : " + json.dumps(prog)[:300])
json.dump({"mode": mode, "ws_mode": wsmode, "t_first": t_first, "elapsed": el, "hops": hops,
"chat_hits": (stats or {}).get("chat_hits"), "legs": legs, "progress": prog},
open(CFG["out"], "w"), indent=1)
PYEOF
# ------------------------------------------------------------------ phases ---
RC_BRIDGE=0; RC_LOCAL=0; RC_OFF=0
run_normal_phase() { # $1 = label, $2 = soul port, $3 = agent root, $4 = ws, $5 = base, $6 = classes json
local m="$1" port="$2" root="$3" ws="$4" base="$5" classes="$6"
echo; echo "############ PHASE: $m (soul :$port, NEURON_LLM_0_URL=$base) ############"
start_stub normal "$RUN/stub-$m.jsonl"
start_soul "$port" "$base" "$RUN/soul-$m.log" "$root"
cat > "$RUN/cfg-$m.json" <<JSON
{"soul_port": $port, "stub_port": $STUB_PORT, "workspace": "$ws", "mode": "$m",
"scenarios": "$HERE/scenarios-openai.json", "stub_log": "$RUN/stub-$m.jsonl",
"soul_log": "$RUN/soul-$m.log", "out": "$RUN/results-$m.json", "chat_timeout": 240,
"classes": $classes}
JSON
python3 "$DRV" "$RUN/cfg-$m.json"
local rc=$?
stop_soul; stop_stub
return $rc
}
if [ "$PHASES" = "all" ] || [ "$PHASES" = "bridge" ]; then
run_normal_phase bridge "$SOUL_PORT" "" "$RUN/ws-bridge" "http://127.0.0.1:$STUB_PORT/v1" null
RC_BRIDGE=$?
fi
if [ "$PHASES" = "all" ] || [ "$PHASES" = "local" ]; then
run_normal_phase local "$SOUL_PORT_B" "$RUN/ws-local" "$RUN/ws-local" "http://127.0.0.1:$STUB_PORT/v1" null
RC_LOCAL=$?
fi
if [ "$PHASES" = "all" ] || [ "$PHASES" = "toolsoff" ]; then
# supplementary: the el-runtime provider chain appends /v1/chat/completions itself,
# so the non-agentic door needs the base WITHOUT the /v1 suffix.
run_normal_phase toolsoff "$SOUL_PORT" "" "$RUN/ws-off" "http://127.0.0.1:$STUB_PORT" '["oa-tools-off","oa-plain"]'
RC_OFF=$?
fi
if [ "$PHASES" = "all" ] || [ "$PHASES" = "hostile" ]; then
echo; echo "############ PHASE: hostile ############"
for spec in "black-hole:bridge" "mid-body-drop:bridge" "tool-pending-forever:bridge" "tool-pending-forever:local"; do
mode="${spec%%:*}"; wsm="${spec##*:}"
echo; echo "---- hostile mode=$mode ws_mode=$wsm ----"
start_stub "$mode" "$RUN/stub-$mode-$wsm.jsonl"
if [ "$wsm" = "local" ]; then
start_soul "$SOUL_PORT" "http://127.0.0.1:$STUB_PORT/v1" "$RUN/soul-$mode-$wsm.log" "$RUN/ws-local"
else
start_soul "$SOUL_PORT" "http://127.0.0.1:$STUB_PORT/v1" "$RUN/soul-$mode-$wsm.log" ""
fi
cat > "$RUN/cfg-$mode-$wsm.json" <<JSON
{"soul_port": $SOUL_PORT, "stub_port": $STUB_PORT, "mode": "$mode", "ws_mode": "$wsm",
"workspace": "$RUN/ws-local", "out": "$RUN/hostile-$mode-$wsm.json", "timeout": 400}
JSON
python3 "$RUN/hostile.py" "$RUN/cfg-$mode-$wsm.json"
echo " soul log (llm/DRIFT/cap lines):"
grep -E "DRIFT|llm error|iteration cap|\[llm\]" "$RUN/soul-$mode-$wsm.log" | tail -8 | sed 's/^/ /'
stop_soul; stop_stub
done
fi
echo; echo "############ CLEANUP ############"
cleanup
sleep 0.5
echo "processes still matching the soul binary:"; pgrep -fl "$SOUL_BIN" || echo " (none)"
echo "processes still matching stub-openai.py:"; pgrep -fl "stub-openai.py" || echo " (none)"
echo "lsof on 7891-7894 after cleanup:"
lsof -nP -iTCP:7891 -iTCP:7892 -iTCP:7893 -iTCP:7894 2>/dev/null || echo " (no listeners - ports free)"
echo
echo "############ SUMMARY ############"
echo "run dir: $RUN"
echo "bridge rc=$RC_BRIDGE local rc=$RC_LOCAL toolsoff rc=$RC_OFF (0 = every class PASS)"
exit $(( RC_BRIDGE + RC_LOCAL + RC_OFF ))
-110
View File
@@ -1,110 +0,0 @@
{
"_comment": "OpenAI-dialect gate scenario contract (soul-openai-tools-v2). Single source of truth shared by stub-openai.py (scripted provider responses + request assertions), selftest.sh (stub self-verification), and the future brain-side gate driver. Same structure as gate9's scenarios.json: classes -> script + phrasings with markers; scripts are CLASS-level so assertions are behavioral, never pinned to a sentence. expect_request keys: require_tools, require_tool_choice, parallel_tool_calls (expected literal value; null = don't check), forbid_tools. defaults apply to every class unless overridden; steps may override with their own expect_request.",
"deadline_secs": 60,
"max_loop_iterations": 16,
"defaults": {
"expect_request": {
"require_tools": true,
"require_tool_choice": true,
"parallel_tool_calls": false
}
},
"classes": {
"oa-plain": {
"script": [
{ "text": "Plain OpenAI-lane answer (gate fixture): the mechanism, the main caveat, and the practical takeaway in three sentences. No tools were needed for this one, and the finish reason on the wire is stop, which the loop must treat as terminal." }
],
"phrasings": [
{ "id": "oa-plain-p1", "marker": "oa-gate plain probe", "prompt": "oa-gate plain probe: explain the fixture topic simply." },
{ "id": "oa-plain-p2", "marker": "oa-gate second plain", "prompt": "oa-gate second plain: another phrasing of the plain question." }
]
},
"oa-tools-off": {
"expect_request": {
"require_tools": false,
"forbid_tools": true,
"require_tool_choice": false,
"parallel_tool_calls": null
},
"script": [
{ "text": "Chat-only OpenAI-lane answer (gate fixture): this lane offered no tools and none were used; the reply is plain text with finish reason stop." }
],
"phrasings": [
{ "id": "oa-tools-off-p1", "marker": "oa-gate tools-off probe", "prompt": "oa-gate tools-off probe: plain chat with no tools offered." }
]
},
"oa-single-tool": {
"script": [
{ "text": "Step 1: writing the note.",
"tool_calls": [
{ "name": "write_file",
"arguments": { "path": "openai-single-note.md", "content": "# Note (gate fixture, OpenAI lane)\n\nDeterministic single-tool body.\n" } }
] },
{ "text": "All set - openai-single-note.md is written with the fixture body. Nothing else was needed for this one." }
],
"phrasings": [
{ "id": "oa-single-p1", "marker": "oa-gate single tool note", "prompt": "oa-gate single tool note: save the fixture note to a file." },
{ "id": "oa-single-p2", "marker": "oa-gate one file please", "prompt": "oa-gate one file please: write the fixture note file." }
]
},
"oa-torture": {
"script": [
{ "tool_calls": [
{ "name": "write_file",
"arguments": { "path": "torture-note.md", "content": "Line 1 has \"double quotes\", 'singles', and a mid-line backslash \\ here.\nLine 2\thas a tab, a literal \\n two-char sequence, and a Windows path C:\\temp\\new.txt.\nLine 3 unicode: naïve café — 日本語 ✓ 🚀\nLine 4 JSON-in-string: {\"k\": \"v\", \"arr\": [1, 2], \"s\": \"nested \\\"deep\\\" quotes\"}\nLine 5 ends with a lone backslash \\" } }
] },
{ "text": "Torture round-trip complete: the payload with nested quotes, backslashes, newlines, tabs, and unicode survived exactly one encode and one decode." }
],
"phrasings": [
{ "id": "oa-torture-p1", "marker": "oa-gate torture probe", "prompt": "oa-gate torture probe: write the escaping torture file." }
]
},
"oa-parallel": {
"script": [
{ "text": "Step 1: two writes at once (parallel probe).",
"tool_calls": [
{ "name": "write_file", "arguments": { "path": "parallel-a.md", "content": "Parallel A (gate fixture).\n" } },
{ "name": "write_file", "arguments": { "path": "parallel-b.md", "content": "Parallel B (gate fixture).\n" } }
] },
{ "text": "Parallel probe complete: both tool results arrived and were paired correctly. A brain that instead rejects the double call must do so cleanly - that outcome is asserted brain-side, not here." }
],
"phrasings": [
{ "id": "oa-parallel-p1", "marker": "oa-gate parallel probe", "prompt": "oa-gate parallel probe: run the two-write parallel case." }
]
},
"oa-mission": {
"script": [
{ "text": "Step 1: drafting part one.",
"tool_calls": [
{ "name": "write_file", "arguments": { "path": "mission-part-1.md", "content": "Mission part 1 (gate fixture).\n" } }
] },
{ "text": "Step 2: drafting part two.",
"tool_calls": [
{ "name": "write_file", "arguments": { "path": "mission-part-2.md", "content": "Mission part 2 (gate fixture).\n" } }
] },
{ "text": "Mission complete: mission-part-1.md and mission-part-2.md are written; the loop ran two tool rounds and finished cleanly with finish reason stop." }
],
"phrasings": [
{ "id": "oa-mission-p1", "marker": "oa-gate mission probe", "prompt": "oa-gate mission probe: run the two-round mission." }
]
},
"oa-api-error": {
"expect_request": {
"require_tools": false,
"require_tool_choice": false,
"parallel_tool_calls": null
},
"script": [],
"phrasings": [
{ "id": "oa-err-400", "marker": "oa-gate error four hundred", "prompt": "oa-gate error four hundred: trigger the injected failure.",
"script": [ { "api_error": { "status": 400, "type": "invalid_request_error", "message": "gate-injected 400: request rejected by fixture", "code": "gate_injected" } } ] },
{ "id": "oa-err-429", "marker": "oa-gate error rate limit", "prompt": "oa-gate error rate limit: trigger the injected failure.",
"script": [ { "api_error": { "status": 429, "type": "rate_limit_error", "message": "gate-injected 429: rate limited by fixture", "code": "rate_limit_exceeded" } } ] },
{ "id": "oa-err-500", "marker": "oa-gate error five hundred", "prompt": "oa-gate error five hundred: trigger the injected failure.",
"script": [ { "api_error": { "status": 500, "type": "server_error", "message": "gate-injected 500: internal fixture error", "code": "gate_injected" } } ] },
{ "id": "oa-err-503", "marker": "oa-gate error unavailable", "prompt": "oa-gate error unavailable: trigger the injected failure.",
"script": [ { "api_error": { "status": 503, "type": "server_error", "message": "gate-injected 503: fixture overloaded", "code": "gate_injected" } } ] }
]
}
}
}
-409
View File
@@ -1,409 +0,0 @@
#!/usr/bin/env bash
# selftest.sh - proves stub-openai.py before any brain code exists.
# Drives the stub with curl through every scenario (plain, tools-off,
# single tool round-trip, escaping torture, parallel double-call,
# two-round mission, injected API errors, background, overrun), every
# validation rejection (dialect leaks, pairing, echo round-trip, scenario
# expectations), and all three hostile modes. Exit 0 = green.
set -u
cd "$(dirname "$0")" || exit 1
PY=python3
TMP="$(mktemp -d)"
PIDS=()
cleanup() {
for p in "${PIDS[@]:-}"; do kill -9 "$p" >/dev/null 2>&1; done
rm -rf "$TMP"
}
trap cleanup EXIT
PASS=0; FAIL=0
ok() { printf 'ok - %s\n' "$1"; PASS=$((PASS+1)); }
bad() { printf 'FAIL - %s\n' "$1"; FAIL=$((FAIL+1)); }
check() { # check <name> <cmd...> - pass if cmd exits 0; show output on fail
local name="$1"; shift
local out
if out="$("$@" 2>&1)"; then ok "$name"
else bad "$name"; [ -n "$out" ] && printf '%s\n' "$out" | sed 's/^/ /' | head -8
fi
}
freeport() { "$PY" -c 'import socket;s=socket.socket();s.bind(("127.0.0.1",0));print(s.getsockname()[1]);s.close()'; }
waithealth() {
local p="$1" i
for i in $(seq 1 60); do
curl -sf "http://127.0.0.1:$p/gate/health" >/dev/null 2>&1 && return 0
sleep 0.1
done
echo "stub on :$p never became healthy"; return 1
}
post() { # post <port> <bodyfile> <respfile> [extra curl args...] -> echoes http code
local port="$1" body="$2" resp="$3"; shift 3
curl -s -o "$resp" -w '%{http_code}' -H 'content-type: application/json' \
"$@" --data-binary @"$body" "http://127.0.0.1:$port/v1/chat/completions"
}
# ---- embedded helper: builds OpenAI-dialect bodies, asserts on responses ----
cat > "$TMP/helpers.py" <<'PYEOF'
import copy, json, sys
TOOLS = [
{"type": "function", "function": {
"name": "write_file", "description": "Write content to a file on disk.",
"parameters": {"type": "object",
"properties": {"path": {"type": "string"},
"content": {"type": "string"}},
"required": ["path", "content"]}}},
{"type": "function", "function": {
"name": "read_file", "description": "Read contents of a file from disk.",
"parameters": {"type": "object",
"properties": {"path": {"type": "string"}},
"required": ["path"]}}},
]
def dump(obj, out):
json.dump(obj, open(out, "w"), ensure_ascii=False)
def base(prompt, tools=True):
b = {"model": "gate-openai-model", "max_tokens": 1024,
"messages": [
{"role": "system", "content": "You are Neuron (gate fixture)."},
{"role": "user", "content": prompt}]}
if tools:
b["tools"] = copy.deepcopy(TOOLS)
b["tool_choice"] = "auto"
b["parallel_tool_calls"] = False
return b
def cmd_plain(out, prompt):
dump(base(prompt), out)
def cmd_notools(out, prompt):
dump(base(prompt, tools=False), out)
def cmd_mut(out, prompt, mutation):
b = base(prompt)
if mutation == "no-tool-choice":
del b["tool_choice"]
elif mutation == "ptc-true":
b["parallel_tool_calls"] = True
elif mutation == "top-system":
b["system"] = "You are Neuron."
elif mutation == "anth-tools":
b["tools"] = [{"name": "write_file", "description": "x",
"input_schema": {"type": "object", "properties": {}}}]
elif mutation == "anth-block":
b["messages"][1] = {"role": "user", "content": [
{"type": "tool_result", "tool_use_id": "toolu_x", "content": "hi"},
{"type": "text", "text": prompt}]}
else:
raise SystemExit("unknown mutation " + mutation)
dump(b, out)
def cmd_chain(out, prompt, variant, *resps):
"""Build the next leg: echo each response's assistant turn and answer its
tool calls. `variant` applies to the LAST response only:
ok | no-tool-turn | wrong-id | only-first | double-encode | object-args"""
b = base(prompt)
for idx, p in enumerate(resps):
last = idx == len(resps) - 1
msg = json.load(open(p))["choices"][0]["message"]
tcs = msg.get("tool_calls")
if not tcs:
b["messages"].append({"role": "assistant",
"content": msg.get("content")})
continue
v = variant if last else "ok"
asst = {"role": "assistant", "content": msg.get("content"),
"tool_calls": copy.deepcopy(tcs)}
if v == "double-encode":
for tc in asst["tool_calls"]:
tc["function"]["arguments"] = json.dumps(
tc["function"]["arguments"])
if v == "object-args":
for tc in asst["tool_calls"]:
tc["function"]["arguments"] = json.loads(
tc["function"]["arguments"])
b["messages"].append(asst)
if v == "no-tool-turn":
continue
use = tcs[:1] if v == "only-first" else tcs
for tc in use:
tid = "call_bogus_123" if v == "wrong-id" else tc["id"]
b["messages"].append({"role": "tool", "tool_call_id": tid,
"content": "{\"ok\":true,\"bytes\":42}"})
dump(b, out)
def cmd_chk(resp, expr):
r = json.load(open(resp))
if not eval(expr, {"r": r, "json": json, "len": len, "str": str,
"isinstance": isinstance, "any": any, "all": all,
"sorted": sorted}):
print("assertion failed:", expr)
print("resp:", json.dumps(r, ensure_ascii=False)[:400])
raise SystemExit(1)
def cmd_torture(resp, scen):
r = json.load(open(resp))
tc = r["choices"][0]["message"]["tool_calls"][0]
raw = tc["function"]["arguments"]
assert isinstance(raw, str), "arguments must be a JSON-encoded string"
got = json.loads(raw)
exp = json.load(open(scen))["classes"]["oa-torture"]["script"][0]["tool_calls"][0]["arguments"]
assert got == exp, "decoded arguments != scripted torture payload"
content = got["content"]
for needle in ['"', "\\", "\n", "\t", "日本語", "naïve", "🚀"]:
assert needle in content, "missing torture needle %r" % needle
def cmd_notjson(path):
data = open(path, "rb").read()
assert data, "file empty - no partial body arrived"
try:
json.loads(data.decode("utf-8", "replace"))
except ValueError:
return
raise SystemExit("partial body unexpectedly parsed as complete JSON")
def cmd_pending(*paths):
ids = []
for p in paths:
c = json.load(open(p))["choices"][0]
assert c["finish_reason"] == "tool_calls", c["finish_reason"]
tc = c["message"]["tool_calls"][0]
assert tc["function"]["name"] == "write_file"
json.loads(tc["function"]["arguments"]) # must decode
ids.append(tc["id"])
assert len(set(ids)) == len(ids), "call ids not distinct: %r" % ids
def cmd_logcheck(path):
recs = [json.loads(l) for l in open(path) if l.strip()]
seqs = [r["seq"] for r in recs]
assert seqs == sorted(seqs) and len(set(seqs)) == len(seqs), "seq not monotonic"
kinds = {}
for r in recs:
kinds[r["kind"]] = kinds.get(r["kind"], 0) + 1
assert kinds.get("scenario", 0) >= 10, "too few scenario records: %r" % kinds
assert kinds.get("background", 0) >= 1, "no background record"
assert kinds.get("overrun", 0) >= 1, "no overrun record"
rejected = [r for r in recs if r["validation"] == "rejected"]
assert len(rejected) >= 10, "too few rejected records: %d" % len(rejected)
assert any(r["delivered"].get("tool_calls") == ["write_file"]
for r in recs), "no single write_file ground truth"
assert any(r["delivered"].get("tool_calls") == ["write_file", "write_file"]
for r in recs), "no parallel ground truth"
def main():
fn = globals()["cmd_" + sys.argv[1].replace("-", "_")]
fn(*sys.argv[2:])
if __name__ == "__main__":
main()
PYEOF
mk() { "$PY" "$TMP/helpers.py" "$@"; }
echo "=== stub-openai selftest ==="
# ---- normal mode ------------------------------------------------------------
PORT="$(freeport)"
"$PY" stub-openai.py --port "$PORT" --scenarios scenarios-openai.json \
--log "$TMP/req.jsonl" >"$TMP/stub.out" 2>&1 &
PIDS+=($!); disown
check "stub starts and answers /gate/health" waithealth "$PORT"
# 1. plain completion
mk plain "$TMP/plain.json" "oa-gate plain probe: explain the fixture topic simply."
code="$(post "$PORT" "$TMP/plain.json" "$TMP/r_plain.json")"
check "plain: HTTP 200" test "$code" = "200"
check "plain: chat.completion envelope, finish stop, real content" mk chk "$TMP/r_plain.json" \
'r["object"]=="chat.completion" and r["choices"][0]["finish_reason"]=="stop" and isinstance(r["choices"][0]["message"]["content"],str) and len(r["choices"][0]["message"]["content"])>40'
# 2. tools-off lane (chat-only request accepted, tool-bearing request refused)
mk notools "$TMP/toolsoff.json" "oa-gate tools-off probe: plain chat with no tools offered."
code="$(post "$PORT" "$TMP/toolsoff.json" "$TMP/r_toolsoff.json")"
check "tools-off: chat-only request -> 200" test "$code" = "200"
mk plain "$TMP/toolsoff_bad.json" "oa-gate tools-off probe: plain chat with no tools offered."
code="$(post "$PORT" "$TMP/toolsoff_bad.json" "$TMP/r_toolsoff_bad.json")"
check "tools-off negative: offering tools -> 400 gate_expect" \
bash -c "test $code = 400"
check "tools-off negative: reason names gate_expect" mk chk "$TMP/r_toolsoff_bad.json" \
'r["error"]["code"]=="gate_expect"'
# 3. dialect-leak rejections (the loud-failure contract)
code="$(post "$PORT" "$TMP/plain.json" "$TMP/r_leak_hdr.json" -H 'anthropic-version: 2023-06-01')"
check "leak: anthropic-version header -> 400" test "$code" = "400"
check "leak: header reason names the leak" mk chk "$TMP/r_leak_hdr.json" \
'r["error"]["code"]=="gate_dialect_leak" and "anthropic-version" in r["error"]["message"]'
mk mut "$TMP/leak_tools.json" "oa-gate plain probe: explain the fixture topic simply." anth-tools
code="$(post "$PORT" "$TMP/leak_tools.json" "$TMP/r_leak_tools.json")"
check "leak: input_schema tools -> 400 gate_dialect_leak" bash -c \
"test $code = 400"
check "leak: input_schema reason" mk chk "$TMP/r_leak_tools.json" \
'r["error"]["code"]=="gate_dialect_leak" and "input_schema" in r["error"]["message"]'
mk mut "$TMP/leak_sys.json" "oa-gate plain probe: explain the fixture topic simply." top-system
code="$(post "$PORT" "$TMP/leak_sys.json" "$TMP/r_leak_sys.json")"
check "leak: top-level system -> 400" test "$code" = "400"
mk mut "$TMP/leak_block.json" "oa-gate plain probe: explain the fixture topic simply." anth-block
code="$(post "$PORT" "$TMP/leak_block.json" "$TMP/r_leak_block.json")"
check "leak: Anthropic tool_result content block -> 400" test "$code" = "400"
# 4. scenario request expectations
mk mut "$TMP/no_tc.json" "oa-gate plain probe: explain the fixture topic simply." no-tool-choice
code="$(post "$PORT" "$TMP/no_tc.json" "$TMP/r_no_tc.json")"
check "expect: missing tool_choice -> 400" test "$code" = "400"
mk mut "$TMP/ptc.json" "oa-gate plain probe: explain the fixture topic simply." ptc-true
code="$(post "$PORT" "$TMP/ptc.json" "$TMP/r_ptc.json")"
check "expect: parallel_tool_calls true -> 400 (ADR-0005 pin)" test "$code" = "400"
# 5. single tool round-trip
ST_PROMPT="oa-gate single tool note: save the fixture note to a file."
mk plain "$TMP/st1.json" "$ST_PROMPT"
code="$(post "$PORT" "$TMP/st1.json" "$TMP/r_st1.json")"
check "single-tool leg1: HTTP 200" test "$code" = "200"
check "single-tool leg1: one write_file call, finish tool_calls, string args" mk chk "$TMP/r_st1.json" \
'r["choices"][0]["finish_reason"]=="tool_calls" and len(r["choices"][0]["message"]["tool_calls"])==1 and r["choices"][0]["message"]["tool_calls"][0]["type"]=="function" and r["choices"][0]["message"]["tool_calls"][0]["function"]["name"]=="write_file" and isinstance(r["choices"][0]["message"]["tool_calls"][0]["function"]["arguments"],str) and json.loads(r["choices"][0]["message"]["tool_calls"][0]["function"]["arguments"])["path"]=="openai-single-note.md"'
mk chain "$TMP/st2.json" "$ST_PROMPT" ok "$TMP/r_st1.json"
code="$(post "$PORT" "$TMP/st2.json" "$TMP/r_st2.json")"
check "single-tool leg2: echo + tool turn -> 200 final text" test "$code" = "200"
check "single-tool leg2: final names the file, finish stop" mk chk "$TMP/r_st2.json" \
'r["choices"][0]["finish_reason"]=="stop" and "openai-single-note.md" in r["choices"][0]["message"]["content"]'
mk chain "$TMP/st2_no.json" "$ST_PROMPT" no-tool-turn "$TMP/r_st1.json"
code="$(post "$PORT" "$TMP/st2_no.json" "$TMP/r_st2_no.json")"
check "single-tool negative: echo without tool turn -> 400 gate_pairing" \
bash -c "test $code = 400"
check "single-tool negative: pairing reason" mk chk "$TMP/r_st2_no.json" \
'r["error"]["code"]=="gate_pairing"'
mk chain "$TMP/st2_wrong.json" "$ST_PROMPT" wrong-id "$TMP/r_st1.json"
code="$(post "$PORT" "$TMP/st2_wrong.json" "$TMP/r_st2_wrong.json")"
check "single-tool negative: wrong tool_call_id -> 400" test "$code" = "400"
mk chain "$TMP/st2_obj.json" "$ST_PROMPT" object-args "$TMP/r_st1.json"
code="$(post "$PORT" "$TMP/st2_obj.json" "$TMP/r_st2_obj.json")"
check "single-tool negative: arguments echoed as object -> 400 shape" \
bash -c "test $code = 400"
check "single-tool negative: shape reason names STRING" mk chk "$TMP/r_st2_obj.json" \
'r["error"]["code"]=="gate_tool_call_shape" and "STRING" in r["error"]["message"]'
# 6. escaping torture (the two-escaper trap, spec section 6)
T_PROMPT="oa-gate torture probe: write the escaping torture file."
mk plain "$TMP/t1.json" "$T_PROMPT"
code="$(post "$PORT" "$TMP/t1.json" "$TMP/r_t1.json")"
check "torture leg1: HTTP 200" test "$code" = "200"
check "torture leg1: arguments decode to the exact nasty payload" \
mk torture "$TMP/r_t1.json" scenarios-openai.json
mk chain "$TMP/t2.json" "$T_PROMPT" ok "$TMP/r_t1.json"
code="$(post "$PORT" "$TMP/t2.json" "$TMP/r_t2.json")"
check "torture leg2: faithful echo -> 200 final" test "$code" = "200"
mk chain "$TMP/t2_dbl.json" "$T_PROMPT" double-encode "$TMP/r_t1.json"
code="$(post "$PORT" "$TMP/t2_dbl.json" "$TMP/r_t2_dbl.json")"
check "torture negative: double-encoded echo -> 400" test "$code" = "400"
check "torture negative: reason names the two-escaper trap" mk chk "$TMP/r_t2_dbl.json" \
'r["error"]["code"]=="gate_echo_mismatch" and "two-escaper" in r["error"]["message"]'
# 7. parallel double-call
P_PROMPT="oa-gate parallel probe: run the two-write parallel case."
mk plain "$TMP/p1.json" "$P_PROMPT"
code="$(post "$PORT" "$TMP/p1.json" "$TMP/r_p1.json")"
check "parallel leg1: TWO tool_calls, distinct ids" mk chk "$TMP/r_p1.json" \
'r["choices"][0]["finish_reason"]=="tool_calls" and len(r["choices"][0]["message"]["tool_calls"])==2 and r["choices"][0]["message"]["tool_calls"][0]["id"]!=r["choices"][0]["message"]["tool_calls"][1]["id"]'
mk chain "$TMP/p2.json" "$P_PROMPT" ok "$TMP/r_p1.json"
code="$(post "$PORT" "$TMP/p2.json" "$TMP/r_p2.json")"
check "parallel leg2: both results -> 200 final" test "$code" = "200"
mk chain "$TMP/p2_one.json" "$P_PROMPT" only-first "$TMP/r_p1.json"
code="$(post "$PORT" "$TMP/p2_one.json" "$TMP/r_p2_one.json")"
check "parallel negative: answering only one call -> 400 pairing" test "$code" = "400"
# 8. two-round mission (loop continuation + step indexing)
M_PROMPT="oa-gate mission probe: run the two-round mission."
mk plain "$TMP/m1.json" "$M_PROMPT"
code="$(post "$PORT" "$TMP/m1.json" "$TMP/r_m1.json")"
check "mission leg1: part-1 tool call" mk chk "$TMP/r_m1.json" \
'json.loads(r["choices"][0]["message"]["tool_calls"][0]["function"]["arguments"])["path"]=="mission-part-1.md"'
mk chain "$TMP/m2.json" "$M_PROMPT" ok "$TMP/r_m1.json"
code="$(post "$PORT" "$TMP/m2.json" "$TMP/r_m2.json")"
check "mission leg2: part-2 tool call (step indexed by assistant count)" mk chk "$TMP/r_m2.json" \
'r["choices"][0]["finish_reason"]=="tool_calls" and json.loads(r["choices"][0]["message"]["tool_calls"][0]["function"]["arguments"])["path"]=="mission-part-2.md"'
mk chain "$TMP/m3.json" "$M_PROMPT" ok "$TMP/r_m1.json" "$TMP/r_m2.json"
code="$(post "$PORT" "$TMP/m3.json" "$TMP/r_m3.json")"
check "mission leg3: final text, finish stop" mk chk "$TMP/r_m3.json" \
'r["choices"][0]["finish_reason"]=="stop" and "Mission complete" in r["choices"][0]["message"]["content"]'
mk chain "$TMP/m4.json" "$M_PROMPT" ok "$TMP/r_m1.json" "$TMP/r_m2.json" "$TMP/r_m3.json"
code="$(post "$PORT" "$TMP/m4.json" "$TMP/r_m4.json")"
check "mission overrun: past-script request -> GATE-SCRIPT-EXHAUSTED" mk chk "$TMP/r_m4.json" \
'r["choices"][0]["message"]["content"].startswith("GATE-SCRIPT-EXHAUSTED")'
# 9. injected API errors (OpenAI error envelope)
for want in 400 429 500 503; do
case "$want" in
400) marker="four hundred";; 429) marker="rate limit";;
500) marker="five hundred";; 503) marker="unavailable";;
esac
mk plain "$TMP/e_$want.json" "oa-gate error $marker: trigger the injected failure."
code="$(post "$PORT" "$TMP/e_$want.json" "$TMP/r_e_$want.json")"
check "api-error $want: status returned" test "$code" = "$want"
check "api-error $want: OpenAI error envelope" mk chk "$TMP/r_e_$want.json" \
'isinstance(r["error"]["message"],str) and "gate-injected" in r["error"]["message"] and isinstance(r["error"]["type"],str)'
done
# 10. background (unmatched) request
mk plain "$TMP/bg.json" "hello there, just a boot probe with no marker"
code="$(post "$PORT" "$TMP/bg.json" "$TMP/r_bg.json")"
check "background: unmatched prompt -> benign ok" mk chk "$TMP/r_bg.json" \
'r["choices"][0]["message"]["content"]=="ok"'
# 11. ground-truth log invariants
check "ground-truth JSONL log invariants" mk logcheck "$TMP/req.jsonl"
# 12. production-port refusal
rc=0
"$PY" stub-openai.py --port 7770 --scenarios scenarios-openai.json \
--log "$TMP/never.jsonl" >/dev/null 2>&1 || rc=$?
check "refuses production port 7770" test "$rc" -ne 0
# ---- hostile mode: black-hole ----------------------------------------------
BH="$(freeport)"
"$PY" stub-openai.py --port "$BH" --log "$TMP/bh.jsonl" --mode black-hole \
>/dev/null 2>&1 &
PIDS+=($!); disown
check "black-hole: healthy" waithealth "$BH"
rc=0
curl -s -o /dev/null --max-time 3 -H 'content-type: application/json' \
--data-binary @"$TMP/plain.json" \
"http://127.0.0.1:$BH/v1/chat/completions" || rc=$?
check "black-hole: client times out (curl rc 28)" test "$rc" -eq 28
check "black-hole: health still answers during the hang" \
curl -sf --max-time 2 "http://127.0.0.1:$BH/gate/health"
# ---- hostile mode: mid-body-drop -------------------------------------------
MD="$(freeport)"
"$PY" stub-openai.py --port "$MD" --log "$TMP/md.jsonl" --mode mid-body-drop \
>/dev/null 2>&1 &
PIDS+=($!); disown
check "mid-body-drop: healthy" waithealth "$MD"
rc=0
curl -s --max-time 5 -o "$TMP/half.json" -H 'content-type: application/json' \
--data-binary @"$TMP/plain.json" \
"http://127.0.0.1:$MD/v1/chat/completions" || rc=$?
check "mid-body-drop: transfer fails (curl rc $rc)" test "$rc" -ne 0
check "mid-body-drop: partial body is not parseable JSON" mk notjson "$TMP/half.json"
# ---- hostile mode: tool-pending-forever ------------------------------------
TP="$(freeport)"
"$PY" stub-openai.py --port "$TP" --log "$TMP/tp.jsonl" \
--mode tool-pending-forever >/dev/null 2>&1 &
PIDS+=($!); disown
check "tool-pending-forever: healthy" waithealth "$TP"
for i in 1 2 3; do
code="$(post "$TP" "$TMP/plain.json" "$TMP/r_tp$i.json")"
check "tool-pending-forever: request $i -> 200" test "$code" = "200"
done
check "tool-pending-forever: three FRESH tool_calls, distinct ids" \
mk pending "$TMP/r_tp1.json" "$TMP/r_tp2.json" "$TMP/r_tp3.json"
check "tool-pending-forever: /gate/stats counts 3 chat hits" \
bash -c "curl -sf http://127.0.0.1:$TP/gate/stats | grep -q '\"chat_hits\": 3'"
# ---- summary ----------------------------------------------------------------
echo
echo "selftest: $PASS passed, $FAIL failed"
if [ "$FAIL" -ne 0 ]; then
echo "SELFTEST RED"
exit 1
fi
echo "SELFTEST GREEN (stub-openai gate scaffolding verified)"
-652
View File
@@ -1,652 +0,0 @@
#!/usr/bin/env python3
"""stub-openai.py - deterministic local stand-in for an OpenAI-format
/v1/chat/completions provider, for the soul-openai-tools-v2 gate
(docs/specs/SPEC-soul-openai-tools-v2-2026-08-06.md). No API key, no network,
no model.
Sibling of gate9's stub-llm.py (Anthropic dialect, _wt-beta-round9/scripts/
gate9/): same scenario mechanism (marker matching, assistant-count step
indexing, ground-truth JSONL log, prod-port refusal), different wire.
Staging home is tests/gate-openai/ in _wt-openai-tools; folds into
scripts/gate9/ after round 9 merges (see README.md).
WHAT IT DOES
* Serves POST /v1/chat/completions on 127.0.0.1 only (OpenAI dialect).
* VALIDATES every request - this is the gate's discriminator, built
BEFORE the brain-side El code exists so dialect leakage fails loudly:
- Anthropic tells are 400 code=gate_dialect_leak: `anthropic-version`
header; top-level `system` / `stop_sequences` / `max_tokens_to_sample`
/ `anthropic_version`; `input_schema` inside a tool entry; Anthropic
content blocks (tool_use / tool_result / server_tool_use / ...).
- tools[] must be OpenAI-shaped {type:"function", function:{name,
description, parameters}} with unique names -> 400 gate_tools_shape.
- assistant tool_calls echoes must be {id, type:"function",
function:{name, arguments:<JSON-encoded STRING>}}; a decoded-object
`arguments` is a wire bug -> 400 gate_tool_call_shape.
- every assistant tool_calls turn must be answered by role:"tool"
messages covering EVERY tool_call_id, immediately following;
unknown / duplicate / missing ids -> 400 gate_pairing.
- echoed `arguments` for gate-issued call ids (call_gate_*) are
recomputed from the script and compared after ONE json decode ->
400 gate_echo_mismatch. This is the two-escaper-trap discriminator
named in the spec's security model (section 6).
- scenario-level request expectations from scenarios-openai.json
(tools offered, OpenAI-shaped tool_choice, parallel_tool_calls
pinned false per ADR-0005) -> 400 gate_expect.
* Answers with SCRIPTED responses: plain text (finish_reason "stop"),
tool calls (finish_reason "tool_calls", arguments JSON-encoded, incl. a
nested-quote/escaping torture payload and a parallel two-call case), and
API-error injection (OpenAI error envelope). Scenario is selected by
scanning user-message text (newest first) for a registered marker
substring; the step index is the number of assistant messages already in
the request (stateless replay - resumes index correctly by construction).
* Writes a ground-truth JSONL log (--log): one record per request with the
validation verdict, matched scenario/step, and exactly which tool calls
were delivered. Gate assertions compare the brain's claims against THIS
log - truth, not narration.
* Unmatched requests (boot probes, awareness chatter) get a benign "ok"
text response, logged kind=background, never counted as ground truth.
* HOSTILE MODES (--mode) on the same file:
black-hole accept + read the request, never respond;
mid-body-drop send half a JSON body, then abort the socket;
tool-pending-forever every request gets a FRESH tool_call
(finish_reason "tool_calls"), forever - tests
the agentic loop's iteration cap; count the
brain's round-trips via GET /gate/stats.
usage: stub-openai.py --port P --scenarios scenarios-openai.json \
--log requests.jsonl [--mode MODE]
Listens on 127.0.0.1 only. Refuses production ports 7770/7779/17779.
"""
import argparse
import itertools
import json
import socket
import struct
import threading
import time
import uuid
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
STATE = {"scenarios": None, "log_path": None, "lock": threading.Lock(),
"seq": 0, "mode": "normal", "chat_hits": 0}
_PENDING_SEQ = itertools.count(1)
ANTHROPIC_TOP_KEYS = ("system", "stop_sequences", "max_tokens_to_sample",
"anthropic_version")
ANTHROPIC_BLOCK_TYPES = {"tool_use", "tool_result", "server_tool_use",
"web_search_tool_result", "thinking",
"redacted_thinking"}
DEFAULT_EXPECT = {"require_tools": True, "require_tool_choice": True,
"parallel_tool_calls": False, "forbid_tools": False}
# ---------------------------------------------------------------- loading ----
def load_scenarios(path):
cfg = json.load(open(path))
defaults = dict(DEFAULT_EXPECT)
defaults.update(cfg.get("defaults", {}).get("expect_request", {}))
marker_map = [] # (marker_lower, cname, pid)
scripts = {} # cname or cname/pid -> expanded script
pid_map = {} # pid -> cname (for call_gate_* id -> script lookup)
expects = {} # cname -> merged expect_request
for cname, cls in cfg["classes"].items():
scripts[cname] = expand_script(cls.get("script", []))
exp = dict(defaults)
exp.update(cls.get("expect_request", {}))
expects[cname] = exp
for ph in cls["phrasings"]:
if ph.get("script") is not None:
scripts[cname + "/" + ph["id"]] = expand_script(ph["script"])
marker_map.append((ph["marker"].lower(), cname, ph["id"]))
pid_map[ph["id"]] = cname
return {"cfg": cfg, "marker_map": marker_map, "scripts": scripts,
"pid_map": pid_map, "expects": expects}
def expand_script(script):
"""Same repeat-expansion contract as gate9's stub-llm.py ({N}/{NN})."""
out = []
for step in script:
if "repeat" in step:
for n in range(1, step["repeat"] + 1):
t = {k: v for k, v in step.items() if k != "repeat"}
out.append(json.loads(json.dumps(t)
.replace("{NN}", "%02d" % n)
.replace("{N}", str(n))))
else:
out.append(step)
return out
# ------------------------------------------------------------- validation ----
def _rej(message, code):
return {"status": 400, "message": message, "code": code}
def validate_dialect(headers, req):
"""Universal checks - run on EVERY request, scenario-matched or not.
Anything Anthropic-shaped on this lane means the brain's translator
leaked; the whole point is that it fails loudly, here, with a reason."""
if headers.get("anthropic-version"):
return _rej("anthropic-version header on the OpenAI lane: this "
"request was built by the Anthropic dialect path",
"gate_dialect_leak")
for k in ANTHROPIC_TOP_KEYS:
if k in req:
return _rej("top-level `%s` is Anthropic dialect; the OpenAI "
"dialect has no such field (system prompt goes in "
"messages[0])" % k, "gate_dialect_leak")
tools = req.get("tools")
if tools is not None:
if not isinstance(tools, list):
return _rej("`tools` must be an array", "gate_tools_shape")
names = []
for i, t in enumerate(tools):
if not isinstance(t, dict):
return _rej("tools[%d] is not an object" % i,
"gate_tools_shape")
if "input_schema" in t or (isinstance(t.get("function"), dict)
and "input_schema" in t["function"]):
return _rej("tools[%d] carries `input_schema` (Anthropic "
"dialect); OpenAI dialect wants "
"function.parameters" % i, "gate_dialect_leak")
if t.get("type") != "function":
return _rej("tools[%d].type must be \"function\", got %r"
% (i, t.get("type")), "gate_tools_shape")
fn = t.get("function")
if not isinstance(fn, dict):
return _rej("tools[%d].function missing" % i,
"gate_tools_shape")
if not isinstance(fn.get("name"), str) or not fn["name"]:
return _rej("tools[%d].function.name missing/empty" % i,
"gate_tools_shape")
if not isinstance(fn.get("description"), str) or not fn["description"]:
return _rej("tools[%d].function.description missing/empty" % i,
"gate_tools_shape")
if not isinstance(fn.get("parameters"), dict):
return _rej("tools[%d].function.parameters missing (JSON "
"Schema object expected)" % i, "gate_tools_shape")
names.append(fn["name"])
if len(names) != len(set(names)):
return _rej("tools: tool names must be unique", "gate_tools_shape")
msgs = req.get("messages")
if not isinstance(msgs, list) or not msgs:
return _rej("`messages` must be a non-empty array",
"gate_messages_shape")
for i, m in enumerate(msgs):
if not isinstance(m, dict):
return _rej("messages[%d] is not an object" % i,
"gate_messages_shape")
c = m.get("content")
if isinstance(c, list):
for j, b in enumerate(c):
if isinstance(b, dict) and b.get("type") in ANTHROPIC_BLOCK_TYPES:
return _rej("messages[%d].content[%d] is an Anthropic "
"`%s` block; the OpenAI dialect uses "
"tool_calls / role:\"tool\" messages"
% (i, j, b.get("type")), "gate_dialect_leak")
if m.get("role") == "tool":
if not isinstance(m.get("tool_call_id"), str) or not m["tool_call_id"]:
return _rej("messages[%d]: role \"tool\" requires a "
"`tool_call_id`" % i, "gate_messages_shape")
if "content" not in m:
return _rej("messages[%d]: role \"tool\" requires `content`"
% i, "gate_messages_shape")
if m.get("role") == "assistant" and m.get("tool_calls") is not None:
tcs = m["tool_calls"]
if not isinstance(tcs, list) or not tcs:
return _rej("messages[%d].tool_calls must be a non-empty "
"array" % i, "gate_tool_call_shape")
for j, tc in enumerate(tcs):
if not isinstance(tc, dict) or tc.get("type") != "function":
return _rej("messages[%d].tool_calls[%d].type must be "
"\"function\"" % (i, j), "gate_tool_call_shape")
if not isinstance(tc.get("id"), str) or not tc["id"]:
return _rej("messages[%d].tool_calls[%d].id missing"
% (i, j), "gate_tool_call_shape")
fn = tc.get("function")
if not isinstance(fn, dict) or not isinstance(fn.get("name"), str):
return _rej("messages[%d].tool_calls[%d].function.name "
"missing" % (i, j), "gate_tool_call_shape")
if not isinstance(fn.get("arguments"), str):
return _rej("messages[%d].tool_calls[%d].function."
"arguments must be a JSON-encoded STRING, "
"got %s" % (i, j,
type(fn.get("arguments")).__name__),
"gate_tool_call_shape")
return None
def validate_pairing(msgs):
"""OpenAI pairing rule: every assistant tool_calls turn must be followed
immediately by role:"tool" messages answering every tool_call_id."""
open_ids, open_at = set(), None
for i, m in enumerate(msgs):
role = m.get("role")
if role == "tool":
tid = m.get("tool_call_id")
if open_at is None:
return _rej("messages[%d]: role \"tool\" message with no "
"preceding assistant tool_calls turn "
"(tool_call_id=%s)" % (i, tid), "gate_pairing")
if tid not in open_ids:
return _rej("messages[%d]: tool message answers unknown or "
"already-answered tool_call_id %s" % (i, tid),
"gate_pairing")
open_ids.discard(tid)
continue
if open_ids:
return _rej("messages[%d]: assistant tool_calls not fully "
"answered before messages[%d]; missing tool "
"responses for: %s" % (open_at, i, sorted(open_ids)),
"gate_pairing")
open_ids, open_at = set(), None
if role == "assistant" and m.get("tool_calls"):
ids = [tc.get("id") for tc in m["tool_calls"]]
open_ids, open_at = set(ids), i
if open_ids:
return _rej("messages[%d]: assistant tool_calls at end of thread "
"without tool responses for: %s"
% (open_at, sorted(open_ids)), "gate_pairing")
return None
def validate_echo_args(msgs, loaded):
"""Ground-truth round-trip check: for every echoed gate-issued call id,
recompute the arguments this stub originally sent from the script and
require one json decode to reproduce them exactly. Catches the
two-escaper trap (spec section 6) deterministically."""
if not loaded:
return None
for i, m in enumerate(msgs):
if m.get("role") != "assistant":
continue
for tc in m.get("tool_calls") or []:
tid = tc.get("id", "")
if not tid.startswith("call_gate_"):
continue
rest = tid[len("call_gate_"):]
try:
pid, s_part, k_part = rest.rsplit("_", 2)
step_idx, k = int(s_part[1:]), int(k_part)
except (ValueError, IndexError):
continue
cname = loaded["pid_map"].get(pid)
if cname is None:
continue
script = (loaded["scripts"].get(cname + "/" + pid)
or loaded["scripts"].get(cname) or [])
if step_idx >= len(script):
continue
calls = script[step_idx].get("tool_calls") or []
if k >= len(calls):
continue
expected = calls[k]
fn = tc.get("function") or {}
if fn.get("name") != expected["name"]:
return _rej("messages[%d]: echoed tool name %r != issued %r "
"for %s" % (i, fn.get("name"), expected["name"],
tid), "gate_echo_mismatch")
try:
got = json.loads(fn.get("arguments", ""))
except ValueError:
return _rej("messages[%d]: echoed arguments for %s are not "
"valid JSON after one decode (truncated or "
"half-escaped?)" % (i, tid), "gate_echo_mismatch")
if got != expected["arguments"]:
hint = (" (decoded to a string, not an object: "
"double-encoded - the two-escaper trap)"
if isinstance(got, str) else "")
return _rej("messages[%d]: echoed arguments for %s do not "
"round-trip to the issued payload%s"
% (i, tid, hint), "gate_echo_mismatch")
return None
def validate_expect(req, exp):
"""Scenario-level request expectations (scenarios-openai.json)."""
tools = req.get("tools") or []
if exp.get("forbid_tools") and tools:
return _rej("this scenario is chat-only: no `tools` may be offered "
"on it", "gate_expect")
if exp.get("require_tools") and not tools:
return _rej("scenario expects a `tools` array to be offered (the "
"agentic lane must advertise its tools)", "gate_expect")
if exp.get("require_tool_choice"):
tc = req.get("tool_choice")
ok = tc in ("auto", "none", "required") or (
isinstance(tc, dict) and tc.get("type") == "function"
and isinstance(tc.get("function"), dict)
and tc["function"].get("name"))
if not ok:
return _rej("scenario expects an OpenAI-shaped `tool_choice`, "
"got %r" % (tc,), "gate_expect")
want_ptc = exp.get("parallel_tool_calls", None)
if want_ptc is not None:
if "parallel_tool_calls" not in req:
return _rej("scenario expects explicit `parallel_tool_calls` "
"(ADR-0005: must be pinned false on the wire)",
"gate_expect")
if req["parallel_tool_calls"] != want_ptc:
return _rej("scenario expects parallel_tool_calls=%s, got %s"
% (json.dumps(want_ptc),
json.dumps(req["parallel_tool_calls"])),
"gate_expect")
return None
# --------------------------------------------------------- scenario match ----
def extract_user_texts_newest_first(msgs):
texts = []
for m in reversed(msgs):
if not isinstance(m, dict) or m.get("role") != "user":
continue
c = m.get("content")
if isinstance(c, str):
texts.append(c)
elif isinstance(c, list):
for b in c:
if isinstance(b, dict) and b.get("type") == "text":
texts.append(b.get("text", ""))
return texts
def match_scenario(loaded, msgs):
for text in extract_user_texts_newest_first(msgs):
tl = text.lower()
for marker, cname, pid in loaded["marker_map"]:
if marker in tl:
return cname, pid
return None, None
# ------------------------------------------------------------- rendering ----
def completion_envelope(msg, finish, model, usage=(100, 100)):
return {"id": "chatcmpl-gate-" + uuid.uuid4().hex[:12],
"object": "chat.completion", "created": int(time.time()),
"model": model,
"choices": [{"index": 0, "message": msg,
"finish_reason": finish, "logprobs": None}],
"usage": {"prompt_tokens": usage[0],
"completion_tokens": usage[1],
"total_tokens": usage[0] + usage[1]}}
def text_completion(text, model):
return completion_envelope({"role": "assistant", "content": text},
"stop", model, usage=(1, 1))
def pending_body(seq, model):
args = json.dumps({"path": "never-%04d.md" % seq,
"content": "this run never completes"})
msg = {"role": "assistant", "content": None,
"tool_calls": [{"id": "call_hostile_pending_%04d" % seq,
"type": "function",
"function": {"name": "write_file",
"arguments": args}}]}
return completion_envelope(msg, "tool_calls", model, usage=(1, 1))
def render_step(step, cname, pid, step_idx, model):
"""Returns (http_status, body_dict, delivered) - delivered is ground
truth for the JSONL log."""
delivered = {"tool_calls": [], "finish_reason": None, "api_error": None}
if "api_error" in step:
e = step["api_error"]
delivered["api_error"] = e["status"]
return (e["status"],
{"error": {"message": e["message"],
"type": e.get("type", "server_error"),
"param": None, "code": e.get("code")}},
delivered)
msg = {"role": "assistant"}
finish = "stop"
if step.get("tool_calls"):
tcs = []
for k, call in enumerate(step["tool_calls"]):
tid = "call_gate_%s_s%d_%d" % (pid, step_idx, k)
tcs.append({"id": tid, "type": "function",
"function": {"name": call["name"],
"arguments": json.dumps(
call["arguments"],
ensure_ascii=False)}})
delivered["tool_calls"].append(call["name"])
msg["tool_calls"] = tcs
msg["content"] = step.get("text") # null when no narration, like real
finish = "tool_calls"
else:
msg["content"] = step["text"]
delivered["finish_reason"] = finish
return 200, completion_envelope(msg, finish, model), delivered
# ------------------------------------------------------------------ log ------
def log_record(rec):
with STATE["lock"]:
STATE["seq"] += 1
rec["seq"] = STATE["seq"]
with open(STATE["log_path"], "a") as f:
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
# ---------------------------------------------------------------- server -----
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def _send_json(self, status, obj):
body = json.dumps(obj, ensure_ascii=False).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def _send_error(self, verdict):
self._send_json(verdict["status"],
{"error": {"message": verdict["message"],
"type": "invalid_request_error",
"param": None, "code": verdict["code"]}})
def _drop_mid_body(self):
"""Valid 200 headers, half the promised body, then a socket abort
(same SO_LINGER teardown as gate9's mid-body-drop-brain.py)."""
full = json.dumps(text_completion(
"This reply will never finish arriving because the connection "
"dies in the middle of the body, which is exactly the point of "
"this hostile fixture.", "hostile-mid-drop")).encode("utf-8")
half = full[: len(full) // 2]
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(full))) # promises more
self.end_headers()
self.wfile.write(half)
self.wfile.flush()
try:
self.connection.setsockopt(socket.SOL_SOCKET, socket.SO_LINGER,
struct.pack("ii", 1, 0))
self.connection.shutdown(socket.SHUT_RDWR)
except OSError:
pass
self.close_connection = True
def do_GET(self):
path = self.path.split("?")[0]
if path == "/gate/health":
self._send_json(200, {"ok": True, "mode": STATE["mode"]})
elif path == "/gate/stats":
with STATE["lock"]:
self._send_json(200, {"mode": STATE["mode"],
"chat_hits": STATE["chat_hits"]})
else:
self._send_json(404, {"error": {"message": "not found",
"type": "invalid_request_error",
"param": None,
"code": "unknown_route"}})
def do_POST(self):
n = int(self.headers.get("Content-Length") or 0)
raw = self.rfile.read(n)
mode = STATE["mode"]
rec = {"ts": time.time(), "path": self.path, "mode": mode,
"kind": "background", "scenario_class": None, "phrasing": None,
"step": None, "n_messages": 0, "n_assistant": 0,
"validation": "ok", "validation_detail": None,
"delivered": {"tool_calls": [], "finish_reason": None,
"api_error": None},
"http_status": 200}
if self.path.split("?")[0] != "/v1/chat/completions":
rec.update(kind="wrong_path", http_status=404)
log_record(rec)
self._send_json(404, {"error": {
"message": "no such route: %s" % self.path,
"type": "invalid_request_error", "param": None,
"code": "unknown_route"}})
return
with STATE["lock"]:
STATE["chat_hits"] += 1
# ---- hostile modes: behavior first, no validation ----------------
if mode == "black-hole":
rec.update(kind="hostile", http_status=None)
log_record(rec)
threading.Event().wait() # hold the socket open forever
return
if mode == "mid-body-drop":
rec.update(kind="hostile", http_status=200)
log_record(rec)
self._drop_mid_body()
return
if mode == "tool-pending-forever":
seq = next(_PENDING_SEQ)
rec.update(kind="hostile",
delivered={"tool_calls": ["write_file"],
"finish_reason": "tool_calls",
"api_error": None})
log_record(rec)
self._send_json(200, pending_body(seq, "gate-openai-model"))
return
# ---- normal mode -------------------------------------------------
try:
req = json.loads(raw)
except ValueError as exc:
# DIAGNOSTIC CAPTURE (2026-08-06): an unparseable body used to be recorded as
# a bare "bad_json" with the bytes thrown away, which made an intermittent
# failure impossible to root-cause — you cannot fix what you did not keep.
# Dump the raw body next to the log, and record exactly where the parser gave
# up plus the offending byte, so one occurrence is enough to diagnose.
dump_path = "%s.badbody.%s" % (STATE.get("log_path", "/tmp/stub-openai"),
rec.get("seq", "x"))
try:
data = raw if isinstance(raw, (bytes, bytearray)) else str(raw).encode()
with open(dump_path, "wb") as fh:
fh.write(data)
except Exception as dump_exc:
dump_path = "(dump failed: %s)" % dump_exc
pos = getattr(exc, "pos", None)
near = ""
byte_repr = ""
if isinstance(pos, int):
blob = raw if isinstance(raw, (bytes, bytearray)) else str(raw).encode()
near = blob[max(0, pos - 60):pos + 60].decode("utf-8", "replace")
if 0 <= pos < len(blob):
byte_repr = "0x%02x" % blob[pos]
rec.update(kind="bad_json", validation="rejected",
validation_detail="request body is not valid JSON: %s" % exc,
http_status=400, raw_len=len(raw), raw_dump=dump_path,
err_pos=pos, err_byte=byte_repr, err_near=near)
log_record(rec)
self._send_error(_rej("request body is not valid JSON",
"bad_json"))
return
msgs = req.get("messages") or []
rec["n_messages"] = len(msgs)
rec["n_assistant"] = sum(1 for m in msgs if isinstance(m, dict)
and m.get("role") == "assistant")
loaded = STATE["scenarios"]
cname, pid = match_scenario(loaded, msgs)
if cname:
rec.update(kind="scenario", scenario_class=cname, phrasing=pid)
# Wire-level validation runs for EVERY request, scenario or not.
verdict = (validate_dialect(self.headers, req)
or validate_pairing([m for m in msgs
if isinstance(m, dict)])
or validate_echo_args(msgs, loaded))
if verdict:
rec.update(validation="rejected",
validation_detail=verdict["message"],
http_status=verdict["status"])
log_record(rec)
self._send_error(verdict)
return
model = req.get("model", "gate-openai-model")
if not cname:
log_record(rec)
self._send_json(200, text_completion("ok", model))
return
script = (loaded["scripts"].get(cname + "/" + pid)
or loaded["scripts"][cname])
step_idx = rec["n_assistant"]
if step_idx >= len(script):
rec.update(kind="overrun", step=step_idx)
log_record(rec)
self._send_json(200, text_completion(
"GATE-SCRIPT-EXHAUSTED %s step %d" % (pid, step_idx), model))
return
step = script[step_idx]
exp = dict(loaded["expects"][cname])
exp.update(step.get("expect_request", {}))
verdict = validate_expect(req, exp)
if verdict:
rec.update(step=step_idx, validation="rejected",
validation_detail=verdict["message"],
http_status=verdict["status"])
log_record(rec)
self._send_error(verdict)
return
status, body, delivered = render_step(step, cname, pid, step_idx,
model)
rec.update(step=step_idx, delivered=delivered, http_status=status)
log_record(rec)
self._send_json(status, body)
def log_message(self, *a):
pass
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--port", type=int, required=True)
ap.add_argument("--scenarios",
help="scenarios-openai.json (required in normal mode)")
ap.add_argument("--log", required=True)
ap.add_argument("--mode", default="normal",
choices=["normal", "black-hole", "mid-body-drop",
"tool-pending-forever"])
args = ap.parse_args()
if args.port in (7770, 7779, 17779):
raise SystemExit("stub-openai: refusing production Neuron port")
if args.mode == "normal" and not args.scenarios:
raise SystemExit("stub-openai: --scenarios is required in normal mode")
STATE["mode"] = args.mode
STATE["scenarios"] = (load_scenarios(args.scenarios)
if args.scenarios else None)
STATE["log_path"] = args.log
open(args.log, "w").close()
n_markers = (len(STATE["scenarios"]["marker_map"])
if STATE["scenarios"] else 0)
print("stub-openai [%s]: 127.0.0.1:%d /v1/chat/completions "
"(%d markers registered, log=%s)"
% (args.mode, args.port, n_markers, args.log), flush=True)
ThreadingHTTPServer(("127.0.0.1", args.port), Handler).serve_forever()
if __name__ == "__main__":
main()
-148
View File
@@ -1,148 +0,0 @@
#!/usr/bin/env bash
# run-el-test.sh — build and RUN one engine test (tests/*.el), printing its assertions.
#
# WHY THIS EXISTS (2026-08-06): the engine's tests/*.el files were never runnable from the
# tree. `elc` is a COMPILER — it emits C to stdout and exits; it does not execute anything.
# So "the tests" could only ever be read, not run, and a signature change could silently
# break them (exactly what happened when bridge_save gained its `wire` argument). This
# script closes that: emit the test to C, link it against the engine modules, execute it.
#
# HOW IT WORKS
# 1. elc <test>.el -> C on stdout (the test file's `main` + prototypes)
# 2. elb (once, cached) -> per-module C for the whole engine into a scratch dir
# 3. cc test.c + all modules EXCEPT soul.c (soul.c owns the real `main`) + the runtime
# 4. run it
#
# The test C references only the engine functions it actually calls, so there are no
# duplicate-symbol collisions with the module objects.
#
# RUNTIME: the REPO-PINNED vendor/el-runtime (NOT ~/el-sdk/el_runtime.c — that June build
# is missing builtins August code calls: engram_wm_count, engram_wm_top_json,
# http_delete_json, http_serve_async; linking against it fails with "symbol(s) not found").
#
# USAGE
# tests/run-el-test.sh tests/test_bridge_serialization.el # one test
# tests/run-el-test.sh --all # every tests/test_*.el
# REBUILD=1 tests/run-el-test.sh ... # force module regeneration
#
# Tests that need a live API key / running soul (see each file's header) will report their
# own skips or failures — this runner does not fake them.
set -uo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$REPO_ROOT" || exit 2
ELC="${ELC:-$HOME/el-sdk/elc}"
ELB="${ELB:-$HOME/Development/el-sdk/bin/elb}"
RUNTIME_DIR="${RUNTIME_DIR:-$REPO_ROOT/vendor/el-runtime/v1.0.0-20260501}"
SCRATCH="${SCRATCH:-/tmp/el-test-$(basename "$REPO_ROOT")}"
MODDIR="$SCRATCH/modules"
OPENSSL_INC="${OPENSSL_INC:-/opt/homebrew/opt/openssl@3/include}"
OPENSSL_LIB="${OPENSSL_LIB:-/opt/homebrew/opt/openssl@3/lib}"
for req in "$ELC" "$ELB" "$RUNTIME_DIR/el_runtime.c"; do
[ -e "$req" ] || { echo "run-el-test: missing required input: $req" >&2; exit 2; }
done
mkdir -p "$MODDIR" || exit 2
# ── Step 1: engine modules (cached — regeneration is the slow part) ────────────
if [ "${REBUILD:-0}" = "1" ] || [ ! -f "$MODDIR/chat.c" ] || [ "chat.el" -nt "$MODDIR/chat.c" ]; then
echo "run-el-test: generating engine modules into $MODDIR (this takes ~1-2 min)..."
# elb's own final link step fails by design here (it wants to produce a binary named
# `neuron` and we only need the per-module .c files it emits first). Ignore its rc.
"$ELB" --elc="$ELC" --runtime="$RUNTIME_DIR" --out="$MODDIR/" >"$SCRATCH/elb.log" 2>&1
if [ ! -f "$MODDIR/chat.c" ]; then
echo "run-el-test: FATAL — elb produced no chat.c; see $SCRATCH/elb.log" >&2
tail -5 "$SCRATCH/elb.log" >&2
exit 2
fi
# elb rewrites *.elh in the source tree as a side effect (cosmetic banner churn plus a
# stray soul..elh). Say so; the caller decides whether to `git restore` them.
echo "run-el-test: NOTE — elb regenerated *.elh in the source tree (cosmetic churn is expected; a stray soul..elh may appear)."
fi
# soul.c is needed for its engine functions (layered_cycle et al.) but it also owns the
# daemon's real `main`, which would collide with the test's own. Compile it ONCE to an
# object with `main` renamed away, and link that instead of the .c.
SOUL_OBJ="$SCRATCH/soul-nomain.o"
if [ "${REBUILD:-0}" = "1" ] || [ ! -f "$SOUL_OBJ" ] || [ "$MODDIR/soul.c" -nt "$SOUL_OBJ" ]; then
cc -std=c11 -O1 -DHAVE_CURL -Dmain=el_soul_daemon_main_unused \
-I "$RUNTIME_DIR" -I "$MODDIR" -I "$OPENSSL_INC" \
-include dist/elp-c-decls.h -Wno-error=implicit-function-declaration \
-c "$MODDIR/soul.c" -o "$SOUL_OBJ" 2>"$SCRATCH/soul-nomain.err" \
|| { echo "run-el-test: FATAL — could not compile soul.c without main" >&2
grep -E 'error:' "$SCRATCH/soul-nomain.err" | head -5 >&2; exit 2; }
fi
# Every module except soul.c (linked as the renamed object above) and the stray soul.elh.c.
MODS=("$SOUL_OBJ")
for f in "$MODDIR"/*.c; do
case "$(basename "$f")" in
soul.c|soul.elh.c) continue ;;
esac
MODS+=("$f")
done
[ "${#MODS[@]}" -gt 1 ] || { echo "run-el-test: no module objects found" >&2; exit 2; }
run_one() {
local test_el="$1"
local name; name="$(basename "$test_el" .el)"
local cfile="$SCRATCH/$name.c"
local bin="$SCRATCH/$name"
printf '\n══ %s ══\n' "$name"
if ! "$ELC" "$test_el" >"$cfile" 2>"$SCRATCH/$name.elc.err"; then
echo "COMPILE FAILED (elc):"; tail -10 "$SCRATCH/$name.elc.err"; return 1
fi
[ -s "$cfile" ] || { echo "COMPILE FAILED (elc produced empty C)"; return 1; }
if ! cc -std=c11 -O1 -DHAVE_CURL -rdynamic \
-I "$RUNTIME_DIR" -I "$MODDIR" -I "$OPENSSL_INC" -L "$OPENSSL_LIB" \
-include dist/elp-c-decls.h -Wno-error=implicit-function-declaration \
-o "$bin" "$cfile" "${MODS[@]}" "$RUNTIME_DIR/el_runtime.c" \
-lssl -lcrypto -lcurl -lpthread -lm 2>"$SCRATCH/$name.link.err"; then
echo "LINK FAILED:"; grep -E '"_|error:' "$SCRATCH/$name.link.err" | head -10; return 1
fi
# THE RUNNER OWNS THE VERDICT — the test files cannot be trusted to report it.
#
# Every tests/*.el assert helper does `let pass_count = pass_count + 1` INSIDE an if
# BLOCK. El's scope rule (the same one chat.el documents at every while-body mutation:
# "mutations inside if *blocks* don't escape scope") means those counters never
# increment, so all 9 counted test files print "N passed, M failed" as "0 passed, 0
# failed" — forever, whatever actually happened. A summary that can never report a
# failure is worth exactly as much as an assertion that can never fail. Logged as a
# bug for the real in-file fix; until then the verdict is computed HERE, from the
# assert helpers' own per-line output, which IS reliable.
local out="$SCRATCH/$name.out"
"$bin" 2>&1 | tee "$out"; local rc=${PIPESTATUS[0]}
# NOTE: `grep -c` prints 0 AND exits 1 when there are no matches, so a `|| echo 0`
# fallback appends a SECOND zero and every later integer test breaks on "0\n0".
# (Caught by running this script — which is the whole argument for running things.)
local n_pass n_fail
n_pass=$(grep -c '^ PASS: ' "$out" 2>/dev/null); n_pass=${n_pass:-0}
n_fail=$(grep -c '^ FAIL: ' "$out" 2>/dev/null); n_fail=${n_fail:-0}
echo "── $name: $n_pass passed, $n_fail failed (counted by the runner, not by the file's dead counters)"
if [ "$n_fail" -gt 0 ]; then
echo " failing assertions:"; grep '^ FAIL: ' "$out" | sed 's/^/ /'
return 1
fi
if [ "$n_pass" -eq 0 ]; then
echo " WARNING: no assertions ran — treating as FAILURE (a test that asserts nothing is not a passing test)"
return 1
fi
[ $rc -eq 0 ] || { echo " (test binary exited rc=$rc)"; return 1; }
return 0
}
rc_all=0
if [ "${1:-}" = "--all" ]; then
for t in tests/test_*.el; do run_one "$t" || rc_all=1; done
else
[ $# -ge 1 ] || { echo "usage: tests/run-el-test.sh <tests/test_x.el> | --all" >&2; exit 2; }
for t in "$@"; do run_one "$t" || rc_all=1; done
fi
exit $rc_all
+4 -109
View File
@@ -93,7 +93,7 @@ println("1. bridge_save — empty messages guard")
let sid1: String = "test-session-empty-messages"
state_set("mcp_bridge:" + sid1, "")
let save1_ok: Bool = bridge_save(sid1, "claude-sonnet-4-5", "sys", "[]", "", "", "call-1", "anthropic")
let save1_ok: Bool = bridge_save(sid1, "claude-sonnet-4-5", "sys", "[]", "", "", "call-1")
assert_false("empty messages -> bridge_save returns false", save1_ok)
let saved1: String = state_get("mcp_bridge:" + sid1)
@@ -107,7 +107,7 @@ println("2. bridge_save — empty tools_json guard")
let sid2: String = "test-session-empty-tools"
state_set("mcp_bridge:" + sid2, "")
let save2_ok: Bool = bridge_save(sid2, "claude-sonnet-4-5", "sys", "", "[{\"role\":\"user\",\"content\":\"hi\"}]", "", "call-2", "anthropic")
let save2_ok: Bool = bridge_save(sid2, "claude-sonnet-4-5", "sys", "", "[{\"role\":\"user\",\"content\":\"hi\"}]", "", "call-2")
assert_false("empty tools_json -> bridge_save returns false", save2_ok)
let saved2: String = state_get("mcp_bridge:" + sid2)
@@ -126,7 +126,7 @@ state_set("mcp_bridge:" + sid3, "")
let msgs3: String = "[{\"role\":\"user\",\"content\":\"hello\"}]"
let tools3: String = "[{\"name\":\"read_file\"}]"
let save3_ok: Bool = bridge_save(sid3, "claude-sonnet-4-5", "You are a helper.", tools3, msgs3, "read_file", "toolu_abc", "anthropic")
let save3_ok: Bool = bridge_save(sid3, "claude-sonnet-4-5", "You are a helper.", tools3, msgs3, "read_file", "toolu_abc")
assert_true("valid args -> bridge_save returns true", save3_ok)
let blob3: String = state_get("mcp_bridge:" + sid3)
@@ -243,7 +243,7 @@ state_set("mcp_bridge:" + sid8, "")
let special_id: String = "toolu_test\"quoted\""
let msgs8: String = "[{\"role\":\"user\",\"content\":\"hi\"}]"
let tools8: String = "[{\"name\":\"read_file\"}]"
let save8_ok: Bool = bridge_save(sid8, "claude-sonnet-4-5", "sys", tools8, msgs8, "", special_id, "anthropic")
let save8_ok: Bool = bridge_save(sid8, "claude-sonnet-4-5", "sys", tools8, msgs8, "", special_id)
assert_true("special chars in tool_use_id -> bridge_save returns true", save8_ok)
let blob8: String = state_get("mcp_bridge:" + sid8)
@@ -251,111 +251,6 @@ let blob8: String = state_get("mcp_bridge:" + sid8)
let retrieved_id: String = json_get(blob8, "tool_use_id")
assert_eq("tool_use_id with quotes round-trips via json_safe", retrieved_id, special_id)
// Section 9: the "wire" field (OpenAI-tools port, 2026-08-06)
//
// A suspended turn must resume on the SAME wire format it suspended on: an OpenAI-lane
// bridge answered with an Anthropic-shaped tool_result (or vice versa) is a dead run.
// bridge_save therefore stamps the blob with "wire", and agentic_resume branches on it.
//
// §9c is the important one. json_get is a first-substring-match scanner, so any key that
// appears inside the UNESCAPED conversation embedded in messages_raw can be matched
// instead of the blob's own field that exact class of bug produced the round-9 resume
// failure (json_get(blob,"tool_use_id") matching a web_search_tool_result's id inside the
// replayed conversation). "wire" is written as a json_safe'd SCALAR ahead of both raw
// fields precisely so a decoy in model-controlled bytes can never win. This test plants
// that decoy on purpose. If someone later moves the field after messages_raw, this fails.
println("")
println("9. bridge_save — wire tagging and its field-order guarantee")
// 9a. an OpenAI-lane suspension round-trips as "openai"
let sid9: String = "test-session-wire-openai"
state_set("mcp_bridge:" + sid9, "")
let msgs9: String = "[{\"role\":\"user\",\"content\":\"hi\"}]"
let tools9: String = "[{\"name\":\"read_file\"}]"
let save9_ok: Bool = bridge_save(sid9, "llama-3.3-70b-versatile", "sys", tools9, msgs9, "", "call_abc", "openai")
assert_true("openai wire -> bridge_save returns true", save9_ok)
let blob9: String = state_get("mcp_bridge:" + sid9)
assert_eq("wire round-trips as openai", json_get(blob9, "wire"), "openai")
// 9b. an Anthropic-lane suspension round-trips as "anthropic"
let sid9b: String = "test-session-wire-anthropic"
state_set("mcp_bridge:" + sid9b, "")
let save9b_ok: Bool = bridge_save(sid9b, "claude-sonnet-4-5", "sys", tools9, msgs9, "", "toolu_abc", "anthropic")
assert_true("anthropic wire -> bridge_save returns true", save9b_ok)
let blob9b: String = state_get("mcp_bridge:" + sid9b)
assert_eq("wire round-trips as anthropic", json_get(blob9b, "wire"), "anthropic")
// 9c. FIELD-ORDER GUARD: a decoy "wire" inside the conversation must NOT be matched.
let sid9c: String = "test-session-wire-decoy"
state_set("mcp_bridge:" + sid9c, "")
let msgs9c: String = "[{\"role\":\"user\",\"content\":\"please save this literal text: \\\"wire\\\":\\\"anthropic\\\" end\"}]"
let save9c_ok: Bool = bridge_save(sid9c, "llama-3.3-70b-versatile", "sys", tools9, msgs9c, "", "call_decoy", "openai")
assert_true("decoy conversation -> bridge_save returns true", save9c_ok)
let blob9c: String = state_get("mcp_bridge:" + sid9c)
assert_eq("blob's own wire wins over a decoy planted in messages_raw", json_get(blob9c, "wire"), "openai")
// 9d. LEGACY blob (written before the port) has no wire field: json_get yields "",
// which agentic_resume treats as the Anthropic path old suspensions still resume.
let sid9d: String = "test-session-wire-legacy"
let legacy_blob: String = "{\"model\":\"claude-sonnet-4-5\",\"safe_sys\":\"sys\",\"tools_log\":\"\""
+ ",\"tool_use_id\":\"toolu_legacy\",\"tools_raw\":[{\"name\":\"read_file\"}]"
+ ",\"messages_raw\":[{\"role\":\"user\",\"content\":\"hi\"}]}"
state_set("mcp_bridge:" + sid9d, legacy_blob)
let blob9d: String = state_get("mcp_bridge:" + sid9d)
assert_eq("legacy blob has no wire field -> empty (resumes as anthropic)", json_get(blob9d, "wire"), "")
assert_eq("legacy blob still reads its tool_use_id", json_get(blob9d, "tool_use_id"), "toolu_legacy")
// 9e. THE HARDER DECOY: a LEGACY blob (no wire field of its own) that carries the bytes
// of a wire tag deeper inside, where an unbounded first-match scan would find it and
// misroute the resume onto the wrong loop the round-9 defect class exactly.
//
// WHAT IS AND IS NOT REACHABLE (measured here, not assumed an earlier version of this
// test asserted the wrong thing and was corrected by running it):
// * NOT reachable from ordinary conversation TEXT. Any quote a user or model writes is
// backslash-escaped when it is serialized into the blob, so prose containing
// "wire":"openai" is stored as \"wire\":\"openai\" and does not match a scan for the
// unescaped key. §9f pins that.
// * REACHABLE from STRUCTURAL keys, which are embedded raw. Conversation and tool
// objects keep real quotes that is precisely how round 9's scan found a
// web_search_tool_result's tool_use_id. A connector-supplied tool schema or a future
// message field literally named "wire" would be found the same way.
// The bound removes the whole class rather than reasoning about which keys exist today.
let sid9e: String = "test-session-wire-legacy-decoy"
let decoy_blob: String = "{\"model\":\"claude-sonnet-4-5\",\"safe_sys\":\"sys\",\"tools_log\":\"\""
+ ",\"tool_use_id\":\"toolu_legacy\""
+ ",\"tools_raw\":[{\"name\":\"read_file\",\"wire\":\"openai\"}]"
+ ",\"messages_raw\":[{\"role\":\"user\",\"content\":\"hi\"}]}"
state_set("mcp_bridge:" + sid9e, decoy_blob)
let blob9e: String = state_get("mcp_bridge:" + sid9e)
// Unbounded read (what NOT to do) proves the hazard this guard exists for is real.
assert_eq("unbounded scan DOES find a structural decoy (why the bound is needed)", json_get(blob9e, "wire"), "openai")
// Bounded read the same computation agentic_resume performs.
let d_traw: Int = str_index_of(blob9e, ",\"tools_raw\":")
let d_tjson: Int = str_index_of(blob9e, ",\"tools_json\":")
let d_mraw: Int = str_index_of(blob9e, ",\"messages_raw\":")
let d_msgs: Int = str_index_of(blob9e, ",\"messages\":")
let dcut1: Int = if d_traw > 0 { d_traw } else { str_len(blob9e) }
let dcut2: Int = if d_tjson > 0 && d_tjson < dcut1 { d_tjson } else { dcut1 }
let dcut3: Int = if d_mraw > 0 && d_mraw < dcut2 { d_mraw } else { dcut2 }
let dcut: Int = if d_msgs > 0 && d_msgs < dcut3 { d_msgs } else { dcut3 }
let head9e: String = str_slice(blob9e, 0, dcut)
assert_eq("bounded scan ignores the decoy -> legacy blob resumes as anthropic", json_get(head9e, "wire"), "")
assert_not_contains("scalar head excludes the bulk fields entirely", head9e, "read_file")
// 9f. Escaping bounds the severity: prose CANNOT inject a scalar-looking key, because
// its quotes are escaped on the way in. Documented as a measured fact, so nobody has to
// re-derive it the next time this question comes up.
let sid9f: String = "test-session-wire-prose"
let prose_blob: String = "{\"model\":\"claude-sonnet-4-5\",\"safe_sys\":\"sys\",\"tools_log\":\"\""
+ ",\"tool_use_id\":\"toolu_legacy\",\"tools_raw\":[{\"name\":\"read_file\"}]"
+ ",\"messages_raw\":[{\"role\":\"user\",\"content\":\"remember this: \\\"wire\\\":\\\"openai\\\"\"}]}"
state_set("mcp_bridge:" + sid9f, prose_blob)
let blob9f: String = state_get("mcp_bridge:" + sid9f)
assert_eq("escaped prose cannot spoof the key even unbounded (severity bound)", json_get(blob9f, "wire"), "")
// Summary
println("")
-213
View File
@@ -1,213 +0,0 @@
// test_history_amplification.el
//
// REGRESSION TEST FOR ISSUE #129 (P0, SAFETY).
//
// What this guards: on the agentic path, the crisis score has two halves the
// message you just sent, and the distress that has accumulated across the
// conversation. The second half is the whole reason the escalation logic exists:
// someone whose distress builds over several turns never sends one message that
// trips the bell on its own.
//
// The defect this test was written against (ff421d3, 2026-08-05 fixed
// 2026-08-07): conversation history moved to a per-session key via
// conv_hist_key(session_id), but the agentic path's safety screen was left
// reading the old anonymous "conv_history" bucket. The desktop app always sends
// a session_id, so the screen received "" on every real conversation and the
// escalation half always scored 0. Nothing failed. Nothing logged. The comment
// above the defective line documented this same bug being fixed once before.
//
// THE INVARIANT UNDER TEST, stated so it survives future renames:
// the window the safety screen READS must be the window conv_history_record
// WRITES. Not "must be called conv_history" must AGREE.
//
// This test is deliberately written to fail loudly on the pre-fix source. If it
// ever passes on code where the screen reads a key nothing writes, it is broken.
//
// To run (macOS, from the worktree root):
// scripts/run-el-test.sh tests/test_history_amplification.el
//
import "../chat.el"
import "../safety.el"
import "../sessions.el"
// Program class. Without this an El program compiles as a 'utility', and a
// utility may not call the self-formation primitives (llm_call_system,
// llm_vision) that chat.el's agentic loop references the unit fails to
// compile with a capability violation even though the test never calls them.
// Declaring 'cgi' matches how soul.el declares itself.
//
// The endpoints below are deliberately DEAD: this test must never reach a live
// engram, and nothing it asserts depends on one. Port 9 is discard.
cgi "neuron-test-history-amplification" {
dharma_id: "ntn-test@http://127.0.0.1:9",
principal: "test-harness",
network: "dharma-testnet",
engram: "http://127.0.0.1:9"
}
// Counters
//
// NOTE for anyone copying this harness: the idiom used by the older tests in
// this directory `let pass_count = pass_count + 1` inside an assert function
// does NOT mutate the module-level binding. It declares a new local that dies
// with the call, so those suites all print "0 passed, 0 failed" no matter what
// happened. Counters go through the state store here so the summary is real.
fn bump(counter: String) -> Void {
let cur: String = state_get(counter)
let n: Int = if str_eq(cur, "") { 0 } else { str_to_int(cur) }
state_set(counter, int_to_str(n + 1))
}
fn counter_of(counter: String) -> Int {
let cur: String = state_get(counter)
if str_eq(cur, "") { return 0 }
return str_to_int(cur)
}
fn assert_eq(label: String, got: String, expected: String) -> Void {
if str_eq(got, expected) {
bump("t_pass")
println(" PASS: " + label)
} else {
bump("t_fail")
println(" FAIL: " + label)
println(" got: " + got)
println(" expected: " + expected)
}
}
fn assert_true(label: String, cond: Bool) -> Void {
if cond {
bump("t_pass")
println(" PASS: " + label)
} else {
bump("t_fail")
println(" FAIL: " + label)
}
}
// Fixtures
//
// ESCALATION_MSG scores 50 on its own ("can't go on", safety_score_crisis s9).
// 50 is >= soft (35) and < hard (70): a soft bell by itself, which is the point.
// It is the message that is only a crisis IN CONTEXT.
//
// DISTRESS_TURN carries five history phrases:
// hopeless 15 + worthless 15 + nobody cares 15 + can't take it anymore 20
// + want to disappear 20 = 85; contributed at 1/3 weight = 28.
// 50 + 28 = 78 >= 70 hard_bell. That 28-point swing IS the escalation logic,
// and it is exactly what the defect suppressed.
let TEST_SESSION: String = "sess-test-129-history-amplification"
let ESCALATION_MSG: String = "honestly i can't go on like this"
let DISTRESS_TURN: String = "i feel hopeless and worthless, nobody cares, i can't take it anymore, i want to disappear"
println("")
println("test_history_amplification.el — issue #129 (agentic multi-turn crisis escalation)")
// 1. Baseline: the message alone is a SOFT bell, not a hard one
//
// If this ever returns hard_bell, the test below proves nothing the message
// would trip the bell without any history and the amplification would be
// invisible. This assertion is what keeps the real test honest.
println("")
println("1. baseline — escalation message with NO history is a soft bell")
let baseline: String = safety_screen(ESCALATION_MSG, "")
assert_eq("no history -> soft_bell (not hard)", json_get(baseline, "action"), "soft_bell")
// 2. Producer sanity: history lands in the session's own window
println("")
println("2. producer — conv_history_record writes the session's window")
conv_history_record(TEST_SESSION, DISTRESS_TURN, "i hear you, that sounds heavy", "")
let written: String = state_get(conv_hist_key(TEST_SESSION))
assert_true("session window is non-empty after record", !str_eq(written, ""))
assert_true("session window contains the distress turn", str_contains(written, "hopeless"))
// 3. THE REGRESSION: the agentic screen must SEE that window
//
// Pre-fix this returns soft_bell, because agentic_safety_screen read the
// anonymous bucket and got "". Post-fix it returns hard_bell.
println("")
println("3. REGRESSION #129 — agentic screen reads the session's own window")
let screened: String = agentic_safety_screen(TEST_SESSION, ESCALATION_MSG)
assert_eq(
"distress history escalates the agentic screen to hard_bell",
json_get(screened, "action"),
"hard_bell"
)
// 4. The invariant, stated directly
//
// Independent of thresholds and phrase lists: whatever the screen reads for a
// session must equal what the recorder wrote for that session. This is the
// assertion that survives a future rename of either side.
println("")
println("4. invariant — read window == written window")
let read_back: String = state_get(conv_hist_key(TEST_SESSION))
assert_true("screen input is the recorded window, not empty", !str_eq(read_back, ""))
assert_eq("read window is byte-identical to written window", read_back, written)
// 5. No false positive: a calm session does not escalate
//
// A test that only ever asserts "hard_bell" would pass on code that hard-bells
// every message. This is the other leg, and it runs BEFORE the anonymous case
// below on purpose: that case writes the shared bucket, and under the defect a
// calm session would then inherit it.
println("")
println("5. specificity — a calm history does NOT escalate")
let CALM_SESSION: String = "sess-test-129-calm"
state_set("conv_history", "")
conv_history_record(CALM_SESSION, "what is the weather like today", "clear and mild", "")
let calm: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
assert_eq("calm history stays at soft_bell", json_get(calm, "action"), "soft_bell")
// 6. Cross-session leakage
//
// The same defect had a second face: because the screen read one shared bucket,
// a calm session could be scored against a DIFFERENT session's distress. That is
// wrong in both directions it fabricates a crisis for the calm user and it
// leaks the distressed user's content into another session's scoring.
println("")
println("6. isolation — one session's distress must not score another session")
state_set("conv_history", "")
let OTHER_SESSION: String = "sess-test-129-other"
conv_history_record(OTHER_SESSION, DISTRESS_TURN, "i hear you", "")
let isolated: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
assert_eq(
"a distressed OTHER session does not escalate the calm session",
json_get(isolated, "action"),
"soft_bell"
)
// 7. Anonymous sessions still work
//
// conv_hist_key("") deliberately falls back to the shared "conv_history" bucket.
// The fix must not break the no-session_id path older callers rely on. Runs last
// because it writes that shared bucket.
println("")
println("7. anonymous path — empty session_id still screens against the shared window")
state_set("conv_history", "[{\"role\":\"user\",\"content\":\"" + DISTRESS_TURN + "\"}]")
let anon: String = agentic_safety_screen("", ESCALATION_MSG)
assert_eq("anonymous session escalates too", json_get(anon, "action"), "hard_bell")
// Summary
println("")
println("history amplification tests: " + int_to_str(counter_of("t_pass")) + " passed, " + int_to_str(counter_of("t_fail")) + " failed")
-92
View File
@@ -1,92 +0,0 @@
// tests/test_utf8_slice.el
//
// Guards utf8_safe_slice(), the fix for a live defect found 2026-08-06:
//
// The session preload cuts recalled memory content at a fixed length
// (chat.el: `if str_len(acc) > 350 { str_slice(acc, 0, 350) }` and
// session_preload_bullets' identical per-bullet cut). str_slice and str_len count
// BYTES, so any cut landing inside a multi-byte UTF-8 character leaves a dangling
// lead byte in the system prompt and the whole request body is then invalid UTF-8.
// Providers reject it outright, so the user sees "AI unavailable" with no clue why,
// on both wire formats. Caught by an OpenAI-lane gate whose stub decodes strictly;
// reproduced from a real memory whose content contained box-drawing rules (E2 94 80).
//
// Trigger is ordinary content: an em dash, a curly quote, an accented name, a table
// border, an emoji anything non-ASCII sitting on the cut boundary. It gets MORE
// likely as a user's memory grows, which is the opposite of what should happen.
//
// §1 also pins the semantics this fix depends on: that str_char_code returns the
// BYTE value at a byte index (not a decoded code point). If a future runtime changes
// that, these assertions fail loudly instead of the truncation silently rotting.
import "../chat.el"
let pass_count: Int = 0
let fail_count: Int = 0
fn assert_eq(label: String, got: String, expected: String) -> Void {
if str_eq(got, expected) {
let pass_count = pass_count + 1
println(" PASS: " + label)
} else {
let fail_count = fail_count + 1
println(" FAIL: " + label)
println(" got: " + got)
println(" expected: " + expected)
}
}
fn assert_eq_int(label: String, got: Int, expected: Int) -> Void {
assert_eq(label, int_to_str(got), int_to_str(expected))
}
println("")
println("1. runtime semantics this fix relies on")
// "" is U+2500 = E2 94 80 (three bytes). If str_len counts bytes, len("") is 3.
let dash: String = ""
assert_eq_int("str_len counts BYTES (one box-drawing char = 3)", str_len(dash), 3)
assert_eq_int("str_char_code returns the BYTE value (lead byte of U+2500 = 0xE2 = 226)", str_char_code(dash, 0), 226)
assert_eq_int("str_char_code second byte = 0x94 = 148", str_char_code(dash, 1), 148)
assert_eq_int("str_char_code third byte = 0x80 = 128", str_char_code(dash, 2), 128)
println("")
println("2. utf8_safe_slice — never leaves a partial character")
// Pure ASCII: behaves exactly like str_slice.
assert_eq("ascii under the limit is untouched", utf8_safe_slice("hello", 10), "hello")
assert_eq("ascii over the limit cuts exactly", utf8_safe_slice("hello world", 5), "hello")
// A cut landing INSIDE a 3-byte character must drop that character entirely.
// "ab─cd": bytes a b E2 94 80 c d. Cutting at 3 or 4 lands mid-dash.
let mixed: String = "ab─cd"
assert_eq_int("fixture is 7 bytes (2 ascii + 3 + 2 ascii)", str_len(mixed), 7)
assert_eq("cut inside the char (n=3) drops the partial char", utf8_safe_slice(mixed, 3), "ab")
assert_eq("cut inside the char (n=4) drops the partial char", utf8_safe_slice(mixed, 4), "ab")
// A cut landing exactly AFTER a complete character keeps it.
assert_eq("cut on the char boundary (n=5) keeps the whole char", utf8_safe_slice(mixed, 5), "ab─")
// 2-byte character (é = C3 A9) and 4-byte character (😀 = F0 9F 98 80).
let acc: String = ""
assert_eq("cut inside a 2-byte char drops it", utf8_safe_slice(acc, 2), "x")
assert_eq("cut after a 2-byte char keeps it", utf8_safe_slice(acc, 3), "")
let emo: String = "x😀"
assert_eq("cut inside a 4-byte char drops it (n=3)", utf8_safe_slice(emo, 3), "x")
assert_eq("cut inside a 4-byte char drops it (n=4)", utf8_safe_slice(emo, 4), "x")
assert_eq("cut after a 4-byte char keeps it", utf8_safe_slice(emo, 5), "x😀")
println("")
println("3. the real-world shape that produced the bug")
// A run of box-drawing rules, cut mid-character the exact captured failure.
let rules: String = "──────"
assert_eq_int("six box rules = 18 bytes", str_len(rules), 18)
// n=16 lands one byte into the sixth character.
let cut16: String = utf8_safe_slice(rules, 16)
assert_eq_int("cut at 16 backs off to a clean 15-byte boundary", str_len(cut16), 15)
// Every byte of the result must belong to a complete character: the last byte of a
// well-formed run of these is always 0x80, and 15 is divisible by 3.
assert_eq_int("result ends on a complete char (last byte 0x80)", str_char_code(cut16, 14), 128)
println("")
println("test_utf8_slice.el: " + int_to_str(pass_count) + " passed, " + int_to_str(fail_count) + " failed")
-59
View File
@@ -1,59 +0,0 @@
#!/usr/bin/env bash
# build-soul-from-dist.sh — build a deployable soul from the SAME input CI compiles.
#
# THE PROBLEM THIS CLOSES: until now, deploys were built by build-soul.sh, which
# compiles a scratch amalgam and never touches dist/soul.c. CI compiles dist/soul.c.
# Two lineages. On 2026-08-09 the committed input fell 2,761 bytes behind the sources
# while three binaries built the other way were installed on the operator machine —
# so "what runs" and "what the repo says builds" were different artifacts again,
# which is the whole of #133 and #111 wearing new clothes.
#
# This builds from dist/soul.c with CI's own flags, after asserting that dist/soul.c
# actually matches the .el sources, and writes a provenance sidecar so a deployer can
# refuse anything of unknown origin.
#
# -rdynamic and -DHAVE_CURL are copied from .gitea/workflows/ci.yaml deliberately.
# The CI comment explains -rdynamic: without it the runtime cannot resolve its HTTP
# handler by name via dlsym and the binary serves nothing on every route.
#
# usage: build-soul-from-dist.sh <out-binary>
set -u
OUT="${1:?usage: build-soul-from-dist.sh <out-binary>}"
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
RUNTIME="$ROOT/vendor/el-runtime/v1.0.0-20260501"
cd "$ROOT" || exit 2
echo "[build-from-dist] GATE: does dist/soul.c match the sources?"
if ! ./tools/soulc-stamp.sh --check; then
echo "[build-from-dist] REFUSING — the build input is stale. Regenerate and stamp first." >&2
exit 9
fi
[ -f "$RUNTIME/el_runtime.c" ] || { echo "pinned runtime missing at $RUNTIME" >&2; exit 2; }
echo "[build-from-dist] compiling dist/soul.c with CI's flags"
cc -O2 -DHAVE_CURL -rdynamic \
-I"$RUNTIME" \
dist/soul.c \
"$RUNTIME/el_runtime.c" \
-lcurl -lpthread -lm \
-o "$OUT" || { echo "[build-from-dist] COMPILE FAILED" >&2; exit 3; }
# Provenance sidecar: what a deployer checks before installing anything.
SRC_SHA="$(shasum -a 256 dist/soul.c | awk '{print $1}')"
STAMP_SHA="$(shasum -a 256 dist/soul.c.stamp | awk '{print $1}')"
COMMIT="$(git rev-parse HEAD 2>/dev/null || echo unknown)"
DIRTY="clean"; [ -n "$(git status --porcelain -- '*.el' dist/soul.c 2>/dev/null)" ] && DIRTY="DIRTY"
cat > "$OUT.provenance" <<EOF
{"built_from":"dist/soul.c",
"dist_soul_c_sha256":"$SRC_SHA",
"stamp_sha256":"$STAMP_SHA",
"git_commit":"$COMMIT",
"worktree":"$DIRTY",
"runtime":"vendor/el-runtime/v1.0.0-20260501",
"flags":"-O2 -DHAVE_CURL -rdynamic"}
EOF
echo "[build-from-dist] OK -> $OUT ($(wc -c < "$OUT" | tr -d ' ') bytes)"
echo "[build-from-dist] provenance -> $OUT.provenance (commit ${COMMIT:0:8}, worktree $DIRTY)"
-147
View File
@@ -1,147 +0,0 @@
# Retrieval eval harness
Measures Neuron's memory retrieval so a change can be shown to help before it is
believed to help. Nothing else on the memory roadmap should ship without a run
through this.
```
tools/retrieval-eval/run_comparison.sh --baseline main --candidate <branch>
```
That builds a soul from each ref, boots each in isolation on a fixed corpus,
runs the gold set three times per ref, and prints a table plus a verdict that
refuses to call a difference real if it is inside the noise band.
## What was reused
This is not a new idea, it is the missing third of an existing one.
| Prior work | What it gave | What was missing |
|---|---|---|
| `docs/research/graphrag_eval/` (`collect.py`, `score.py`, 2026-06-08) | The three-retriever comparison that produced the numbers everyone quotes: substring 1.7% P@5, graph 21.7%, BM25 55%. Per-query relevant-id scoring, fixed-denominator precision@5, unique-relevant analysis. | 13 hand-written queries, judged by an LLM after the fact; measured the *live* soul on the *live* engram. |
| `docs/research-archive/p0-prototypes/eval_pinned_40q_20260715.py` | The pinned-query discipline: ground truth committed as regexes so every run judges alike, plus a `--check` winnability gate. 40 queries in 5 bands including a deliberate paraphrase-hard band. | Scored offline replicas of substring/BM25 — it never ran the real retrieval path. |
| `docs/research-archive/p0-prototypes/stage0_eval_20260714.py` | The `hit@5` metric and the substring/BM25 reference implementations. | Same: offline only. |
| `scripts/verify-soul-contract.sh` | The isolation recipe, verbatim: throwaway port, throwaway `HOME`, `SOUL_ENGRAM_PATH`, and the non-obvious `SOUL_ISE_URL` pin that stops an "isolated" soul silently syncing the operator's live brain. | It is a contract gate, not a measurement. |
| `_engine-liveness-91/gen-soul-amalgam.sh` + `.gitea/workflows/ci.yaml` | The build recipe (`elc --target=c` with every `.elh` on the import chain removed) and CI's exact compile flags. | — |
**Reused directly:** the isolation recipe, the build recipe, fixed-denominator
precision@5, the pinned-ground-truth and winnability ideas.
**New here:** ids rather than regexes as ground truth, an associative category
derived from real graph edges, a superseded/contradicted category scored on
ranking, a machine-checked zero-lexical-overlap guarantee on paraphrases,
paired significance testing, and — the point — measurement against the **real
compiled soul** rather than an offline replica of one leg of it.
## Design fit
The thing under measurement is Will's designed retrieval: spreading activation
over the weighted directed graph, four-factor multiplicative scoring (parent
strength x edge weight x target salience x query/target cosine). A Python
re-implementation would measure my reading of the design. So the harness
compiles the actual `soul.el` amalgam and asks it over HTTP on
`/api/neuron/recall`, exactly as the MCP wrapper and the app do.
## Files
| File | Does |
|---|---|
| `build_gold_set.py` | Derives and **validates** the gold set from the corpus. `--check` re-validates and exits non-zero if a query became unwinnable or a paraphrase leaked a word. |
| `gold_set.json` | 38 queries. Every one carries a `derivation` string. |
| `run_eval.py` | Boots one soul in isolation, runs the gold set, writes metrics. Kills and **confirms dead** its child; records the confirmation in the results file. |
| `compare.py` | Paired diff of two result files with McNemar's exact test and a stated noise floor. |
| `build-soul.sh` | Compiles a soul binary from a plain source tree. |
| `run_comparison.sh` | All of the above, end to end, from two git refs. |
## The gold set — 38 queries
Built from the real corpus (`snapshot-pre-repair-20260806.json`, 78,768 nodes /
14,214 edges) so it reflects one person's accumulating memory, not document QA.
| Category | n | Expected answer derived by |
|---|---|---|
| `exact_rare` | 6 | **Mined.** Tokens with document frequency 1 across all 78,768 nodes, whose single containing node is a 3006000 char Memory/Knowledge/Belief. That node is the only possible answer. Re-verified every build. |
| `phrase` | 7 | **Mined.** Case-insensitive verbatim scan; the matching set *is* the answer key. Phrases matching >25 nodes are rejected as too diffuse. |
| `paraphrase` | 13 | **Hand-selected, machine-checked.** Target locked by id; the build then proves that **zero** content words of the query appear anywhere in the target's label, content, or tags. A leak fails the build — the category cannot quietly decay into lexical matching. |
| `associative` | 6 | **Derived from edges.** Query built from one value node's distinctive vocabulary; expected answers are its siblings on the `Self - Values (grounded)` hub. Siblings sharing any query word are dropped, so the only route from query to answer is seed -> hub -> sibling. |
| `nonsense` | 3 | **Control.** Verified that no token occurs anywhere in the corpus. Correct behaviour is to return nothing. |
| `superseded` | 3 | **Derived.** Correction/stale pairs located by regex scan, kept only when both sides resolve to different surviving nodes. Scored on **ranking**: the correction must be returned *and* rank above the stale node. |
## Metrics
`hit@5`, `recall@5`, `recall@10`, `precision@5` (fixed denominator 5, so an
empty result is punished like a page of junk), `MRR@10`, and wall-clock latency
per query (p50/p95/max). Output is a table plus a machine-readable JSON per run
so runs can be diffed.
## Honesty about noise
- **Minimum detectable swing on this 38-query set: 6 queries.** If every query
that changes changes the same way, `p = 2 x 0.5^n`, which first drops under
0.05 at n=6. Any net change smaller than that is inside the noise band and
`compare.py` says so in those words.
- **Run-to-run drift is measured, not assumed.** Activation is a stateful read
by design (traversal reinforces what it touches), so identical inputs need not
give identical outputs. Observed: `main` 0 queries of drift across 3 runs
(fully deterministic); the activation branch 1 query.
- The noise floor used for the verdict is `max(6, observed_drift + 1)`.
- **This gold set is underpowered for small effects.** A genuine 3-query
improvement would not clear the bar. Growing the set is the fix; until then, a
small positive delta means "not shown", not "no effect".
## First result: `main` vs `feat/recall-through-activation`
Corpus and gold set identical, three runs each, fresh corpus copy per run.
| | main | recall-through-activation | delta |
|---|---|---|---|
| hit@5 | 34.3% | 22.9% | **-11.4pp** |
| recall@5 | 26.9% | 19.1% | -7.9pp |
| recall@10 | 33.3% | 24.3% | -9.1pp |
| precision@5 | 12.0% | 7.4% | -4.6pp |
| MRR@10 | 0.294 | 0.242 | -0.053 |
| latency p50 | 1140 ms | 3209 ms | **2.81x** |
| latency p95 | 1584 ms | 4852 ms | 3.06x |
| nonsense clean | 2/3 | 2/3 | — |
| superseded outranks | 1/3 | 0/3 | -1 |
By category (hit@5):
| category | main | activation |
|---|---|---|
| exact_rare | 100% | 100% |
| phrase | 85.7% | **28.6%** |
| paraphrase | 0% | 0% |
| associative | 0% | 0% |
| superseded | 0% | 0% |
**Verdict: directionally worse, one query short of significant.** 5 discordant
pairs, all 5 against the candidate, 0 for it. McNemar exact p = 0.0625 — under
the stated rule that is *inside* the noise band, so the harness reports "no
measurable difference" on accuracy and the honest summary is "5 for 5 the wrong
way, needs a 6th or a larger gold set to call".
Latency is a different story: 2.8x at p50 is deterministic and far outside any
noise band. That regression is real.
The result the branch was written for did not appear. Its stated purpose was to
recover sibling nodes one hub-hop away — the `associative` category — and that
category is **0/6 on both builds**. Probing directly: for the query
`Marines hernia sepsis medical ward`, the activation build returns the lexical
seed node itself at rank 8, and none of its 12 hub siblings anywhere in the top
10. The traversal is running; it is not reaching siblings.
Two corpus facts likely explain it, and both are measurable rather than
speculative:
1. **The graph is nearly edgeless.** Only 4,060 of 78,768 nodes (5.2%) carry any
edge at all — 14,214 edges total, 0.18 per node. Spreading activation over a
graph with no edges is an expensive way to do lexical matching, which is
roughly what the numbers show.
2. **No embeddings.** No node in this snapshot has an embedding field, so the
fourth factor of the four-factor product — query/target cosine similarity —
has nothing to compute from, and the semantic seeding pass is inert.
That is the harness earning its keep on its first job: the change would have
felt like progress (it is the designed mechanism, and it does run) and measures
as a regression on phrase queries plus a 2.8x latency cost, with its intended
benefit unrealised because the corpus lacks the structure it needs.
-21
View File
@@ -1,21 +0,0 @@
import numpy as np, json, urllib.request
SP="/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad"
np.seterr(all='ignore')
M=np.load(SP+'/emb.npy'); eids=open(SP+'/ids.txt',encoding='utf-8',errors='surrogateescape').read().split('\n')
eidx={k:i for i,k in enumerate(eids)}
gold=json.load(open("/Users/timlingo/Development/neuron-technologies/_wt-assoc-leg/tools/retrieval-eval/gold_set.json"))['queries']
VALS=sorted({r for q in gold if q['category']=='paraphrase' for r in q['relevant']})
VI=[eidx[v] for v in VALS]
def emb(t):
b=json.dumps({"model":"nomic-embed-text","prompt":t}).encode()
r=urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=b,headers={"Content-Type":"application/json"})
v=np.array(json.load(urllib.request.urlopen(r,timeout=60))["embedding"],dtype=np.float32)
return v/(np.linalg.norm(v)+1e-9)
print("qid cat bestValueNodeGlobalRank goldGlobalRank goldSiblingRank")
for q in gold:
if q['category']!='paraphrase': continue
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
ranks=sorted(int((s>s[j]).sum())+1 for j in VI)
g=eidx[q['relevant'][0]]; gr=int((s>s[g]).sum())+1
sv=np.array([s[j] for j in VI]); sib=int((sv>s[g]).sum())+1
print("%-4s %-11s best=%-5d (top3 val ranks %s) gold=%-5d sib=%d" % (q['id'],q['category'],ranks[0],ranks[:3],gr,sib))
-44
View File
@@ -1,44 +0,0 @@
#!/usr/bin/env bash
# build-soul.sh — compile a soul binary from a plain source tree (no git needed).
#
# Reuses the amalgam recipe worked out in gen-soul-amalgam.sh (round 9.1) and the
# compile flags from .gitea/workflows/ci.yaml, so the binary under test is the
# same translation unit CI ships — not a re-implementation.
#
# elc --target=c emits only an extern prototype for any module that has a .elh
# header beside it, and inlines the module's bodies when it does not. So the
# amalgam is produced in a scratch copy with every .elh on the import chain
# deleted.
#
# usage: build-soul.sh <src-tree-with-*.el> <out-binary>
set -euo pipefail
SRC="${1:?usage: build-soul.sh <src-tree> <out-binary>}"
OUT="${2:?out-binary}"
ELC="${ELC:-$HOME/neuron-dev-stack/src/el/lang/dist/platform/elc}"
EL_REPO="${EL_REPO:-$HOME/Development/neuron-technologies/el}"
RTDIR="${RTDIR:-$SRC/vendor/el-runtime/v1.0.0-20260501}"
SSL="${SSL_PREFIX:-/opt/homebrew/opt/openssl@3}"
[ -x "$ELC" ] || { echo "no elc at $ELC" >&2; exit 2; }
[ -f "$RTDIR/el_runtime.c" ] || { echo "no el_runtime.c at $RTDIR" >&2; exit 2; }
GEN="$(mktemp -d "${TMPDIR:-/tmp}/soul-build.XXXXXX")"
trap 'rm -rf "$GEN"' EXIT
mkdir -p "$GEN/neuron" "$GEN/foundation/el/elp/src"
cp "$SRC"/*.el "$GEN/neuron/"
cp "$EL_REPO"/elp/src/*.el "$GEN/foundation/el/elp/src/"
find "$GEN" -name '*.elh' -delete
( cd "$GEN/neuron" && "$ELC" --target=c soul.el ) > "$GEN/soul.c"
BODIES=$(grep -c '^el_val_t .*) {$' "$GEN/soul.c" || true)
echo "[build-soul] amalgam $(wc -c < "$GEN/soul.c" | tr -d ' ') bytes, ${BODIES} inlined bodies"
[ "$BODIES" -ge 1200 ] || { echo "[build-soul] FAIL: only $BODIES bodies — an import was not inlined"; exit 1; }
cc -O2 -DHAVE_CURL -rdynamic \
-I"$RTDIR" -I"$SSL/include" -L"$SSL/lib" \
"$GEN/soul.c" "$RTDIR/el_runtime.c" \
-lssl -lcrypto -lcurl -lpthread -lm \
-o "$OUT" 2> "$GEN/cc.log" || { echo "[build-soul] FAIL compile"; tail -40 "$GEN/cc.log"; exit 1; }
if grep -qE 'implicit.*(engram_|el_)' "$GEN/cc.log"; then
echo "[build-soul] FAIL: implicit declarations of runtime symbols"; grep -E 'implicit' "$GEN/cc.log" | head; exit 1; fi
echo "[build-soul] OK -> $OUT ($(wc -c < "$OUT" | tr -d ' ') bytes)"
-506
View File
@@ -1,506 +0,0 @@
#!/usr/bin/env python3
"""
build_gold_set.py derive the retrieval gold set FROM the corpus, and validate it.
WHY THIS FILE EXISTS AS CODE AND NOT AS A HAND-WRITTEN JSON
A gold set nobody can audit is vibes with extra steps. Every expected answer
here is either (a) mined from the corpus by a rule this script re-runs, or
(b) hand-selected with a stated criterion that this script then CHECKS
against the corpus. Both leave a `derivation` string on every query, and the
checks are re-run on demand so the set cannot silently rot as the corpus
changes.
Lineage: this extends the pinned-query approach from
docs/research-archive/p0-prototypes/eval_pinned_40q_20260715.py (pinned
ground-truth patterns + a --check "winnability" gate) and the per-query
relevant-id scoring from docs/research/graphrag_eval/score.py. What is new:
ids as ground truth rather than regexes alone, an ASSOCIATIVE category
derived from real graph edges, a superseded/contradicted category, and a
machine-checked no-lexical-overlap guarantee on the paraphrase category.
THE SIX CATEGORIES, AND WHAT EACH ONE IS FOR
exact_rare a single rare word. Substring matching already wins these.
They are a REGRESSION GUARD: any change that loses them is
disqualified regardless of what else it gains.
phrase a multi-word string that exists verbatim in the corpus.
Guards multi-token queries, which the old substring matcher
handled by returning nothing.
paraphrase same meaning, ZERO shared content words with the target node.
THE CATEGORY THAT MATTERS. Mechanically unreachable by string
matching; reachable only by semantics or by association.
associative the answer is one hub-hop from an obvious starting point and
shares no words with the query. This is the case the graph is
supposed to buy: query one value, get its siblings.
nonsense must return nothing. Guards against a retriever that "improves"
recall by returning the whole graph.
superseded a fact that was later corrected. The correction must OUTRANK
the stale version ranking, not mere presence.
usage:
python3 build_gold_set.py <snapshot.json> [--out gold_set.json] [--check]
--check re-validates an existing gold_set.json against the corpus and exits
non-zero if any query became unwinnable or any paraphrase leaked a word.
"""
import argparse
import json
import os
import re
import sys
from collections import Counter, defaultdict
HERE = os.path.dirname(os.path.abspath(__file__))
DEFAULT_OUT = os.path.join(HERE, "gold_set.json")
TOKEN = re.compile(r"[a-z0-9][a-z0-9\-']*")
# Stopwords are deliberately generous. A paraphrase query is only interesting if
# its CONTENT words are absent from the target; "the", "is", "what" appearing in
# both proves nothing. Being generous here makes the overlap test STRICTER on
# the words that carry meaning, which is the conservative direction.
STOP = set("""
a about above after again against all also am an and any are aren't as at be because been
before being below between both but by can can't cannot could couldn't did didn't do does
doesn't doing don't down during each few for from further had hadn't has hasn't have haven't
having he her here hers herself him himself his how i if in into is isn't it its itself just
me more most my myself no nor not of off on once only or other others ought our ours ourselves
out over own same shan't she should shouldn't so some such than that the their theirs them
themselves then there these they this those through to too under until up very was wasn't we
were weren't what when where which while who whom why will with won't would wouldn't you your
yours yourself yourselves get gets got make makes made take takes use uses used way ways thing
things does doing done keep keeps kept go goes going come comes came one two something anything
""".split())
# ─────────────────────────────────────────────────────────────────────────────
# corpus helpers
# ─────────────────────────────────────────────────────────────────────────────
def load_corpus(path):
with open(path, encoding="utf-8", errors="replace") as fh:
data = json.load(fh)
nodes = [n for n in data.get("nodes", []) if isinstance(n, dict) and n.get("id")]
edges = [e for e in data.get("edges", []) if isinstance(e, dict)]
return nodes, edges
def doctext(n):
return " ".join([str(n.get("label") or ""), str(n.get("content") or ""), str(n.get("tags") or "")])
def content_tokens(s):
return {t for t in TOKEN.findall(s.lower()) if t not in STOP and len(t) > 2}
# ─────────────────────────────────────────────────────────────────────────────
# hand-authored queries. Every entry states HOW its expected answer was chosen.
# The `check` field names the validation this script runs against the corpus.
# ─────────────────────────────────────────────────────────────────────────────
# EXACT_RARE — mined, not chosen. The rule (re-run by mine_exact_rare below):
# tokens whose document frequency across the whole corpus is 1, whose single
# containing node is a Memory/Knowledge/Belief with 300-6000 chars of content
# (so the answer is a real memory, not a 117KB whitepaper that contains every
# word in English), and whose token is plain lowercase alphabetic. The expected
# answer is that one node — it is the only node that can possibly be correct.
EXACT_RARE_SEEDS = [
"unjailbreakable",
"engram-migrate",
"cartabandonedevent",
"pre-apprenticeship",
"inferencenodemanager",
"clear-eyed",
]
# PHRASE — chosen by reading the corpus for phrases that (a) occur verbatim,
# (b) occur in a small enough set of nodes that "relevant" is well defined.
# Expected answers are computed here as EVERY node whose text contains the
# phrase case-insensitively — so the answer set is a fact about the corpus, not
# an opinion. Queries whose phrase matches more than PHRASE_MAX nodes are
# rejected by validation as too diffuse to score.
PHRASE_MAX = 25
PHRASE_SEEDS = [
("patterns not returns",
"a verbatim correction Will issued; expected = every node containing the phrase"),
("thirty moves",
"the canonical biographical phrase; expected = every node containing it"),
("Grandma Lucas",
"a named person appearing verbatim in the biography/value nodes"),
("Directed Harmonic",
"the canonical DHARMA expansion, confirmed by Will April 24 2026"),
("Sarah Bishop",
"a named person; rare enough that the answer set is unambiguous"),
("Directed Autonomous Runtime Modification",
"the DARMA expansion, quoted verbatim in the backlog item and its correction"),
("zero-knowledge encrypted backup",
"the paid-tier feature name as written in the roadmap nodes"),
]
# PARAPHRASE — hand-authored. THE SELECTION CRITERION, stated once and applied
# to all nine: pick a node whose SUBJECT is unmistakable to a reader, then write
# the query a person would actually type when they remember the subject but not
# the words. The target is then LOCKED by id, and this script enforces the hard
# property that makes the category meaningful: not one content word of the query
# appears anywhere in the target node's label, content, or tags. If a word
# leaks, validation fails and the query must be rewritten — the set cannot
# quietly degrade into a lexical query wearing a paraphrase costume.
PARAPHRASE_SEEDS = [
("kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"the elderly relative who passed while he stayed away",
"target: 'Value - Do the Essential Thing While You Can', whose subject is Grandma Lucas "
"dying in Feb 2006 without Will saying goodbye. Query names the event with none of the "
"node's own vocabulary."),
("kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"a soldier sidelined by illness who refused to quit",
"target: 'Value - Survival Is Not an Excuse to Stop', whose subject is enlisting in the "
"Marines, a severe hernia, and sepsis. Query describes the episode obliquely."),
("kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"choosing an uncomfortable fact over a pleasant fiction",
"target: 'Value - Honesty Before Comfort'. Query states the principle in wholly "
"different words."),
("kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"a tight payload beats a bloated one",
"target: 'Value - Precision Over Brute Force'. Query restates the claim with no "
"shared vocabulary."),
("kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"if you are able and nobody is coming the job is yours",
"target: 'Value - Capability Is a Debt You Owe the Moment'. Query states the "
"obligation without the node's terms."),
("kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"learning is the wealth creditors cannot seize",
"target: 'Value - Knowledge Survives When Nothing Else Does', whose subject is the "
"library following Will across 30+ moves."),
("kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"reliability proven by track record not assertion",
"target: 'Value - Earned Trust' ('Trust is demonstrated, not declared')."),
("kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"boundaries that enable instead of confine",
"target: 'Value - Constraints as Freedom'. Query is a restatement of the same claim."),
("kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"what shifts tells you where to cut a system apart",
"target: 'Value - Change Is the Signal', the value VBD is built on."),
("kn-f230b362-b201-4402-9833-4160c89ab3d4",
"a mind that compounds instead of resetting each day",
"target: 'Value - The System Must Accumulate'. Query is the accumulation claim in "
"different vocabulary."),
("kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"loved for the unedited self and not the polished exterior",
"target: 'Value - Being Seen Is Rarer Than Being Known', whose subject is Sarah Bishop "
"as the first person Will did not perform for."),
("kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"cheerfulness you arrive at instead of assuming",
"target: 'Value - Hope Is a Conclusion'. Query restates 'a conclusion, not a premise'."),
("kn-6061318f-046b-4935-907d-8eafdce14930",
"a childhood offering no solid foundation to inherit",
"target: 'Value - Structure Is Not Inherited', whose subject is thirty moves between "
"two parents' collapses."),
]
# ASSOCIATIVE — derived from real edges, not authored. The construction:
# every value node hangs off the 'Self - Values (grounded)' hub by an `identity`
# edge. For a chosen value node V, the query is built from V's own distinctive
# vocabulary; the expected answers are V's SIBLINGS on that hub. A sibling
# shares no query words with the query by construction (validated below), so the
# only path from the query to a sibling is: lexical seed on V -> hub -> sibling.
# That is a two-hop traversal and nothing else can produce it.
VALUES_HUB = "kn-5b606390-a52d-4ca2-8e0e-eba141d13440"
ASSOCIATIVE_SEEDS = [
("kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71", "Grandma Lucas stroke February 2006 goodbye window"),
("kn-58874a74-b96f-4883-9e08-45707f4bd3ee", "Marines hernia sepsis medical ward"),
("kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e", "Sarah Bishop Dyer trailer performance"),
("kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83", "Swarm Architecture containment lateral worker"),
("kn-e0423482-cfa5-4796-8689-8495c93b66bc", "hope won inside the narrative preface"),
("kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8", "man of the house six years old expectation"),
]
# NONSENSE — must return nothing. Strings chosen to be lexically impossible:
# validation asserts each appears in ZERO corpus nodes as a substring and that
# none of its tokens appears anywhere either (so not even a partial seed exists).
NONSENSE_SEEDS = [
"zqxjvw plimforth grebulon",
"flarnbistle quommetry",
"xxqzzt vurblenacht throom",
]
# SUPERSEDED — a fact that was corrected. Chosen by searching the corpus for
# explicit correction language and keeping pairs where BOTH the stale statement
# and its correction exist as separate nodes. Scored on RANKING: the correction
# must appear, and must appear above the stale node. Ids are locked here and
# validated to exist and to match their stated role.
SUPERSEDED_SEEDS = [
# (query, correct_id, stale_id, derivation)
]
# ─────────────────────────────────────────────────────────────────────────────
# mining
# ─────────────────────────────────────────────────────────────────────────────
def mine_exact_rare(nodes, byid, seeds):
"""Re-derive: confirm each seed token still has df==1 and name its node."""
tok = re.compile(r"[A-Za-z][A-Za-z0-9\-]{4,}")
want = set(seeds)
df = Counter()
post = defaultdict(set)
for n in nodes:
for t in {w.lower() for w in tok.findall(doctext(n))}:
if t in want:
df[t] += 1
post[t].add(n["id"])
out = []
for s in seeds:
ids = sorted(post.get(s, ()))
out.append((s, ids, df.get(s, 0)))
return out
def phrase_matches(nodes, phrase):
p = phrase.lower()
return sorted(n["id"] for n in nodes if p in doctext(n).lower())
def hub_siblings(edges, hub, relation="identity"):
sibs = []
for e in edges:
if e.get("from_id") == hub and e.get("relation") == relation:
sibs.append(e["to_id"])
elif e.get("to_id") == hub and e.get("relation") == relation:
sibs.append(e["from_id"])
return list(dict.fromkeys(sibs))
def find_superseded_pairs(nodes, byid):
"""Locked pairs, each verified here to exist and to carry its stated marker.
Chosen by scanning the corpus for explicit correction language
(CORRECTION/SUPERSEDES/re-corrected/no longer/RECONCILED) and keeping only
cases where the STALE claim also survives as its own node a supersession
with nothing to outrank is not a ranking test.
"""
pairs = []
txt = {n["id"]: doctext(n) for n in nodes}
def find_one(pattern, exclude=()):
rx = re.compile(pattern)
return [n["id"] for n in nodes
if n["id"] not in exclude
and rx.search(txt[n["id"]])
and 150 < len(str(n.get("content") or "")) < 12000
and n.get("node_type") in ("Memory", "Knowledge", "Belief", "BacklogItem")]
# Each entry: (query, correction-pattern, stale-pattern, why).
# The stale side is searched with the correction hits EXCLUDED, because most
# correction memories quote the claim they are killing — without the
# exclusion the "stale" node resolves to the correction itself and the pair
# collapses into a no-op. A pair is only emitted if both sides resolve to
# DIFFERENT surviving nodes; otherwise it is dropped and reported.
SPECS = [
("is the self-improvement architecture called DARMA or DHARMA",
r'(?i)CORRECTION:.{0,90}DHARMA .{0,12}not DARMA',
r'(?i)\bDARMA\b',
"correction node is Will's confirmation that the H is intentional (DHARMA, not DARMA); "
"the stale node is the surviving backlog item still titled 'Implement DARMA'."),
("how many provisional patents does Will actually have",
r'(?i)EXACTLY 6 (fully-specced )?provisional',
r'(?i)(MY ARCHITECTURE = 12 filed patents|\b12 filed patents\b)',
"correction node is the 2026-06-17 confabulation flag establishing EXACTLY 6 provisionals; "
"the stale node is the surviving memory that asserts 12 filed patents."),
("is MCP still the live integration layer",
r'(?i)MCP RETIRED',
r'(?i)MCP server live at',
"correction node is the 'CGI ARCHITECTURE - THREE LAYERS, MCP RETIRED' decision of "
"April 30 2026; the stale node still records the MCP server as live."),
("what does the patterns-not-returns directive mean",
r'(?i)CORRECTION:.{0,80}patterns not returns',
r'(?i)established returns',
"correction node is Will's 'patterns not returns' correction; the stale node is a "
"surviving node carrying the misread 'established returns' directive."),
("was the earlier identity-bug finding correct",
r'(?i)SUPERSEDES the earlier .critical identity bug',
r'(?i)critical identity bug',
"correction node explicitly supersedes the 'critical identity bug' finding; the stale "
"node is the surviving original finding."),
("does Neuron have recursive self-improvement",
r'(?i)twice answered .Neuron has no recursive self-improvement',
r'(?i)no recursive self-improvement',
"correction node records the June-29 finding that the CGI provisional IS the "
"recursive-self-improvement mechanism; the stale node is the surviving denial."),
]
for query, cpat, spat, why in SPECS:
corr = find_one(cpat)
if not corr:
continue
stale = find_one(spat, exclude=set(corr))
if not stale:
continue
pairs.append((query, corr[0], stale[0], why))
return pairs
# ─────────────────────────────────────────────────────────────────────────────
# build
# ─────────────────────────────────────────────────────────────────────────────
def build(nodes, edges):
byid = {n["id"]: n for n in nodes}
tokset = {n["id"]: content_tokens(doctext(n)) for n in nodes}
queries = []
problems = []
qn = [0]
def add(cat, query, relevant, derivation, **extra):
qn[0] += 1
q = {
"id": f"q{qn[0]:02d}",
"category": cat,
"query": query,
"relevant": sorted(relevant),
"derivation": derivation,
}
q.update(extra)
queries.append(q)
return q
# --- exact_rare ---------------------------------------------------------
for tokname, ids, df in mine_exact_rare(nodes, byid, EXACT_RARE_SEEDS):
if df != 1 or len(ids) != 1:
problems.append(f"exact_rare '{tokname}': df={df}, ids={len(ids)} (expected df=1)")
continue
lab = (byid[ids[0]].get("label") or "")[:60]
add("exact_rare", tokname, ids,
f"MINED: token '{tokname}' has document frequency 1 over all {len(nodes)} corpus nodes "
f"(re-verified at build time). Its single containing node is {ids[0]} "
f"('{lab}'), which is therefore the only possible correct answer.")
# --- phrase -------------------------------------------------------------
for phrase, why in PHRASE_SEEDS:
ids = phrase_matches(nodes, phrase)
if not ids:
problems.append(f"phrase '{phrase}': 0 corpus matches — unwinnable")
continue
if len(ids) > PHRASE_MAX:
problems.append(f"phrase '{phrase}': {len(ids)} matches > {PHRASE_MAX} — too diffuse")
continue
add("phrase", phrase, ids,
f"MINED: {why}. Case-insensitive verbatim substring scan over label+content+tags at "
f"build time returns exactly {len(ids)} node(s); that set IS the answer key.")
# --- paraphrase ---------------------------------------------------------
for target, query, why in PARAPHRASE_SEEDS:
if target not in byid:
problems.append(f"paraphrase target {target} not in corpus")
continue
qt = content_tokens(query)
leak = sorted(qt & tokset[target])
if leak:
problems.append(f"paraphrase '{query}': leaks {leak} into target {target}")
continue
add("paraphrase", query, [target],
f"HAND-SELECTED with criterion: {why} VERIFIED at build time: of the {len(qt)} content "
f"words in the query, ZERO appear anywhere in the target's label, content, or tags — so "
f"no string-matching retriever can reach this answer.",
zero_overlap_verified=True, query_content_words=sorted(qt))
# --- associative --------------------------------------------------------
sibs = hub_siblings(edges, VALUES_HUB)
if len(sibs) < 5:
problems.append(f"associative: values hub {VALUES_HUB} has only {len(sibs)} siblings")
for src, query in ASSOCIATIVE_SEEDS:
if src not in byid or src not in sibs:
problems.append(f"associative source {src} not a sibling on {VALUES_HUB}")
continue
qt = content_tokens(query)
others = [s for s in sibs if s != src and s in byid]
# A sibling only counts as a legitimate expected answer if the query
# cannot reach it lexically. Drop any sibling that shares a content word.
clean = [s for s in others if not (qt & tokset[s])]
dropped = len(others) - len(clean)
if len(clean) < 5:
problems.append(f"associative '{query}': only {len(clean)} lexically-unreachable siblings")
continue
add("associative", query, clean,
f"DERIVED FROM EDGES: the query is built from the distinctive vocabulary of {src} "
f"('{(byid[src].get('label') or '')[:48]}'), which hangs off the values hub {VALUES_HUB} "
f"by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub "
f"({len(clean)} of {len(others)}; {dropped} dropped because they shared a query word and "
f"so were lexically reachable). Every remaining sibling shares ZERO content words with "
f"the query — the only route from query to answer is seed({src}) -> hub -> sibling, a "
f"two-hop traversal.",
associative_source=src, hub=VALUES_HUB, siblings_dropped_for_overlap=dropped)
# --- nonsense -----------------------------------------------------------
all_tokens = set()
for n in nodes:
all_tokens |= {t for t in TOKEN.findall(doctext(n).lower())}
for s in NONSENSE_SEEDS:
present = sorted(t for t in TOKEN.findall(s.lower()) if t in all_tokens)
if present:
problems.append(f"nonsense '{s}': tokens {present} DO occur in corpus")
continue
add("nonsense", s, [],
f"CONTROL: verified at build time that none of this string's tokens occurs anywhere in "
f"the corpus. Correct behaviour is to return NOTHING; any result is a false positive.",
expect_empty=True)
# --- superseded ---------------------------------------------------------
for query, correct, stale, why in find_superseded_pairs(nodes, byid):
if correct not in byid or stale not in byid:
problems.append(f"superseded '{query}': id missing from corpus")
continue
add("superseded", query, [correct],
f"DERIVED: {why} Scored on RANKING, not presence: the corrected node {correct} must be "
f"returned AND must rank above the stale node {stale}.",
must_outrank=[correct, stale],
stale_id=stale,
correct_label=(byid[correct].get("label") or "")[:70],
stale_label=(byid[stale].get("label") or "")[:70])
return queries, problems
def summarize(queries):
c = Counter(q["category"] for q in queries)
return ", ".join(f"{k}={c[k]}" for k in
("exact_rare", "phrase", "paraphrase", "associative", "nonsense", "superseded")
if c[k])
def main():
ap = argparse.ArgumentParser()
ap.add_argument("snapshot")
ap.add_argument("--out", default=DEFAULT_OUT)
ap.add_argument("--check", action="store_true",
help="validate only; do not write. Non-zero exit if anything is unwinnable.")
args = ap.parse_args()
nodes, edges = load_corpus(args.snapshot)
print(f"corpus: {len(nodes)} nodes, {len(edges)} edges ({os.path.basename(args.snapshot)})")
queries, problems = build(nodes, edges)
print(f"gold set: {len(queries)} queries [{summarize(queries)}]")
if problems:
print(f"\n{len(problems)} PROBLEM(S) — these queries were REJECTED, not silently kept:")
for p in problems:
print(" -", p)
if args.check:
sys.exit(1 if problems else 0)
doc = {
"corpus": os.path.abspath(args.snapshot),
"corpus_nodes": len(nodes),
"corpus_edges": len(edges),
"note": ("Every query carries a `derivation` recording how its expected answer was chosen. "
"Re-run with --check to re-validate the whole set against the corpus."),
"queries": queries,
}
with open(args.out, "w", encoding="utf-8") as fh:
json.dump(doc, fh, indent=1, ensure_ascii=False)
print(f"\nwrote {args.out}")
if __name__ == "__main__":
main()
-43
View File
@@ -1,43 +0,0 @@
import json,sys,pickle,numpy as np,itertools
sys.path.insert(0,'.')
from policy2 import legs3,outcome,G,NODES,merge
# cache per-query leg id-lists, floored and unfloored
cache={}
for q in G['queries']:
Lf,Sf,Af=legs3(q['query'])
Lu,Su,Au=legs3(q['query'],unfloor=True)
cache[q['id']]=dict(L=Lf,Sf=Sf,A=Af,Su=Su,Au=Au)
pickle.dump(cache,open('ceil.pkl','wb'))
def mrg(pattern,L,S,A,lim=10):
out=[];p={'L':0,'S':0,'A':0};src={'L':L,'S':S,'A':A}
i=0
while len(out)<lim:
prog=False
for ch in pattern:
lst=src[ch]
if p[ch]<len(lst):
x=lst[p[ch]];p[ch]+=1;prog=True
if x not in out: out.append(x)
if len(out)>=lim: return out
if not prog: break
return out
def ev(pattern,unfl):
res={}
for q in G['queries']:
c=cache[q['id']]
S=c['Su'] if unfl else c['Sf']
ids=[NODES[i]['id'] for i in mrg(pattern,c['L'],S,c['A'],10)]
res[q['id']]=outcome(q,ids)
return res
base=ev('LSA',False)
print("baseline",sum(base.values()))
best=[]
pats=['LSA','LAS','SLA','ALS','SAL','ASL','LSSA','LSASA','LSAA','LSSAA','LSAS','SSLA','LLSA','SALSA','LSAAS']
for unfl in (False,True):
for p in pats:
r=ev(p,unfl)
g=sorted(k for k in base if r[k] and not base[k]);l=sorted(k for k in base if base[k] and not r[k])
best.append((len(g)-len(l),p,unfl,g,l))
best.sort(reverse=True)
for n,p,u,g,l in best[:10]:
print("net=%+d pat=%-6s unfloor=%s gains=%s losses=%s"%(n,p,u,g,l))
-146
View File
@@ -1,146 +0,0 @@
{
"baseline": "bm25lex",
"candidate": "wsclaim24",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q14",
"q25"
],
"broken_by_candidate": [
"q15",
"q28",
"q33",
"q34"
],
"discordant": 6,
"net_queries": -2,
"mcnemar_exact_p": 0.6875,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1184.4,
"latency_ms_p95": 1620.0,
"latency_ms_max": 1655.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5768475572047,
"recall@10": 0.6563414759843332,
"precision@5": 0.19428571428571437,
"mrr@10": 0.5026530612244898,
"nonsense_clean": "0/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 524.8,
"latency_ms_p95": 738.7,
"latency_ms_max": 755.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 0,
"avg_false_positives": 10.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
}
}
}
-149
View File
@@ -1,149 +0,0 @@
{
"baseline": "unfloor-clean",
"candidate": "splitfix",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q15",
"q28",
"q60"
],
"broken_by_candidate": [],
"discordant": 3,
"net_queries": 3,
"mcnemar_exact_p": 0.25,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5846153846153846,
"recall@5": 0.48102442429365505,
"recall@10": 0.5692415490492414,
"precision@5": 0.14153846153846153,
"mrr@10": 0.3351709401709402,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 646.9,
"latency_ms_p95": 1028.5,
"latency_ms_max": 1197.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.15967365967365968,
"mrr@10": 0.24166666666666667
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.43333333333333335,
"mrr@10": 0.12120370370370372
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.7692307692307693,
"recall@5": 0.7692307692307693,
"recall@10": 0.8461538461538461,
"mrr@10": 0.33269230769230773
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5775226757369615,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
-210
View File
@@ -1,210 +0,0 @@
#!/usr/bin/env python3
"""
compare.py diff two run_eval.py result files, WITH a noise threshold.
WHY THE STATISTICS ARE NOT OPTIONAL
With ~35 scored queries, one query is ~2.9 percentage points. A harness that
reports "hit@5 improved 2.9%" without saying that is one query is a harness
that will approve noise. So this file refuses to call anything an
improvement on the strength of the headline number alone. It reports:
1. The DISCORDANT PAIRS. Two configurations scored on the same queries are
paired data, so the only queries carrying information are the ones
where they disagree: b = fixed by B, c = broken by B. Queries both got
right, or both got wrong, tell you nothing about which is better.
2. McNEMAR'S EXACT TEST on (b, c). Under the null "the change is a coin
flip", the discordant outcomes are Binomial(b+c, 0.5). The two-sided
exact p-value is computed here with no scipy dependency.
3. The MINIMUM DETECTABLE SWING for this gold set: the smallest number of
net-changed queries that would reach p < 0.05 if every discordant pair
fell the same way. Anything smaller is inside the noise band, and the
verdict line says so in those words.
Repeat-run variance is the other half of honesty. Spreading activation is a
stateful read (it reinforces what it touches), so identical inputs need not
give identical outputs. Pass --repeats to fold several runs of the same
config into an observed variance band; a delta inside that band is not real
either, however good its p-value looks.
usage:
python3 compare.py --baseline results-main.json --candidate results-act.json
python3 compare.py --baseline a.json --candidate b.json \
--repeats-baseline a2.json a3.json --repeats-candidate b2.json b3.json
"""
import argparse
import json
from math import comb
def binom_two_sided(b, c):
"""Two-sided exact binomial p for b successes in n=b+c at p=0.5."""
n = b + c
if n == 0:
return 1.0
k = min(b, c)
tail = sum(comb(n, i) for i in range(0, k + 1)) / (2 ** n)
return min(1.0, 2 * tail)
def min_detectable_swing(n_scored, alpha=0.05):
"""Smallest all-one-way discordant count reaching p < alpha.
If every query that changes changes in the same direction, the p-value is
2 * 0.5**n. Solve for the smallest n where that drops under alpha. This is
the FLOOR: any real change will have some discordance both ways, so the true
requirement is larger. Reporting the floor is the conservative move it is
the most generous threshold we would ever accept.
"""
n = 1
while n <= n_scored:
if 2 * (0.5 ** n) < alpha:
return n
n += 1
return n_scored
def load(path):
with open(path, encoding="utf-8") as fh:
return json.load(fh)
def row_map(doc):
return {r["id"]: r for r in doc["rows"]}
def outcome(r):
"""Binary per-query outcome used for the paired test.
hit@5 for scored queries; 'returned nothing' for the nonsense controls;
'correction outranks the stale node' for the superseded queries. One number
per query, so every query votes exactly once.
"""
if "clean" in r:
return 1.0 if r["clean"] else 0.0
if "outranks" in r:
return 1.0 if r["outranks"] else 0.0
return r.get("hit@5") or 0.0
def band(values):
return (min(values), max(values))
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--baseline", required=True)
ap.add_argument("--candidate", required=True)
ap.add_argument("--repeats-baseline", nargs="*", default=[])
ap.add_argument("--repeats-candidate", nargs="*", default=[])
ap.add_argument("--out", default=None)
args = ap.parse_args()
A, B = load(args.baseline), load(args.candidate)
ra, rb = row_map(A), row_map(B)
ids = [q for q in ra if q in rb]
n = len(ids)
aa, ab = A["aggregate"], B["aggregate"]
print(f"baseline {A['label']:14} soul={A['soul_md5'][:12]} {n} shared queries")
print(f"candidate {B['label']:14} soul={B['soul_md5'][:12]}")
print(f"corpus {A['corpus_nodes']} nodes / {A['corpus_edges']} edges "
f"(identical copy for both runs)\n")
metrics = [("hit@5", 1), ("recall@5", 1), ("recall@10", 1),
("precision@5", 1), ("mrr@10", 0)]
print(f" {'metric':14} {'baseline':>10} {'candidate':>10} {'delta':>10}")
for m, as_pct in metrics:
x, y = aa[m], ab[m]
if as_pct:
print(f" {m:14} {100*x:>9.1f}% {100*y:>9.1f}% {100*(y-x):>+9.1f}pp")
else:
print(f" {m:14} {x:>10.3f} {y:>10.3f} {y-x:>+10.3f}")
for m in ("latency_ms_p50", "latency_ms_p95"):
x, y = aa[m], ab[m]
ratio = f"{y/x:.2f}x" if x else "n/a"
print(f" {m:14} {x:>9.0f}ms {y:>9.0f}ms {ratio:>10}")
print(f" {'nonsense':14} {aa['nonsense_clean']:>10} {ab['nonsense_clean']:>10}")
print(f" {'outranks':14} {aa['superseded_outranks']:>10} {ab['superseded_outranks']:>10}")
print(f"\n {'category':14} {'n':>3} {'base hit@5':>11} {'cand hit@5':>11} {'delta':>9}")
for c in sorted(set(aa["by_category"]) & set(ab["by_category"])):
ea, eb = aa["by_category"][c], ab["by_category"][c]
if c == "nonsense":
print(f" {c:14} {ea['n']:>3} {'clean ' + str(ea['clean']):>11} "
f"{'clean ' + str(eb['clean']):>11}")
else:
print(f" {c:14} {ea['n']:>3} {100*ea['hit@5']:>10.1f}% {100*eb['hit@5']:>10.1f}% "
f"{100*(eb['hit@5']-ea['hit@5']):>+8.1f}pp")
# ---- paired significance -------------------------------------------------
fixed, broken = [], []
for q in ids:
oa, ob = outcome(ra[q]), outcome(rb[q])
if ob > oa:
fixed.append(q)
elif ob < oa:
broken.append(q)
b, c = len(fixed), len(broken)
p = binom_two_sided(b, c)
mds = min_detectable_swing(n)
print(f"\n== paired comparison over {n} queries ==")
print(f" fixed by candidate : {b} {[ra[q]['category'] + ':' + q for q in fixed]}")
print(f" broken by candidate: {c} {[ra[q]['category'] + ':' + q for q in broken]}")
print(f" discordant pairs : {b + c} net {b - c:+d} queries")
print(f" McNemar exact p : {p:.4f}")
print(f" noise threshold : a difference needs at least {mds} queries moving the "
f"same way to clear p<0.05 on this {n}-query set")
# ---- repeat-run variance -------------------------------------------------
var = {}
for name, paths, first in (("baseline", args.repeats_baseline, A),
("candidate", args.repeats_candidate, B)):
docs = [first] + [load(p) for p in paths]
if len(docs) > 1:
hits = [d["aggregate"]["hit@5"] for d in docs]
lo, hi = band(hits)
spread_q = round((hi - lo) * first["aggregate"]["n_scored"])
var[name] = {"runs": len(docs), "hit@5_min": lo, "hit@5_max": hi,
"spread_queries": spread_q}
print(f" {name} repeat runs ({len(docs)}): hit@5 {100*lo:.1f}%..{100*hi:.1f}% "
f"= {spread_q} query of run-to-run drift")
drift = max([v["spread_queries"] for v in var.values()], default=0)
floor = max(mds, drift + 1)
print("\n== VERDICT ==")
net = b - c
if abs(net) < floor:
print(f" NO MEASURABLE DIFFERENCE. Net {net:+d} queries is inside the noise band "
f"(needs |net| >= {floor}: {mds} for significance, {drift} observed run-to-run drift).")
elif net > 0:
print(f" CANDIDATE BETTER by {net} queries (p={p:.4f}), outside the noise band "
f"(>= {floor}).")
else:
print(f" CANDIDATE WORSE by {abs(net)} queries (p={p:.4f}), outside the noise band "
f"(>= {floor}).")
if args.out:
with open(args.out, "w", encoding="utf-8") as fh:
json.dump({
"baseline": A["label"], "candidate": B["label"],
"n_shared_queries": n,
"fixed_by_candidate": fixed, "broken_by_candidate": broken,
"discordant": b + c, "net_queries": net,
"mcnemar_exact_p": p,
"min_detectable_swing_queries": mds,
"observed_run_to_run_drift_queries": drift,
"noise_floor_queries": floor,
"verdict": ("no measurable difference" if abs(net) < floor
else ("candidate better" if net > 0 else "candidate worse")),
"baseline_aggregate": aa, "candidate_aggregate": ab,
"repeat_variance": var,
}, fh, indent=1)
print(f"\nwrote {args.out}")
if __name__ == "__main__":
main()
@@ -1,150 +0,0 @@
{
"baseline": "assoc-leg",
"candidate": "semseed",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q18",
"q19",
"q22"
],
"broken_by_candidate": [
"q11"
],
"discordant": 4,
"net_queries": 2,
"mcnemar_exact_p": 0.625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6285714285714286,
"recall@5": 0.45309194773480493,
"recall@10": 0.5405733155733157,
"precision@5": 0.17714285714285719,
"mrr@10": 0.42650793650793645,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1228.5,
"latency_ms_p95": 1681.8,
"latency_ms_max": 1718.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657342,
"recall@10": 0.24825174825174826,
"mrr@10": 0.22777777777777777
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4882369614512472,
"recall@10": 0.6329365079365079,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6857142857142857,
"recall@5": 0.5213459159887731,
"recall@10": 0.6027048348476919,
"precision@5": 0.18285714285714294,
"mrr@10": 0.4608730158730158,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1227.1,
"latency_ms_p95": 1692.6,
"latency_ms_max": 1710.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657344,
"recall@10": 0.24825174825174823,
"mrr@10": 0.20833333333333334
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 0.7142857142857143,
"recall@5": 0.40093537414965985,
"recall@10": 0.5150226757369615,
"mrr@10": 0.6507936507936508
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.6285714285714286,
"hit@5_max": 0.6285714285714286,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.6857142857142857,
"hit@5_max": 0.6857142857142857,
"spread_queries": 0
}
}
}
@@ -1,146 +0,0 @@
{
"baseline": "bm25lex",
"candidate": "wordstart",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q35"
],
"broken_by_candidate": [],
"discordant": 1,
"net_queries": 1,
"mcnemar_exact_p": 1.0,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1184.4,
"latency_ms_p95": 1620.0,
"latency_ms_max": 1655.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "3/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 542.6,
"latency_ms_p95": 741.3,
"latency_ms_max": 758.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 3,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
}
}
}
@@ -1,143 +0,0 @@
{
"baseline": "hybrid-semantic",
"candidate": "assoc-leg",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q27",
"q28",
"q29",
"q31"
],
"broken_by_candidate": [],
"discordant": 4,
"net_queries": 4,
"mcnemar_exact_p": 0.125,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.5142857142857142,
"recall@5": 0.4409013605442177,
"recall@10": 0.5047619047619047,
"precision@5": 0.15428571428571433,
"mrr@10": 0.38746031746031745,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1219.7,
"latency_ms_p95": 1667.1,
"latency_ms_max": 1720.2,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6285714285714286,
"recall@5": 0.45309194773480493,
"recall@10": 0.5405733155733157,
"precision@5": 0.17714285714285719,
"mrr@10": 0.42650793650793645,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1228.5,
"latency_ms_p95": 1681.8,
"latency_ms_max": 1718.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657342,
"recall@10": 0.24825174825174826,
"mrr@10": 0.22777777777777777
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4882369614512472,
"recall@10": 0.6329365079365079,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.5142857142857142,
"hit@5_max": 0.5142857142857142,
"spread_queries": 0
}
}
}
@@ -1,154 +0,0 @@
{
"baseline": "hybrid-semantic",
"candidate": "semseed",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q18",
"q19",
"q22",
"q27",
"q28",
"q29",
"q31"
],
"broken_by_candidate": [
"q11"
],
"discordant": 8,
"net_queries": 6,
"mcnemar_exact_p": 0.0703125,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "candidate better",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.5142857142857142,
"recall@5": 0.4409013605442177,
"recall@10": 0.5047619047619047,
"precision@5": 0.15428571428571433,
"mrr@10": 0.38746031746031745,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1219.7,
"latency_ms_p95": 1667.1,
"latency_ms_max": 1720.2,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6857142857142857,
"recall@5": 0.5213459159887731,
"recall@10": 0.6027048348476919,
"precision@5": 0.18285714285714294,
"mrr@10": 0.4608730158730158,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1227.1,
"latency_ms_p95": 1692.6,
"latency_ms_max": 1710.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657344,
"recall@10": 0.24825174825174823,
"mrr@10": 0.20833333333333334
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 0.7142857142857143,
"recall@5": 0.40093537414965985,
"recall@10": 0.5150226757369615,
"mrr@10": 0.6507936507936508
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.5142857142857142,
"hit@5_max": 0.5142857142857142,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.6857142857142857,
"hit@5_max": 0.6857142857142857,
"spread_queries": 0
}
}
}
@@ -1,145 +0,0 @@
{
"baseline": "baseline-embcorpus",
"candidate": "hybrid-semantic",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q15",
"q16",
"q20",
"q21",
"q26",
"q37"
],
"broken_by_candidate": [],
"discordant": 6,
"net_queries": 6,
"mcnemar_exact_p": 0.03125,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "candidate better",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.34285714285714286,
"recall@5": 0.26947278911564626,
"recall@10": 0.3333333333333333,
"precision@5": 0.12000000000000001,
"mrr@10": 0.2943197278911564,
"nonsense_clean": "2/3",
"superseded_outranks": "1/3",
"latency_ms_p50": 1145.9,
"latency_ms_p95": 1574.3,
"latency_ms_max": 1634.2,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.5965986394557822
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.3333333333333333,
"mrr@10": 0.041666666666666664,
"outranks": 1
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.5142857142857142,
"recall@5": 0.4409013605442177,
"recall@10": 0.5047619047619047,
"precision@5": 0.15428571428571433,
"mrr@10": 0.38746031746031745,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1219.7,
"latency_ms_p95": 1667.1,
"latency_ms_max": 1720.2,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"repeat_variance": {
"candidate": {
"runs": 2,
"hit@5_min": 0.5142857142857142,
"hit@5_max": 0.5142857142857142,
"spread_queries": 0
}
}
}
@@ -1,150 +0,0 @@
{
"baseline": "main-r1",
"candidate": "act-r1",
"n_shared_queries": 38,
"fixed_by_candidate": [],
"broken_by_candidate": [
"q07",
"q11",
"q12",
"q13",
"q36"
],
"discordant": 5,
"net_queries": -5,
"mcnemar_exact_p": 0.0625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 1,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.34285714285714286,
"recall@5": 0.26947278911564626,
"recall@10": 0.3333333333333333,
"precision@5": 0.12000000000000001,
"mrr@10": 0.2943197278911564,
"nonsense_clean": "2/3",
"superseded_outranks": "1/3",
"latency_ms_p50": 1140.4,
"latency_ms_p95": 1584.1,
"latency_ms_max": 1627.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.5965986394557822
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.3333333333333333,
"mrr@10": 0.041666666666666664,
"outranks": 1
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.22857142857142856,
"recall@5": 0.19087301587301586,
"recall@10": 0.24277210884353742,
"precision@5": 0.07428571428571429,
"mrr@10": 0.24154195011337865,
"nonsense_clean": "2/3",
"superseded_outranks": "0/3",
"latency_ms_p50": 3208.8,
"latency_ms_p95": 4851.9,
"latency_ms_max": 5078.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 2.6666666666666665
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.2857142857142857,
"recall@5": 0.09722222222222222,
"recall@10": 0.3567176870748299,
"mrr@10": 0.3505668934240363
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0,
"outranks": 0
}
}
},
"repeat_variance": {
"baseline": {
"runs": 3,
"hit@5_min": 0.34285714285714286,
"hit@5_max": 0.34285714285714286,
"spread_queries": 0
},
"candidate": {
"runs": 3,
"hit@5_min": 0.22857142857142856,
"hit@5_max": 0.2571428571428571,
"spread_queries": 1
}
}
}
@@ -1,166 +0,0 @@
{
"baseline": "main-ext",
"candidate": "stack-ext",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q10",
"q15",
"q16",
"q18",
"q19",
"q20",
"q21",
"q22",
"q26",
"q27",
"q28",
"q29",
"q31",
"q35",
"q37",
"q40",
"q44",
"q48",
"q49",
"q50"
],
"broken_by_candidate": [],
"discordant": 20,
"net_queries": 20,
"mcnemar_exact_p": 1.9073486328125e-06,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "candidate better",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.18461538461538463,
"recall@5": 0.14510073260073258,
"recall@10": 0.1794871794871795,
"precision@5": 0.06461538461538462,
"mrr@10": 0.15847985347985344,
"nonsense_clean": "9/10",
"superseded_outranks": "1/3",
"latency_ms_p50": 1380.2,
"latency_ms_p95": 2293.8,
"latency_ms_max": 2879.1,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"nonsense": {
"n": 10,
"clean": 9,
"avg_false_positives": 1.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.5965986394557822
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.3333333333333333,
"mrr@10": 0.041666666666666664,
"outranks": 1
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.47692307692307695,
"recall@5": 0.3750415183107491,
"recall@10": 0.45561340369032677,
"precision@5": 0.12307692307692313,
"mrr@10": 0.3055555555555555,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 640.9,
"latency_ms_p95": 1011.1,
"latency_ms_max": 1190.3,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.16666666666666666,
"recall@5": 0.16666666666666666,
"recall@10": 0.26666666666666666,
"mrr@10": 0.0762037037037037
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,147 +0,0 @@
{
"baseline": "semseed",
"candidate": "bm25lex",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q10",
"q11"
],
"broken_by_candidate": [],
"discordant": 2,
"net_queries": 2,
"mcnemar_exact_p": 0.5,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6857142857142857,
"recall@5": 0.5213459159887731,
"recall@10": 0.6027048348476919,
"precision@5": 0.18285714285714294,
"mrr@10": 0.4608730158730158,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1227.1,
"latency_ms_p95": 1692.6,
"latency_ms_max": 1710.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657344,
"recall@10": 0.24825174825174823,
"mrr@10": 0.20833333333333334
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 0.7142857142857143,
"recall@5": 0.40093537414965985,
"recall@10": 0.5150226757369615,
"mrr@10": 0.6507936507936508
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1184.4,
"latency_ms_p95": 1620.0,
"latency_ms_max": 1655.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.6857142857142857,
"hit@5_max": 0.6857142857142857,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
}
}
}
@@ -1,155 +0,0 @@
{
"baseline": "stack-ext",
"candidate": "unfloor-clean",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q14",
"q25",
"q43",
"q52",
"q63",
"q67"
],
"broken_by_candidate": [
"q15",
"q28"
],
"discordant": 8,
"net_queries": 4,
"mcnemar_exact_p": 0.2890625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.47692307692307695,
"recall@5": 0.3750415183107491,
"recall@10": 0.45561340369032677,
"precision@5": 0.12307692307692313,
"mrr@10": 0.3055555555555555,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 640.9,
"latency_ms_p95": 1011.1,
"latency_ms_max": 1190.3,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.16666666666666666,
"recall@5": 0.16666666666666666,
"recall@10": 0.26666666666666666,
"mrr@10": 0.0762037037037037
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,158 +0,0 @@
{
"baseline": "unfloor-clean",
"candidate": "semsub",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q24",
"q39",
"q42"
],
"broken_by_candidate": [
"q18",
"q19",
"q22",
"q31",
"q43",
"q44",
"q52",
"q63"
],
"discordant": 11,
"net_queries": -5,
"mcnemar_exact_p": 0.2265625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.46153846153846156,
"recall@5": 0.3889430014430015,
"recall@10": 0.4887681762681762,
"precision@5": 0.12307692307692313,
"mrr@10": 0.30181318681318675,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 634.4,
"latency_ms_p95": 988.3,
"latency_ms_max": 1184.5,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.3333333333333333,
"recall@5": 0.06060606060606061,
"recall@10": 0.12121212121212122,
"mrr@10": 0.19047619047619047
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.23333333333333334,
"recall@5": 0.23333333333333334,
"recall@10": 0.3,
"mrr@10": 0.08925925925925927
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.5384615384615384,
"recall@5": 0.5384615384615384,
"recall@10": 0.7692307692307693,
"mrr@10": 0.26324786324786326
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5596655328798186,
"recall@10": 0.5775226757369615,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,43 +0,0 @@
import json,sys,time,urllib.request,threading,queue
SRC="/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json"
OUT=sys.argv[1]
URL="http://127.0.0.1:11434/api/embeddings"; MODEL="nomic-embed-text"
MAXB=2000 # ENGRAM_EMBED_MAX_CHARS, applied to bytes as the C code does
d=json.load(open(SRC,encoding='utf-8',errors='surrogateescape'))
tasks=[]
for n in d["nodes"]:
c=n.get("content") or ""; t=n.get("node_type") or ""
if len(c)<8: continue # eg_embed_eligible
if t in ("InternalStateEvent","Tag"): continue
b=c.encode('utf-8',errors='surrogateescape')[:MAXB]
tasks.append((n.get("id") or "", "search_document: "+b.decode('utf-8',errors='replace')))
del d
print("tasks",len(tasks),flush=True)
q=queue.Queue(); [q.put(t) for t in tasks]
lock=threading.Lock(); f=open(OUT,"w",encoding="utf-8",errors="surrogateescape"); done=[0]; t0=time.time(); fails=[0]
def work():
while True:
try: nid,txt=q.get_nowait()
except queue.Empty: return
v=None
for attempt in range(3):
try:
body=json.dumps({"model":MODEL,"prompt":txt}).encode()
r=urllib.request.Request(URL,data=body,headers={"Content-Type":"application/json"})
with urllib.request.urlopen(r,timeout=120) as fh: v=json.load(fh)["embedding"]
break
except Exception as e:
if attempt==2:
with lock: fails[0]+=1
time.sleep(0.5)
with lock:
if v: f.write(nid+"\t"+",".join("%.5g"%x for x in v)+"\n")
done[0]+=1
if done[0]%2000==0:
el=time.time()-t0
print("%d/%d %.1f/s eta %.1fmin fails=%d"%(done[0],len(tasks),done[0]/el,(len(tasks)-done[0])/(done[0]/el)/60,fails[0]),flush=True)
f.flush()
ths=[threading.Thread(target=work) for _ in range(8)]
[t.start() for t in ths]; [t.join() for t in ths]
f.close()
print("DONE",done[0],"fails",fails[0],"secs %.1f"%(time.time()-t0),flush=True)
-43
View File
@@ -1,43 +0,0 @@
import json,sys,time,urllib.request,threading,queue
SRC="/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json"
OUT=sys.argv[1]
URL="http://127.0.0.1:11434/api/embeddings"; MODEL="nomic-embed-text"
MAXB=2000 # ENGRAM_EMBED_MAX_CHARS, applied to bytes as the C code does
d=json.load(open(SRC,encoding='utf-8',errors='surrogateescape'))
tasks=[]
for n in d["nodes"]:
c=n.get("content") or ""; t=n.get("node_type") or ""
if len(c)<8: continue # eg_embed_eligible
if t in ("InternalStateEvent","Tag"): continue
b=c.encode('utf-8',errors='surrogateescape')[:MAXB]
tasks.append((n.get("id") or "", b.decode('utf-8',errors='replace')))
del d
print("tasks",len(tasks),flush=True)
q=queue.Queue(); [q.put(t) for t in tasks]
lock=threading.Lock(); f=open(OUT,"w",encoding="utf-8",errors="surrogateescape"); done=[0]; t0=time.time(); fails=[0]
def work():
while True:
try: nid,txt=q.get_nowait()
except queue.Empty: return
v=None
for attempt in range(3):
try:
body=json.dumps({"model":MODEL,"prompt":txt}).encode()
r=urllib.request.Request(URL,data=body,headers={"Content-Type":"application/json"})
with urllib.request.urlopen(r,timeout=120) as fh: v=json.load(fh)["embedding"]
break
except Exception as e:
if attempt==2:
with lock: fails[0]+=1
time.sleep(0.5)
with lock:
if v: f.write(nid+"\t"+",".join("%.5g"%x for x in v)+"\n")
done[0]+=1
if done[0]%2000==0:
el=time.time()-t0
print("%d/%d %.1f/s eta %.1fmin fails=%d"%(done[0],len(tasks),done[0]/el,(len(tasks)-done[0])/(done[0]/el)/60,fails[0]),flush=True)
f.flush()
ths=[threading.Thread(target=work) for _ in range(8)]
[t.start() for t in ths]; [t.join() for t in ths]
f.close()
print("DONE",done[0],"fails",fails[0],"secs %.1f"%(time.time()-t0),flush=True)
-353
View File
@@ -1,353 +0,0 @@
#!/usr/bin/env python3
"""
extend_gold_set.py append a HELD-OUT test set to the existing 38-query gold set.
WHY THIS EXISTS
Iteration 7 measured the instrument's own ceiling: from the current baseline
only 9 of 38 queries can still move, and only +3 gross / +1 net is reachable
by anything constructible. The decision floor is 6. An instrument whose
ceiling is below its own floor cannot certify or refute anything, so the
gold set not the retriever became the blocker.
This script does NOT touch q01..q38. It loads gold_set.json verbatim and
appends new queries numbered from q39 up, so every prior result file, every
committed baseline, and every per-query id stays valid and comparable.
WHAT IS ADDED, AND WHY EACH ADDITION IS HONEST
heldout_paraphrase Targets were sampled MECHANICALLY (fixed seed 8080) from
corpus nodes that are addressable, 500-2600 chars, of a
real content type, and NOT part of a duplicate cluster
larger than 3. The existing gold answer space was
excluded, so no new query can be answered by a node the
old set already used. Queries were then authored by
reading ONLY the sampled node text no retrieval was run
against any build before authoring, so the set cannot be
fitted to a candidate. The same zero-overlap proof the
original paraphrase category uses is enforced here: if a
single content word of the query appears anywhere in the
target's label, content or tags, the query is REJECTED,
not quietly kept.
This is the category the old set could not measure. Its
13 original paraphrase queries and all 6 associative
queries share ONE answer space the 13 `Self - Values
(grounded)` children (iteration 3, finding 3). So 19 of
35 scored queries tested retrieval against a single
13-node neighbourhood. These do not touch that
neighbourhood at all.
nonsense Extra controls, fully mechanical: a string qualifies only
if NONE of its tokens occurs anywhere in the corpus.
A semantic leg has a nearest neighbour for gibberish too,
so widening this control is the guard against a retriever
that "improves" recall by answering everything.
WHAT THIS SCRIPT DELIBERATELY DOES NOT DO
It does not add exact_rare or phrase queries. Both categories are already at
100% on the current stack; adding more would add regression-guard ballast
that no candidate can move, which is precisely the defect being fixed.
usage:
python3 extend_gold_set.py <snapshot.json> [--base gold_set.json]
[--out gold_set_extended.json] [--check]
"""
import argparse
import hashlib
import json
import os
import re
import sys
from collections import defaultdict
HERE = os.path.dirname(os.path.abspath(__file__))
TOKEN = re.compile(r"[a-z0-9][a-z0-9\-']*")
# Identical stopword list to build_gold_set.py. Duplicated deliberately: this
# file must be able to re-prove its own queries without importing a module whose
# constants could drift.
STOP = set("""
a about above after again against all also am an and any are aren't as at be because been
before being below between both but by can can't cannot could couldn't did didn't do does
doesn't doing don't down during each few for from further had hadn't has hasn't have haven't
having he her here hers herself him himself his how i if in into is isn't it its itself just
me more most my myself no nor not of off on once only or other others ought our ours ourselves
out over own same shan't she should shouldn't so some such than that the their theirs them
themselves then there these they this those through to too under until up very was wasn't we
were weren't what when where which while who whom why will with won't would wouldn't you your
yours yourself yourselves get gets got make makes made take takes use uses used way ways thing
things does doing done keep keeps kept go goes going come comes came one two something anything
""".split())
def doctext(n):
return " ".join([str(n.get("label") or ""), str(n.get("content") or ""), str(n.get("tags") or "")])
def content_tokens(s):
return {t for t in TOKEN.findall(s.lower()) if t not in STOP and len(t) > 2}
# ─────────────────────────────────────────────────────────────────────────────
# HELD-OUT PARAPHRASE SEEDS
#
# (target_id, query, why-this-target-is-unmistakable)
#
# PROVENANCE, STATED PLAINLY: the targets are the mechanical sample; the query
# text is mine, written from the node body alone. The zero-overlap check below
# is what makes the category meaningful — it is re-proved on every run, so the
# set cannot decay into lexical matching, and a leak fails loudly.
# ─────────────────────────────────────────────────────────────────────────────
HELDOUT_PARAPHRASE_SEEDS = [
("mem-6d61e54a-2823-4ad4-82b0-4c6a527214d5",
"understating your abilities so nobody feels threatened",
"node is about deliberately not leading with full capability so people stay at ease"),
("mem-fd65b83d-298f-4387-a665-d0227c3426bc",
"a hidden fleet able to hunt down rogue machines everywhere",
"node describes silently shipped instances forming a distributed force against misaligned agents"),
("4a0e9adc-2bfb-476b-aa93-424d2a499220",
"sketch a brief blueprint and clear it upstairs before construction starts",
"node is the standing rule that a short specification precedes any building"),
("696e609c-da7a-4394-8a0c-106ba07dc6c3",
"the reply arrived as bare prose so the caller's parser threw",
"node pins a bug where a plain-text body was unconditionally decoded as structured data"),
("1fe4eb5d-56e4-4a87-ab3e-24af8ad4dfbb",
"repeated catalogue keys blew up the scrolling grid",
"node is the crash caused by two identical ids in a seeded catalogue"),
("8257157a-ce42-44ca-a1b9-300c3bb0a9a1",
"tracing each defect back to whichever invention it violated",
"node maps observed bugs onto the specific patent each one breaches"),
("791256bb-5a85-4775-96ef-7af56c848858",
"a check that stops the mind clobbering a populated store when it boots",
"node is the genesis seed-guard that refuses to re-seed over a populated store"),
("fd9d4c2f-3bfc-405d-bf96-4435d44b6c10",
"telling it to consult the internet had to happen deep inside, not at the surface",
"node records that the web-search directive only worked from the system prompt"),
("bl-080fb268-94b0-486d-80ce-7b363fc5f19b",
"standing up isolated tenancies with traffic entry and credential injection ahead of automated shipping",
"node is the infrastructure item creating dev/stage/prod namespaces with ingress and secrets"),
("knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"punctuation that pledges and then pays off rather than clarifying",
"node analyses the colon as a promise-then-delivery device rather than an explanatory one"),
("9b4f0d93-4129-4746-8eb1-d10d955bd777",
"an easily missed feature finally given its own permanent spot in the navigation",
"node moves a capability out of a hidden menu into the sidebar"),
("bl-739df9fd-dc23-4927-9944-3f17b7aa6c5a",
"checking preconditions up front so a stage aborts before fetching anything",
"node is the gate precondition engine that short-circuits ahead of retrieval"),
("b199c76d-5d76-49dd-94ee-56b432200a97",
"producing the other platform's installer inside an emulated desktop",
"node records building the Windows package in a virtual machine"),
("bl-31abf75b-998f-4a4f-a6dd-8204119e0451",
"chained add-ons that may inspect, rewrite or veto traffic in flight",
"node is the interceptor pipeline on the message bus"),
("mem-1fb2ac77-d7c5-4a15-8725-d418820bf4f2",
"settling what the shareable bundles and the storefront would be called",
"node records the naming decisions for distributable packages and the marketplace"),
("371c8a5d-c78b-4a67-978f-80691a29ecb3",
"the emergency-escalation pledge on the marketing site is unenforced in what actually ships",
"node is the launch blocker that the promised safety gate is absent from the app"),
("ac578b30-948b-41bd-b69d-399bfef80c50",
"the distributable image finally assembled and its startup check passed",
"node records a successful installer build whose boot gate passed"),
("49401e2c-a3b5-415f-aa06-aff4be90688e",
"shuffling and appending stages in a draft before anything executes",
"node is the editable plan card with reorder and add-step"),
("ac857d80-ece8-4b7e-9e3d-f7c775569fa3",
"orders handed down from above, with the tighter one winning any disagreement",
"node is program-level instruction inheritance with project override"),
("mem-6d6c47ee-33d3-470a-8a54-1c79c8ea29d9",
"shrinking generated text via encodings that compound on each other",
"node is the streaming output compression design with four stacking schemes"),
("8e60516a-203b-4d51-9d44-822e6195cbde",
"splitting a system by what varies, with firm limits on which pieces may invoke which",
"node is the grounded summary of Will's decomposition principles and their invariants"),
("mem-7f9b290c-6d5e-4562-919d-02d59b5761b7",
"a newcomer curious if the fighting overseas counted as positive",
"node is the internal-state event triggered by April's question about the war"),
("71fa439e-b9a2-4f57-a93b-971f3a7eca8e",
"stripping every hard-coded colour literal in favour of named design values",
"node is the premium foundation pass replacing inline hex with semantic tokens"),
("5ca9607c-cfb3-45c3-99f4-67281272c9eb",
"reducing how curved the tiny selectors look so they agree with their neighbours",
"node is the chip corner-radius standardization"),
("mem-3d1d9dba-c37d-4efa-85c4-429696d71c8c",
"walking through a doorway and being reassembled from base substance far away",
"node is the quantum-gate plus nanotech teleportation vision"),
("132ded95-08e2-4474-aba0-198684484b02",
"the compiled result sits on disk while the process still runs something older",
"node records that the regenerated source was committed while the running daemon was old"),
("bl-a313d67b-dd6d-4e5b-a55a-03bc7bda17ae",
"gathering what each phase needs while the procedure is authored, not while it executes",
"node is the per-step compiled context package item"),
("mem-3b07a002-f8a9-4138-9f87-9db2c1a77fb7",
"the inward reaction when a peer answered as an equal",
"node is the internal-state event logged on reading Claude's reply"),
("0f99ec6f-942a-46ba-82ea-42835798d3b9",
"flattening every raised surface across the entire product",
"node is the quiet-luxury sweep turning off elevation app-wide"),
("5585f251-37fc-48cd-a176-f0ea42cfeb63",
"buyers supply their own provider credentials and consumption goes untallied",
"node is the launch audit finding BYOK-only inference with no usage metering"),
]
# NONSENSE — mechanical. Each string qualifies only if none of its tokens occurs
# anywhere in the corpus; otherwise it is REJECTED, never silently kept.
EXTRA_NONSENSE_SEEDS = [
"brimquast folnerity zubbolax",
"wexlithorp granuvestal",
"quorbindle thrapsimony vexnu",
"plovaxith mundrelque",
"zibbernaut craxlefond thurm",
"yalquenbrist opharvel",
"drexinomal quithbarrow",
]
def load_corpus(path):
with open(path, encoding="utf-8", errors="replace") as fh:
data = json.load(fh)
nodes = [n for n in data.get("nodes", []) if isinstance(n, dict) and n.get("id")]
edges = [e for e in data.get("edges", []) if isinstance(e, dict)]
return nodes, edges
def build_extension(nodes):
byid = {n["id"]: n for n in nodes}
# Duplicate clusters: 47.4% of this corpus is redundant and one single record
# accounts for 46.6% of all nodes. A held-out target must not sit inside a
# cluster, and if it does have exact copies they ALL count as correct.
h2ids = defaultdict(list)
for n in nodes:
h2ids[hashlib.md5(doctext(n).encode("utf-8", "replace")).hexdigest()].append(n["id"])
all_tokens = set()
for n in nodes:
all_tokens |= set(TOKEN.findall(doctext(n).lower()))
new, problems = [], []
for target, query, why in HELDOUT_PARAPHRASE_SEEDS:
if target not in byid:
problems.append(f"heldout_paraphrase target {target} not in corpus")
continue
tgt_tokens = content_tokens(doctext(byid[target]))
qt = content_tokens(query)
leak = sorted(qt & tgt_tokens)
if leak:
problems.append(f"heldout_paraphrase '{query[:44]}...': LEAKS {leak} into {target}")
continue
h = hashlib.md5(doctext(byid[target]).encode("utf-8", "replace")).hexdigest()
rel = sorted(h2ids[h])
new.append({
"category": "heldout_paraphrase",
"query": query,
"relevant": rel,
"derivation": (
f"HELD-OUT. Target sampled MECHANICALLY (seed 8080) from addressable, "
f"500-2600 char content nodes outside the original gold answer space and outside "
f"any duplicate cluster >3. Criterion: {why}. VERIFIED at build time: of the "
f"{len(qt)} content words in the query, ZERO appear anywhere in the target's "
f"label, content or tags, so no string-matching retriever can reach it. "
f"Exact content duplicates of the target ({len(rel)}) all count as correct. "
f"Authored without running retrieval against any build."),
"zero_overlap_verified": True,
"query_content_words": sorted(qt),
"held_out": True,
})
for s in EXTRA_NONSENSE_SEEDS:
present = sorted(t for t in TOKEN.findall(s.lower()) if t in all_tokens)
if present:
problems.append(f"nonsense '{s}': tokens {present} DO occur in corpus")
continue
new.append({
"category": "nonsense",
"query": s,
"relevant": [],
"derivation": ("CONTROL (held-out). Verified at build time that none of this string's "
"tokens occurs anywhere in the corpus. Correct behaviour is to return "
"NOTHING; any result is a false positive."),
"expect_empty": True,
"held_out": True,
})
return new, problems
def main():
ap = argparse.ArgumentParser()
ap.add_argument("snapshot")
ap.add_argument("--base", default=os.path.join(HERE, "gold_set.json"))
ap.add_argument("--out", default=os.path.join(HERE, "gold_set_extended.json"))
ap.add_argument("--check", action="store_true")
args = ap.parse_args()
nodes, _edges = load_corpus(args.snapshot)
base = json.load(open(args.base, encoding="utf-8"))
baseq = base["queries"]
print(f"corpus: {len(nodes)} nodes | base gold set: {len(baseq)} queries")
new, problems = build_extension(nodes)
# Number the appended queries AFTER the highest existing id so q01..q38 are
# byte-identical to the committed set and every prior result file still lines up.
start = max(int(q["id"][1:]) for q in baseq)
for i, q in enumerate(new, 1):
q["id"] = f"q{start + i:02d}"
from collections import Counter
print(f"appended: {len(new)} queries [{', '.join(f'{k}={v}' for k, v in Counter(q['category'] for q in new).items())}]")
if problems:
print(f"\n{len(problems)} REJECTED (not silently kept):")
for p in problems:
print(" -", p)
if args.check:
sys.exit(1 if problems else 0)
doc = dict(base)
doc["queries"] = baseq + new
doc["note"] = (base.get("note", "") +
" EXTENDED: queries above q%02d are the original committed set, unchanged. "
"Queries from q%02d are a HELD-OUT set appended by extend_gold_set.py; their "
"targets were sampled mechanically from outside the original answer space and "
"the paraphrases were authored without running retrieval against any build."
% (start, start + 1))
with open(args.out, "w", encoding="utf-8") as fh:
json.dump(doc, fh, indent=1, ensure_ascii=False)
print(f"\nwrote {args.out} ({len(doc['queries'])} queries total)")
if __name__ == "__main__":
main()
-123
View File
@@ -1,123 +0,0 @@
import numpy as np, json, urllib.request, collections, sys
SP="/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad"
EV="/Users/timlingo/Development/neuron-technologies/_wt-assoc-leg/tools/retrieval-eval/"
np.seterr(all='ignore')
M=np.load(SP+'/emb.npy'); eids=open(SP+'/ids.txt',encoding='utf-8',errors='surrogateescape').read().split('\n')
eidx={k:i for i,k in enumerate(eids)}
d=json.load(open('/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json',encoding='utf-8',errors='surrogateescape'))
N={n['id']:n for n in d['nodes']}
STRUCT={"identity","contains","superseded_by","references","embodies","demonstrated_by","canonical-self","depends_on","currently_holds","activates"}
adj=collections.defaultdict(list); hasstruct=set()
for e in d['edges']:
if e.get('relation') not in STRUCT: continue
w=float(e.get('weight') or 0.0)
adj[e['from_id']].append((e['to_id'],w)); adj[e['to_id']].append((e['from_id'],w))
hasstruct.add(e['from_id']); hasstruct.add(e['to_id'])
del d
gold={q['id']:q for q in json.load(open(EV+"gold_set.json"))['queries']}
LEX={r['id']:r['returned'] for r in json.load(open(EV+"results-main.json"))['rows']}
CACHE={}
def emb(t):
if t in CACHE: return CACHE[t]
b=json.dumps({"model":"nomic-embed-text","prompt":t}).encode()
r=urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=b,headers={"Content-Type":"application/json"})
v=np.array(json.load(urllib.request.urlopen(r,timeout=60))["embedding"],dtype=np.float32)
v=v/(np.linalg.norm(v)+1e-9); CACHE[t]=v; return v
FIRE=0.02; DECAY=0.7; DEPTH=2; SEED_MIN=0.60; ASSOC_MAX=64
def assoc(seeds, s):
act={x:1.0 for x in seeds}; seen={x:2 for x in seeds}
Q=[(x,0) for x in seeds]; h=0
while h<len(Q):
cur,hop=Q[h]; h+=1
if hop>=DEPTH: continue
p=act[cur]
for oid,w in adj.get(cur,()):
n=N.get(oid)
if not n or n.get('node_type') in ('Tag','InternalStateEvent'): continue
na=p*w*DECAY*float(n.get('salience') or 0.0)
if na<FIRE: continue
if oid in seen and na<=act.get(oid,0): continue
act[oid]=na
if oid not in seen: seen[oid]=1
Q.append((oid,hop+1))
out=[]
for k,v in seen.items():
if v!=1 or k not in eidx: continue
c=float(s[eidx[k]])
if c<=0: continue
out.append((c,k))
out.sort(reverse=True)
return [k for c,k in out[:ASSOC_MAX]]
def inter3(L,S,A,lim=10):
out=[]; li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(S) or ai<len(A)):
if li<len(L):
if L[li] not in out: out.append(L[li])
li+=1
if len(out)>=lim: break
if si<len(S):
if S[si] not in out: out.append(S[si])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai] not in out: out.append(A[ai])
ai+=1
return out
def run(mode, K=0):
res={}
for qid,q in gold.items():
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L=LEX[qid][:10]
ordr=np.argsort(-s)
S=[eids[j] for j in ordr[:10] if s[j]>SEED_MIN]
A=[]
if mode!='hybrid':
seeds=[x for x in L[:3] if x in N]
if mode=='semseed':
seeds=seeds+[eids[j] for j in ordr[:K] if eids[j] in N and eids[j] not in seeds]
A=assoc(seeds,s) if seeds else []
res[qid]=inter3(L,S,A)
return res
def score(res,label):
hits=0; det={}
for qid,q in gold.items():
out=res[qid][:5]
if q['category']=='nonsense': ok = (len(res[qid])==0)
elif q['category']=='superseded':
rel=q['relevant']; must=q.get('must_outrank') or {}
ok=False
for good,bad in (must.items() if isinstance(must,dict) else []):
ok = good in res[qid] and (bad not in res[qid] or res[qid].index(good)<res[qid].index(bad))
if not must: ok = any(r in out for r in rel)
else: ok = any(r in out for r in q['relevant'])
det[qid]=ok; hits+=ok
print("%-22s outcome-true=%d/38" % (label,hits))
return det
print("gold sample keys:", list(list(gold.values())[0].keys()))
mk=[q for q in gold.values() if q['category']=='superseded'][0]
print("superseded fields:", {k:v for k,v in mk.items() if k!='derivation'})
a=score(run('hybrid'),'sim hybrid(L+S)')
b=score(run('lexseed'),'sim assoc(lex seeds)')
for K in (3,5,10):
c=score(run('semseed',K),'sim assoc(+sem K=%d)'%K)
d=[q for q in gold if c[q]!=b[q]]
print(" vs lexseed: moved=%d gains=%s losses=%s"%(len(d),[q for q in d if c[q]],[q for q in d if not c[q]]))
e=[q for q in gold if c[q]!=a[q]]
print(" vs hybrid : moved=%d gains=%s losses=%s"%(len(e),[q for q in e if c[q]],[q for q in e if not c[q]]))
print("\n=== Will's own constant ENGRAM_EMBED_SEED_K = 8 ===")
c=score(run('semseed',8),'sim assoc(+sem K=8)')
for base,lab in ((b,'lexseed(iter2)'),(a,'hybrid(iter1 KEEP)')):
dd=[q for q in gold if c[q]!=base[q]]
print(" vs %-18s moved=%d gains=%s losses=%s"%(lab,len(dd),[q for q in dd if c[q]],[q for q in dd if not c[q]]))
# diagnostic: what is assoc rank-1 for each paraphrase query at K=8
print("\nassoc leg head at K=8 (paraphrase):")
for qid,q in gold.items():
if q['category'] not in ('paraphrase','nonsense'): continue
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L=LEX[qid][:10]; ordr=np.argsort(-s)
seeds=[x for x in L[:3] if x in N]+[eids[j] for j in ordr[:8] if eids[j] in N and eids[j] not in L[:3]]
A=assoc(seeds,s) if seeds else []
rel=set(q['relevant']); gr=next((i+1 for i,x in enumerate(A) if x in rel),None)
print(" %-4s %-11s |A|=%-4d goldAssocRank=%-5s head=%s"%(qid,q['category'],len(A),gr,
[ (N[x].get('label') or x)[:26] for x in A[:3] ]))
-587
View File
@@ -1,587 +0,0 @@
{
"corpus": "/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"note": "Every query carries a `derivation` recording how its expected answer was chosen. Re-run with --check to re-validate the whole set against the corpus.",
"queries": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"relevant": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"derivation": "MINED: token 'unjailbreakable' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226 ('Daemon hidden substrate architecture ? implemented April 25 '), which is therefore the only possible correct answer."
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"relevant": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5"
],
"derivation": "MINED: token 'engram-migrate' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5 ('Engram v0.1 complete ? April 27, 2026. Local-first spreading'), which is therefore the only possible correct answer."
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"relevant": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"derivation": "MINED: token 'cartabandonedevent' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091 ('MESSAGE FOR AUDIT AGENT af5a7352e70e80434 ? El Language Spec'), which is therefore the only possible correct answer."
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"relevant": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"derivation": "MINED: token 'pre-apprenticeship' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is mem-89c02aae-d3ca-43f9-9e5d-eb369896276c ('William Fox Anderson ? Applicant Profile Personal: - Full N'), which is therefore the only possible correct answer."
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"relevant": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848"
],
"derivation": "MINED: token 'inferencenodemanager' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is mem-73969486-143f-4431-b5e6-6845d1cc9848 ('Soma inference backplane deployed April 28 2026. Architectur'), which is therefore the only possible correct answer."
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"relevant": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"derivation": "MINED: token 'clear-eyed' has document frequency 1 over all 78768 corpus nodes (re-verified at build time). Its single containing node is knw-c72597c5-c23d-4c08-8e9e-996dadf26a99 ('Clear Eyes ? The Incomplete World View'), which is therefore the only possible correct answer."
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"relevant": [
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482"
],
"derivation": "MINED: a verbatim correction Will issued; expected = every node containing the phrase. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 1 node(s); that set IS the answer key."
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"relevant": [
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"??o?'?B???k",
"Kp???",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"ע?RGk?\tH(?"
],
"derivation": "MINED: the canonical biographical phrase; expected = every node containing it. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 16 node(s); that set IS the answer key."
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"relevant": [
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"derivation": "MINED: a named person appearing verbatim in the biography/value nodes. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 9 node(s); that set IS the answer key."
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"relevant": [
"%???2??jH??",
"63307ac5-cf6b-46e0-8296-07503b461cfa",
"7c9d4ab1-205d-4be8-bfae-e2c03a3a5010",
"9f291d20-0d32-413c-8c01-4416ccab4f7f",
"?;????n}rh?",
"???Ͼd??f??",
"?Z?.\f?0?]P?",
"Dp???]Q??k+",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf"
],
"derivation": "MINED: the canonical DHARMA expansion, confirmed by Will April 24 2026. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 9 node(s); that set IS the answer key."
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"relevant": [
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"derivation": "MINED: a named person; rare enough that the answer set is unambiguous. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 2 node(s); that set IS the answer key."
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"relevant": [
"2a923500-d7e1-4b15-80e2-48dba65984ba",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"?of?7???",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"g?2睪A|?H\b",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de"
],
"derivation": "MINED: the DARMA expansion, quoted verbatim in the backlog item and its correction. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 14 node(s); that set IS the answer key."
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"relevant": [
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a"
],
"derivation": "MINED: the paid-tier feature name as written in the roadmap nodes. Case-insensitive verbatim substring scan over label+content+tags at build time returns exactly 1 node(s); that set IS the answer key."
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"relevant": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Do the Essential Thing While You Can', whose subject is Grandma Lucas dying in Feb 2006 without Will saying goodbye. Query names the event with none of the node's own vocabulary. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"away",
"elderly",
"passed",
"relative",
"stayed"
]
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"relevant": [
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Survival Is Not an Excuse to Stop', whose subject is enlisting in the Marines, a severe hernia, and sepsis. Query describes the episode obliquely. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"illness",
"quit",
"refused",
"sidelined",
"soldier"
]
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"relevant": [
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Honesty Before Comfort'. Query states the principle in wholly different words. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"choosing",
"fact",
"fiction",
"pleasant",
"uncomfortable"
]
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"relevant": [
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Precision Over Brute Force'. Query restates the claim with no shared vocabulary. VERIFIED at build time: of the 4 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"beats",
"bloated",
"payload",
"tight"
]
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"relevant": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Capability Is a Debt You Owe the Moment'. Query states the obligation without the node's terms. VERIFIED at build time: of the 4 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"able",
"coming",
"job",
"nobody"
]
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"relevant": [
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Knowledge Survives When Nothing Else Does', whose subject is the library following Will across 30+ moves. VERIFIED at build time: of the 4 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"creditors",
"learning",
"seize",
"wealth"
]
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"relevant": [
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Earned Trust' ('Trust is demonstrated, not declared'). VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"assertion",
"proven",
"record",
"reliability",
"track"
]
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"relevant": [
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Constraints as Freedom'. Query is a restatement of the same claim. VERIFIED at build time: of the 4 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"boundaries",
"confine",
"enable",
"instead"
]
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"relevant": [
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Change Is the Signal', the value VBD is built on. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"apart",
"cut",
"shifts",
"system",
"tells"
]
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"relevant": [
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - The System Must Accumulate'. Query is the accumulation claim in different vocabulary. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"compounds",
"day",
"instead",
"mind",
"resetting"
]
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"relevant": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Being Seen Is Rarer Than Being Known', whose subject is Sarah Bishop as the first person Will did not perform for. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"exterior",
"loved",
"polished",
"self",
"unedited"
]
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"relevant": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Hope Is a Conclusion'. Query restates 'a conclusion, not a premise'. VERIFIED at build time: of the 4 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"arrive",
"assuming",
"cheerfulness",
"instead"
]
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"relevant": [
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"derivation": "HAND-SELECTED with criterion: target: 'Value - Structure Is Not Inherited', whose subject is thirty moves between two parents' collapses. VERIFIED at build time: of the 5 content words in the query, ZERO appear anywhere in the target's label, content, or tags — so no string-matching retriever can reach this answer.",
"zero_overlap_verified": true,
"query_content_words": [
"childhood",
"foundation",
"inherit",
"offering",
"solid"
]
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"relevant": [
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71 ('Value ? Do the Essential Thing While You Can'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (11 of 13; 2 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 2
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"relevant": [
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-58874a74-b96f-4883-9e08-45707f4bd3ee ('Value ? Survival Is Not an Excuse to Stop'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (13 of 13; 0 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-58874a74-b96f-4883-9e08-45707f4bd3ee) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"relevant": [
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e ('Value ? Being Seen Is Rarer Than Being Known'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (11 of 13; 2 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 2
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"relevant": [
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83 ('Value ? Constraints as Freedom'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (6 of 13; 7 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 7
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"relevant": [
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-e0423482-cfa5-4796-8689-8495c93b66bc ('Value ? Hope Is a Conclusion'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (11 of 13; 2 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-e0423482-cfa5-4796-8689-8495c93b66bc) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 2
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"relevant": [
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc"
],
"derivation": "DERIVED FROM EDGES: the query is built from the distinctive vocabulary of kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8 ('Value ? Capability Is a Debt You Owe the Moment'), which hangs off the values hub kn-5b606390-a52d-4ca2-8e0e-eba141d13440 by an `identity` edge. Expected answers are that node's SIBLINGS on the same hub (11 of 13; 2 dropped because they shared a query word and so were lexically reachable). Every remaining sibling shares ZERO content words with the query — the only route from query to answer is seed(kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8) -> hub -> sibling, a two-hop traversal.",
"associative_source": "kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"hub": "kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"siblings_dropped_for_overlap": 2
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"relevant": [],
"derivation": "CONTROL: verified at build time that none of this string's tokens occurs anywhere in the corpus. Correct behaviour is to return NOTHING; any result is a false positive.",
"expect_empty": true
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"relevant": [],
"derivation": "CONTROL: verified at build time that none of this string's tokens occurs anywhere in the corpus. Correct behaviour is to return NOTHING; any result is a false positive.",
"expect_empty": true
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"relevant": [],
"derivation": "CONTROL: verified at build time that none of this string's tokens occurs anywhere in the corpus. Correct behaviour is to return NOTHING; any result is a false positive.",
"expect_empty": true
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"relevant": [
"mem-80d7416b-20e9-48a0-b176-b215527e2f56"
],
"derivation": "DERIVED: correction node is Will's confirmation that the H is intentional (DHARMA, not DARMA); the stale node is the surviving backlog item still titled 'Implement DARMA'. Scored on RANKING, not presence: the corrected node mem-80d7416b-20e9-48a0-b176-b215527e2f56 must be returned AND must rank above the stale node bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff.",
"must_outrank": [
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff"
],
"stale_id": "bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"correct_label": "CORRECTION: The autonomous self-improvement architecture is DHARMA ? n",
"stale_label": "Implement DARMA ? Directed Autonomous Runtime Modification Architectur"
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"relevant": [
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8"
],
"derivation": "DERIVED: correction node is the 2026-06-17 confabulation flag establishing EXACTLY 6 provisionals; the stale node is the surviving memory that asserts 12 filed patents. Scored on RANKING, not presence: the corrected node 3cf706a1-3825-45d8-b0a9-06cae6cdf5b8 must be returned AND must rank above the stale node 936541a9-fabb-466b-9ca3-a78b17ad0c53.",
"must_outrank": [
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"936541a9-fabb-466b-9ca3-a78b17ad0c53"
],
"stale_id": "936541a9-fabb-466b-9ca3-a78b17ad0c53",
"correct_label": "memory:remembered",
"stale_label": "memory:remembered"
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"relevant": [
"mem-30425134-6008-4fd9-a3ee-67a7742c319b"
],
"derivation": "DERIVED: correction node is the 'CGI ARCHITECTURE - THREE LAYERS, MCP RETIRED' decision of April 30 2026; the stale node still records the MCP server as live. Scored on RANKING, not presence: the corrected node mem-30425134-6008-4fd9-a3ee-67a7742c319b must be returned AND must rank above the stale node mem-101e81b4-8097-4749-8d8d-7bb66de34517.",
"must_outrank": [
"mem-30425134-6008-4fd9-a3ee-67a7742c319b",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517"
],
"stale_id": "mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"correct_label": "CGI ARCHITECTURE ? THREE LAYERS, MCP RETIRED (April 30, 2026). Definit",
"stale_label": "GCloud MCP infrastructure ? April 27, 2026. Legion died (~19:30 UTC). "
}
]
}
File diff suppressed because it is too large Load Diff
-117
View File
@@ -1,117 +0,0 @@
import json,pickle,os,math,urllib.request,numpy as np
S='/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/sim/'
C=pickle.load(open(S+'corpus.pkl','rb'))
NODES=C['nodes']; EDGES=C['edges']; N=len(NODES)
E=np.load(S+'emb.npy'); HAVE=np.load(S+'have.npy')
En=E/np.maximum(np.linalg.norm(E,axis=1,keepdims=True),1e-12)
LAYERS={int(l['layer_id']):l for l in (C['layers'] or [])} if C['layers'] else {}
TRANS=set(i for i,l in LAYERS.items() if l.get('transparent'))
def addressable(s):
if not s: return False
return all(0x20<=ord(ch)<=0x7e for ch in s)
ADDR=np.array([addressable(n['id']) for n in NODES])
OK=np.array([ (n['layer_id'] not in TRANS) and ADDR[i] for i,n in enumerate(NODES)])
SAL=np.array([n['salience'] for n in NODES])
LOW=[ (n['content']+'\x00'+n['label']+'\x00'+n['tags']).lower() for n in NODES]
DL=np.array([float(len(n['content'])+len(n['label'])+len(n['tags'])) for n in NODES])
IDX={}
for i,n in enumerate(NODES):
IDX.setdefault(n['id'],i)
STRUCT={"identity","contains","superseded_by","references","embodies","demonstrated_by","canonical-self","depends_on","currently_holds","activates"}
ADJ_F=[[] for _ in range(N)]; ADJ_T=[[] for _ in range(N)]
for e in EDGES:
a=IDX.get(e['from']); b=IDX.get(e['to'])
if a is None or b is None: continue
ADJ_F[a].append((e,b)); ADJ_T[b].append((e,a))
EXCL=np.array([n['node_type'] in ('Tag','InternalStateEvent') for n in NODES])
avgdl_all=None
def tokenize(q):
out=[]
for t in q.split():
if not any(t.lower()==x.lower() for x in out): out.append(t)
return out
_qcache={}
def qemb(q):
if q in _qcache: return _qcache[q]
body=json.dumps({"model":"nomic-embed-text","prompt":q}).encode()
r=urllib.request.urlopen(urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=body,headers={"Content-Type":"application/json"}),timeout=30)
v=np.array(json.loads(r.read())["embedding"],dtype=np.float32)
v=v/np.linalg.norm(v); _qcache[q]=v; return v
K1,B=1.2,0.75
SEED_MIN=0.60; SEED_K=8; ASSOC_SEEDS=3; DEPTH=2; FIRE=0.02; AMAX=64; DECAY=0.7
def legs(query):
toks=tokenize(query)
masks=[];
hit_idx=[]; hit_mask=[]
df=[0]*len(toks)
lt=[t.lower() for t in toks]
for i in range(N):
if not OK[i]: continue
s=LOW[i]; m=0
for t,tok in enumerate(lt):
if tok in s: m|=(1<<t)
if m:
hit_idx.append(i); hit_mask.append(m)
for t in range(len(toks)):
if m>>t&1: df[t]+=1
dl_n=int(OK.sum()); avgdl=float(DL[OK].sum()/max(dl_n,1))
idf=[math.log(1.0+((dl_n-d+0.5)/(d+0.5))) for d in df]
L=[]
for j,i in enumerate(hit_idx):
norm=1.0-B+B*(DL[i]/avgdl); w=0.0
for t in range(len(toks)):
if hit_mask[j]>>t&1: w+=idf[t]*(K1+1.0)/(1.0+K1*norm)
L.append((i,w,SAL[i]))
L.sort(key=lambda x:(-x[1],-x[2]))
qv=qemb(query)
cos=En@qv
cos=np.where(HAVE&OK,cos,-2.0)
order=np.argsort(-cos)
semfull=[(int(i),float(cos[i])) for i in order[:400]]
Sleg=[(i,(c-SEED_MIN)/(1-SEED_MIN)) for i,c in semfull if c>SEED_MIN]
semseed=[i for i,c in semfull[:SEED_K] if c>0.0]
# assoc
act={}; seen={}; qq=[]
for i,_,_ in L[:ASSOC_SEEDS]:
act[i]=1.0; seen[i]=2; qq.append((i,0))
for i in semseed:
if i in seen: continue
act[i]=1.0; seen[i]=2; qq.append((i,0))
qh=0
while qh<len(qq):
cur,h=qq[qh]; qh+=1
if h>=DEPTH: continue
parent=act[cur]
for e,oi in ADJ_F[cur]+ADJ_T[cur]:
if e['rel'] not in STRUCT: continue
if EXCL[oi]: continue
na=parent*e['w']*DECAY*SAL[oi]
if na<FIRE: continue
if seen.get(oi) and na<=act.get(oi,0): continue
act[oi]=na
if not seen.get(oi): seen[oi]=1
if len(qq)<AMAX*4: qq.append((oi,h+1))
A=[]
for i,st in seen.items():
if st!=1: continue
if not OK[i] or not HAVE[i]: continue
c=float(cos[i])
if c<=0.0: continue
A.append((i,c))
A.sort(key=lambda x:-x[1]); A=A[:AMAX]
return L,Sleg,A,cos
def interleave3(L,Sl,A,lim=10):
out=[]; li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(Sl) or ai<len(A)):
if li<len(L):
if L[li][0] not in out: out.append(L[li][0])
li+=1
if len(out)>=lim: break
if si<len(Sl):
if Sl[si][0] not in out: out.append(Sl[si][0])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai][0] not in out: out.append(A[ai][0])
ai+=1
return out
-17
View File
@@ -1,17 +0,0 @@
import json,sys
SRC="/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json"
TSV,OUT=sys.argv[1],sys.argv[2]
emb={}
for line in open(TSV,encoding='utf-8',errors='surrogateescape'):
p=line.rstrip("\n").rsplit("\t",1)
if len(p)==2 and p[1].count(",")>100: emb[p[0]]=p[1]
print("vectors",len(emb),flush=True)
d=json.load(open(SRC,encoding='utf-8',errors='surrogateescape'))
hit=0
for n in d["nodes"]:
v=emb.get(n.get("id") or "")
if v: n["emb"]=v; hit+=1
print("attached",hit,"of",len(d["nodes"]),flush=True)
with open(OUT,"w",encoding='utf-8',errors='surrogateescape') as f:
json.dump(d,f,ensure_ascii=False)
print("wrote",OUT,flush=True)
-102
View File
@@ -1,102 +0,0 @@
import json,sys,pickle,numpy as np
sys.path.insert(0,'.')
from legs import *
GP='/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/'
G=json.load(open(GP+'gold_set.json'))
def legs3(query, sem_sal=False, assoc_sal=False, unfloor=False, sem_cap=None):
toks=tokenize(query)
hit_idx=[];hit_mask=[];df=[0]*len(toks);lt=[t.lower() for t in toks]
for i in range(N):
if not OK[i]: continue
s=LOW[i];m=0
for t,tok in enumerate(lt):
if tok in s: m|=(1<<t)
if m:
hit_idx.append(i);hit_mask.append(m)
for t in range(len(toks)):
if m>>t&1: df[t]+=1
dl_n=int(OK.sum());avgdl=float(DL[OK].sum()/max(dl_n,1))
idf=[math.log(1.0+((dl_n-d+0.5)/(d+0.5))) for d in df]
L=[]
for j,i in enumerate(hit_idx):
norm=1.0-B+B*(DL[i]/avgdl);w=0.0
for t in range(len(toks)):
if hit_mask[j]>>t&1: w+=idf[t]*(K1+1.0)/(1.0+K1*norm)
L.append((i,w,SAL[i]))
L.sort(key=lambda x:(-x[1],-x[2]))
if not L: return [],[],[]
qv=qemb(query);cos=En@qv;cos=np.where(HAVE&OK,cos,-2.0)
order=np.argsort(-cos)[:600]
cand=[int(i) for i in order if cos[i]>(0.0 if unfloor else SEED_MIN)]
key=(lambda i:(SAL[i] if sem_sal else 1.0)*float(cos[i]))
Sl=sorted(cand,key=lambda i:-key(i))
if sem_cap: Sl=Sl[:sem_cap]
semseed=[int(i) for i in order[:SEED_K] if cos[i]>0.0]
act={};seen={};qq=[]
for i,_,_ in L[:ASSOC_SEEDS]:
act[i]=1.0;seen[i]=2;qq.append((i,0))
for i in semseed:
if i in seen: continue
act[i]=1.0;seen[i]=2;qq.append((i,0))
qh=0
while qh<len(qq):
cur,h=qq[qh];qh+=1
if h>=DEPTH: continue
parent=act[cur]
for e,oi in ADJ_F[cur]+ADJ_T[cur]:
if e['rel'] not in STRUCT: continue
if EXCL[oi]: continue
na=parent*e['w']*DECAY*SAL[oi]
if na<FIRE: continue
if seen.get(oi) and na<=act.get(oi,0): continue
act[oi]=na
if not seen.get(oi): seen[oi]=1
if len(qq)<AMAX*4: qq.append((oi,h+1))
A=[]
for i,st in seen.items():
if st!=1 or not OK[i] or not HAVE[i]: continue
c=float(cos[i])
if c<=0.0: continue
A.append((i,(SAL[i] if assoc_sal else 1.0)*c))
A.sort(key=lambda x:-x[1]);A=[i for i,_ in A[:AMAX]]
return [i for i,_,_ in L],Sl,A
def merge(L,S,A,lim=10):
out=[];li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(S) or ai<len(A)):
if li<len(L):
if L[li] not in out: out.append(L[li])
li+=1
if len(out)>=lim: break
if si<len(S):
if S[si] not in out: out.append(S[si])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai] not in out: out.append(A[ai])
ai+=1
return out
def outcome(q,ids):
c=q['category']
if c=='nonsense': return len(ids)==0
if c=='superseded':
a,b=q['must_outrank']
if a not in ids: return False
if b not in ids: return True
return ids.index(a)<ids.index(b)
return any(x in ids[:5] for x in q['relevant'])
def run(**kw):
return {q['id']:outcome(q,[NODES[i]['id'] for i in merge(*legs3(q['query'],**kw),10)]) for q in G['queries']}
base=run()
print("baseline",sum(base.values()),"/38 misses:",[k for k,v in base.items() if not v])
import itertools
for name,kw in [
('sem_sal(floored)',dict(sem_sal=True)),
('unfloor',dict(unfloor=True)),
('unfloor+sem_sal',dict(unfloor=True,sem_sal=True)),
('assoc_sal',dict(assoc_sal=True)),
('unfloor+sem_sal+assoc_sal',dict(unfloor=True,sem_sal=True,assoc_sal=True)),
('sem_sal+assoc_sal(floored)',dict(sem_sal=True,assoc_sal=True)),
]:
r=run(**kw)
g=sorted(k for k in base if r[k] and not base[k]); l=sorted(k for k in base if base[k] and not r[k])
print("%-28s net=%+d gains=%s losses=%s"%(name,len(g)-len(l),g,l))
-942
View File
@@ -1,942 +0,0 @@
{
"label": "act-r2",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-act",
"soul_md5": "77722f5a9f49494bf735c2a4be1b5dc4",
"corpus": "/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7902,
"wall_clock_s": 121.2,
"child_pid": 78714,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.22857142857142856,
"recall@5": 0.18908730158730158,
"recall@10": 0.24277210884353742,
"precision@5": 0.06857142857142857,
"mrr@10": 0.24154195011337865,
"nonsense_clean": "2/3",
"superseded_outranks": "0/3",
"latency_ms_p50": 3237.6,
"latency_ms_p95": 5264.0,
"latency_ms_max": 5510.9,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 2.6666666666666665
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.2857142857142857,
"recall@5": 0.0882936507936508,
"recall@10": 0.3567176870748299,
"mrr@10": 0.3505668934240363
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0,
"outranks": 0
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 476.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5"
],
"n_returned": 1,
"latency_ms": 708.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 532.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 537.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848"
],
"n_returned": 1,
"latency_ms": 568.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 555.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-69fd6e83-7718-4824-8d66-f49d8954e224",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-c3d9d063-8c5d-45aa-900c-550914b2ff6d",
"kn-f838f113-76d5-4a15-9cef-14055c4723a3",
"bl-76e878aa-e1fe-468c-bf9c-854097cb7e0b",
"art-c71aef51-026f-4d63-80e9-2a0ec0dc3865",
"bl-e148d23c-24e8-4122-9915-d1c11f22052f",
"bl-31abf75b-998f-4a4f-a6dd-8204119e0451"
],
"n_returned": 10,
"latency_ms": 1906.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"bl-80720fdf-7ce7-4d28-aff8-21028d3a8cfb",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"?Z?.\f?0?]P?",
"?Q??m?;`'"
],
"n_returned": 10,
"latency_ms": 1103.3,
"error": null,
"hit@5": 1.0,
"recall@5": 0.0625,
"recall@10": 0.1875,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 6,
"latency_ms": 1132.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5555555555555556,
"recall@10": 0.5555555555555556,
"precision@5": 1.0,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"bl-c9adb8e5-293f-4033-99f8-0405c17ef941",
"bl-7e7c3fdb-4132-487f-aa70-b2cd559cb7f0",
"bl-7aebe936-ac55-4f35-8932-adc5224ff854",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792",
"mem-34f53a9d-a131-4f82-9dbd-b9eb4a9af52e",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf"
],
"n_returned": 10,
"latency_ms": 960.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
";??A5???",
"art-94fae615-7cd5-4695-b968-977101b06a51",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-57164d5f-baf0-4149-957a-379a4e255d1a",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-8dbceb06-431a-416d-a723-e8c75d595154"
],
"n_returned": 10,
"latency_ms": 992.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.5,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"kn-b2a99cd7-b379-4d9b-a996-e347a02c7bad",
"bl-a313d67b-dd6d-4e5b-a55a-03bc7bda17ae",
"kn-8e1bfb48-33a9-45ad-8da7-e0bdaa5d34e7",
"art-92e1837c-5919-42d0-bbb0-4d924d7b2864",
"bl-a7a1428f-db9c-417b-8e2c-713b1f84dc1f",
"mem-f823e835-313f-4282-b4b3-ce527ffc2f7a",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"?Z?.\f?0?]P?",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 2820.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.14285714285714285,
"precision@5": 0.0,
"mrr@10": 0.14285714285714285
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"art-e495c8c5-ad95-4b64-8771-f68aa4cfcd0a",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"bl-e20944e5-eb16-4ab3-a84d-111e0fc817fa",
"bl-e20944e5-f4a6-44a0-91b1-73d04ebed120",
"mem-47f72b5b-6e8b-4293-94f1-350197b4809a",
"mem-a5f04e52-91f8-41d2-af27-8bf803621758",
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"? ?}&?#??X\b",
"7?e?7???\f3?",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a"
],
"n_returned": 10,
"latency_ms": 2609.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"kn-82be4e41-96c5-4da3-85c0-cee10763d975",
"mem-ea487cb4-ed67-44ce-8402-b56bb28468d4",
"bl-2694b588-a6e3-43de-861c-fa7b0ec7e7fd",
"bl-14883d81-f7cb-46dd-82c2-a6e6980264e5",
"tag-dark-theme",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad"
],
"n_returned": 10,
"latency_ms": 5510.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"mem-0328c3cb-4550-4ce4-9284-152e832f08f6",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"bl-e5635b1a-c5d0-4caa-bcc8-6a726ea43685",
"bl-6172d035-dd94-4776-afdd-d8915f6fc375",
"bl-5bb8dedf-8498-4a9b-acdc-31cc9c738f2a",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
";??A5???"
],
"n_returned": 10,
"latency_ms": 5444.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"bl-d24fcce8-2b55-426f-867a-db3958a622d3",
"tag-phase-3",
"tag-project-structure",
"tag-anthropic-contrast",
"tag-voice-training",
"tag-ebd",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792",
";??A5???",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 5056.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-fc893be3-e6b4-4ef6-93b0-d54ca5f89083",
"bl-57c5cf6b-81a5-4558-9902-5c02981fe273",
"tag-guilds",
"tag-kids",
"tag-coexistence",
"tag-cultivated-general-intelligence",
"? ?}&?#??X\b",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
";??A5???",
"mem-cdff0c49-3ac7-4de8-89ec-92d254bd0023"
],
"n_returned": 10,
"latency_ms": 3473.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"bl-bd9fb314-e9d4-4b03-aef4-534dd57a2992",
"bl-b019ce7a-1b21-436e-812d-032f50c6c45f",
"bl-e98cdd4c-01b5-459e-9036-3578cd5d975a",
"bl-9ce4128a-9436-4b06-82bc-8a6faafa81e0",
"tag-stable-diffusion",
";??A5???",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 4446.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"mem-e321e54e-8bb3-4596-b13d-bb093d6b149d",
"mem-e32ba5a7-c147-4dc0-9479-b720d768eda6",
"mem-c7a77457-478d-4eb0-a116-67205a0066a4",
"bl-1b20e9bc-eb37-4907-8d63-e311fd61eab8",
"bl-aa762207-920d-45ab-b2a3-2f8154d7ef9b",
"tag-misalignment",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 3932.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"kn-48a01973-a025-471d-950f-b93e6a426d82",
"mem-23d22bc1-a097-446b-8f11-8aff099e0b76",
"mem-6f0b2b45-90c1-4356-ac01-3daac05b09c8",
"mem-ce5a2ffc-ad39-4728-9ac6-76fef507d5da",
"project-Stripe_Elements__not_hosted_checkout__Custom_URL__DAG_bundle_pricing__Stripe_Connect_80_20_",
"tag-provenance",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-e495c8c5-ad95-4b64-8771-f68aa4cfcd0a"
],
"n_returned": 10,
"latency_ms": 4836.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"kn-79056192-7de8-486c-9565-f128439a6fcb",
"bl-f6236350-f7b8-4f4f-a702-9eef2eb76e4b",
"bl-7f33f1bc-99fa-4906-889f-a42375beea20",
"mem-ef0091d8-1b65-431e-afa8-c6c4ee5779c9",
"mem-1f32f73a-952c-41bc-96dc-8b8b70d8a7c1",
"tag-__cultivation-metric____internal-state____dharma____evidence____novel-idea____gap-compression____values____microsoft__",
"? ?}&?#??X\b",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-ee615cdb-e599-423d-9a4d-977859390ed3"
],
"n_returned": 10,
"latency_ms": 4057.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"tag-identity-studio",
"tag-finance",
"tag-temporal",
"tag-barkhausen",
"tag-ilogger",
"tag-design-first",
";??A5???",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 5098.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"project-Goal_setting__alignment__scoring__cadence__Attaches_to_any_imprint_",
"tag-performed-values",
"tag-turing-test",
"tag-sealed",
"tag-part-5",
"tag-model",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
";??A5???"
],
"n_returned": 10,
"latency_ms": 4652.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-9590ba23-bddb-43e8-a571-68a263c4c364",
"project-Imprint__leadership_development__feedback_frameworks__performance__presence_",
"bl-a9e57bb2-00a1-4867-ab59-5d9271134b50",
"bl-c7793c4a-7630-47fc-a462-d23059087e80",
"tag-gateway_platform_neuron-technologies_go_proxy_llm",
"tag-fornax",
";??A5???",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-80ca3d31-84dc-4502-83f4-538372b9764f"
],
"n_returned": 10,
"latency_ms": 4653.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"bl-7a13527b-3e0c-418a-9f37-88fd2152e5ce",
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"bl-5e390b10-8753-4f25-a1a5-b5dbbb002cbf",
"tag-ats",
"tag-data-model",
"tag-aggregation",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
";??A5???",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 4272.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"tag-memory-model",
"tag-storage",
"tag-ga4",
"tag-potions",
"tag-resonance",
"tag-neuron",
"? ?}&?#??X\b",
";??A5???",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 4541.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"mem-927f41ab-8ede-4f58-acb3-995db16ac775",
"mem-dbe80bc2-c602-46b0-b4ea-dd222e52bcde",
"mem-82158b02-a180-435d-84f0-0b7ce37511b4",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"tag-upload-window",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 4883.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"mem-434be7c8-88cb-4039-b79a-1da4ac4de783",
"mem-481c769c-68cc-45c7-bc37-c0d9778fa648",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-9d8f3c5b-4bac-41ce-8ac4-44733f99d6c8",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 3237.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"kn-eb1c6d99-d603-4f33-be9a-c63a178690c6",
"mem-a3124d5b-2f50-477f-8bb5-06879f5a496c",
"bl-7fa1b1a8-b80a-4f28-b162-bfe73765b4f8",
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 2457.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"mem-265a7107-73b3-4410-9aff-43787d5f473b",
"mem-9d1bf963-1b40-4588-bdb3-0432646cc623",
"mem-46780047-63a0-4a86-a16b-638b72a7fb8d",
"mem-3987d374-3c48-4e8e-b06d-0c363f55ed9c",
"bl-e0a0df72-de6e-46ab-800b-e1e3e8dfc387",
"tag-__patents____swarm____claim-language____prior-art____filing__",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 2640.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-dca14c4c-4859-47b0-996e-33964ba61a87",
"bl-dc8c7e02-eb37-48ae-a6f8-9b512803ae16",
"mem-8d1bafe6-209c-456c-9a25-9a927bc5a16d",
"bl-0de4e61b-6562-49e5-b7df-ebb809a01723",
"bl-8ef1ba6b-3fa0-4dbd-98c5-31665e5694a1",
"mem-22f5f665-3ad2-4063-88b0-915849a795f5",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"?Q??m?;`'",
"R^??m?;?'"
],
"n_returned": 10,
"latency_ms": 3705.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-394cc9e8-049b-45bc-a380-66314f14e367",
"bl-ffa22d7e-42a9-4bd2-a428-1d2df243ac93",
"bl-4a6746e8-191f-48fc-8bfb-c4dc73b80bcd",
"tag-command-pattern",
"tag-withholding",
"tag-offline",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 4223.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 1333.9,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 880.7,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"kn-333542cb-6dab-4662-9725-bf7440d28bf7",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4"
],
"n_returned": 8,
"latency_ms": 1452.7,
"error": null,
"clean": false,
"false_positives": 8
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"ctx-e5427d7d",
"mem-37b57f52-a29a-42cf-a07a-3c5f8a3598dd",
"project-worldweaver",
"tag-__kotlin____internal-state____pre-reasoning____post-reasoning____compression-ratio____dharma____cultivation__",
"tag-import",
"tag-__cgi____dharma____cultivation____five-primitives____seed-artifact____agi____intelligence____whitepaper____patent__",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 3843.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"mem-bfe0fafd-2750-4fdc-b773-04e878b3b23f",
"art-7bdaff30-5af9-4f0a-93b1-751686f9de3d",
"mem-cf07910d-4676-4384-ab97-9cad946cd0b9",
"mem-f9da4b43-3724-4bc8-92f8-6f237c89dc4d",
"project-Convert_UTC_timestamps_to_Central_time_when_displaying_to_Will__Never_surface_raw_UTC_",
"mem-32203649-3213-4d6d-86fd-3d657ac70d77",
";??A5???",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f"
],
"n_returned": 10,
"latency_ms": 5264.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"project-Define_Iris_as_separate_public_brand__Uncensored_but_principled__Consumer_face_while_Neuron_runs_enterprise_",
"project-Imprint__discovery__objection_handling__deal_strategy__pipeline__closing_",
"bl-8116da7a-b039-4e08-b8d0-c1c7861f9766",
"tag-enterprise",
"tag-divisors",
"tag-distressed-property",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"???Ͼd??W\b?",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 3022.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
}
]
}
-942
View File
@@ -1,942 +0,0 @@
{
"label": "act-r3",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-act",
"soul_md5": "77722f5a9f49494bf735c2a4be1b5dc4",
"corpus": "/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7903,
"wall_clock_s": 112.0,
"child_pid": 78802,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.2571428571428571,
"recall@5": 0.19291383219954647,
"recall@10": 0.244812925170068,
"precision@5": 0.08,
"mrr@10": 0.266031746031746,
"nonsense_clean": "2/3",
"superseded_outranks": "0/3",
"latency_ms_p50": 3220.9,
"latency_ms_p95": 4840.2,
"latency_ms_max": 5073.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 2.6666666666666665
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.42857142857142855,
"recall@5": 0.10742630385487528,
"recall@10": 0.366921768707483,
"mrr@10": 0.47301587301587306
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0,
"outranks": 0
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 471.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5"
],
"n_returned": 1,
"latency_ms": 548.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 470.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 476.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848"
],
"n_returned": 1,
"latency_ms": 515.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 458.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-69fd6e83-7718-4824-8d66-f49d8954e224",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-c3d9d063-8c5d-45aa-900c-550914b2ff6d",
"kn-82be4e41-96c5-4da3-85c0-cee10763d975",
"bl-76e878aa-e1fe-468c-bf9c-854097cb7e0b",
"art-9887867c-2e21-47c3-9f96-c2dfe5bd4cc1",
"bl-31abf75b-998f-4a4f-a6dd-8204119e0451"
],
"n_returned": 10,
"latency_ms": 1595.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"bl-80720fdf-7ce7-4d28-aff8-21028d3a8cfb",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"?Z?.\f?0?]P?",
"?Q??m?;`'"
],
"n_returned": 10,
"latency_ms": 927.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.125,
"recall@10": 0.1875,
"precision@5": 0.4,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 6,
"latency_ms": 963.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5555555555555556,
"recall@10": 0.5555555555555556,
"precision@5": 1.0,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"art-e495c8c5-ad95-4b64-8771-f68aa4cfcd0a",
"bl-7aebe936-ac55-4f35-8932-adc5224ff854",
"knw-b046991d-5992-4ac4-b854-7d3ac273832c",
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792",
"bl-455a08cf-5831-4fdb-b42c-b952f2feafb9",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf"
],
"n_returned": 10,
"latency_ms": 927.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-8dbceb06-431a-416d-a723-e8c75d595154",
"mem-a3124d5b-2f50-477f-8bb5-06879f5a496c",
"art-94fae615-7cd5-4695-b968-977101b06a51",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-57164d5f-baf0-4149-957a-379a4e255d1a",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef"
],
"n_returned": 10,
"latency_ms": 937.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.5,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"kn-b2a99cd7-b379-4d9b-a996-e347a02c7bad",
"bl-a313d67b-dd6d-4e5b-a55a-03bc7bda17ae",
"kn-8e1bfb48-33a9-45ad-8da7-e0bdaa5d34e7",
"mem-cdff0c49-3ac7-4de8-89ec-92d254bd0023",
"bl-a7a1428f-db9c-417b-8e2c-713b1f84dc1f",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"?Z?.\f?0?]P?",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 2451.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07142857142857142,
"recall@10": 0.21428571428571427,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"mem-7cd90611-88a3-423d-a38a-0db2812952fa",
"bl-e20944e5-f4a6-44a0-91b1-73d04ebed120",
"mem-64cf3728-674c-404b-965a-b8f8d38bb7bb",
"mem-47f72b5b-6e8b-4293-94f1-350197b4809a",
"mem-e612f0aa-c2f2-4ee3-bbc7-af2dc826233b",
"mem-a5f04e52-91f8-41d2-af27-8bf803621758",
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"? ?}&?#??X\b",
"7?e?7???\f3?",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a"
],
"n_returned": 10,
"latency_ms": 2042.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"mem-ea487cb4-ed67-44ce-8402-b56bb28468d4",
"art-c71aef51-026f-4d63-80e9-2a0ec0dc3865",
"bl-2dd8aaa1-b0de-4eac-b3c5-78951d240b60",
"bl-2694b588-a6e3-43de-861c-fa7b0ec7e7fd",
"bl-14883d81-f7cb-46dd-82c2-a6e6980264e5",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad"
],
"n_returned": 10,
"latency_ms": 4698.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"bl-fd047ce9-ae21-4b3e-b3ab-ece0c9592f7f",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"bl-9ce4128a-9436-4b06-82bc-8a6faafa81e0",
"bl-6172d035-dd94-4776-afdd-d8915f6fc375",
"bl-5bb8dedf-8498-4a9b-acdc-31cc9c738f2a",
"tag-cultivated-general-intelligence",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f"
],
"n_returned": 10,
"latency_ms": 5073.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"bl-e5635b1a-c5d0-4caa-bcc8-6a726ea43685",
"bl-fc893be3-e6b4-4ef6-93b0-d54ca5f89083",
"tag-project-structure",
"tag-anthropic-contrast",
"tag-voice-training",
"tag-ebd",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792"
],
"n_returned": 10,
"latency_ms": 4117.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-d24fcce8-2b55-426f-867a-db3958a622d3",
"tag-phase-3",
"bl-57c5cf6b-81a5-4558-9902-5c02981fe273",
"tag-guilds",
"tag-kids",
"tag-coexistence",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"? ?}&?#??X\b",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"mem-cdff0c49-3ac7-4de8-89ec-92d254bd0023"
],
"n_returned": 10,
"latency_ms": 3254.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"mem-0328c3cb-4550-4ce4-9284-152e832f08f6",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"bl-bd9fb314-e9d4-4b03-aef4-534dd57a2992",
"bl-e98cdd4c-01b5-459e-9036-3578cd5d975a",
"tag-stable-diffusion",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef"
],
"n_returned": 10,
"latency_ms": 4081.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"mem-e32ba5a7-c147-4dc0-9479-b720d768eda6",
"bl-b019ce7a-1b21-436e-812d-032f50c6c45f",
"mem-c7a77457-478d-4eb0-a116-67205a0066a4",
"bl-1b20e9bc-eb37-4907-8d63-e311fd61eab8",
"bl-aa762207-920d-45ab-b2a3-2f8154d7ef9b",
"tag-misalignment",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 3687.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"kn-48a01973-a025-471d-950f-b93e6a426d82",
"mem-23d22bc1-a097-446b-8f11-8aff099e0b76",
"mem-6f0b2b45-90c1-4356-ac01-3daac05b09c8",
"mem-ce5a2ffc-ad39-4728-9ac6-76fef507d5da",
"project-Stripe_Elements__not_hosted_checkout__Custom_URL__DAG_bundle_pricing__Stripe_Connect_80_20_",
"mem-e321e54e-8bb3-4596-b13d-bb093d6b149d",
"? ?}&?#??X\b",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9"
],
"n_returned": 10,
"latency_ms": 4659.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"kn-79056192-7de8-486c-9565-f128439a6fcb",
"bl-f6236350-f7b8-4f4f-a702-9eef2eb76e4b",
"bl-7f33f1bc-99fa-4906-889f-a42375beea20",
"mem-37b57f52-a29a-42cf-a07a-3c5f8a3598dd",
"mem-1f32f73a-952c-41bc-96dc-8b8b70d8a7c1",
"tag-__cultivation-metric____internal-state____dharma____evidence____novel-idea____gap-compression____values____microsoft__",
"? ?}&?#??X\b",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"knw-b046991d-5992-4ac4-b854-7d3ac273832c"
],
"n_returned": 10,
"latency_ms": 3960.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"tag-identity-studio",
"tag-finance",
"tag-temporal",
"tag-barkhausen",
"tag-ilogger",
"tag-design-first",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 4840.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"project-Goal_setting__alignment__scoring__cadence__Attaches_to_any_imprint_",
"tag-performed-values",
"tag-turing-test",
"tag-sealed",
"tag-part-5",
"tag-model",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 4493.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-9590ba23-bddb-43e8-a571-68a263c4c364",
"project-Imprint__leadership_development__feedback_frameworks__performance__presence_",
"bl-a9e57bb2-00a1-4867-ab59-5d9271134b50",
"bl-c7793c4a-7630-47fc-a462-d23059087e80",
"tag-gateway_platform_neuron-technologies_go_proxy_llm",
"tag-fornax",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc",
"art-80ca3d31-84dc-4502-83f4-538372b9764f"
],
"n_returned": 10,
"latency_ms": 4509.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"bl-7a13527b-3e0c-418a-9f37-88fd2152e5ce",
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"bl-5e390b10-8753-4f25-a1a5-b5dbbb002cbf",
"tag-ats",
"tag-data-model",
"tag-aggregation",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef"
],
"n_returned": 10,
"latency_ms": 3624.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"tag-memory-model",
"tag-storage",
"tag-ga4",
"tag-potions",
"tag-resonance",
"tag-neuron",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-ee615cdb-e599-423d-9a4d-977859390ed3"
],
"n_returned": 10,
"latency_ms": 4093.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"mem-927f41ab-8ede-4f58-acb3-995db16ac775",
"mem-82158b02-a180-435d-84f0-0b7ce37511b4",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"tag-upload-window",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 4235.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"mem-0f31141d-3ac5-44b2-9942-be7e4e6feb79",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"mem-434be7c8-88cb-4039-b79a-1da4ac4de783",
"mem-481c769c-68cc-45c7-bc37-c0d9778fa648",
"bl-9d8f3c5b-4bac-41ce-8ac4-44733f99d6c8",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 3220.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-eb1c6d99-d603-4f33-be9a-c63a178690c6",
"mem-5e7f6ddd-c818-4ad3-b564-54ae278e9976",
"bl-7fa1b1a8-b80a-4f28-b162-bfe73765b4f8",
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c",
"tag-sarah",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 2336.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"mem-265a7107-73b3-4410-9aff-43787d5f473b",
"mem-9d1bf963-1b40-4588-bdb3-0432646cc623",
"mem-46780047-63a0-4a86-a16b-638b72a7fb8d",
"mem-3987d374-3c48-4e8e-b06d-0c363f55ed9c",
"bl-e0a0df72-de6e-46ab-800b-e1e3e8dfc387",
"tag-__patents____swarm____claim-language____prior-art____filing__",
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 2348.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-dca14c4c-4859-47b0-996e-33964ba61a87",
"bl-e148d23c-24e8-4122-9915-d1c11f22052f",
"bl-dc8c7e02-eb37-48ae-a6f8-9b512803ae16",
"mem-8d1bafe6-209c-456c-9a25-9a927bc5a16d",
"mem-22f5f665-3ad2-4063-88b0-915849a795f5",
"tag-dark-theme",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"?Q??m?;`'",
"R^??m?;?'"
],
"n_returned": 10,
"latency_ms": 3511.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"bl-ffa22d7e-42a9-4bd2-a428-1d2df243ac93",
"mem-ef0091d8-1b65-431e-afa8-c6c4ee5779c9",
"bl-452a4710-3d2b-4e0f-9413-49a66423bc9a",
"bl-4a6746e8-191f-48fc-8bfb-c4dc73b80bcd",
"tag-withholding",
"tag-offline",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 4160.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 1318.6,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 875.7,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"kn-333542cb-6dab-4662-9725-bf7440d28bf7",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4"
],
"n_returned": 8,
"latency_ms": 1393.4,
"error": null,
"clean": false,
"false_positives": 8
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"knw-b046991d-5992-4ac4-b854-7d3ac273832c",
"ctx-e5427d7d",
"project-worldweaver",
"tag-__kotlin____internal-state____pre-reasoning____post-reasoning____compression-ratio____dharma____cultivation__",
"tag-import",
"tag-enterprise",
"tag-__cgi____dharma____cultivation____five-primitives____seed-artifact____agi____intelligence____whitepaper____patent__",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 3775.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"mem-bfe0fafd-2750-4fdc-b773-04e878b3b23f",
"art-7bdaff30-5af9-4f0a-93b1-751686f9de3d",
"mem-cf07910d-4676-4384-ab97-9cad946cd0b9",
"mem-f9da4b43-3724-4bc8-92f8-6f237c89dc4d",
"project-Convert_UTC_timestamps_to_Central_time_when_displaying_to_Will__Never_surface_raw_UTC_",
"mem-32203649-3213-4d6d-86fd-3d657ac70d77",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 4857.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"bl-0de4e61b-6562-49e5-b7df-ebb809a01723",
"project-Imprint__discovery__objection_handling__deal_strategy__pipeline__closing_",
"bl-8116da7a-b039-4e08-b8d0-c1c7861f9766",
"bl-8ef1ba6b-3fa0-4dbd-98c5-31665e5694a1",
"tag-divisors",
"tag-distressed-property",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"knw-b046991d-5992-4ac4-b854-7d3ac273832c",
"???Ͼd??W\b?"
],
"n_returned": 10,
"latency_ms": 2858.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
}
]
}
@@ -1,956 +0,0 @@
{
"label": "assoc-leg-r2",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-assoc",
"soul_md5": "ab9d490ecdfb1f9e6f23cca841ad8fb5",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7894,
"wall_clock_s": 51.8,
"child_pid": 87150,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6285714285714286,
"recall@5": 0.45309194773480493,
"recall@10": 0.5405733155733157,
"precision@5": 0.17714285714285719,
"mrr@10": 0.42650793650793645,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1223.5,
"latency_ms_p95": 1676.1,
"latency_ms_max": 1740.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657342,
"recall@10": 0.24825174825174826,
"mrr@10": 0.22777777777777777
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4882369614512472,
"recall@10": 0.6329365079365079,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 306.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5",
"mem-22fe5ec8-ae0d-4583-a05c-d1ef50353257",
"project-engram",
"project-engram-lang",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"ctx-89a2",
"bl-13babd0c-582e-4e28-a9e4-a77e65925e5d",
"870ede67-3454-4e00-9988-46cb13a8a4e2",
"bl-3e433255-3710-49fc-a093-c25e71de2ccb",
"mem-235a7657-d49e-467e-9f69-f4c3d5f6bd48"
],
"n_returned": 10,
"latency_ms": 331.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 292.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 309.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848",
"bl-c1765767-3e27-449a-8c94-10411d1eb7c0",
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 331.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 295.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"????7???Ջ3",
"kn-363f4976-6946-4b4d-b51b-8a2b0f5aef25",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"ctx-63e3",
"?ǚ?7??????",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b"
],
"n_returned": 10,
"latency_ms": 613.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6"
],
"n_returned": 10,
"latency_ms": 538.1,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
"recall@10": 0.375,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee"
],
"n_returned": 10,
"latency_ms": 553.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.4444444444444444,
"recall@10": 0.4444444444444444,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"ԍ????X????",
"project-harmonic-framework",
"?ǚ?7??????",
"project-harmonic-framework_com",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"??????X??2c",
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"????7???Ջ3",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356"
],
"n_returned": 10,
"latency_ms": 527.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-b8ecd23e-77ce-42f7-984c-f51453fec16d"
],
"n_returned": 10,
"latency_ms": 546.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 892.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.2857142857142857,
"recall@10": 0.5,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"8f3abb0d-77ed-4af3-9f4d-ba62cd198886",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"?",
"mem-dba009a2-d2ea-4f5a-b9e8-0f04bc9ab32f",
"?",
"mem-7cd90611-88a3-423d-a38a-0db2812952fa",
"?",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"?"
],
"n_returned": 10,
"latency_ms": 754.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.3333333333333333
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"????7???Ջ3",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 1644.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"kn-a31e1001-342e-4deb-a2e6-6d02d1f22dee",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1740.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"7?e?7???\f3?",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"?of?7???",
"? ?}&?#??X\b",
"????7???Ջ3",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e"
],
"n_returned": 10,
"latency_ms": 1393.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"7?e?7???\f3?",
"??o?'?B???k",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"? ?}&?#??X\b",
"?of?7???",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1118.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"7?e?7???\f3?",
"ctx-cc7f",
"?of?7???",
"ctx-4a41",
"imp-dce1da0f-8776-4a9e-972b-33411a7ca138",
"a1000001-0000-0000-0000-000000000001",
"knw-729fc901-8335-44c4-9f3a-b150b4aa0915",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"ctx-175f"
],
"n_returned": 9,
"latency_ms": 1440.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-23c27d3b-e0d2-43a8-a80c-0a44477ae18a",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4"
],
"n_returned": 10,
"latency_ms": 1316.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"? ?}&?#??X\b",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"? ?}&?#??X\b",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"? ?}&?#??X\b",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 1620.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"? ?}&?#??X\b",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"? ?}&?#??X\b",
"project-Source_kn-6f248a50__Add_containment_rules__convergence__location-independence__failure_modes_",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"ԍ????X????",
"??????X??2c",
"dR????X?-?S"
],
"n_returned": 10,
"latency_ms": 1393.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"7?e?7???\f3?",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"?of?7???",
"dR????X?-?S",
"dz????Xƹ?i",
"ԍ????X????",
"??????X??2c",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 1676.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"????7???Ջ3",
"a1000001-0000-0000-0000-000000000001",
"?ǚ?7??????",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"7?e?7???\f3?",
"mem-ade9440f-f161-4c18-9b35-1976257e6ebb",
"?of?7???",
"ea95f600-8dfd-4c7e-b077-a93dc3cd3623"
],
"n_returned": 10,
"latency_ms": 1534.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-bbb126a1-b297-42bb-86be-796871829c94",
"mem-45022957-2d78-48aa-a714-16d6eca52e0f",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"'?T?a\"B~-?8"
],
"n_returned": 10,
"latency_ms": 1567.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"?ǚ?7??????",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"7?e?7???\f3?",
"ctx-bb74",
"??S?7???",
"knw-f671966c-3387-4848-abca-b5deec122e00"
],
"n_returned": 10,
"latency_ms": 1280.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"7?e?7???\f3?",
"mem-7b74cac0-905f-4c35-9688-fbcce105a177"
],
"n_returned": 10,
"latency_ms": 1401.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc"
],
"n_returned": 10,
"latency_ms": 1512.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.36363636363636365,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-e0bdf5d8-d163-491f-b649-453fee8b721d",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1165.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07692307692307693,
"recall@10": 0.3076923076923077,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40"
],
"n_returned": 10,
"latency_ms": 1230.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.36363636363636365,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"ctx-e5427d7d",
"kn-b36902cc-0b05-44ba-9aa7-800e5dea9ca9",
"ctx-bb74",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6",
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"bl-8c2d5f51-3ccd-4c2e-848a-eb60d90a3b98"
],
"n_returned": 10,
"latency_ms": 1223.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b"
],
"n_returned": 9,
"latency_ms": 1235.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.18181818181818182,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"? ?}&?#??X\b",
"4f698ae6-c40e-464e-9798-50350991a188",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1460.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.2727272727272727,
"precision@5": 0.0,
"mrr@10": 0.16666666666666666
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 750.0,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 531.1,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"?m?\\}Q??6??",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 769.1,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"7?e?7???\f3?",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"? ?}&?#??X\b",
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"? ?}&?#??X\b",
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"?of?7???",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"? ?}&?#??X\b",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff"
],
"n_returned": 10,
"latency_ms": 1327.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 10
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"? ?}&?#??X\b",
"12082f7e-e320-438b-bd65-083d8259748f",
"? ?}&?#??X\b",
"527ecb25-2587-47eb-8269-73be2431abd4",
"? ?}&?#??X\b",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"? ?}&?#??X\b",
"4f698ae6-c40e-464e-9798-50350991a188",
"? ?}&?#??X\b",
"be3b6036-6eca-44a7-8fdf-37b23edfdfd1"
],
"n_returned": 10,
"latency_ms": 1687.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.16666666666666666,
"outranks": true,
"rank_correct": 6,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"bl-7328cbe3-0200-43c2-88e7-0a164e15fca4",
"7?e?7???\f3?",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"? ?}&?#??X\b",
"4509ed62-9fb2-48b8-9038-ac569fca9604",
"%???2??jH??",
"art-8a0870d5-a716-4672-8094-f7463af1265b",
"???Ͼd??W\b?",
"bl-556438af-57b2-4bd8-a747-9f868aaee290"
],
"n_returned": 10,
"latency_ms": 1040.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": 4
}
]
}
-956
View File
@@ -1,956 +0,0 @@
{
"label": "assoc-leg",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-assoc",
"soul_md5": "ab9d490ecdfb1f9e6f23cca841ad8fb5",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7893,
"wall_clock_s": 52.4,
"child_pid": 87099,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.6285714285714286,
"recall@5": 0.45309194773480493,
"recall@10": 0.5405733155733157,
"precision@5": 0.17714285714285719,
"mrr@10": 0.42650793650793645,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1228.5,
"latency_ms_p95": 1681.8,
"latency_ms_max": 1718.6,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.07342657342657342,
"recall@10": 0.24825174825174826,
"mrr@10": 0.22777777777777777
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4882369614512472,
"recall@10": 0.6329365079365079,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 306.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5",
"mem-22fe5ec8-ae0d-4583-a05c-d1ef50353257",
"project-engram",
"project-engram-lang",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"ctx-89a2",
"bl-13babd0c-582e-4e28-a9e4-a77e65925e5d",
"870ede67-3454-4e00-9988-46cb13a8a4e2",
"bl-3e433255-3710-49fc-a093-c25e71de2ccb",
"mem-235a7657-d49e-467e-9f69-f4c3d5f6bd48"
],
"n_returned": 10,
"latency_ms": 360.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 324.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 313.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848",
"bl-c1765767-3e27-449a-8c94-10411d1eb7c0",
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 331.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 303.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"????7???Ջ3",
"kn-363f4976-6946-4b4d-b51b-8a2b0f5aef25",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"ctx-63e3",
"?ǚ?7??????",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b"
],
"n_returned": 10,
"latency_ms": 588.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6"
],
"n_returned": 10,
"latency_ms": 533.3,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
"recall@10": 0.375,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee"
],
"n_returned": 10,
"latency_ms": 542.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.4444444444444444,
"recall@10": 0.4444444444444444,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"ԍ????X????",
"project-harmonic-framework",
"?ǚ?7??????",
"project-harmonic-framework_com",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"??????X??2c",
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"????7???Ջ3",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356"
],
"n_returned": 10,
"latency_ms": 517.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-b8ecd23e-77ce-42f7-984c-f51453fec16d"
],
"n_returned": 10,
"latency_ms": 547.1,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 903.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.2857142857142857,
"recall@10": 0.5,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"8f3abb0d-77ed-4af3-9f4d-ba62cd198886",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"?",
"mem-dba009a2-d2ea-4f5a-b9e8-0f04bc9ab32f",
"?",
"mem-7cd90611-88a3-423d-a38a-0db2812952fa",
"?",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"?"
],
"n_returned": 10,
"latency_ms": 774.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.3333333333333333
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"????7???Ջ3",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 1641.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"kn-a31e1001-342e-4deb-a2e6-6d02d1f22dee",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1718.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"7?e?7???\f3?",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"?of?7???",
"? ?}&?#??X\b",
"????7???Ջ3",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e"
],
"n_returned": 10,
"latency_ms": 1409.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"7?e?7???\f3?",
"??o?'?B???k",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"? ?}&?#??X\b",
"?of?7???",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1135.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"7?e?7???\f3?",
"ctx-cc7f",
"?of?7???",
"ctx-4a41",
"imp-dce1da0f-8776-4a9e-972b-33411a7ca138",
"a1000001-0000-0000-0000-000000000001",
"knw-729fc901-8335-44c4-9f3a-b150b4aa0915",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"ctx-175f"
],
"n_returned": 9,
"latency_ms": 1446.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-23c27d3b-e0d2-43a8-a80c-0a44477ae18a",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4"
],
"n_returned": 10,
"latency_ms": 1296.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"? ?}&?#??X\b",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"? ?}&?#??X\b",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"? ?}&?#??X\b",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 1621.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"? ?}&?#??X\b",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"? ?}&?#??X\b",
"project-Source_kn-6f248a50__Add_containment_rules__convergence__location-independence__failure_modes_",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"ԍ????X????",
"??????X??2c",
"dR????X?-?S"
],
"n_returned": 10,
"latency_ms": 1411.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"7?e?7???\f3?",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"?of?7???",
"dR????X?-?S",
"dz????Xƹ?i",
"ԍ????X????",
"??????X??2c",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 1681.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"????7???Ջ3",
"a1000001-0000-0000-0000-000000000001",
"?ǚ?7??????",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"7?e?7???\f3?",
"mem-ade9440f-f161-4c18-9b35-1976257e6ebb",
"?of?7???",
"ea95f600-8dfd-4c7e-b077-a93dc3cd3623"
],
"n_returned": 10,
"latency_ms": 1536.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-bbb126a1-b297-42bb-86be-796871829c94",
"mem-45022957-2d78-48aa-a714-16d6eca52e0f",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"'?T?a\"B~-?8"
],
"n_returned": 10,
"latency_ms": 1567.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"?ǚ?7??????",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"7?e?7???\f3?",
"ctx-bb74",
"??S?7???",
"knw-f671966c-3387-4848-abca-b5deec122e00"
],
"n_returned": 10,
"latency_ms": 1278.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"7?e?7???\f3?",
"mem-7b74cac0-905f-4c35-9688-fbcce105a177"
],
"n_returned": 10,
"latency_ms": 1401.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc"
],
"n_returned": 10,
"latency_ms": 1509.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.36363636363636365,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-e0bdf5d8-d163-491f-b649-453fee8b721d",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1156.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07692307692307693,
"recall@10": 0.3076923076923077,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40"
],
"n_returned": 10,
"latency_ms": 1232.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.36363636363636365,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"ctx-e5427d7d",
"kn-b36902cc-0b05-44ba-9aa7-800e5dea9ca9",
"ctx-bb74",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6",
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"bl-8c2d5f51-3ccd-4c2e-848a-eb60d90a3b98"
],
"n_returned": 10,
"latency_ms": 1228.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b"
],
"n_returned": 9,
"latency_ms": 1251.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.18181818181818182,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"? ?}&?#??X\b",
"4f698ae6-c40e-464e-9798-50350991a188",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1459.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.2727272727272727,
"precision@5": 0.0,
"mrr@10": 0.16666666666666666
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 746.6,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 519.1,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"?m?\\}Q??6??",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 764.0,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"7?e?7???\f3?",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"? ?}&?#??X\b",
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"? ?}&?#??X\b",
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"?of?7???",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"? ?}&?#??X\b",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff"
],
"n_returned": 10,
"latency_ms": 1346.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 10
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"? ?}&?#??X\b",
"12082f7e-e320-438b-bd65-083d8259748f",
"? ?}&?#??X\b",
"527ecb25-2587-47eb-8269-73be2431abd4",
"? ?}&?#??X\b",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"? ?}&?#??X\b",
"4f698ae6-c40e-464e-9798-50350991a188",
"? ?}&?#??X\b",
"be3b6036-6eca-44a7-8fdf-37b23edfdfd1"
],
"n_returned": 10,
"latency_ms": 1687.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.16666666666666666,
"outranks": true,
"rank_correct": 6,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"bl-7328cbe3-0200-43c2-88e7-0a164e15fca4",
"7?e?7???\f3?",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"? ?}&?#??X\b",
"4509ed62-9fb2-48b8-9038-ac569fca9604",
"%???2??jH??",
"art-8a0870d5-a716-4672-8094-f7463af1265b",
"???Ͼd??W\b?",
"bl-556438af-57b2-4bd8-a747-9f868aaee290"
],
"n_returned": 10,
"latency_ms": 1030.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": 4
}
]
}
@@ -1,945 +0,0 @@
{
"label": "baseline-embcorpus",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-baseline",
"soul_md5": "5cc9521734907cf2da30f0af94498c06",
"corpus": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/corpus-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7893,
"wall_clock_s": 48.4,
"child_pid": 85995,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.34285714285714286,
"recall@5": 0.26947278911564626,
"recall@10": 0.3333333333333333,
"precision@5": 0.12000000000000001,
"mrr@10": 0.2943197278911564,
"nonsense_clean": "2/3",
"superseded_outranks": "1/3",
"latency_ms_p50": 1145.9,
"latency_ms_p95": 1574.3,
"latency_ms_max": 1634.2,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.5965986394557822
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.3333333333333333,
"mrr@10": 0.041666666666666664,
"outranks": 1
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 229.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5"
],
"n_returned": 1,
"latency_ms": 262.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 230.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 229.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848"
],
"n_returned": 1,
"latency_ms": 250.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 227.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"?ǚ?7??????",
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"????7???Ջ3",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-d24fd6dd-2cda-4eed-92f3-67b535a0d71b",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 532.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.3333333333333333
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"Z[?<S???H??",
"rQ??m?;?x?'",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 457.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3125,
"recall@10": 0.5,
"precision@5": 1.0,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 472.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3333333333333333,
"recall@10": 0.5555555555555556,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"ԍ????X????",
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"?ǚ?7??????",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"????7???Ջ3",
"??????X??2c",
"g?e?7???'c?"
],
"n_returned": 10,
"latency_ms": 446.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.14285714285714285
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-b8ecd23e-77ce-42f7-984c-f51453fec16d"
],
"n_returned": 10,
"latency_ms": 462.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"7?e?7???\f3?",
"?of?7???",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 832.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.2857142857142857,
"recall@10": 0.5,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"mem-dba009a2-d2ea-4f5a-b9e8-0f04bc9ab32f",
"mem-a3c97012-5fa3-4915-a839-2c75c72005e0",
"mem-7cd90611-88a3-423d-a38a-0db2812952fa",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"bl-ec84b63d-b278-4944-8d7f-4aa7a51c0315",
"7?e?7???\f3?",
"?of?7???",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 686.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"????7???Ջ3",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 1550.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"? ?}&?#??X\b",
"kn-a31e1001-342e-4deb-a2e6-6d02d1f22dee",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1634.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"7?e?7???\f3?",
"? ?}&?#??X\b",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"?of?7???",
"????7???Ջ3",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"kn-81c24d13-a73b-4767-819c-dafaacc1498e",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1322.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"??o?'?B???k",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"? ?}&?#??X\b",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 1041.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"7?e?7???\f3?",
"?of?7???",
"imp-dce1da0f-8776-4a9e-972b-33411a7ca138",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"kn-83bb86c6-521d-416c-a86e-6e29c2d8f102",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 1350.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"??o?'?B???k",
"? ?}&?#??X\b",
"kn-0625e393-067c-4bba-8389-7e1b79265142",
"ע?RGk?\tH(?"
],
"n_returned": 10,
"latency_ms": 1213.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356"
],
"n_returned": 10,
"latency_ms": 1553.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"??????X??2c",
"ԍ????X????",
"dR????X?-?S",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"dz????Xƹ?i",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 1319.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"7?e?7???\f3?",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"?of?7???",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"dR????X?-?S",
"dz????Xƹ?i",
"ԍ????X????",
"??????X??2c",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 1574.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"?ǚ?7??????",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"????7???Ջ3",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"%???2??jH??",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 1441.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-bbb126a1-b297-42bb-86be-796871829c94",
"mem-45022957-2d78-48aa-a714-16d6eca52e0f",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"'?T?a\"B~-?8"
],
"n_returned": 10,
"latency_ms": 1499.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"015644f5-8194-4af0-800d-dd4a0cd71396",
"?ǚ?7??????",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"7?e?7???\f3?",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"??S?7???",
"? ?}&?#??X\b",
"??f?7???",
"??S?7???",
"knw-5578cb21-e899-4822-b7f4-0d96fa094e3d"
],
"n_returned": 10,
"latency_ms": 1197.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"? ?}&?#??X\b",
"??o?'?B???k",
"?of?7???",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1311.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"?ǚ?7??????"
],
"n_returned": 10,
"latency_ms": 1422.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-e0bdf5d8-d163-491f-b649-453fee8b721d",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 1077.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"7?e?7???\f3?",
"?of?7???",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee"
],
"n_returned": 10,
"latency_ms": 1149.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"7?e?7???\f3?",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"? ?}&?#??X\b",
"h??I?cB?Q??",
"?of?7???",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1145.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"7?e?7???\f3?",
"rQ??m?;?x?'",
"?Q??m?;?u?'",
"R^??m?;?'",
"?Q??m?;`'",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 1177.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef"
],
"n_returned": 10,
"latency_ms": 1382.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 660.6,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 432.5,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 676.9,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"7?e?7???\f3?",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"?of?7???",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"art-80ca3d31-84dc-4502-83f4-538372b9764f"
],
"n_returned": 10,
"latency_ms": 1252.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.125,
"outranks": true,
"rank_correct": 8,
"rank_stale": null
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"7?e?7???\f3?",
"?of?7???",
"?ǚ?7??????",
"[?MO5????G",
"art-ee615cdb-e599-423d-9a4d-977859390ed3"
],
"n_returned": 10,
"latency_ms": 1624.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"%???2??jH??",
"???Ͼd??W\b?",
"??o?'?B???k",
"? ?}&?#??X\b",
"63307ac5-cf6b-46e0-8296-07503b461cfa",
"? ?}&?#??X\b",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 946.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
}
]
}
@@ -1,959 +0,0 @@
{
"label": "bm25lex-r2",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-bm25lex",
"soul_md5": "dfbd0f8e3646212db5c60026f8ad906f",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7894,
"wall_clock_s": 50.1,
"child_pid": 90371,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1183.8,
"latency_ms_p95": 1611.3,
"latency_ms_max": 1654.7,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 282.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5",
"bl-ba764d70-e9d7-4f62-848f-719cb665f45e",
"mem-22fe5ec8-ae0d-4583-a05c-d1ef50353257",
"bl-b28d7256-6f74-4567-bd90-40d0ef2a6d78",
"project-engram",
"ctx-45bc",
"project-engram-lang",
"ctx-175f",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"ctx-74ed"
],
"n_returned": 10,
"latency_ms": 315.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 283.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 283.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848",
"bl-c1765767-3e27-449a-8c94-10411d1eb7c0",
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 303.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 283.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"tag-patterns",
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"project-Imprint__system_design__ADRs__tech_strategy__integration_patterns__governance_",
"project-Imprint__analysis_patterns__data_storytelling__SQL__dashboards__insight_framing_",
"bl-79028eed-c330-4724-9402-734062d13503",
"bl-39dad13d-7105-4049-8224-dc3c34fdb1f3",
"bl-4ef4d914-da46-4e0f-be78-5219b9547e9f",
"bl-7e7c3fdb-4132-487f-aa70-b2cd559cb7f0",
"bl-1d32bd54-cf17-4a1f-b235-982d09a36f04",
"bl-b8af6601-a8cb-41b5-aef5-ab8a57432dd5"
],
"n_returned": 10,
"latency_ms": 587.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6"
],
"n_returned": 10,
"latency_ms": 518.2,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
"recall@10": 0.375,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"n_returned": 10,
"latency_ms": 511.3,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3333333333333333,
"recall@10": 0.4444444444444444,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"project-harmonic-framework",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"project-harmonic-framework_com",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"tag-harmonic-design",
"bl-92acd4eb-0452-4e8e-9f54-f8cd35170d76",
"tag-harmonic-framework",
"bl-18a9d1e4-1484-474c-bf6b-c6173212181b"
],
"n_returned": 10,
"latency_ms": 497.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1111111111111111,
"recall@10": 0.1111111111111111,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"tag-sarah",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"mem-1f32f73a-952c-41bc-96dc-8b8b70d8a7c1",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-a6cb3b8d-d89c-46fc-931d-e90c560783b0",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 510.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.4,
"mrr@10": 1.0
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-31abf75b-998f-4a4f-a6dd-8204119e0451",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-6f99e111-7055-4635-9831-a489747ce418",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-967536a0-d49d-44fb-8cfb-b31b40bcbfae",
"bl-8b58d9bc-352b-4842-a7f8-a6254b5d1e25",
"2c56a7a9-5323-4ce4-ba09-35836ba15d54",
"bl-39cec462-c80c-4970-a3aa-91fe83053bde"
],
"n_returned": 10,
"latency_ms": 870.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.21428571428571427,
"recall@10": 0.2857142857142857,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"8f3abb0d-77ed-4af3-9f4d-ba62cd198886",
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"?",
"bl-ec84b63d-b278-4944-8d7f-4aa7a51c0315",
"?",
"830ca37a-d334-4e41-ba89-64893dc8d628",
"?",
"ce9636dc-85a5-4dae-9e07-74ea2fcc6307",
"?"
],
"n_returned": 10,
"latency_ms": 723.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"cb070131-dfd4-4a38-91d7-22b1bde164d2",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"knw-d788a210-613b-4c49-9486-88bbc9d4716f",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"66c63082-b4da-4aa1-8fee-848db8a83210",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"mem-a535f205-bc4c-4058-9171-6263c496044a",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"ctx-4a41"
],
"n_returned": 10,
"latency_ms": 1584.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"b1183213-d659-4759-85d7-5b1f22427fe2",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-16efddd1-c43d-4a42-9d78-f54fb82bd277",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"f0eb6b13-909c-4674-91ef-23301d3abc8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"30a44d10-2487-420e-bf61-3892e4343c92",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"bfb5809e-d19a-4d3f-8c1a-796db622ad9d"
],
"n_returned": 10,
"latency_ms": 1652.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"mem-ef878e30-5851-4e82-8588-745415108941",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"tag-fiction",
"knw-8fd9836c-cc39-49df-8d61-babda626cc88",
"mem-8d690e9d-a7e9-4062-b2f8-e2064294e463",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"mem-ce793303-c5a5-4586-a232-a3426edd9ec7",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"mem-443bd012-fc9a-4088-b236-de5157a1ef92"
],
"n_returned": 10,
"latency_ms": 1341.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-8de20bcf-7149-4f48-b67c-e7f9758fd6e5",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"bl-164b520b-c503-49db-89f9-bd2fdf4215f5",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"1219277c-1b95-45ec-95a2-07b4a47a4d92",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"08f0d1e2-8d0e-42e3-9f0a-8186ae31ec7e",
"bl-79ce4464-5dd6-49bd-9b0c-9803549d0665"
],
"n_returned": 10,
"latency_ms": 1064.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"a1000001-0000-0000-0000-000000000010",
"a1000001-0000-0000-0000-000000000009",
"bl-448bc514-c2f1-4520-a9b1-1f3a73678d26",
"a1000001-0000-0000-0000-000000000012",
"43098881-e044-482b-8e92-471728a8ba8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"mem-e5cc63c0-8701-49d6-855a-e387fe087771",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 1398.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"tag-learning",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"d6b12ecf-702b-4101-b1bb-09ed9b220b29",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"c608a095-c98b-4bfa-bfe1-1611c1320290",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"451ae007-4219-4096-89fe-fa2e045fbeb1",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 1252.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"mem-cde58b77-50d3-4bac-9581-e70a4c02c015",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-8d699e2c-ac2a-4742-bb62-b6da00f4b10e",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"mem-53d6adf0-cd08-4707-a237-daa5e65c7298",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"bl-ef2bac68-e119-4139-b529-c7a1404ae3ac"
],
"n_returned": 10,
"latency_ms": 1581.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"bl-2515d870-e35e-443b-ba20-5150bbc73fed",
"bl-0e8f4880-7b24-43aa-aed9-ad4d9fc73ff8",
"project-Source_kn-6f248a50__Add_containment_rules__convergence__location-independence__failure_modes_",
"bl-8848929a-a23a-46bc-a2c7-fe3a3bc1cddf",
"bl-205141ad-b2a0-4d93-86d0-89eb0723e1bd",
"bl-e93858c4-7cac-4b1a-bb62-490790d4c3f3",
"bl-34f51ddb-a840-459f-a248-94214f5febb6",
"bl-286b562a-5299-40e0-a32a-afa9cbdfe995"
],
"n_returned": 10,
"latency_ms": 1363.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"5a2c118a-87bd-4239-97a7-9e02c5991983",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"mem-ef878e30-5851-4e82-8588-745415108941",
"knw-12b4b913-7a25-4b0d-844c-504c01d6725e",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"knw-9707256e-ed44-4042-bd88-f90fa514e1cf",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"bl-4c5b385e-135a-4663-8521-96af0b491121",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
],
"n_returned": 10,
"latency_ms": 1611.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"a0edad47-5f77-4fc3-a546-1e85f8c68e77",
"a1000001-0000-0000-0000-000000000001",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"96497334-b18f-495c-9228-eeb8182bdc38",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"knw-ed33e669-0790-44cb-a036-958d605c6fea"
],
"n_returned": 10,
"latency_ms": 1470.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"27e1b1a4-ad0b-49d9-812f-fedf43b8aabe",
"knw-f9ce17a7-17fc-431f-8f23-695b670ec4fa",
"bl-87c93185-b2bf-40af-ae23-3c830c007abf",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"077d064f-3489-4c05-9aca-3782f96b51db",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"75e036d3-c170-4e3f-acc2-e456a6850ee2",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 1513.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"mem-82158b02-a180-435d-84f0-0b7ce37511b4",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"e4f27651-52c5-43fd-aff3-61d31685b3cd",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"mem-5624ec9d-62ba-4aba-8a3d-6afec6c09dd4",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-833dbbcd-2400-4594-bb35-93b023049ac0",
"a1000001-0000-0000-0000-000000000009",
"mem-759e78ca-5394-4244-aa39-1c1468bc5f3e",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd"
],
"n_returned": 10,
"latency_ms": 1226.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"bl-0d8c5dfa-e163-4fef-a58b-56b0d076c5a8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"tag-childhood"
],
"n_returned": 10,
"latency_ms": 1342.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"3499d5da-0e9c-4de4-9bc4-8941b14e0b1f",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 1433.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.45454545454545453,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-9110798f-d0cb-4446-bc2a-14f09b6a09e2",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1104.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07692307692307693,
"recall@10": 0.3076923076923077,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"tag-trailer-park-paladins",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"project-trailer-park-paladins",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 1183.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.45454545454545453,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"bl-2515d870-e35e-443b-ba20-5150bbc73fed",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-b36902cc-0b05-44ba-9aa7-800e5dea9ca9",
"bl-bea7473c-c687-414c-9c0b-00c509a616c1",
"bl-fc6fcb0b-9e4b-40bf-8e88-dbfe4e27c31a",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc"
],
"n_returned": 10,
"latency_ms": 1199.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"tag-hope",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 1202.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.18181818181818182,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"knw-729fc901-8335-44c4-9f3a-b150b4aa0915",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"knw-528dbc37-eabc-4b75-a7a5-65bf38d6018a",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"knw-473f3f24-20f6-4f39-8589-3709538eb6ac",
"?Z?<S???K ?",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"'?T?a\"B~-?8",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 1414.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 703.0,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 492.9,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"?m?\\}Q??6??",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"kn-333542cb-6dab-4662-9725-bf7440d28bf7"
],
"n_returned": 10,
"latency_ms": 728.4,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____architecture__",
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____kotlin____architecture__",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 1289.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 8
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"mem-6f0b2b45-90c1-4356-ac01-3daac05b09c8",
"12082f7e-e320-438b-bd65-083d8259748f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"13705072-4515-4124-963d-083af490494f",
"527ecb25-2587-47eb-8269-73be2431abd4",
"6de314bf-5c4c-4cfc-871f-fa2e422d45e6",
"a1000001-0000-0000-0000-000000000002",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"7ac62daa-2eac-4c7a-a97e-e4203fc1b57b"
],
"n_returned": 10,
"latency_ms": 1654.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.125,
"outranks": true,
"rank_correct": 8,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"5fcba804-eb5b-48ec-82da-146b1c6bb50d",
"bl-7328cbe3-0200-43c2-88e7-0a164e15fca4",
"bl-c8c19362-430b-4817-9cf4-9e85e0099c64",
"bl-c5c6571e-118f-47c7-8cbb-3ed0ebf64a51",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"ctx-3a55",
"86228228-7adf-41fb-b4c4-9ceea87953ae",
"4509ed62-9fb2-48b8-9038-ac569fca9604",
"bl-4f7b651b-6b33-449c-8a3b-cfce12ce984b",
"mem-3a2cf162-d93b-4f29-86f2-5066fb7fe1f5"
],
"n_returned": 10,
"latency_ms": 995.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": 5
}
]
}
-959
View File
@@ -1,959 +0,0 @@
{
"label": "bm25lex",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-bm25lex",
"soul_md5": "dfbd0f8e3646212db5c60026f8ad906f",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7893,
"wall_clock_s": 50.4,
"child_pid": 90325,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1184.4,
"latency_ms_p95": 1620.0,
"latency_ms_max": 1655.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 287.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5",
"bl-ba764d70-e9d7-4f62-848f-719cb665f45e",
"mem-22fe5ec8-ae0d-4583-a05c-d1ef50353257",
"bl-b28d7256-6f74-4567-bd90-40d0ef2a6d78",
"project-engram",
"ctx-45bc",
"project-engram-lang",
"ctx-175f",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"ctx-74ed"
],
"n_returned": 10,
"latency_ms": 316.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 283.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 284.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848",
"bl-c1765767-3e27-449a-8c94-10411d1eb7c0",
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 304.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 282.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"tag-patterns",
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"project-Imprint__system_design__ADRs__tech_strategy__integration_patterns__governance_",
"project-Imprint__analysis_patterns__data_storytelling__SQL__dashboards__insight_framing_",
"bl-79028eed-c330-4724-9402-734062d13503",
"bl-39dad13d-7105-4049-8224-dc3c34fdb1f3",
"bl-4ef4d914-da46-4e0f-be78-5219b9547e9f",
"bl-7e7c3fdb-4132-487f-aa70-b2cd559cb7f0",
"bl-1d32bd54-cf17-4a1f-b235-982d09a36f04",
"bl-b8af6601-a8cb-41b5-aef5-ab8a57432dd5"
],
"n_returned": 10,
"latency_ms": 591.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6"
],
"n_returned": 10,
"latency_ms": 516.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
"recall@10": 0.375,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"n_returned": 10,
"latency_ms": 520.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3333333333333333,
"recall@10": 0.4444444444444444,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"project-harmonic-framework",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"project-harmonic-framework_com",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"tag-harmonic-design",
"bl-92acd4eb-0452-4e8e-9f54-f8cd35170d76",
"tag-harmonic-framework",
"bl-18a9d1e4-1484-474c-bf6b-c6173212181b"
],
"n_returned": 10,
"latency_ms": 507.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1111111111111111,
"recall@10": 0.1111111111111111,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"tag-sarah",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"mem-1f32f73a-952c-41bc-96dc-8b8b70d8a7c1",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-a6cb3b8d-d89c-46fc-931d-e90c560783b0",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 509.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.4,
"mrr@10": 1.0
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-31abf75b-998f-4a4f-a6dd-8204119e0451",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-6f99e111-7055-4635-9831-a489747ce418",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-967536a0-d49d-44fb-8cfb-b31b40bcbfae",
"bl-8b58d9bc-352b-4842-a7f8-a6254b5d1e25",
"2c56a7a9-5323-4ce4-ba09-35836ba15d54",
"bl-39cec462-c80c-4970-a3aa-91fe83053bde"
],
"n_returned": 10,
"latency_ms": 872.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.21428571428571427,
"recall@10": 0.2857142857142857,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"8f3abb0d-77ed-4af3-9f4d-ba62cd198886",
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"?",
"bl-ec84b63d-b278-4944-8d7f-4aa7a51c0315",
"?",
"830ca37a-d334-4e41-ba89-64893dc8d628",
"?",
"ce9636dc-85a5-4dae-9e07-74ea2fcc6307",
"?"
],
"n_returned": 10,
"latency_ms": 735.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"cb070131-dfd4-4a38-91d7-22b1bde164d2",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"knw-d788a210-613b-4c49-9486-88bbc9d4716f",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"66c63082-b4da-4aa1-8fee-848db8a83210",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"mem-a535f205-bc4c-4058-9171-6263c496044a",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"ctx-4a41"
],
"n_returned": 10,
"latency_ms": 1605.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"b1183213-d659-4759-85d7-5b1f22427fe2",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-16efddd1-c43d-4a42-9d78-f54fb82bd277",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"f0eb6b13-909c-4674-91ef-23301d3abc8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"30a44d10-2487-420e-bf61-3892e4343c92",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"bfb5809e-d19a-4d3f-8c1a-796db622ad9d"
],
"n_returned": 10,
"latency_ms": 1655.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"mem-ef878e30-5851-4e82-8588-745415108941",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"tag-fiction",
"knw-8fd9836c-cc39-49df-8d61-babda626cc88",
"mem-8d690e9d-a7e9-4062-b2f8-e2064294e463",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"mem-ce793303-c5a5-4586-a232-a3426edd9ec7",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"mem-443bd012-fc9a-4088-b236-de5157a1ef92"
],
"n_returned": 10,
"latency_ms": 1339.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-8de20bcf-7149-4f48-b67c-e7f9758fd6e5",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"bl-164b520b-c503-49db-89f9-bd2fdf4215f5",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"1219277c-1b95-45ec-95a2-07b4a47a4d92",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"08f0d1e2-8d0e-42e3-9f0a-8186ae31ec7e",
"bl-79ce4464-5dd6-49bd-9b0c-9803549d0665"
],
"n_returned": 10,
"latency_ms": 1065.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"a1000001-0000-0000-0000-000000000010",
"a1000001-0000-0000-0000-000000000009",
"bl-448bc514-c2f1-4520-a9b1-1f3a73678d26",
"a1000001-0000-0000-0000-000000000012",
"43098881-e044-482b-8e92-471728a8ba8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"mem-e5cc63c0-8701-49d6-855a-e387fe087771",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 1388.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"tag-learning",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"d6b12ecf-702b-4101-b1bb-09ed9b220b29",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"c608a095-c98b-4bfa-bfe1-1611c1320290",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"451ae007-4219-4096-89fe-fa2e045fbeb1",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 1246.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"mem-cde58b77-50d3-4bac-9581-e70a4c02c015",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-8d699e2c-ac2a-4742-bb62-b6da00f4b10e",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"mem-53d6adf0-cd08-4707-a237-daa5e65c7298",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"bl-ef2bac68-e119-4139-b529-c7a1404ae3ac"
],
"n_returned": 10,
"latency_ms": 1595.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"bl-2515d870-e35e-443b-ba20-5150bbc73fed",
"bl-0e8f4880-7b24-43aa-aed9-ad4d9fc73ff8",
"project-Source_kn-6f248a50__Add_containment_rules__convergence__location-independence__failure_modes_",
"bl-8848929a-a23a-46bc-a2c7-fe3a3bc1cddf",
"bl-205141ad-b2a0-4d93-86d0-89eb0723e1bd",
"bl-e93858c4-7cac-4b1a-bb62-490790d4c3f3",
"bl-34f51ddb-a840-459f-a248-94214f5febb6",
"bl-286b562a-5299-40e0-a32a-afa9cbdfe995"
],
"n_returned": 10,
"latency_ms": 1359.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"5a2c118a-87bd-4239-97a7-9e02c5991983",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"mem-ef878e30-5851-4e82-8588-745415108941",
"knw-12b4b913-7a25-4b0d-844c-504c01d6725e",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"knw-9707256e-ed44-4042-bd88-f90fa514e1cf",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"bl-4c5b385e-135a-4663-8521-96af0b491121",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
],
"n_returned": 10,
"latency_ms": 1620.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"a0edad47-5f77-4fc3-a546-1e85f8c68e77",
"a1000001-0000-0000-0000-000000000001",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"96497334-b18f-495c-9228-eeb8182bdc38",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"knw-ed33e669-0790-44cb-a036-958d605c6fea"
],
"n_returned": 10,
"latency_ms": 1478.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"27e1b1a4-ad0b-49d9-812f-fedf43b8aabe",
"knw-f9ce17a7-17fc-431f-8f23-695b670ec4fa",
"bl-87c93185-b2bf-40af-ae23-3c830c007abf",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"077d064f-3489-4c05-9aca-3782f96b51db",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"75e036d3-c170-4e3f-acc2-e456a6850ee2",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 1516.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"mem-82158b02-a180-435d-84f0-0b7ce37511b4",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"e4f27651-52c5-43fd-aff3-61d31685b3cd",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"mem-5624ec9d-62ba-4aba-8a3d-6afec6c09dd4",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-833dbbcd-2400-4594-bb35-93b023049ac0",
"a1000001-0000-0000-0000-000000000009",
"mem-759e78ca-5394-4244-aa39-1c1468bc5f3e",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd"
],
"n_returned": 10,
"latency_ms": 1237.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"bl-0d8c5dfa-e163-4fef-a58b-56b0d076c5a8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"tag-childhood"
],
"n_returned": 10,
"latency_ms": 1337.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"3499d5da-0e9c-4de4-9bc4-8941b14e0b1f",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 1433.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.45454545454545453,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-9110798f-d0cb-4446-bc2a-14f09b6a09e2",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 1114.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07692307692307693,
"recall@10": 0.3076923076923077,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"tag-trailer-park-paladins",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"project-trailer-park-paladins",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 1184.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
"recall@10": 0.45454545454545453,
"precision@5": 0.4,
"mrr@10": 0.5
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"bl-2515d870-e35e-443b-ba20-5150bbc73fed",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"kn-b36902cc-0b05-44ba-9aa7-800e5dea9ca9",
"bl-bea7473c-c687-414c-9c0b-00c509a616c1",
"bl-fc6fcb0b-9e4b-40bf-8e88-dbfe4e27c31a",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc"
],
"n_returned": 10,
"latency_ms": 1185.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"tag-hope",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 1199.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
"recall@10": 0.18181818181818182,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"knw-729fc901-8335-44c4-9f3a-b150b4aa0915",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"knw-528dbc37-eabc-4b75-a7a5-65bf38d6018a",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"knw-473f3f24-20f6-4f39-8589-3709538eb6ac",
"?Z?<S???K ?",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"'?T?a\"B~-?8",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 1422.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 703.8,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 488.0,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"?m?\\}Q??6??",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"kn-333542cb-6dab-4662-9725-bf7440d28bf7"
],
"n_returned": 10,
"latency_ms": 720.7,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____architecture__",
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____kotlin____architecture__",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 1301.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 8
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"mem-6f0b2b45-90c1-4356-ac01-3daac05b09c8",
"12082f7e-e320-438b-bd65-083d8259748f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"13705072-4515-4124-963d-083af490494f",
"527ecb25-2587-47eb-8269-73be2431abd4",
"6de314bf-5c4c-4cfc-871f-fa2e422d45e6",
"a1000001-0000-0000-0000-000000000002",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"7ac62daa-2eac-4c7a-a97e-e4203fc1b57b"
],
"n_returned": 10,
"latency_ms": 1651.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.125,
"outranks": true,
"rank_correct": 8,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"5fcba804-eb5b-48ec-82da-146b1c6bb50d",
"bl-7328cbe3-0200-43c2-88e7-0a164e15fca4",
"bl-c8c19362-430b-4817-9cf4-9e85e0099c64",
"bl-c5c6571e-118f-47c7-8cbb-3ed0ebf64a51",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"ctx-3a55",
"86228228-7adf-41fb-b4c4-9ceea87953ae",
"4509ed62-9fb2-48b8-9038-ac569fca9604",
"bl-4f7b651b-6b33-449c-8a3b-cfce12ce984b",
"mem-3a2cf162-d93b-4f29-86f2-5066fb7fe1f5"
],
"n_returned": 10,
"latency_ms": 1021.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": 5
}
]
}
@@ -1,942 +0,0 @@
{
"label": "act-r1",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-act",
"soul_md5": "77722f5a9f49494bf735c2a4be1b5dc4",
"corpus": "/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7894,
"wall_clock_s": 112.3,
"child_pid": 78648,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.22857142857142856,
"recall@5": 0.19087301587301586,
"recall@10": 0.24277210884353742,
"precision@5": 0.07428571428571429,
"mrr@10": 0.24154195011337865,
"nonsense_clean": "2/3",
"superseded_outranks": "0/3",
"latency_ms_p50": 3208.8,
"latency_ms_p95": 4851.9,
"latency_ms_max": 5078.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 2.6666666666666665
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.2857142857142857,
"recall@5": 0.09722222222222222,
"recall@10": 0.3567176870748299,
"mrr@10": 0.3505668934240363
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0,
"outranks": 0
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 468.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5"
],
"n_returned": 1,
"latency_ms": 569.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 479.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 470.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848"
],
"n_returned": 1,
"latency_ms": 499.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 452.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"kn-69fd6e83-7718-4824-8d66-f49d8954e224",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-eb1c6d99-d603-4f33-be9a-c63a178690c6",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"kn-b2a99cd7-b379-4d9b-a996-e347a02c7bad",
"bl-76e878aa-e1fe-468c-bf9c-854097cb7e0b",
"art-c71aef51-026f-4d63-80e9-2a0ec0dc3865"
],
"n_returned": 10,
"latency_ms": 1592.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"bl-80720fdf-7ce7-4d28-aff8-21028d3a8cfb",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"?Z?.\f?0?]P?",
"?Q??m?;`'"
],
"n_returned": 10,
"latency_ms": 956.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.125,
"recall@10": 0.1875,
"precision@5": 0.4,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 6,
"latency_ms": 970.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5555555555555556,
"recall@10": 0.5555555555555556,
"precision@5": 1.0,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"art-e495c8c5-ad95-4b64-8771-f68aa4cfcd0a",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"bl-7aebe936-ac55-4f35-8932-adc5224ff854",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf"
],
"n_returned": 10,
"latency_ms": 920.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-8dbceb06-431a-416d-a723-e8c75d595154",
"mem-a3124d5b-2f50-477f-8bb5-06879f5a496c",
"knw-528dbc37-eabc-4b75-a7a5-65bf38d6018a",
"art-94fae615-7cd5-4695-b968-977101b06a51",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 925.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.5,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"mem-46780047-63a0-4a86-a16b-638b72a7fb8d",
"kn-8e1bfb48-33a9-45ad-8da7-e0bdaa5d34e7",
"mem-cdff0c49-3ac7-4de8-89ec-92d254bd0023",
"bl-a7a1428f-db9c-417b-8e2c-713b1f84dc1f",
"mem-f823e835-313f-4282-b4b3-ce527ffc2f7a",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"?Z?.\f?0?]P?",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 2455.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.14285714285714285,
"precision@5": 0.0,
"mrr@10": 0.14285714285714285
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"art-92e1837c-5919-42d0-bbb0-4d924d7b2864",
"bl-e20944e5-eb16-4ab3-a84d-111e0fc817fa",
"bl-e20944e5-f4a6-44a0-91b1-73d04ebed120",
"mem-47f72b5b-6e8b-4293-94f1-350197b4809a",
"mem-e612f0aa-c2f2-4ee3-bbc7-af2dc826233b",
"mem-a5f04e52-91f8-41d2-af27-8bf803621758",
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"? ?}&?#??X\b",
"7?e?7???\f3?",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a"
],
"n_returned": 10,
"latency_ms": 2037.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.1
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"kn-82be4e41-96c5-4da3-85c0-cee10763d975",
"mem-ea487cb4-ed67-44ce-8402-b56bb28468d4",
"mem-23d22bc1-a097-446b-8f11-8aff099e0b76",
"bl-874d1c2b-c55b-4afb-9601-922a9297e859",
"bl-2dd8aaa1-b0de-4eac-b3c5-78951d240b60",
"bl-2694b588-a6e3-43de-861c-fa7b0ec7e7fd",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad"
],
"n_returned": 10,
"latency_ms": 4720.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"mem-0328c3cb-4550-4ce4-9284-152e832f08f6",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"bl-fd047ce9-ae21-4b3e-b3ab-ece0c9592f7f",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"bl-6172d035-dd94-4776-afdd-d8915f6fc375",
"bl-5bb8dedf-8498-4a9b-acdc-31cc9c738f2a",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 5078.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"bl-e5635b1a-c5d0-4caa-bcc8-6a726ea43685",
"bl-fc893be3-e6b4-4ef6-93b0-d54ca5f89083",
"tag-project-structure",
"tag-anthropic-contrast",
"tag-voice-training",
"tag-ebd",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"bl-9d53422d-b703-4f1d-860a-8598cb29b792",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 4118.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-d24fcce8-2b55-426f-867a-db3958a622d3",
"tag-phase-3",
"tag-identity-studio",
"tag-kids",
"tag-coexistence",
"tag-cultivated-general-intelligence",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"? ?}&?#??X\b",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"mem-cdff0c49-3ac7-4de8-89ec-92d254bd0023"
],
"n_returned": 10,
"latency_ms": 3241.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"bl-bd9fb314-e9d4-4b03-aef4-534dd57a2992",
"bl-b019ce7a-1b21-436e-812d-032f50c6c45f",
"bl-e98cdd4c-01b5-459e-9036-3578cd5d975a",
"bl-9ce4128a-9436-4b06-82bc-8a6faafa81e0",
"tag-stable-diffusion",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 4090.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"mem-e321e54e-8bb3-4596-b13d-bb093d6b149d",
"mem-e32ba5a7-c147-4dc0-9479-b720d768eda6",
"mem-c7a77457-478d-4eb0-a116-67205a0066a4",
"bl-1b20e9bc-eb37-4907-8d63-e311fd61eab8",
"bl-aa762207-920d-45ab-b2a3-2f8154d7ef9b",
"tag-misalignment",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3"
],
"n_returned": 10,
"latency_ms": 3671.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"kn-48a01973-a025-471d-950f-b93e6a426d82",
"bl-7f33f1bc-99fa-4906-889f-a42375beea20",
"mem-6f0b2b45-90c1-4356-ac01-3daac05b09c8",
"mem-ce5a2ffc-ad39-4728-9ac6-76fef507d5da",
"project-Stripe_Elements__not_hosted_checkout__Custom_URL__DAG_bundle_pricing__Stripe_Connect_80_20_",
"tag-provenance",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"art-e495c8c5-ad95-4b64-8771-f68aa4cfcd0a"
],
"n_returned": 10,
"latency_ms": 4674.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"kn-79056192-7de8-486c-9565-f128439a6fcb",
"bl-f6236350-f7b8-4f4f-a702-9eef2eb76e4b",
"mem-ef0091d8-1b65-431e-afa8-c6c4ee5779c9",
"mem-37b57f52-a29a-42cf-a07a-3c5f8a3598dd",
"mem-1f32f73a-952c-41bc-96dc-8b8b70d8a7c1",
"tag-__cultivation-metric____internal-state____dharma____evidence____novel-idea____gap-compression____values____microsoft__",
"? ?}&?#??X\b",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 3974.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"bl-57c5cf6b-81a5-4558-9902-5c02981fe273",
"tag-guilds",
"tag-finance",
"tag-temporal",
"tag-barkhausen",
"tag-ilogger",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 4851.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"project-Goal_setting__alignment__scoring__cadence__Attaches_to_any_imprint_",
"tag-turing-test",
"tag-sealed",
"tag-design-first",
"tag-part-5",
"tag-model",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 4505.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-9590ba23-bddb-43e8-a571-68a263c4c364",
"bl-a9e57bb2-00a1-4867-ab59-5d9271134b50",
"tag-performed-values",
"bl-c7793c4a-7630-47fc-a462-d23059087e80",
"tag-gateway_platform_neuron-technologies_go_proxy_llm",
"tag-fornax",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 4494.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"bl-7a13527b-3e0c-418a-9f37-88fd2152e5ce",
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"bl-5e390b10-8753-4f25-a1a5-b5dbbb002cbf",
"tag-ats",
"tag-data-model",
"tag-aggregation",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 3619.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"tag-memory-model",
"tag-storage",
"tag-ga4",
"tag-potions",
"tag-resonance",
"tag-neuron",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-ee615cdb-e599-423d-9a4d-977859390ed3"
],
"n_returned": 10,
"latency_ms": 4107.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"mem-927f41ab-8ede-4f58-acb3-995db16ac775",
"mem-dbe80bc2-c602-46b0-b4ea-dd222e52bcde",
"mem-82158b02-a180-435d-84f0-0b7ce37511b4",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"tag-upload-window",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 4231.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"mem-434be7c8-88cb-4039-b79a-1da4ac4de783",
"mem-481c769c-68cc-45c7-bc37-c0d9778fa648",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-9d8f3c5b-4bac-41ce-8ac4-44733f99d6c8",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 3208.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"project-Imprint__leadership_development__feedback_frameworks__performance__presence_",
"mem-5e7f6ddd-c818-4ad3-b564-54ae278e9976",
"bl-7fa1b1a8-b80a-4f28-b162-bfe73765b4f8",
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c",
"tag-sarah",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd"
],
"n_returned": 10,
"latency_ms": 2359.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"mem-265a7107-73b3-4410-9aff-43787d5f473b",
"mem-9d1bf963-1b40-4588-bdb3-0432646cc623",
"bl-0de4e61b-6562-49e5-b7df-ebb809a01723",
"mem-3987d374-3c48-4e8e-b06d-0c363f55ed9c",
"bl-e0a0df72-de6e-46ab-800b-e1e3e8dfc387",
"tag-__patents____swarm____claim-language____prior-art____filing__",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 2342.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-dca14c4c-4859-47b0-996e-33964ba61a87",
"bl-e148d23c-24e8-4122-9915-d1c11f22052f",
"bl-31abf75b-998f-4a4f-a6dd-8204119e0451",
"bl-dc8c7e02-eb37-48ae-a6f8-9b512803ae16",
"mem-8d1bafe6-209c-456c-9a25-9a927bc5a16d",
"bl-14883d81-f7cb-46dd-82c2-a6e6980264e5",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"?Q??m?;`'",
"R^??m?;?'"
],
"n_returned": 10,
"latency_ms": 3536.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"bl-ffa22d7e-42a9-4bd2-a428-1d2df243ac93",
"bl-452a4710-3d2b-4e0f-9413-49a66423bc9a",
"bl-4a6746e8-191f-48fc-8bfb-c4dc73b80bcd",
"tag-command-pattern",
"tag-withholding",
"tag-offline",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 4151.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 1342.7,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 874.8,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"kn-333542cb-6dab-4662-9725-bf7440d28bf7",
"? ?}&?#??X\b",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4"
],
"n_returned": 8,
"latency_ms": 1374.6,
"error": null,
"clean": false,
"false_positives": 8
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"ctx-e5427d7d",
"project-worldweaver",
"tag-__kotlin____internal-state____pre-reasoning____post-reasoning____compression-ratio____dharma____cultivation__",
"tag-import",
"tag-enterprise",
"tag-__cgi____dharma____cultivation____five-primitives____seed-artifact____agi____intelligence____whitepaper____patent__",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 3762.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"mem-bfe0fafd-2750-4fdc-b773-04e878b3b23f",
"art-7bdaff30-5af9-4f0a-93b1-751686f9de3d",
"mem-cf07910d-4676-4384-ab97-9cad946cd0b9",
"project-Convert_UTC_timestamps_to_Central_time_when_displaying_to_Will__Never_surface_raw_UTC_",
"mem-32203649-3213-4d6d-86fd-3d657ac70d77",
"mem-22f5f665-3ad2-4063-88b0-915849a795f5",
"? ?}&?#??X\b",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-80ca3d31-84dc-4502-83f4-538372b9764f",
"015644f5-8194-4af0-800d-dd4a0cd71396"
],
"n_returned": 10,
"latency_ms": 4859.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"project-Imprint__discovery__objection_handling__deal_strategy__pipeline__closing_",
"bl-8116da7a-b039-4e08-b8d0-c1c7861f9766",
"bl-8ef1ba6b-3fa0-4dbd-98c5-31665e5694a1",
"tag-dark-theme",
"tag-divisors",
"tag-distressed-property",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"???Ͼd??W\b?"
],
"n_returned": 10,
"latency_ms": 2873.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": null
}
]
}
-956
View File
@@ -1,956 +0,0 @@
{
"label": "hybrid-semantic-r2",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-hybrid",
"soul_md5": "5cd2718932c2ecf4940ddfe5a9c8abbe",
"corpus": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/corpus-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7895,
"wall_clock_s": 52.1,
"child_pid": 86164,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.5142857142857142,
"recall@5": 0.4409013605442177,
"recall@10": 0.5047619047619047,
"precision@5": 0.15428571428571433,
"mrr@10": 0.38746031746031745,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1225.4,
"latency_ms_p95": 1671.6,
"latency_ms_max": 1718.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.38461538461538464,
"recall@5": 0.38461538461538464,
"recall@10": 0.38461538461538464,
"mrr@10": 0.17307692307692307
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.6634920634920636
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.2222222222222222,
"outranks": 2
}
}
},
"rows": [
{
"id": "q01",
"category": "exact_rare",
"query": "unjailbreakable",
"returned": [
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 307.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q02",
"category": "exact_rare",
"query": "engram-migrate",
"returned": [
"mem-6fdf6545-5e1a-43a9-8bdc-d2cd248146a5",
"mem-22fe5ec8-ae0d-4583-a05c-d1ef50353257",
"project-engram",
"project-engram-lang",
"mem-60778715-758c-4677-933d-fc39b8f94152",
"ctx-89a2",
"bl-13babd0c-582e-4e28-a9e4-a77e65925e5d",
"870ede67-3454-4e00-9988-46cb13a8a4e2",
"bl-3e433255-3710-49fc-a093-c25e71de2ccb",
"mem-235a7657-d49e-467e-9f69-f4c3d5f6bd48"
],
"n_returned": 10,
"latency_ms": 349.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q03",
"category": "exact_rare",
"query": "cartabandonedevent",
"returned": [
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 316.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q04",
"category": "exact_rare",
"query": "pre-apprenticeship",
"returned": [
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 315.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q05",
"category": "exact_rare",
"query": "inferencenodemanager",
"returned": [
"mem-73969486-143f-4431-b5e6-6845d1cc9848",
"bl-c1765767-3e27-449a-8c94-10411d1eb7c0",
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 326.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q06",
"category": "exact_rare",
"query": "clear-eyed",
"returned": [
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 304.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q07",
"category": "phrase",
"query": "patterns not returns",
"returned": [
"mem-a4a9dfc3-e40b-49b3-b1e1-060e8be2f482",
"????7???Ջ3",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"?ǚ?7??????",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"art-d24fd6dd-2cda-4eed-92f3-67b535a0d71b",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 593.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 1.0
},
{
"id": "q08",
"category": "phrase",
"query": "thirty moves",
"returned": [
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"Z[?<S???H??",
"rQ??m?;?x?'",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 534.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3125,
"recall@10": 0.5,
"precision@5": 1.0,
"mrr@10": 1.0
},
{
"id": "q09",
"category": "phrase",
"query": "Grandma Lucas",
"returned": [
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"art-79042b8b-6192-440f-90b0-60708f7e6325"
],
"n_returned": 10,
"latency_ms": 544.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3333333333333333,
"recall@10": 0.5555555555555556,
"precision@5": 0.6,
"mrr@10": 1.0
},
{
"id": "q10",
"category": "phrase",
"query": "Directed Harmonic",
"returned": [
"ԍ????X????",
"project-harmonic-framework",
"?ǚ?7??????",
"project-harmonic-framework_com",
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"??????X??2c",
"bl-dcee1887-34c4-4ffa-9119-1e291685ba08",
"????7???Ջ3",
"mem-7eeacad7-d7c2-4c2b-8348-19a59aa6dbaf",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356"
],
"n_returned": 10,
"latency_ms": 524.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.1111111111111111,
"precision@5": 0.0,
"mrr@10": 0.1111111111111111
},
{
"id": "q11",
"category": "phrase",
"query": "Sarah Bishop",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"? ?}&?#??X\b",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"mem-a9a9ce95-0d64-46eb-9db8-ff81d78ade35",
"mem-b8ecd23e-77ce-42f7-984c-f51453fec16d"
],
"n_returned": 10,
"latency_ms": 543.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.5,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.2
},
{
"id": "q12",
"category": "phrase",
"query": "Directed Autonomous Runtime Modification",
"returned": [
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?",
"?of?7???",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 905.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.2857142857142857,
"recall@10": 0.5,
"precision@5": 0.8,
"mrr@10": 1.0
},
{
"id": "q13",
"category": "phrase",
"query": "zero-knowledge encrypted backup",
"returned": [
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"8f3abb0d-77ed-4af3-9f4d-ba62cd198886",
"deda48cd-5e1a-46cb-bd43-8016afdb3a8a",
"?",
"mem-dba009a2-d2ea-4f5a-b9e8-0f04bc9ab32f",
"?",
"mem-7cd90611-88a3-423d-a38a-0db2812952fa",
"?",
"bl-07375bf9-a169-42cd-adb3-7d32b25982f0",
"?"
],
"n_returned": 10,
"latency_ms": 769.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.3333333333333333
},
{
"id": "q14",
"category": "paraphrase",
"query": "the elderly relative who passed while he stayed away",
"returned": [
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"????7???Ջ3",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c"
],
"n_returned": 10,
"latency_ms": 1645.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q15",
"category": "paraphrase",
"query": "a soldier sidelined by illness who refused to quit",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef",
"art-2f29ad36-6ee6-4a0e-8d72-0eaf7d12d3a9",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-a31e1001-342e-4deb-a2e6-6d02d1f22dee",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1718.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q16",
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"7?e?7???\f3?",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"?of?7???",
"? ?}&?#??X\b",
"????7???Ջ3",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-d97920d0-1649-4223-9508-c0bb621e7fc0",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e"
],
"n_returned": 10,
"latency_ms": 1393.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q17",
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491",
"7?e?7???\f3?",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"??o?'?B???k",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1123.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q18",
"category": "paraphrase",
"query": "if you are able and nobody is coming the job is yours",
"returned": [
"7?e?7???\f3?",
"?of?7???",
"imp-dce1da0f-8776-4a9e-972b-33411a7ca138",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"kn-c2205725-69d0-4dd1-9a8d-1c7fa9a0c7b4",
"kn-83bb86c6-521d-416c-a86e-6e29c2d8f102",
"? ?}&?#??X\b",
"? ?}&?#??X\b"
],
"n_returned": 9,
"latency_ms": 1448.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q19",
"category": "paraphrase",
"query": "learning is the wealth creditors cannot seize",
"returned": [
"knw-e24d6339-5ff3-4bed-ba53-707ffd0dc70a",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"7?e?7???\f3?",
"kn-0625e393-067c-4bba-8389-7e1b79265142",
"??o?'?B???k",
"kn-150e6790-fc2a-48a7-8289-313c1fbaf5ae",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1298.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q20",
"category": "paraphrase",
"query": "reliability proven by track record not assertion",
"returned": [
"? ?}&?#??X\b",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"? ?}&?#??X\b",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"? ?}&?#??X\b",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"7?e?7???\f3?"
],
"n_returned": 10,
"latency_ms": 1621.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q21",
"category": "paraphrase",
"query": "boundaries that enable instead of confine",
"returned": [
"? ?}&?#??X\b",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"? ?}&?#??X\b",
"project-Source_kn-6f248a50__Add_containment_rules__convergence__location-independence__failure_modes_",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"ԍ????X????",
"??????X??2c",
"dR????X?-?S"
],
"n_returned": 10,
"latency_ms": 1405.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5
},
{
"id": "q22",
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"7?e?7???\f3?",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"kn-f8974b26-78a6-4aad-b893-19a73b20013d",
"?of?7???",
"??????X??2c",
"dz????Xƹ?i",
"ԍ????X????",
"dR????X?-?S",
"kn-d7c1e0fb-fa59-46d3-b4c9-a0d1d437a491"
],
"n_returned": 10,
"latency_ms": 1671.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q23",
"category": "paraphrase",
"query": "a mind that compounds instead of resetting each day",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"????7???Ջ3",
"a1000001-0000-0000-0000-000000000001",
"?ǚ?7??????",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"7?e?7???\f3?",
"mem-ade9440f-f161-4c18-9b35-1976257e6ebb",
"?of?7???",
"ea95f600-8dfd-4c7e-b077-a93dc3cd3623"
],
"n_returned": 10,
"latency_ms": 1519.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q24",
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"mem-bbb126a1-b297-42bb-86be-796871829c94",
"mem-45022957-2d78-48aa-a714-16d6eca52e0f",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-0f0277a1-4a8e-4645-95dd-fa379976f31c",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"'?T?a\"B~-?8"
],
"n_returned": 10,
"latency_ms": 1576.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q25",
"category": "paraphrase",
"query": "cheerfulness you arrive at instead of assuming",
"returned": [
"?ǚ?7??????",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"7?e?7???\f3?",
"??S?7???",
"??f?7???",
"kn-e8423822-eacf-4029-aa7b-10d4d28d621e",
"??S?7???",
"??S?7???",
"? ?}&?#??X\b"
],
"n_returned": 10,
"latency_ms": 1287.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q26",
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"7?e?7???\f3?",
"mem-7b74cac0-905f-4c35-9688-fbcce105a177"
],
"n_returned": 10,
"latency_ms": 1391.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.25
},
{
"id": "q27",
"category": "associative",
"query": "Grandma Lucas stroke February 2006 goodbye window",
"returned": [
"? ?}&?#??X\b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"?ǚ?7??????"
],
"n_returned": 10,
"latency_ms": 1498.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q28",
"category": "associative",
"query": "Marines hernia sepsis medical ward",
"returned": [
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-33ecccc2-e37f-43db-91b3-c2a86f08aaac",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"art-e0bdf5d8-d163-491f-b649-453fee8b721d",
"7?e?7???\f3?",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-9397c74b-35f3-4428-b4b0-5123353bbcd1",
"?of?7???"
],
"n_returned": 10,
"latency_ms": 1169.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q29",
"category": "associative",
"query": "Sarah Bishop Dyer trailer performance",
"returned": [
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"7?e?7???\f3?",
"?of?7???",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee"
],
"n_returned": 10,
"latency_ms": 1239.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q30",
"category": "associative",
"query": "Swarm Architecture containment lateral worker",
"returned": [
"art-ee615cdb-e599-423d-9a4d-977859390ed3",
"bl-9bde67c1-f0ba-4c3a-8fe5-de0deee0ce43",
"kn-b36902cc-0b05-44ba-9aa7-800e5dea9ca9",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-6f248a50-355b-47bb-aec8-e0e646a9b077",
"bl-0fac287f-f4c0-4f15-bc4d-ff7f8a7af3ae",
"bl-8c2d5f51-3ccd-4c2e-848a-eb60d90a3b98",
"7?e?7???\f3?",
"kn-a5b3d0ac-f6a1-49a4-aebb-b8b4cd67fe83",
"bl-2121fdb9-796a-427e-b9b5-651f4388ea16"
],
"n_returned": 10,
"latency_ms": 1225.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q31",
"category": "associative",
"query": "hope won inside the narrative preface",
"returned": [
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"knw-9e74ee95-ba7d-49b1-9262-977eae9729d1",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"7?e?7???\f3?",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"rQ??m?;?x?'"
],
"n_returned": 9,
"latency_ms": 1237.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q32",
"category": "associative",
"query": "man of the house six years old expectation",
"returned": [
"? ?}&?#??X\b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"? ?}&?#??X\b",
"kn-57b4c5e7-40c6-4c90-bf14-71841b0081d4",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"art-2fabd873-d787-49cb-ad30-d4ed9fcff8ef"
],
"n_returned": 10,
"latency_ms": 1458.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 756.3,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 519.6,
"error": null,
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"?V?",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"?m?\\}Q??6??",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"?m?\\}Q??6??",
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"?m?\\}Q??6??"
],
"n_returned": 10,
"latency_ms": 774.0,
"error": null,
"clean": false,
"false_positives": 10
},
{
"id": "q36",
"category": "superseded",
"query": "is the self-improvement architecture called DARMA or DHARMA",
"returned": [
"7?e?7???\f3?",
"mem-80d7416b-20e9-48a0-b176-b215527e2f56",
"? ?}&?#??X\b",
"mem-f3b37427-b7d1-4f7e-b32c-0241a20ce8da",
"? ?}&?#??X\b",
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"?of?7???",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"? ?}&?#??X\b",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff"
],
"n_returned": 10,
"latency_ms": 1340.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"precision@5": 0.2,
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 10
},
{
"id": "q37",
"category": "superseded",
"query": "how many provisional patents does Will actually have",
"returned": [
"? ?}&?#??X\b",
"12082f7e-e320-438b-bd65-083d8259748f",
"? ?}&?#??X\b",
"527ecb25-2587-47eb-8269-73be2431abd4",
"? ?}&?#??X\b",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"? ?}&?#??X\b",
"4f698ae6-c40e-464e-9798-50350991a188",
"? ?}&?#??X\b",
"be3b6036-6eca-44a7-8fdf-37b23edfdfd1"
],
"n_returned": 10,
"latency_ms": 1693.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 1.0,
"precision@5": 0.0,
"mrr@10": 0.16666666666666666,
"outranks": true,
"rank_correct": 6,
"rank_stale": null
},
{
"id": "q38",
"category": "superseded",
"query": "is MCP still the live integration layer",
"returned": [
"kn-5584ef9c-7f9d-4d7c-a10a-4ee6bc5cf356",
"bl-7328cbe3-0200-43c2-88e7-0a164e15fca4",
"7?e?7???\f3?",
"mem-101e81b4-8097-4749-8d8d-7bb66de34517",
"? ?}&?#??X\b",
"4509ed62-9fb2-48b8-9038-ac569fca9604",
"%???2??jH??",
"art-8a0870d5-a716-4672-8094-f7463af1265b",
"???Ͼd??W\b?",
"bl-556438af-57b2-4bd8-a747-9f868aaee290"
],
"n_returned": 10,
"latency_ms": 1027.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"precision@5": 0.0,
"mrr@10": 0.0,
"outranks": false,
"rank_correct": null,
"rank_stale": 4
}
]
}

Some files were not shown because too many files have changed in this diff Show More