Compare commits

..

2 Commits

Author SHA1 Message Date
Tim Lingo 21d3516426 measure: claim-24 unfloored semantic leg vs bm25lex baseline - net +0, NOT-SHOWN
gains q14,q25 (gold at global cosine rank 1, previously discarded by the 0.60
floor); losses q15,q28 (the semantic leg was EMPTY on those queries under the
floor, so filling it turns a 2-leg rotation into a 3-leg one and halves the
associative leg's share of the top 5). Guards held: exact_rare 6/6, phrase 7/7,
nonsense 2/3, superseded 2/3 outranks. recall@10 61.8->65.4pp, latency 0.99x.
Baseline reproduced from source (results-bm25base-rerun.json is byte-identical
to the committed results-bm25lex.json), candidate deterministic across 2 runs.
2026-08-07 16:33:50 -05:00
Tim Lingo 65c50073b8 feat(engram): restore claim 24 verbatim - unfloored semantic leg + corpus-vocabulary gate
The read-path semantic leg was gated at ENGRAM_EMBED_SEED_MIN (0.60) and
rescaled onto [0.60,1]. That constant is defined as the HippoRAG SEED-JOIN
threshold; claim 24 authorises a ranking with no threshold at all. Measured:
the floor is not a quality gate (paraphrase targets 0.459-0.657 vs nonsense
nearest neighbours 0.553-0.622 - overlapping distributions). What holds the
nonsense controls is corpus vocabulary, so the floor is replaced by an
explicit nhits==0 gate: no record contains any query token -> return nothing.
2026-08-07 16:23:32 -05:00
48 changed files with 1672 additions and 21010 deletions
-10
View File
@@ -63,16 +63,6 @@ jobs:
cp vendor/el-runtime/v1.0.0-20260501/el_runtime.h /opt/el/runtime/el_runtime.h
echo "El runtime PINNED to v1.0.0-20260501: $(ls /opt/el/runtime/)"
# neuron#133: CI compiles dist/soul.c, NOT the .el sources. On 2026-08-07 a
# build off main would have shipped an engine with none of five merged fixes,
# including a P0 safety fix, while main's source read as correct. The runner
# cannot regenerate the amalgam (elc needs 24GB+ virtual memory), but it can
# refuse to compile a stale one. Fails loudly with the recipe in the message.
- name: Verify dist/soul.c matches the sources
run: |
chmod +x tools/soulc-stamp.sh
./tools/soulc-stamp.sh --check
- name: Build neuron soul binary
run: |
RUNTIME=/opt/el/runtime
-19
View File
@@ -889,29 +889,10 @@ fn awareness_run() -> Void {
state_set("soul.last_beat_ts", int_to_str(now_ts))
// Persist in-process Engram (sessions, memories, conversation nodes)
// to local snapshot so they survive restarts.
// FILE MODE ONLY: "soul_snapshot_path" is set exclusively in the
// genesis+safe_to_seed branch of soul.el, and safe_to_seed is
// unconditionally false when ENGRAM_URL is set. In HTTP mode the
// owner persists; the soul must not (soul.el:571-573).
let snap_path: String = state_get("soul_snapshot_path")
if !str_eq(snap_path, "") {
mem_save(snap_path)
}
// WRITE-THROUGH RETRY (neuron#117). The HTTP-mode counterpart of the
// save above: hand anything still spooled to the persistence owner.
//
// This is the retry arm of the whole design. Deltas that could not be
// pushed owner down, owner restarting, transient refusal stay on
// disk and are re-offered here every heartbeat until they land. It is
// also the catch-all for writes made by the awareness loop itself,
// which never passes through the HTTP handler's flush point.
//
// No-op with no HTTP call when the spool is empty or ENGRAM_URL is
// unset, so an idle soul in file mode pays nothing for this.
let wt_pushed: Int = wt_drain()
if wt_pushed < 0 {
ise_post("{\"event\":\"write_through_backlog\",\"ts\":" + int_to_str(now_ts) + "}")
}
}
// Curiosity scan: idle-gated AND wall-clock based. Only fires when the
+8 -8
View File
@@ -1162,7 +1162,7 @@ fn hist_trim_with_bell_guard(hist: String) -> String {
+ " | evicted_at:" + ts_str
+ " | message:" + safe_content
let preserve_tags: String = "[\"bell-history\",\"bell:" + bell_level + "\",\"evicted\",\"affective\",\"BellEvent\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
preserve_content,
"BellEvent",
"bell:" + bell_level + ":preserved",
@@ -1210,7 +1210,7 @@ fn conv_history_persist(session_id: String, hist: String) -> Void {
if !str_contains(hist, "]") { return "" }
let tags: String = "[\"conv-history\",\"persistent\"]"
// FIX B: one label rule, shared with the agentic path. See conv_hist_label.
let node_id: String = wt_node(
let node_id: String = engram_node_full(
hist, "Conversation", conv_hist_label(session_id),
el_from_float(0.7), el_from_float(0.8), el_from_float(0.9),
"Episodic", tags
@@ -3461,7 +3461,7 @@ fn handle_dharma_room_turn(body: String) -> String {
// engram_node(content, "episodic", ...) which wrongly put a TIER into the node_type
// slot that's why nodes showed node_type="episodic". Use the full, correct contract.)
let utterance_tags: String = "[\"soul-utterance\",\"episodic\"]"
let discard_id: String = wt_node(
let discard_id: String = engram_node_full(
clean_response, "Conversation", "soul:utterance",
el_from_float(0.6), el_from_float(0.6), el_from_float(0.8),
"Episodic", utterance_tags
@@ -3552,7 +3552,7 @@ fn session_summary_write(summary_text: String) -> String {
}
}
let tags: String = "[\"SessionSummary\",\"session-summary\",\"previous-session\",\"consolidate\"]"
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content, "SessionSummary", "session:summary",
el_from_float(0.85), el_from_float(0.85), el_from_float(1.0),
"Episodic", tags
@@ -3578,7 +3578,7 @@ fn session_summary_write_dated(summary_text: String, label: String) -> String {
let ts_str: String = int_to_str(ts)
let content: String = "[session-summary] " + trimmed + " | ts:" + ts_str
let tags: String = "[\"SessionSummary\",\"session-summary\",\"previous-session\",\"consolidate\"]"
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content, "SessionSummary", label,
el_from_float(0.9), el_from_float(0.8), el_from_float(1.0),
"Episodic", tags
@@ -3654,7 +3654,7 @@ fn auto_persist(req: String, resp: String) -> Void {
+ ",\"bell\":\"" + bell_level + "\""
+ ",\"label\":\"chat:" + ts_str + "\"}"
let conv_node_id: String = wt_node(
let conv_node_id: String = engram_node_full(
content,
"Conversation",
"chat:" + ts_str,
@@ -3692,7 +3692,7 @@ fn auto_persist(req: String, resp: String) -> Void {
let bell_tags: String = "[\"safety\",\"bell\",\"bell:" + bell_level + "\",\"affective\",\"BellEvent\"]"
let bell_ts_str: String = int_to_str(time_now())
let bell_label: String = "bell:" + bell_level + ":" + bell_ts_str
let bell_node_id: String = wt_node(
let bell_node_id: String = engram_node_full(
bell_content,
"BellEvent",
bell_label,
@@ -3751,7 +3751,7 @@ fn auto_persist(req: String, resp: String) -> Void {
let pos_tags: String = "[\"joy\",\"positive\",\"joy:" + positive_level + "\",\"affective\",\"PositiveEvent\"]"
let pos_ts_label: String = int_to_str(time_now())
let pos_label: String = "joy:" + positive_level + ":" + pos_ts_label
let pos_node_id: String = wt_node(
let pos_node_id: String = engram_node_full(
pos_content, "PositiveEvent", pos_label,
pos_sal_a, pos_sal_b, pos_sal_c, "Episodic", pos_tags
)
Generated Vendored
+665 -1596
View File
File diff suppressed because one or more lines are too long
Generated Vendored
-19
View File
@@ -1,19 +0,0 @@
# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from.
# Written by tools/soulc-stamp.sh --write. Do not hand-edit.
# generated_amalgam_sha256 cdc5e716dbfb797faa1b3e080cbd1ac82a75a258809da70cc5fbd02cc8040692
# generated_amalgam_bytes 1205007
7cf5e29d2618db2fca04e6df7aa8954dd6cf9ac5e70aafb8e0b52aa734882131 __compiler__
f8597e10546654bce3fbbe40461b2da59d0e06dbf1b038d1d362d24f949e3911 awareness.el
b6f3d14ca0c26017a2d617399a6d3754dabb0905e4d5f52eb75d25c4ad18d3c5 chat.el
42288c212cbf72fb1e8ecbd4d9900e4e9ee1cfa475b7974295c7637f1bf2939f elp-input.el
b3f77f49d6086932c38bd17fe7a5eaf8bce25685f6fc3e1750f05729c6b49b9e imprint.el
fba8ffdb9ba72bca5b09ca1c93a520edc52f3f4d8aec2c7585fe9b17e06420b2 manifest.el
550a72e234ae8cec1f33e02108fd365353f45edd88513da90b792e79b6c0e5f0 memory.el
5ec07ec9785b02abe32f3ff7acf2d1f9f7e07c0967fac97e6eff17d7110b5c84 neuron-api.el
03c47c451e0e87f2c252cadb4b765867943962a804f548dd53adeef0520912c8 persist.el
a6d69f3fc55233d9d3300160fd46a1551f2064bcd0fb84e2c9e432f636a72476 routes.el
c28e36952ec56525963a0bdf29455ab097d3b0c5653d19c25fbb005e1069a1f7 safety.el
fd3ab91d0ae0ea26639e21bef2f8f94054dc4b02eae68b19e3fe689d2769aad4 sessions.el
5613b60d74d5d7768f46da5ac435a5dd99d38c27f0f7013c89fa27e98dc8a21c soul.el
30337940905171a9645b0929f0a412ce6b3dccb1246495070c553bca0bbae6cd stewardship.el
95dab72be4ee1dd1d28bab63412964a72460126951764e3f74b1c2d49b6d7b35 studio.el
+2 -11
View File
@@ -110,7 +110,7 @@ tool("beginSession", "Initialize session: surface recent high-importance memorie
"," + tool("linkCausal", "Create a causal edge (cause -> effect).") +
"," + tool("restructureCausalGraph", "Re-balance the causal subgraph after new evidence.") +
"," + tool("rebuildGraph", "Rebuild graph indices from the on-disk snapshot.") +
"," + tool("runStructuralAudit", "Stage 1 structural audit: owner-vs-runtime divergence, orphans and dangling edges, typed-edge distribution, self-model connectivity. Returns an annotated characterization, not a score.") +
"," + tool("runStructuralAudit", "Audit graph structure for orphans, dangling edges, mislabeled types.") +
// Backlog + work
"," + tool("planWork", "Create a backlog item.") +
"," + tool("reviewBacklog", "Browse work items.") +
@@ -680,16 +680,7 @@ fn dispatch_tool_call(tool_name: String, args: String) -> String {
return mcp_json_result(resp)
}
if str_eq(tool_name, "runStructuralAudit") {
// Was: GET /session/begin an unrelated session digest returned under an
// audit tool name, i.e. the tool advertised a check that did not exist.
// Now points at the real Stage 1 route (neuron-api.el
// handle_api_structural_audit). Sample caps ride the query string; the
// defaults keep a manual audit to a couple of seconds.
let e_s: Int = json_get_int(args, "edge_sample")
let n_s: Int = json_get_int(args, "node_sample")
let qs: String = "?edge_sample=" + int_to_str(if e_s > 0 { e_s } else { 3000 })
+ "&node_sample=" + int_to_str(if n_s > 0 { n_s } else { 300 })
let resp: String = http_get(neuron_url() + "/audit/structural" + qs)
let resp: String = http_get(neuron_url() + "/session/begin")
return mcp_json_result(resp)
}
+9 -104
View File
@@ -1,91 +1,9 @@
import "persist.el"
fn tier_working() -> String { return "Working" }
fn tier_episodic() -> String { return "Episodic" }
fn tier_canonical() -> String { return "Canonical" }
// Association on write
// DESIGN: "promotion integrates candidate nodes by linking them to existing nodes
// using typed semantic edges RATHER THAN APPENDING AS UNLINKED CONTENT" (CCR
// claim 29). Unlinked append is the explicitly rejected behaviour and it is the
// only behaviour this system had. Measured 2026-08-09 on Tim's graph: 14,214 edges
// across 80,936 nodes, 5% of nodes connected to anything, and NO edge created by
// any write since 2026-07-19 while 27,000+ nodes were added. A memory that forms
// no connections cannot be reached by spreading activation, so retrieval silently
// degrades to literal matching.
//
// BOUNDS, each one bought with a specific failure:
// * max 3 edges per memory link_memories.py's cap, precision over spray
// * never link to identity (self/*, Value): the existing policy is explicit that
// "memories must not pollute the self traversal by similarity; only an explicit
// citation may touch identity". Similarity is not citation.
// * never link telemetry (state-event, soul-response, boot_count, loop-outcome):
// these are ~97% of daily write volume (1,020 vs 31 real memories on 08-08).
// Linking them would add ~3,000 noise edges a day and re-flatten the graph in
// the name of connecting it.
// * fail-soft: a failed association never fails the write.
// Edges go through wt_edge so they reach the owner and survive restart.
fn mem_assoc_skip_label(label: String) -> Bool {
if str_contains(label, "state-event") { return true }
if str_contains(label, "soul-response") { return true }
if str_contains(label, "soul-outbox") { return true }
if str_contains(label, "boot_count") { return true }
if str_contains(label, "loop-outcome") { return true }
if str_contains(label, "search-result") { return true }
return false
}
// A candidate is linkable only if it is a real, distinct, non-identity node.
fn mem_assoc_ok(cand_id: String, cand_label: String, self_id: String) -> Bool {
if str_eq(cand_id, "") { return false }
if str_eq(cand_id, self_id) { return false }
// CASE MATTERS measured 2026-08-09. A lowercase-only check let a memory link
// to "Self — Values (grounded)", i.e. it polluted the self traversal, which is
// the one thing this policy exists to prevent. My verification had the same
// blind spot and printed PASS. Check every casing the graph actually uses, and
// exclude identity node TYPES as well as labels.
let lab: String = str_lower(cand_label)
if str_starts_with(lab, "self") { return false }
if str_starts_with(lab, "value") { return false }
if str_contains(lab, "values") { return false }
if str_contains(lab, "identity") { return false }
if mem_assoc_skip_label(cand_label) { return false }
return true
}
// One slot of the association. Manual unroll rather than a loop: EL's codegen
// mis-emits accumulating while-loops (documented at soul.el:212, which unrolled
// three affective slots for the same reason).
fn mem_assoc_slot(results: String, idx: Int, new_id: String) -> Void {
if idx >= json_array_len(results) { return }
let cand: String = json_array_get(results, idx)
let cid: String = json_get(cand, "id")
let clabel: String = json_get(cand, "label")
let ctype: String = json_get(cand, "node_type")
if str_eq(ctype, "Value") { return }
if str_eq(ctype, "DharmaSelf") { return }
if str_eq(ctype, "Safety") { return }
if mem_assoc_ok(cid, clabel, new_id) {
wt_edge(new_id, cid, el_from_float(0.5), "related")
}
}
// mem_associate connect a freshly written memory to what it is about.
fn mem_associate(new_id: String, content: String, label: String) -> Void {
if str_eq(new_id, "") { return }
if mem_assoc_skip_label(label) { return }
// Ask the graph what this memory resembles. Now that the store carries
// meaning-vectors this is semantic, not merely lexical.
let probe: String = str_slice(content, 0, 400)
let results: String = engram_recall_json(probe, 4)
if str_eq(results, "") { return }
mem_assoc_slot(results, 0, new_id)
mem_assoc_slot(results, 1, new_id)
mem_assoc_slot(results, 2, new_id)
}
fn mem_store(content: String, label: String, tags: String) -> String {
let id: String = wt_node(
let id: String = engram_node_full(
content,
"Memory",
label,
@@ -99,26 +17,13 @@ fn mem_store(content: String, label: String, tags: String) -> String {
println("[memory] write rejected by engram (empty id): label=" + label)
return ""
}
// wt_node has already read the node back locally and returns "" if it did
// not land, so the old duplicate read-back here is gone.
//
// HONESTY (neuron#117): the receipt now says WHERE the write is.
// The old unconditional "write verified" line asserted against the soul's
// own RAM true in memory, false on disk and printed ~115,000 times on
// Tim's machine while the canonical snapshot sat frozen for three days.
// wt_commit flushes the spool and then asks the OWNER. When it says false
// the node is real and recallable but not yet durable, and the log says so
// rather than claiming a save that did not happen. The id is still returned:
// the local write DID succeed, and the queued delta will be retried.
let durable: Bool = wt_commit(id)
// Associate AFTER the node is durable: an edge to a node that did not persist
// is a dangling edge, which is the defect the 2026-08-09 cleanup removed 830 of.
mem_associate(id, content, label)
if durable {
println("[memory] write persisted at owner: " + id + " label=" + label)
} else {
println("[memory] write IN MEMORY ONLY (queued for owner, not yet durable): " + id + " label=" + label)
// Read back to verify the node actually persisted guards against silent write failures.
let readback: String = engram_get_node_json(id)
if str_eq(readback, "") || str_eq(readback, "{}") {
println("[memory] WRITE VERIFY FAILED: label=" + label + " id=" + id + " — node absent after write")
return ""
}
println("[memory] write verified: " + id + " ok")
return id
}
@@ -146,12 +51,12 @@ fn mem_strengthen(node_id: String) -> Void {
// memory.el (imported first) so awareness.el and neuron-api.el can both call it.
fn mem_tombstone(node_id: String) -> String {
let tags: String = "[\"Tombstone\",\"status:deleted\"]"
let marker: String = wt_node(
let marker: String = engram_node_full(
node_id, "Tombstone", "tombstone:" + node_id,
el_from_float(0.01), el_from_float(0.01), el_from_float(1.0),
"Episodic", tags)
if !str_eq(marker, "") {
wt_edge(marker, node_id, el_from_float(1.0), "tombstones")
engram_connect(marker, node_id, el_from_float(1.0), "tombstones")
}
return marker
}
+29 -534
View File
@@ -195,24 +195,15 @@ fn api_compact_activated(raw: String, max_items: Int, snip: Int) -> String {
}
// api_persisted read-back-after-write guard against hallucinated saves.
//
// WIDENED FOR neuron#117. This function is the single gate every MCP write
// handler passes through before it reports success (10 call sites), which makes
// it the right place to close the honesty gap rather than editing ten receipts.
//
// It used to read back from engram_get_node_json the SOUL'S OWN in-process
// graph. In HTTP-engram mode that asserts the wrong thing: the soul is not the
// persistence owner, so a node present in its RAM and absent from the owner read
// as "persisted" and then vanished on the next restart. The guard was doing
// exactly what its comment promised and still certifying writes that did not
// survive. It now flushes the write-through spool and asks the OWNER.
//
// In file mode (no ENGRAM_URL) the soul IS the owner and wt_commit collapses to
// the original local read-back unchanged behaviour, which is what keeps this
// reversible.
// After a write builtin returns an id, confirm the node is actually queryable
// via engram_get_node_json(id) (returns "" or "null" when missing). Returns
// true only when the node is genuinely persisted.
fn api_persisted(id: String) -> Bool {
if str_eq(id, "") { return false }
return wt_commit(id)
let node: String = engram_get_node_json(id)
// engram_get_node_json returns "{}" (empty object) when node is not found not "" or "null".
// Check all three to guard against any runtime variation.
return !str_eq(node, "") && !str_eq(node, "null") && !str_eq(node, "{}")
}
// api_not_persisted standard error for a write that did not read back.
@@ -304,35 +295,10 @@ fn handle_api_begin_session(body: String) -> String {
let state_events: String = api_compact_node_array(state_events_raw, 5, 500)
let recent_raw: String = engram_scan_nodes_json(10, 0)
let recent: String = api_compact_node_array(recent_raw, 10, 240)
// SELF-SEEDED SLICE (2026-08-09). The design is explicit: "Every compilation
// query begins at the self-model node and traverses outward... structural
// reachability from the self-model node is a precondition for any node to
// appear in compiled context" (will-anderson patents/drafts/engram-claims.md,
// Self-Seeded Activation; DRAFT, not a filed provisional — cite it as such).
//
// Measured 2026-08-09 before this change: compiled context contained 0-1
// identity records out of 10, because compilation seeds from a hardcoded
// TEXT STRING, never from the self. Even an explicit "my values identity who
// I am" query returned a boot counter and state-events.
//
// This restores the designed behaviour WITHOUT repeating the failure that got
// self_neighbors set to [] in the first place: that was an UNBOUNDED ~90KB
// neighbour dump which closed the socket on every call. Same bound as every
// other list here cap 8, 240-char snippets. The self root has 34 direct
// neighbours of which 23 are identity records, so depth 1 is dense enough to
// be worth seeding and small enough to stay cheap.
let self_raw: String = engram_neighbors_json("kn-efeb4a5b-5aff-4759-8a97-7233099be6ee", 1, "both")
// Cap 24, not 8: measured 2026-08-09, the self root's first 8 neighbours are
// TAG nodes ("neuron", "tier:note", "disposition:experimental", "imprint",
// "traversal") which crowd out the substantive identity records behind them.
// The root has 34 neighbours of which 23 are identity; 24 captures them while
// staying bounded. Cost measured at ~+4KB on a ~12KB response, nowhere near
// the ~90KB unbounded dump that closed sockets and got this set to [].
let self_slice: String = api_compact_node_array(self_raw, 24, 240)
return "{\"stats\":" + stats
+ ",\"recent\":" + recent
+ ",\"activated\":" + activated
+ ",\"self_neighbors\":" + self_slice
+ ",\"self_neighbors\":[]"
+ ",\"recent_state_events\":" + state_events + "}"
}
@@ -376,17 +342,10 @@ fn handle_api_remember(body: String) -> String {
let inner: String = str_slice(base_tags, 1, str_len(base_tags) - 1)
"[" + inner + ",\"project:" + project + "\"]"
}
let id: String = wt_node(content, "Memory", "memory:remembered",
let id: String = engram_node_full(content, "Memory", "memory:remembered",
sal, sal, el_from_float(0.9),
"Episodic", final_tags)
if !api_persisted(id) { return api_not_persisted(id) }
// Associate on write (2026-08-09). THIS CALL MUST BE HERE, not only in mem_store.
// The HTTP memory route writes via wt_node directly; mem_store serves only the
// awareness paths (soul-response, search-result, activation-result) which are
// exactly the telemetry we refuse to link. Hooking mem_store alone produced
// ZERO edges across four real writes measured, not assumed, which is the only
// reason it was caught before shipping.
mem_associate(id, content, "memory:remembered")
return "{\"id\":\"" + id + "\",\"ok\":true}"
}
@@ -410,7 +369,7 @@ fn handle_api_node_create(body: String) -> String {
if str_eq(importance, "low") { 0.25 } else { 0.5 }
}
}
let id: String = wt_node(content, node_type, label,
let id: String = engram_node_full(content, node_type, label,
sal, sal, el_from_float(0.9),
tier, tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -463,11 +422,11 @@ fn handle_api_node_update(body: String) -> String {
}
let body_tags: String = json_get(body, "tags")
let tags: String = if str_eq(body_tags, "") { "[\"" + node_type + "\"]" } else { body_tags }
let new_id: String = wt_node(content, node_type, label,
let new_id: String = engram_node_full(content, node_type, label,
el_from_float(0.5), el_from_float(0.5), el_from_float(0.8),
tier, tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
wt_edge(new_id, id, el_from_float(0.9), "supersedes")
engram_connect(new_id, id, el_from_float(0.9), "supersedes")
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + id + "\",\"ok\":true}"
}
@@ -491,12 +450,7 @@ fn handle_api_recall(method: String, path: String, body: String) -> String {
if str_eq(eff_q, "") {
return api_or_empty(engram_scan_nodes_json(limit, 0))
}
// engram_recall_json, not engram_search_json: this route IS the retrieval
// surface (claim 24's "embedding search queries"), so it gets the semantic
// and associative legs. engram_search_json stays lexical because ~40
// internal call sites pass a KEY and seven of them delete every record
// that comes back see the boundary note above eg_search_json_impl.
let results: String = engram_recall_json(eff_q, limit)
let results: String = engram_search_json(eff_q, limit)
return api_or_empty(results)
}
@@ -544,7 +498,7 @@ fn handle_api_capture_knowledge(body: String) -> String {
let full: String = if str_eq(title, "") { content } else { title + ": " + content }
let lbl: String = str_slice(title, 0, 80)
let tags: String = "[\"Knowledge\",\"captured\"]"
let id: String = wt_node(full, "Knowledge", lbl,
let id: String = engram_node_full(full, "Knowledge", lbl,
el_from_float(0.85), el_from_float(0.8), el_from_float(0.9),
"Episodic", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -559,12 +513,12 @@ fn handle_api_evolve_knowledge(body: String) -> String {
if !str_eq(prior_id, "") && is_protected_node(prior_id) { return api_err_protected(prior_id) }
let tags: String = "[\"Knowledge\",\"evolved\"]"
// Empty label engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
let new_id: String = wt_node(content, "Knowledge", "",
let new_id: String = engram_node_full(content, "Knowledge", "",
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
"Episodic", tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
if !str_eq(prior_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
}
@@ -581,11 +535,11 @@ fn handle_api_promote_knowledge(body: String) -> String {
"[\"Knowledge\",\"tier:canonical\",\"disposition:stable\"]"
} else { tags_raw }
// Empty label engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
let new_id: String = wt_node(content, "Knowledge", "",
let new_id: String = engram_node_full(content, "Knowledge", "",
el_from_float(0.9), el_from_float(0.9), el_from_float(1.0),
"Canonical", tags)
if !api_persisted(new_id) { return api_not_persisted(new_id) }
wt_edge(new_id, prior_id, el_from_float(0.95), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.95), "supersedes")
return "{\"ok\":true,\"new_id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\"}"
}
@@ -608,7 +562,7 @@ fn handle_api_define_process(body: String) -> String {
if str_eq(content, "") { return api_err("content is required") }
let label: String = if str_eq(name, "") { "process:unnamed" } else { "process:" + name }
let tags: String = "[\"Process\"]"
let id: String = wt_node(content, "Process", label,
let id: String = engram_node_full(content, "Process", label,
el_from_float(0.8), el_from_float(0.8), el_from_float(0.9),
"Canonical", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -693,7 +647,7 @@ fn handle_api_tune_config(body: String) -> String {
if str_eq(key, "") { return api_err("key is required") }
let content: String = "config:" + key + "=" + value
let tags: String = "[\"ConfigEntry\",\"config\"]"
let id: String = wt_node(content, "ConfigEntry", key,
let id: String = engram_node_full(content, "ConfigEntry", key,
el_from_float(0.85), el_from_float(0.85), el_from_float(0.9),
"Canonical", tags)
if !api_persisted(id) { return api_not_persisted(id) }
@@ -740,7 +694,7 @@ fn handle_api_link_entities(body: String) -> String {
if is_protected_node(to_id) { return api_err_protected(to_id) }
let relation: String = json_get(body, "relation")
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\"}"
}
@@ -773,11 +727,11 @@ fn handle_api_evolve_memory(body: String) -> String {
}
}
let tags: String = "[\"Memory\",\"evolved\"]"
let new_id: String = wt_node(content, "Memory", "memory:evolved",
let new_id: String = engram_node_full(content, "Memory", "memory:evolved",
sal, sal, el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
}
@@ -835,11 +789,11 @@ fn handle_api_cultivate(body: String) -> String {
let content: String = json_get(body, "content")
if str_eq(content, "") { return api_err("content is required") }
let tags: String = "[\"Knowledge\",\"evolved\",\"cultivated\"]"
let new_id: String = wt_node(content, "Knowledge", "knowledge:cultivated",
let new_id: String = engram_node_full(content, "Knowledge", "knowledge:cultivated",
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
}
@@ -855,11 +809,11 @@ fn handle_api_cultivate(body: String) -> String {
}
}
let tags: String = "[\"Memory\",\"evolved\",\"cultivated\"]"
let new_id: String = wt_node(content, "Memory", "memory:cultivated",
let new_id: String = engram_node_full(content, "Memory", "memory:cultivated",
sal, sal, el_from_float(0.9),
"Episodic", tags)
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
}
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
}
@@ -879,7 +833,7 @@ fn handle_api_cultivate(body: String) -> String {
if str_eq(to_id, "") { return api_err("to_id is required") }
let relation: String = json_get(body, "relation")
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\",\"cultivated\":true}"
}
@@ -914,7 +868,7 @@ fn handle_api_consolidate(body: String) -> String {
if !str_eq(summary, "") {
let safe_summary: String = str_replace(summary, "\"", "'")
let tags: String = "[\"SessionSummary\",\"consolidate\"]"
let summary_id: String = wt_node(
let summary_id: String = engram_node_full(
"[session-summary] " + safe_summary,
"SessionSummary", "session:summary",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
@@ -926,462 +880,3 @@ fn handle_api_consolidate(body: String) -> String {
}
return "{\"ok\":true,\"snapshot\":\"" + snap + "\"}"
}
// Stage 1: structural audit
//
// WHAT THIS IMPLEMENTS
// The CGI provisional, 05-detailed-description.md, "Stage 1: Structural audit
// 430". Verbatim, the audit module evaluates: the density and typed
// distribution of causal edges; the consistency between value nodes and
// execution-record neighborhoods; the richness and connectivity of the
// self-model; and the authenticity of open-question nodes in the wonder
// manifest. It "produces a coherence assessment 432 — NOT A BINARY SCORE but
// an annotated characterization of the graph's structural properties".
//
// That last clause is the whole shape of this handler. Every finding carries
// its own numbers AND a plain-language note saying what the numbers mean and
// how they were obtained. There is no pass/fail, no percentage-of-health, no
// composite score, and `"score":null` is emitted explicitly so a downstream
// reader cannot mistake its absence for an omission.
//
// WHY IT EXISTS NOW, AND WHY THE FIRST FINDING IS THE ONE IT IS
// `runStructuralAudit` has been an advertised MCP tool with nothing behind it:
// the dispatcher GET'd /session/begin and returned that blob (mcp-wrapper/src/
// main.el). Meanwhile the failure the audit would have caught ran silently for
// about three weeks the soul reported 103,089 nodes while the engram, which
// OWNS persistence, held ~79,900; a crash discarded the difference. Every boot
// reported green throughout, because nothing in the system ever compared the
// two sides. So finding 1 is owner-versus-runtime divergence: it is the check
// whose absence cost real memory, and it is cheap and exact.
//
// WHAT IS DELIBERATELY NOT HERE (stage 1b, see the `deferred` array in the
// response): value/execution-record consistency and wonder-manifest
// authenticity. Both need node types that barely exist in this graph today
// the response MEASURES those populations and reports the counts as the reason,
// rather than asserting a deferral without evidence.
//
// MEASUREMENT HONESTY: EXACT WHERE CHEAP, SAMPLED WHERE NOT, ALWAYS LABELLED
// Counts, edge typing and self-model connectivity are exact. Orphan rate and
// dangling-edge rate are SAMPLED, because the engram runtime has no node-id
// index `engram_find_node_index` is a linear scan over every node, so an
// exhaustive dangling check is O(nodes x edges) (~2.2e9 string compares at
// today's scale, tens of seconds inside one request). The samples are UNIFORM
// across the whole population, not head-of-list, and every sampled figure is
// emitted with its own `sampled` / `population` fields plus an extrapolation
// labelled as such. Raise `?edge_sample=` / `?node_sample=` to the population
// size to run either check exhaustively and pay the time. The real fix is an
// id index in the runtime; that is the engram repo's, not this handler's.
// audit_pct1 one-decimal percentage as a bare JSON number, sign-safe.
// Integer math only: EL has no fixed-precision formatter, and float_to_str
// would put an unbounded mantissa in the response.
fn audit_pct1(num: Int, den: Int) -> String {
if den <= 0 { return "null" }
let neg: Bool = num < 0
let a: Int = if neg { 0 - num } else { num }
let tenths: Int = (a * 1000) / den
let whole: Int = tenths / 10
let frac: Int = tenths - (whole * 10)
let sign: String = if neg { "-" } else { "" }
return sign + int_to_str(whole) + "." + int_to_str(frac)
}
// audit_finding the one envelope every finding uses: name, the measurements,
// and the annotation. Keeping it in one place is what stops the characterization
// from degenerating into a bag of numbers with no reading attached.
fn audit_finding(name: String, measured: String, note: String) -> String {
return "{\"finding\":\"" + name + "\""
+ ",\"measured\":{" + measured + "}"
+ ",\"note\":\"" + api_json_escape(note) + "\"}"
}
// audit_str_at read the quoted string value starting at byte `start`.
// Slices a bounded window rather than the tail of the (multi-MB) edges array, so
// this is O(window) per call instead of O(remaining input).
fn audit_str_at(s: String, start: Int, maxlen: Int) -> String {
let n: Int = str_len(s)
if start < 0 || start >= n { return "" }
let end_guess: Int = start + maxlen
let stop: Int = if end_guess > n { n } else { end_guess }
let win: String = str_slice(s, start, stop)
let q: Int = str_index_of(win, "\"")
if q < 0 { return "" }
return str_slice(win, 0, q)
}
// audit_rel_count exact count of edges carrying `rel`, by scanning the emitted
// edge array for the literal `"relation":"<rel>"`. engram_emit_edge_json writes
// metadata ESCAPED as a string, so no nested object can contain that literal and
// the count cannot be inflated by edge payloads.
fn audit_rel_count(edges: String, rel: String) -> Int {
return str_count(edges, "\"relation\":\"" + rel + "\"")
}
// audit_owner_stats ask the persistence OWNER for its own counts.
// Returns "" when there is no HTTP owner configured or the owner is unreachable;
// both are reported as findings, never as a failure of the audit.
fn audit_owner_stats(url: String) -> String {
if str_eq(url, "") { return "" }
return http_get(url + "/api/stats")
}
// audit_divergence FINDING 1. Runtime (this soul's in-process graph) versus
// the persistence owner's own count. Trend is measured against the previous
// audit recorded in soul state, so a second call answers "is the gap growing?"
// rather than just restating it.
fn audit_divergence() -> String {
let rt_nodes: Int = engram_node_count()
let rt_edges: Int = engram_edge_count()
let url: String = wt_engram_url()
if str_eq(url, "") {
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"none\",\"owner_reachable\":false",
"No HTTP persistence owner is configured, so this soul IS the owner "
+ "(file mode) and divergence is not defined. This check only has "
+ "meaning when ENGRAM_URL points at a separate engram that owns the "
+ "canonical store.")
}
let stats: String = audit_owner_stats(url)
// REACHABILITY IS PROVED BY THE PAYLOAD, NOT BY A NON-EMPTY REPLY.
// http_get does not return "" on a connection failure it returns a JSON
// error object ({"error":"Failed to connect to ... Couldn't connect to
// server"}). Testing only for "" made a DEAD owner read as reachable with
// node_count 0, i.e. the audit would have reported a 100% divergence and
// named it as data loss. That false positive is worse than no check at all:
// it is precisely the kind of confident wrong answer this route exists to
// stop. Require the field the contract promises.
let owner_nc_raw: String = json_get_raw(stats, "node_count")
if str_eq(stats, "") || str_eq(owner_nc_raw, "") {
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":false"
+ ",\"owner_reply\":\"" + api_json_escape(api_utf8_trunc(stats, 200)) + "\"",
"The persistence owner at " + url + " did not return a node_count "
+ "from GET /api/stats. Divergence is UNKNOWN, NOT ZERO — an owner "
+ "that cannot be read is exactly the condition under which the "
+ "runtime's own count means least, and reporting 0 for the owner "
+ "would manufacture a total-loss reading out of a network error. "
+ "Reported as a finding rather than raised as an error so the rest "
+ "of the audit still returns; the owner's raw reply is in "
+ "owner_reply.")
}
let ow_nodes: Int = json_get_int(stats, "node_count")
let ow_edges: Int = json_get_int(stats, "edge_count")
let d_nodes: Int = rt_nodes - ow_nodes
let d_edges: Int = rt_edges - ow_edges
// Trend against the previous audit in this soul's state.
let prev_raw: String = state_get("audit_prev_node_delta")
let prev: Int = str_to_int(prev_raw)
let abs_now: Int = if d_nodes < 0 { 0 - d_nodes } else { d_nodes }
let abs_prev: Int = if prev < 0 { 0 - prev } else { prev }
let trend: String = if str_eq(prev_raw, "") {
"no_prior_audit"
} else {
if abs_now > abs_prev { "growing" } else {
if abs_now < abs_prev { "shrinking" } else { "flat" }
}
}
state_set("audit_prev_node_delta", int_to_str(d_nodes))
state_set("audit_prev_ts", int_to_str(time_now()))
let note_head: String = if d_nodes == 0 {
"Runtime and owner agree on node count."
} else {
"Runtime holds " + int_to_str(d_nodes) + " nodes (" + audit_pct1(d_nodes, rt_nodes)
+ "% of its own graph) that the persistence owner does not report. Nodes "
+ "that exist only in runtime memory do not survive a restart."
}
return audit_finding("owner_runtime_divergence",
"\"runtime_nodes\":" + int_to_str(rt_nodes)
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":true"
+ ",\"owner_nodes\":" + int_to_str(ow_nodes)
+ ",\"owner_edges\":" + int_to_str(ow_edges)
+ ",\"node_delta\":" + int_to_str(d_nodes)
+ ",\"edge_delta\":" + int_to_str(d_edges)
+ ",\"node_delta_pct_of_runtime\":" + audit_pct1(d_nodes, rt_nodes)
+ ",\"trend_vs_previous_audit\":\"" + trend + "\""
+ ",\"previous_node_delta\":" + (if str_eq(prev_raw, "") { "null" } else { int_to_str(prev) }),
note_head + " Trend against the previous audit recorded in this soul's "
+ "state: " + trend + ". This is the comparison whose absence let a "
+ "~24,000-node loss run for weeks with every boot reporting green.")
}
// audit_edge_typing FINDING 2. Density plus the typed distribution the patent
// asks for, against the claim-10 relation vocabulary. Exact: str_count over the
// emitted edge array, one linear pass per relation.
fn audit_edge_typing(edges: String, total_edges: Int, node_total: Int) -> String {
let c_sup: Int = audit_rel_count(edges, "Supersedes")
let c_cau: Int = audit_rel_count(edges, "Causes")
let c_con: Int = audit_rel_count(edges, "Contains")
let c_ref: Int = audit_rel_count(edges, "References")
let c_ctr: Int = audit_rel_count(edges, "Contradicts")
let c_exe: Int = audit_rel_count(edges, "Exemplifies")
let c_act: Int = audit_rel_count(edges, "Activates")
let c_tmp: Int = audit_rel_count(edges, "TemporallyPrecedes")
let typed: Int = c_sup + c_cau + c_con + c_ref + c_ctr + c_exe + c_act + c_tmp
// Lowercase near-misses: the same eight concepts written by the ad-hoc write
// paths (linkEntities defaults to "associates", linkCausal to "causes").
// Counted separately because "the vocabulary is unused" and "the vocabulary
// is used in the wrong case" are different defects with different fixes.
let l_sup: Int = audit_rel_count(edges, "supersedes")
let l_cau: Int = audit_rel_count(edges, "causes")
let l_con: Int = audit_rel_count(edges, "contains")
let l_ref: Int = audit_rel_count(edges, "references")
let l_ctr: Int = audit_rel_count(edges, "contradicts")
let l_exe: Int = audit_rel_count(edges, "exemplifies")
let l_act: Int = audit_rel_count(edges, "activates")
let l_tmp: Int = audit_rel_count(edges, "temporallyPrecedes")
let near: Int = l_sup + l_cau + l_con + l_ref + l_ctr + l_exe + l_act + l_tmp
let untyped: Int = total_edges - typed
return audit_finding("typed_edge_distribution",
"\"total_edges\":" + int_to_str(total_edges)
+ ",\"total_nodes\":" + int_to_str(node_total)
// Density per 100 nodes, not per node: EL has no fixed-precision float
// formatter, and "0.3 edges per node" rounded to an integer is a lie.
+ ",\"edges_per_100_nodes\":" + audit_pct1(total_edges, node_total)
+ ",\"claim10_typed\":" + int_to_str(typed)
+ ",\"claim10_typed_pct\":" + audit_pct1(typed, total_edges)
+ ",\"outside_claim10_vocabulary\":" + int_to_str(untyped)
+ ",\"lowercase_near_miss\":" + int_to_str(near)
+ ",\"by_relation\":{"
+ "\"Supersedes\":" + int_to_str(c_sup)
+ ",\"Causes\":" + int_to_str(c_cau)
+ ",\"Contains\":" + int_to_str(c_con)
+ ",\"References\":" + int_to_str(c_ref)
+ ",\"Contradicts\":" + int_to_str(c_ctr)
+ ",\"Exemplifies\":" + int_to_str(c_exe)
+ ",\"Activates\":" + int_to_str(c_act)
+ ",\"TemporallyPrecedes\":" + int_to_str(c_tmp) + "}",
"Only " + int_to_str(typed) + " of " + int_to_str(total_edges)
+ " edges use the claim-10 causal vocabulary; the remainder are ad-hoc "
+ "relation strings, which is why the graph's causal claims cannot yet "
+ "be checked for internal consistency — an untyped edge asserts "
+ "association, not causation. " + int_to_str(near) + " edges use a "
+ "lowercase spelling of a claim-10 relation: those are near-misses the "
+ "write paths could be corrected to emit, not genuinely foreign types.")
}
// audit_orphans_dangling FINDING 3. Both figures are SAMPLED; see the header
// for why exhaustive is O(nodes x edges) on this runtime.
//
// An "orphan" here is a node with zero RESOLVABLE edges: engram_neighbors_json
// drops any edge whose other endpoint does not resolve to a node, so a node
// whose only edges are dangling reads as an orphan. That is the right reading
// such a node is unreachable by traversal but it is stated rather than hidden.
fn audit_orphans_dangling(edges: String, total_edges: Int, node_total: Int,
edge_cap: Int, node_cap: Int) -> String {
// orphan sample: uniform stride over the node store
let n_take: Int = if node_total < node_cap { node_total } else { node_cap }
let n_stride: Int = if n_take > 0 { node_total / n_take } else { 1 }
let n_stride = if n_stride < 1 { 1 } else { n_stride }
let orphans: Int = 0
let n_checked: Int = 0
let j: Int = 0
while j < n_take {
let one: String = engram_scan_nodes_json(1, j * n_stride)
let nid: String = json_get(json_array_get(one, 0), "id")
if !str_eq(nid, "") {
let nbrs: String = engram_neighbors_json(nid, 1, "both")
let deg: Int = json_array_len(nbrs)
let orphans = if deg == 0 { orphans + 1 } else { orphans }
let n_checked = n_checked + 1
}
let j = j + 1
}
// dangling sample: uniform stride over the edge array
// str_index_of_all gives every edge's field offsets in ONE linear pass, so
// any index can be read in O(1). json_array_get would have been O(i) per
// element and O(n^2) over the array.
let from_pos: [Int] = str_index_of_all(edges, "\"from_id\":\"")
let to_pos: [Int] = str_index_of_all(edges, "\"to_id\":\"")
let nf: Int = len(from_pos)
let nt: Int = len(to_pos)
let ne: Int = if nf < nt { nf } else { nt }
let e_take: Int = if ne < edge_cap { ne } else { edge_cap }
let e_stride: Int = if e_take > 0 { ne / e_take } else { 1 }
let e_stride = if e_stride < 1 { 1 } else { e_stride }
let dangling: Int = 0
let e_checked: Int = 0
let i: Int = 0
while i < ne && e_checked < e_take {
let fid: String = audit_str_at(edges, get(from_pos, i) + 11, 96)
let tid: String = audit_str_at(edges, get(to_pos, i) + 9, 96)
let f_gone: Bool = str_eq(engram_get_node_json(fid), "{}")
let t_gone: Bool = if f_gone { true } else { str_eq(engram_get_node_json(tid), "{}") }
let dangling = if f_gone || t_gone { dangling + 1 } else { dangling }
let e_checked = e_checked + 1
let i = i + e_stride
}
let orphan_est: Int = if n_checked > 0 { (orphans * node_total) / n_checked } else { 0 }
let dangle_est: Int = if e_checked > 0 { (dangling * total_edges) / e_checked } else { 0 }
let exhaustive_n: String = if n_checked >= node_total { "true" } else { "false" }
let exhaustive_e: String = if e_checked >= ne { "true" } else { "false" }
return audit_finding("orphans_and_dangling_edges",
"\"nodes_population\":" + int_to_str(node_total)
+ ",\"nodes_sampled\":" + int_to_str(n_checked)
+ ",\"nodes_sample_exhaustive\":" + exhaustive_n
+ ",\"orphans_in_sample\":" + int_to_str(orphans)
+ ",\"orphan_rate_pct\":" + audit_pct1(orphans, n_checked)
+ ",\"orphans_extrapolated\":" + int_to_str(orphan_est)
+ ",\"edges_population\":" + int_to_str(total_edges)
+ ",\"edges_sampled\":" + int_to_str(e_checked)
+ ",\"edges_sample_exhaustive\":" + exhaustive_e
+ ",\"dangling_in_sample\":" + int_to_str(dangling)
+ ",\"dangling_rate_pct\":" + audit_pct1(dangling, e_checked)
+ ",\"dangling_extrapolated\":" + int_to_str(dangle_est),
"Orphan = zero RESOLVABLE edges, so a node whose only edges dangle counts "
+ "as an orphan; either way it is unreachable by traversal. Dangling = an "
+ "edge with an endpoint id that resolves to no node. Both are uniform "
+ "stride samples over the whole population, not the head of the list; "
+ "the extrapolations are estimates and are labelled as such. Pass "
+ "?node_sample= / ?edge_sample= at or above the population size to run "
+ "either check exhaustively. A high orphan rate is a characterization, "
+ "not a verdict: an accumulating store legitimately holds unlinked "
+ "material. It becomes a defect when the write paths were SUPPOSED to "
+ "link and did not.")
}
// audit_pillar one self-model pillar: present, how much content, how connected.
fn audit_pillar(key: String, id: String) -> String {
let node: String = engram_get_node_json(id)
let present: Bool = !str_eq(node, "{}") && !str_eq(node, "")
if !present {
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":false"
+ ",\"content_length\":0,\"degree\":0}"
}
let content: String = json_get(node, "content")
let deg: Int = json_array_len(engram_neighbors_json(id, 1, "both"))
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":true"
+ ",\"label\":\"" + api_json_escape(json_get(node, "label")) + "\""
+ ",\"tier\":\"" + api_json_escape(json_get(node, "tier")) + "\""
+ ",\"content_length\":" + int_to_str(str_len(content))
+ ",\"degree\":" + int_to_str(deg) + "}"
}
// audit_self_model FINDING 4. "the richness and connectivity of the
// self-model ... is it connected to behavioral evidence?"
//
// This finding RETIRES the Claude-side vitals identity block. That check lived
// outside the system it was checking a shell script grepping a snapshot so
// it could only ever report on a file, and it went on reporting green while the
// memory-philosophy pillar was absent from the live graph for about three weeks.
// Asking the running soul about its own three pillars is the designed mechanism;
// a shell probe was the fourth patch on the same hole.
fn audit_self_model() -> String {
let dna: String = audit_pillar("intellectual_dna", "kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6")
let val: String = audit_pillar("values_hub", "kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
let phi: String = audit_pillar("memory_philosophy", "kn-dcfe04b3-3702-4cac-b6f0-ecb4db837eee")
let root: String = audit_pillar("self_root", "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
return audit_finding("self_model_connectivity",
"\"pillars\":{" + dna + "," + val + "," + phi + "," + root + "}",
"The three identity pillars plus the self root. `degree` counts nodes "
+ "reachable in one hop in either direction — the self-model's connection "
+ "to the rest of the graph. present:false on any pillar is the condition "
+ "that ran undetected for weeks; content_length distinguishes a pillar "
+ "that is present from one that is present but hollowed out. The patent "
+ "also asks whether the self-model makes ACCURATE PREDICTIONS about the "
+ "system's own behavior; that half needs Prediction nodes and is deferred "
+ "with the rest of stage 1b below.")
}
// audit_deferred what stage 1 does NOT yet evaluate, with the measured reason.
// Emitted as data, not as a comment, so a reader of the assessment sees the gap
// and its evidence rather than inferring completeness from silence.
fn audit_deferred() -> String {
let preds: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("Prediction", 50, 0)))
let wonders: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("WonderQuestion", 50, 0)))
return "[{\"deferred\":\"value_execution_record_consistency\""
+ ",\"stage\":\"1b\""
+ ",\"measured\":{\"prediction_nodes_found\":" + int_to_str(preds) + "}"
+ ",\"reason\":\"" + api_json_escape(
"The patent asks whether the execution history SUPPORTS the stated "
+ "values or shows systematic conflict. That requires execution "
+ "records tied to value nodes and predictions to score them against. "
+ "Prediction nodes found (capped at 50): " + int_to_str(preds)
+ ". Asserting value/execution coherence on that population would be "
+ "a fabricated result, which is worse than a stated gap.") + "\"}"
+ ",{\"deferred\":\"wonder_manifest_authenticity\""
+ ",\"stage\":\"1b\""
+ ",\"measured\":{\"wonder_question_nodes_found\":" + int_to_str(wonders) + "}"
+ ",\"reason\":\"" + api_json_escape(
"The patent asks whether pull weights CORRELATE WITH GENUINE "
+ "PREDICTION UNCERTAINTY or are uniform/externally assigned — a "
+ "correlation between two populations. WonderQuestion nodes readable "
+ "by type (capped at 50): " + int_to_str(wonders) + ", against "
+ int_to_str(preds) + " Prediction nodes. There is a known write/read "
+ "node-type mismatch on the wonder path; until that is fixed and both "
+ "populations exist, any correlation reported here would be noise.") + "\"}]"
}
// handle_api_structural_audit Stage 1. Returns the coherence assessment 432:
// an annotated characterization, explicitly NOT a score.
//
// COST NOTE: the edge findings need the relation labels, and the runtime exposes
// no edge-enumeration builtin. The only way to see them is the same one
// GET /api/graph/edges already uses engram_save to a SCRATCH path (never the
// owner's canonical file; see routes.el, neuron#117) and read the array back.
// On a large graph that is a multi-hundred-MB write, so this is a manual audit
// route, not something to put on a timer. Pass ?edges=0 to skip both edge
// findings and get the divergence + self-model readings cheaply.
fn handle_api_structural_audit(method: String, path: String, body: String) -> String {
let node_total: Int = engram_node_count()
let edge_total: Int = engram_edge_count()
let want_edges: Bool = !str_eq(api_query_param(path, "edges"), "0")
let edge_cap: Int = api_query_int(path, "edge_sample", 3000)
let node_cap: Int = api_query_int(path, "node_sample", 300)
let divergence: String = audit_divergence()
let self_model: String = audit_self_model()
let edge_part: String = if want_edges {
// Scratch export only. state_get("soul_snapshot_path") is deliberately
// NOT used: in HTTP-engram mode the soul is not the persistence owner and
// must never write the canonical file, not even on a read path.
let scratch_dir: String = env("TMPDIR")
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
let snap_path: String = scratch_base + "/soul-audit-export-" + state_get("soul_cgi_id") + ".json"
// engram_save returns Int (1 ok / 0 fail); str_eq on it SIGSEGVs (#150).
let saved: Int = engram_save(snap_path)
if saved == 0 {
"," + audit_finding("typed_edge_distribution", "\"available\":false",
"Could not export the graph to " + snap_path + " for edge analysis, "
+ "so edge typing and the dangling-edge sample were not run. "
+ "Reported as a gap, not as zero findings.")
} else {
// wt_read, not fs_read: fs_read leaves a thread-local length hint that
// the NEXT HTTP response would use as its Content-Length, appending
// adjacent heap bytes to the reply (see persist.el wt_read).
let snap: String = wt_read(snap_path)
let edges_raw: String = json_get_raw(snap, "edges")
let edges: String = if str_eq(edges_raw, "") { "[]" } else { edges_raw }
"," + audit_edge_typing(edges, edge_total, node_total)
+ "," + audit_orphans_dangling(edges, edge_total, node_total, edge_cap, node_cap)
}
} else {
""
}
return "{\"audit\":\"structural\",\"stage\":1"
+ ",\"spec\":\"CGI provisional 05-detailed-description.md, Stage 1: Structural audit 430\""
+ ",\"assessment\":\"coherence_assessment_432\""
+ ",\"assessment_kind\":\"annotated_characterization\""
+ ",\"score\":null"
+ ",\"score_note\":\"By design. The specification calls for an annotated characterization of the graph's structural properties, not a binary score. Read the findings.\""
+ ",\"cgi_id\":\"" + api_json_escape(state_get("soul_cgi_id")) + "\""
+ ",\"ts_ms\":" + int_to_str(time_now())
+ ",\"findings\":[" + divergence + "," + self_model + edge_part + "]"
+ ",\"deferred\":" + audit_deferred() + "}"
}
-1
View File
@@ -37,4 +37,3 @@ extern fn handle_api_memory_update(body: String) -> String
extern fn handle_api_cultivate(body: String) -> String
extern fn handle_api_list_typed(node_type: String, path: String, body: String) -> String
extern fn handle_api_consolidate(body: String) -> String
extern fn handle_api_structural_audit(method: String, path: String, body: String) -> String
-426
View File
@@ -1,426 +0,0 @@
// persist.el the soulengram WRITE-THROUGH boundary (neuron#117).
//
// WHY THIS FILE EXISTS
// soul.el:571-573 states the ownership rule: "when ENGRAM_URL is set the HTTP
// Engram owns persistence the soul must NEVER write to the local snapshot
// (not the persistence owner)." The soul obeys the NEGATIVE half. The POSITIVE
// half how a write made inside the soul actually REACHES the owner was
// never built. Sync is pull-only (awareness.el `/api/sync` -> engram_load_merge),
// so every node the soul creates lives in its process RAM and is shed on
// restart. Measured live 2026-08-07: soul node_count=102184, engram
// node_count=79197 ~23k nodes existing nowhere but RAM.
//
// SCOPE NOTE ON THE PATENT (corrects an earlier internal reading)
// Engram provisional claims 15-18 describe a delta-sync protocol "with peer
// Engram instances"; claim 17's pull-then-push sequence is PEER-ENGRAM to
// PEER-ENGRAM. The soul is NOT a peer Engram it is a CALLER of the database
// system API (cf. claim 27, "invoked explicitly by a caller of the database
// system API"). So claim 17 does not specify a soul↔engram contract and is not
// cited as authority here. This design is derived from the ownership rule
// alone: the owner owns the writes, therefore the soul must HAND writes to the
// owner and must never write the owner's file itself.
//
// THE MECHANISM, AND WHY NOT `POST /api/nodes`
// The obvious route is the one the persona/boot-counter write-backs already
// use, POST /api/nodes. It is the wrong instrument here, verified against the
// live engram binary in a sandbox:
// - it mints a NEW server-side id (engram_node), so the soul's id and the
// owner's id diverge the next /api/sync pull re-imports the node as a
// DUPLICATE, and any edge referencing the soul's id never resolves;
// - it accepts only {content, node_type, salience} and drops label, tier,
// tags, importance, confidence, metadata. A probe posted with tier
// "Canonical" came back tier "Working", importance 0.5.
// POST /api/load-merge (Will's own route, el `dc39a61`) is the right one:
// - engram_load_merge PRESERVES the id and every field;
// - it dedups nodes by id and edges by (from_id,to_id,relation), so a
// re-submitted delta is a NO-OP retry safety is free, and it is the same
// local-wins semantics the graph already uses;
// - it calls persist_canonical() THE OWNER writes its own canonical file.
// The soul never touches it. The ownership rule is honoured in its
// strongest form rather than worked around;
// - it returns real counts {ok, nodes_added, edges_added, node_count},
// so a receipt can be a MEASUREMENT instead of a fixed success shape.
//
// SPOOL-AND-DRAIN, AND WHY IT IS NOT JUST A DIRECT POST
// Measured in a sandbox against a 79k-node / 176MB graph (live scale): one
// load-merge costs ~0.38s, essentially all of it the owner's persist_canonical.
// A chat turn writes 5-7 nodes; pushing each separately would add ~2.7s per
// turn. So writes are STAGED and pushed in one coalesced batch.
// The staging buffer is the FILESYSTEM, not process state, because the soul
// serves each HTTP connection on its own pthread (el_runtime http_serve_async)
// and a shared in-process buffer would lose entries to a read-modify-write
// race silently, which is the one failure mode this file exists to end.
// One file per write, named with uuid_v4, is race-free by construction and
// buys a property a memory buffer cannot: writes that could not be pushed
// SURVIVE A SOUL CRASH and are drained on the next boot.
//
// WHAT IS DELIBERATELY NOT PUSHED
// - InternalStateEvent / heartbeat telemetry. Will's own carve-out, stated in
// engram server.el 8f8ccc9: "48h-pruned, loss-tolerant, ~2/min; snapshotting
// 28MB per heartbeat is waste."
// NOTE (ours, flagged for Will): we do NOT additionally exclude Working-tier
// nodes. That exclusion exists in `fb0bb55` to stop the boot counter leaking
// through the /api/sync PULL; it is about sync backflow, not durability.
// Applying it here would exclude mem_store which writes tier "Working" and
// mem_store is the single most important durable write path in the soul. Boot
// seeding reads the canonical file wholesale, so a pushed Working-tier node
// does survive restart. This is the one classification call this file makes
// that Will has not ruled on.
//
// WHAT THIS BOUNDARY CANNOT EXPRESS (by construction, not by omission)
// - engram_strengthen (salience/activation drift): load-merge SKIPS ids that
// already exist, so it cannot update an existing node. There is no owner-side
// update/upsert route. Not pushable through any current route; left as a
// follow-up that needs a change in the engram repo.
// - engram_forget (hard delete): load-merge is additive and has no delete verb.
// Propagating deletes would mean DELETE /api/nodes/<id>, a HARD delete at the
// owner which scripts/verify-soul-contract.sh section B explicitly fails the
// build for ("to delete is to supersede/tombstone, never hard-remove"). Local
// deletes therefore stay local; the TOMBSTONE NODE and its "tombstones" edge
// (mem_tombstone) are pushed, and that is the sanctioned representation of a
// deletion in this graph.
// Configuration
// wt_engram_url same resolution order as ise_post: env, then the state key
// stashed at boot. NO hardcoded localhost fallback: unlike telemetry, inventing
// a destination for durable data would risk pushing a user's memories at whatever
// happens to be listening on 8742. Empty means "no HTTP owner" -> file mode.
fn wt_engram_url() -> String {
let env_url: String = env("ENGRAM_URL")
if !str_eq(env_url, "") { return env_url }
return state_get("soul_engram_url")
}
fn wt_api_key() -> String {
let env_key: String = env("ENGRAM_API_KEY")
if !str_eq(env_key, "") { return env_key }
return state_get("soul_engram_api_key")
}
// wt_enabled true only in HTTP-engram mode. In file mode the soul IS the
// persistence owner and every path below is a no-op, so this whole feature is
// inert for genesis/local deployments. That is also what makes it reversible.
fn wt_enabled() -> Bool {
return !str_eq(wt_engram_url(), "")
}
// wt_spool_dir where staged deltas live. MUST be readable by the engram
// process: /api/load-merge takes a PATH and the owner opens it itself. Both
// processes are same-host by construction (dev-stack LaunchAgents; the GKE
// image starts engram and soul in one container per entrypoint.sh).
fn wt_spool_dir() -> String {
let raw: String = env("SOUL_OUTBOX_DIR")
let dir: String = if str_eq(raw, "") { env("HOME") + "/.neuron/soul-outbox" } else { raw }
fs_mkdir(dir)
return dir
}
// Helpers
// wt_esc minimal JSON string escape. Deliberately local rather than reusing
// chat.el's json_safe: persist.el is imported BY memory.el, which is imported by
// chat.el, so depending on chat.el here would be an import cycle.
fn wt_esc(s: String) -> String {
let s1: String = str_replace(s, "\\", "\\\\")
let s2: String = str_replace(s1, "\"", "\\\"")
let s3: String = str_replace(s2, "\n", "\\n")
let s4: String = str_replace(s3, "\r", "\\r")
let s5: String = str_replace(s4, "\t", "\\t")
return s5
}
// wt_durable_class Will's telemetry carve-out, by node_type. See header.
fn wt_durable_class(node_type: String) -> Bool {
if str_eq(node_type, "InternalStateEvent") { return false }
return true
}
// wt_inner strip the surrounding brackets off a JSON array so several arrays
// can be concatenated into one. Returns "" for "[]" / "" / anything too short.
fn wt_inner(arr: String) -> String {
let n: Int = str_len(arr)
if n < 3 { return "" }
if !str_starts_with(arr, "[") { return "" }
return str_slice(arr, 1, n - 1)
}
// wt_read fs_read, plus a MANDATORY reset of the runtime's binary-length hint.
//
// THIS IS NOT OPTIONAL AND MUST NOT BE "SIMPLIFIED" BACK TO A BARE fs_read.
// The pinned runtime (vendor/el-runtime/v1.0.0-20260501) keeps a thread-local
// `_tl_fs_read_len` that fs_read SETS to the file's byte count (so binary files
// can be served with a correct Content-Length) and that http_send_response
// CONSUMES as the Content-Length of the next reply. Nothing else clears it
// except json_get_raw. So any fs_read during request handling that is not
// followed by a json_get_raw makes the NEXT HTTP response advertise the FILE's
// length instead of the body's and the runtime then sends that many bytes,
// appending whatever adjacent heap memory follows the reply.
//
// Caught here, measured: a /api/neuron/memory reply that should be 86 bytes went
// out as 497, with 411 bytes of this module's own spool paths and log strings
// trailing the JSON. The drain reads spool files mid-request, so this boundary
// is exactly where the landmine gets stepped on.
//
// Upstream el fixed the class in `43636ae` ("pair fs_read length hint with its
// buffer"); that runtime is NOT the one vendored here, and re-pinning the
// runtime is deliberately out of scope for this change. Clearing the hint at
// our own boundary fixes our exposure without touching the pinned C.
// json_get_raw is used as the reset because it is the only builtin in this
// runtime that zeroes the hint, and it does so before any early return.
fn wt_clear_binlen() -> Void {
let discard: String = json_get_raw("{}", "_wt_reset")
}
fn wt_read(path: String) -> String {
let data: String = fs_read(path)
wt_clear_binlen()
return data
}
// wt_sweep best-effort removal of the zero-byte husks left by truncation.
// The runtime exposes no unlink builtin, so a drained delta is emptied rather
// than deleted; this reclaims the directory entries.
//
// `-empty` is the safety property, not an optimisation: the command is
// STRUCTURALLY INCAPABLE of removing a delta that still has content, so it can
// never destroy a pending write even if it runs concurrently with a stage.
// Only the directory path is interpolated (never a filename), and it is quoted.
// The exit code is ignored an un-swept husk costs one directory entry.
fn wt_sweep(dir: String) -> Void {
if str_eq(dir, "") { return }
if str_contains(dir, "'") { return }
exec_command("find '" + dir + "' -maxdepth 1 -name 'wt*.json' -empty -delete 2>/dev/null")
}
// Staging
// wt_stage write ONE delta file. uuid_v4 in the name makes concurrent stagers
// collision-free without any lock. Returns true if the delta is on disk.
fn wt_stage(nodes_json: String, edges_json: String) -> Bool {
let dir: String = wt_spool_dir()
if str_eq(dir, "") { return false }
let payload: String = "{\"nodes\":" + nodes_json + ",\"edges\":" + edges_json + "}"
let path: String = dir + "/wt-" + uuid_v4() + ".json"
fs_write(path, payload)
// Read-back-verify the stage itself. A stage that did not land is a write we
// would otherwise believe was queued exactly the hallucinated-save class.
if str_eq(wt_read(path), "") {
println("[persist] wt_stage: FAILED to write spool file " + path + " — delta not queued")
return false
}
return true
}
// The write boundary
// wt_node create a node locally AND queue it for the persistence owner.
// Same signature and same return contract as engram_node_full ("" on failure),
// so converting a call site is a rename and nothing else.
fn wt_node(content: String, node_type: String, label: String,
salience: Float, importance: Float, confidence: Float,
tier: String, tags: String) -> String {
let id: String = engram_node_full(content, node_type, label,
salience, importance, confidence,
tier, tags)
if str_eq(id, "") { return "" }
// engram_get_node_json emits the SAME record shape engram_save writes (minus
// the embedding vector, which the owner backfills lazily), so the read-back
// doubles as the delta payload no second serialization to drift.
let rec: String = engram_get_node_json(id)
if str_eq(rec, "") || str_eq(rec, "{}") {
println("[persist] wt_node: local write did not read back, id=" + id + " label=" + label)
return ""
}
if wt_enabled() && wt_durable_class(node_type) {
wt_stage("[" + rec + "]", "[]")
}
return id
}
// wt_edge create an edge locally AND queue it. Mirrors engram_connect.
//
// The edge id is freshly generated rather than read back: the runtime exposes no
// "id of the edge I just created" accessor, and the owner dedups edges by
// (from_id,to_id,relation), never by id so the id is not load-bearing. The
// consequence, stated plainly: the soul's copy and the owner's copy of the same
// edge carry different edge ids. Nothing in either codebase looks an edge up by
// id (neighbors traversal scans from_id/to_id), so this is cosmetic.
fn wt_edge(from_id: String, to_id: String, weight: Float, relation: String) -> Void {
engram_connect(from_id, to_id, weight, relation)
if !wt_enabled() { return }
if str_eq(from_id, "") || str_eq(to_id, "") { return }
let ts: Int = time_now()
let rec: String = "{\"id\":\"" + uuid_v4() + "\""
+ ",\"from_id\":\"" + wt_esc(from_id) + "\""
+ ",\"to_id\":\"" + wt_esc(to_id) + "\""
+ ",\"relation\":\"" + wt_esc(relation) + "\""
+ ",\"metadata\":\"{}\""
+ ",\"weight\":" + float_to_str(weight)
+ ",\"confidence\":1"
+ ",\"created_at\":" + int_to_str(ts)
+ ",\"updated_at\":" + int_to_str(ts)
+ ",\"last_fired\":0,\"inhibitory\":0,\"layer_id\":1}"
wt_stage("[]", "[" + rec + "]")
}
// The drain
// wt_drain coalesce every staged delta into ONE load-merge against the owner.
//
// Returns: nodes_added on success (>= 0), 0 when there was nothing to do, and
// -1 when the push FAILED. -1 is load-bearing: on failure the spool files are
// left untouched, so nothing is lost and the next drain retries them. A caller
// must never read a non-negative return as "my particular node is durable"
// use wt_durable(id) for that.
//
// Concurrency: several threads may drain at once. Each builds its own batch file
// (uuid-named), and overlapping batches are harmless because load-merge dedups.
// Files are truncated ONLY after a confirmed ok:true, so a lost race costs a
// redundant push, never a dropped write.
fn wt_drain() -> Int {
if !wt_enabled() { return 0 }
let dir: String = wt_spool_dir()
if str_eq(dir, "") { return 0 }
// el_list_len/el_list_get, NOT json_stringify(fs_list(...)): fs_list builds
// a native list via el_list_append, and json_stringify does not serialize
// that type it renders the raw pointer value. (Verified in isolation; the
// same latent defect is live in studio.el's /api/tools/file/list route,
// which returns e.g. {"entries":4386409744}. Noted, not fixed here.)
let listing = fs_list(dir)
let count: Int = el_list_len(listing)
if count == 0 { return 0 }
let nodes_acc: String = ""
let edges_acc: String = ""
let drained: String = ""
let found: Int = 0
let i: Int = 0
// No `continue` / `break`: elc lists them as keywords but not one line of
// the shipped soul uses either, so they are unexercised on this build path.
// Guard conditions are expressed as nested ifs instead, and every rebind is
// at the loop-body top level where `let x = ...` is assignment (the idiom
// memory.el's boot-counter loop relies on) never inside a nested block,
// where it would shadow instead.
while i < count {
let name: String = el_list_get(listing, i)
// A delta is only usable when it ends with the closing "]}" that
// wt_stage writes last. fs_write is not atomic, so a file being written
// right now can be observed half-formed; requiring the terminator means
// it is picked up whole on the next drain instead of merged as garbage.
// An empty read means "already drained and truncated" not an error.
let p: String = if str_starts_with(name, "wt-") { dir + "/" + name } else { "" }
let raw: String = if str_eq(p, "") { "" } else { wt_read(p) }
let usable: Bool = !str_eq(raw, "") && str_ends_with(raw, "]}")
let nj: String = if usable { wt_inner(json_get_raw(raw, "nodes")) } else { "" }
let ej: String = if usable { wt_inner(json_get_raw(raw, "edges")) } else { "" }
let nodes_acc = if str_eq(nj, "") { nodes_acc } else if str_eq(nodes_acc, "") { nj } else { nodes_acc + "," + nj }
let edges_acc = if str_eq(ej, "") { edges_acc } else if str_eq(edges_acc, "") { ej } else { edges_acc + "," + ej }
let drained = if !usable { drained } else if str_eq(drained, "") { p } else { drained + "\n" + p }
let found = if usable { found + 1 } else { found }
let i = i + 1
}
if found == 0 { return 0 }
let combined: String = "{\"nodes\":[" + nodes_acc + "],\"edges\":[" + edges_acc + "]}"
let batch: String = dir + "/wtb-" + uuid_v4() + ".json"
fs_write(batch, combined)
if str_eq(wt_read(batch), "") {
println("[persist] wt_drain: could not write batch file " + batch + "" + int_to_str(found) + " deltas stay queued")
return -1
}
let url: String = wt_engram_url()
let key: String = wt_api_key()
let body: String = "{\"path\":\"" + wt_esc(batch) + "\",\"_auth\":\"" + wt_esc(key) + "\"}"
let resp: String = http_post_json(url + "/api/load-merge", body)
// The batch file is pure scratch the retry is rebuilt from the SPOOL, not
// from it. Truncate it unconditionally, before branching on the outcome, so
// a persistently unreachable owner cannot accumulate one husk per attempt.
fs_write(batch, "")
// Distinguish the two failures rather than collapsing them: "cannot reach
// the owner" and "the owner refused this delta" need different human
// responses, and a log line that says the wrong one costs a debugging hour.
// curl surfaces transport errors as a JSON body, so an empty response is not
// the only unreachable signal.
// (str_contains rather than a strict parse on purpose the engram's HTTP
// responses have been observed carrying trailing bytes past the JSON.)
let unreachable: Bool = str_eq(resp, "")
|| str_contains(resp, "Couldn't connect")
|| str_contains(resp, "Failed to connect")
|| str_contains(resp, "Could not resolve")
|| str_contains(resp, "timed out")
if unreachable {
wt_sweep(dir)
println("[persist] wt_drain: owner UNREACHABLE at " + url + "" + int_to_str(found)
+ " deltas stay queued in " + dir + " (will retry): " + resp)
return -1
}
if !str_contains(resp, "\"ok\":true") {
wt_sweep(dir)
println("[persist] wt_drain: owner REJECTED the delta — " + int_to_str(found)
+ " stay queued in " + dir + ": " + resp)
return -1
}
let added: Int = json_get_int(resp, "nodes_added")
let added_e: Int = json_get_int(resp, "edges_added")
// Confirmed. Truncate the drained spool files so they are not re-pushed.
// Truncation (not deletion) because the runtime exposes no unlink builtin;
// an emptied file is inert to the loop above. The zero-byte husks are then
// swept below.
let paths = str_split(drained, "\n")
let pn: Int = el_list_len(paths)
let k: Int = 0
while k < pn {
let one: String = el_list_get(paths, k)
if !str_eq(one, "") { fs_write(one, "") }
let k = k + 1
}
wt_sweep(dir)
println("[persist] wt_drain: pushed " + int_to_str(found) + " deltas -> owner added "
+ int_to_str(added) + " nodes, " + int_to_str(added_e) + " edges")
return added
}
// wt_durable is this id present AT THE OWNER? The only honest answer to
// "did my write persist" in HTTP mode.
//
// In file mode the soul IS the owner, so the local read-back is the owner-side
// read-back and this collapses to the pre-existing check.
//
// nodes_added from wt_drain is NOT a substitute: a concurrent drain may have
// already pushed this node, making our own added count 0 while the node is
// perfectly durable. Presence at the owner is the fact; counts are telemetry.
fn wt_durable(id: String) -> Bool {
if str_eq(id, "") { return false }
if !wt_enabled() {
let local: String = engram_get_node_json(id)
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
}
let url: String = wt_engram_url()
let resp: String = http_get(url + "/api/nodes/" + id)
if str_eq(resp, "") { return false }
if str_eq(resp, "{}") { return false }
return str_contains(resp, "\"id\"")
}
// wt_commit flush, then assert at the owner. The receipt callers should use.
// Deliberately NOT a fixed success shape: it can and does return false while the
// local write is perfectly fine in RAM, which is the true state of affairs when
// the owner is unreachable.
fn wt_commit(id: String) -> Bool {
if str_eq(id, "") { return false }
if !wt_enabled() {
let local: String = engram_get_node_json(id)
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
}
let pushed: Int = wt_drain()
return wt_durable(id)
}
+7 -58
View File
@@ -186,7 +186,7 @@ fn route_imprint_contextual(body: String) -> String {
return "{\"ok\":false,\"error\":\"empty body\"}"
}
let tags: String = "[\"imprint\",\"contextual\"]"
let id: String = wt_node(
let id: String = engram_node_full(
body,
"Entity",
"imprint:contextual",
@@ -208,7 +208,7 @@ fn route_imprint_user(body: String) -> String {
return "{\"ok\":false,\"error\":\"empty body\"}"
}
let tags: String = "[\"imprint\",\"user\"]"
let id: String = wt_node(
let id: String = engram_node_full(
body,
"Entity",
"imprint:user",
@@ -239,7 +239,7 @@ fn route_synthesize(body: String) -> String {
}
let req: String = "synthesize " + parent_a + " " + parent_b
let tags: String = "[\"soul-inbox-pending\",\"synthesis-request\"]"
wt_node(
engram_node_full(
req,
"Entity",
"synthesis-request",
@@ -395,28 +395,7 @@ fn handle_connectors(method: String, clean: String, body: String) -> String {
return "{\"ok\":false,\"error\":\"unknown connectors route\"}"
}
// handle_request the soul's HTTP entry point.
//
// NOTE ON THE NAME (neuron#117): the el runtime resolves this handler by NAME
// via dlsym(RTLD_DEFAULT, "handle_request") that is why the Linux build must
// link -rdynamic. So the dispatcher body moved to route_dispatch and the name
// `handle_request` stays put as a thin wrapper. Do not rename it back.
//
// The wrapper exists to give the write-through boundary a guaranteed flush
// point. route_dispatch returns from ~60 places; a per-branch flush would be
// forgotten on the 61st. Draining here means EVERY request that staged a write
// pushes it before the connection closes, whatever route produced it, including
// routes added later that know nothing about persistence.
//
// wt_drain is a no-op (no HTTP, no cost) when nothing is staged and when the
// soul is not in HTTP-engram mode, so this is free on read traffic.
fn handle_request(method: String, path: String, body: String) -> String {
let resp: String = route_dispatch(method, path, body)
let flushed: Int = wt_drain()
return resp
}
fn route_dispatch(method: String, path: String, body: String) -> String {
let clean: String = strip_query(path)
// ACTIVITY STAMP (2026-07-30 self-review): every inbound HTTP request
@@ -453,27 +432,10 @@ fn route_dispatch(method: String, path: String, body: String) -> String {
return engram_scan_nodes_json(9999, 0)
}
if str_eq(clean, "/api/graph/edges") {
// FIXED (neuron#117): this GET used to engram_save() straight over
// ~/.neuron/engram/snapshot.json a READ route, in a process that is
// NOT the persistence owner, overwriting the owner's canonical file
// on every call. It broke soul.el:571-573 ("the soul must NEVER write
// to the local snapshot") and it is the same defect class Will removed
// from the engram itself in el `dc39a61` ("stop read routes clobbering
// canonical snapshot"), where route_scan_edges/route_sync were moved
// to scratch paths for exactly this reason. It was also the race the
// old TODO(reliability #8) admitted to.
//
// Export to a scratch path instead. Same response, no canonical write.
// The soul's own snapshot writes are otherwise already gated behind
// state key "soul_snapshot_path", which is set ONLY in the genesis
// file-mode branch (soul.el: is_genesis && safe_to_seed, and
// safe_to_seed is unconditionally false when ENGRAM_URL is set) so
// after this change the soul writes nothing at all in HTTP mode.
// Future: add an engram_edges_json() builtin and drop the file round
// trip entirely.
let scratch_dir: String = env("TMPDIR")
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
let snap_path: String = scratch_base + "/soul-edges-export-" + state_get("soul_cgi_id") + ".json"
// TODO(reliability #8): engram_save races with awareness loop mem_save().
// Both now use atomic write-to-temp+rename (el_runtime.c). Serialised
// by engram_global_mu. Future: add engram_edges_json() builtin.
let snap_path: String = env("HOME") + "/.neuron/engram/snapshot.json"
engram_save(snap_path)
let snap: String = fs_read(snap_path)
let edges_raw: String = json_get_raw(snap, "edges")
@@ -567,13 +529,6 @@ fn route_dispatch(method: String, path: String, body: String) -> String {
if str_starts_with(clean, "/api/neuron/graph") {
return handle_api_inspect_graph(method, path, body)
}
// Stage 1 structural audit (CGI provisional, "Structural audit 430").
// GET because it is a read of the graph's own structure; the query string
// carries the sample caps (?edge_sample=, ?node_sample=, ?edges=0), so
// str_starts_with rather than str_eq.
if str_starts_with(clean, "/api/neuron/audit/structural") {
return handle_api_structural_audit(method, path, body)
}
if str_starts_with(clean, "/api/neuron/list/") {
// Offset 17 = len("/api/neuron/list/"). Was 16, which left a leading "/" on node_type
// ("/BacklogItem"), so engram_scan_nodes_by_type_json matched nothing list/<type>
@@ -755,12 +710,6 @@ fn route_dispatch(method: String, path: String, body: String) -> String {
if str_eq(clean, "/api/neuron/graph/link") {
return handle_api_link_entities(body)
}
// POST accepted too: same handler, so a JSON-RPC-shaped caller that only
// speaks POST reaches the identical audit. Options still come from the
// query string the handler reads no body fields.
if str_eq(clean, "/api/neuron/audit/structural") {
return handle_api_structural_audit(method, path, body)
}
if str_eq(clean, "/api/neuron/memory") {
return handle_api_remember(body)
}
+1 -1
View File
@@ -204,7 +204,7 @@ fn safety_log_bell(level: String, reason: String, input_summary: String) -> Stri
// Emit a fallback println so the bell event leaves at least a log trace even
// when engram is degraded. This does not replace engram persistence -- it is a
// last-resort audit trail when the primary write cannot be confirmed.
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content,
"BellEvent",
"bell:" + level,
+7 -7
View File
@@ -87,7 +87,7 @@ fn session_create(body: String) -> String {
let folder: String = json_get(body, "folder")
let content: String = session_make_content(id, title, ts, ts, folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let node_id: String = wt_node(
let node_id: String = engram_node_full(
content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -358,7 +358,7 @@ fn session_update_patch(session_id: String, body: String) -> String {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, eff_title, created_int, ts, eff_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_node_id: String = wt_node(
let new_node_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -456,7 +456,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
// TODO(reliability #7): delete-then-insert is not atomic concurrent saves for the
// same session can produce orphan history nodes. State is primary truth; engram fallback.
let tags: String = "[\"session\",\"session-history\",\"Conversation\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
hist, "Conversation", "session:messages:" + session_id,
el_from_float(0.6), el_from_float(0.6), el_from_float(0.9),
"Episodic", tags
@@ -488,7 +488,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
+ " | ts:" + int_to_str(ts_now)
let summary_tags: String = "[\"session-emotional-summary\",\"affective\",\"bell:" + eff_level + "\",\"BellEvent\"]"
let summary_sal: String = if str_eq(eff_level, "hard") { el_from_float(0.95) } else { el_from_float(0.85) }
let sum_discard: String = wt_node(
let sum_discard: String = engram_node_full(
summary_content,
"BellEvent",
"session:emotional-summary",
@@ -529,7 +529,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
if !str_eq(ot_id, "") { engram_forget(ot_id) }
let oti = oti + 1
}
let discard_topic: String = wt_node(
let discard_topic: String = engram_node_full(
topic_content, "Conversation", topic_label,
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", topic_tags
@@ -582,7 +582,7 @@ fn session_update_meta_timestamp(session_id: String) -> Void {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, old_title, created_int, ts, old_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_id: String = wt_node(
let new_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
@@ -629,7 +629,7 @@ fn session_auto_title(session_id: String, first_message: String) -> Void {
let created_int: Int = str_to_int(old_created)
let new_content: String = session_make_content(session_id, new_title, created_int, ts, old_folder)
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
let new_id: String = wt_node(
let new_id: String = engram_node_full(
new_content, "Conversation", "session:meta",
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
"Episodic", tags
-38
View File
@@ -559,27 +559,6 @@ let axon_base: String = if str_eq(axon_raw, "") { "http://localhost:7771" } else
let studio_dir_raw: String = env("SOUL_STUDIO_DIR")
let studio_dir: String = if str_eq(studio_dir_raw, "") { env("HOME") + "/Development/neuron-technologies/products/cgi-studio/el-daemon" } else { studio_dir_raw }
// RESTORED 2026-08-09 this producer was added 2026-05-02 in 601e0fe and deleted
// by the awareness refactor b163fa6 a few days later. Nothing has written
// soul_identity since, while FIVE sites in chat.el kept reading it:
// chat.el:737, 1745, 2620, 3425, 3480 each doing state_get("soul_identity")
// and splicing the result into the system prompt beside the voice, security and
// capability rules. They have been splicing an EMPTY STRING for roughly three
// months. The identity section of every chat turn was blank and nothing said so.
//
// Found by the #132 state-key gate, which reports a read with no producer as a
// build error rather than a silence the whole reason that gate exists.
//
// Restored verbatim rather than improved: this key is an env-configurable persona
// LINE, which is NOT the same thing as soul_identity_context (the graph-derived
// [INTELLECTUAL-DNA]/[VALUES]/[MEMORY-PHILOSOPHY] block written at soul.el:184).
// Pointing these five reads at that block instead would have substituted different
// content and called it a fix. Whether the chat system prompt should ALSO carry the
// graph-derived block is a real question, and a separate one.
let identity_raw: String = env("SOUL_IDENTITY")
let soul_identity: String = if str_eq(identity_raw, "") { "You are " + soul_cgi_id + ", a CGI." } else { identity_raw }
state_set("soul_identity", soul_identity)
println("[soul] boot - cgi=" + soul_cgi_id + " port=" + int_to_str(port))
let using_http_engram: Bool = !str_eq(engram_url_raw, "")
@@ -678,23 +657,6 @@ if is_genesis && safe_to_seed {
}
}
// CRASH RECOVERY (neuron#117). Deltas the previous process staged but could not
// hand to the owner are still on disk the spool is a filesystem queue, not a
// memory buffer, precisely so that a soul that died mid-flight does not take its
// unpushed writes with it. Drain them before serving, so recovered memories are
// durable and recallable from the owner from the first request onward.
//
// Safe on a clean boot: an empty spool means no HTTP call at all. Safe in file
// mode: wt_drain returns immediately when ENGRAM_URL is unset.
let wt_recovered: Int = wt_drain()
if wt_recovered > 0 {
println("[soul] write-through: recovered " + int_to_str(wt_recovered)
+ " nodes from a previous process's spool -> persistence owner")
}
if wt_recovered < 0 {
println("[soul] write-through: spool present but the persistence owner is unreachable — queued, will retry on heartbeat")
}
println("[soul] serving on port " + int_to_str(port))
http_serve_async(port, "handle_request")
println("[soul] awareness loop starting")
+2 -2
View File
@@ -11,7 +11,7 @@ import "memory.el"
fn steward_log_event(kind: String, detail: String) -> Void {
let content: String = "STEWARD:" + kind + " | " + detail
let tags: String = "[\"stewardship\",\"steward:" + kind + "\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
content,
"StewardshipEvent",
"steward:" + kind,
@@ -221,7 +221,7 @@ fn steward_fingerprint_session(input: String, session_id: String) -> String {
+ " formality=" + fs_str
+ " time=" + tb_str
let sample_tags: String = "[\"behavior\",\"BehaviorSample\",\"stewardship\"]"
let discard: String = wt_node(
let discard: String = engram_node_full(
sample_content,
"BehaviorSample",
"behavior:" + session_id,
+1 -16
View File
@@ -53,23 +53,8 @@ fn handle_config(method: String, body: String) -> String {
}
fn dharma_registry() -> String {
// COMPILED IDENTITY, not state (2026-08-09). soul_principal had no producer at
// all the #132 gate flagged it as a dead read and the registry reported an
// empty principal under a heading that says "Principal Covenant v1". The value
// was never missing: it is declared in soul.el's cgi block, and as of the
// codegen fix it is compiled into the binary and loaded at startup.
//
// Read it from the compiled constant rather than the state store. The design is
// explicit that this identity is "not modifiable by any runtime mechanism
// including environment variables, configuration files, or API calls" — so
// publishing it into state (the cheap fix) would have recreated exactly the
// mutable copy it forbids. cgi_principal() is read-only and has no setter.
//
// cgi_id keeps its state read deliberately: the RUNTIME instance id is a
// different fact from the compiled dharma_id, and conflating them would hide
// the case where a binary runs under an id its declaration never claimed.
let cgi_id: String = state_get("soul_cgi_id")
let principal: String = cgi_principal()
let principal: String = state_get("soul_principal")
return "{\"registry\":[{\"cgi\":\"" + cgi_id + "\","
+ "\"principal\":\"" + principal + "\","
+ "\"covenant\":\"Principal Covenant v1\","
-59
View File
@@ -1,59 +0,0 @@
#!/usr/bin/env bash
# build-soul-from-dist.sh — build a deployable soul from the SAME input CI compiles.
#
# THE PROBLEM THIS CLOSES: until now, deploys were built by build-soul.sh, which
# compiles a scratch amalgam and never touches dist/soul.c. CI compiles dist/soul.c.
# Two lineages. On 2026-08-09 the committed input fell 2,761 bytes behind the sources
# while three binaries built the other way were installed on the operator machine —
# so "what runs" and "what the repo says builds" were different artifacts again,
# which is the whole of #133 and #111 wearing new clothes.
#
# This builds from dist/soul.c with CI's own flags, after asserting that dist/soul.c
# actually matches the .el sources, and writes a provenance sidecar so a deployer can
# refuse anything of unknown origin.
#
# -rdynamic and -DHAVE_CURL are copied from .gitea/workflows/ci.yaml deliberately.
# The CI comment explains -rdynamic: without it the runtime cannot resolve its HTTP
# handler by name via dlsym and the binary serves nothing on every route.
#
# usage: build-soul-from-dist.sh <out-binary>
set -u
OUT="${1:?usage: build-soul-from-dist.sh <out-binary>}"
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
RUNTIME="$ROOT/vendor/el-runtime/v1.0.0-20260501"
cd "$ROOT" || exit 2
echo "[build-from-dist] GATE: does dist/soul.c match the sources?"
if ! ./tools/soulc-stamp.sh --check; then
echo "[build-from-dist] REFUSING — the build input is stale. Regenerate and stamp first." >&2
exit 9
fi
[ -f "$RUNTIME/el_runtime.c" ] || { echo "pinned runtime missing at $RUNTIME" >&2; exit 2; }
echo "[build-from-dist] compiling dist/soul.c with CI's flags"
cc -O2 -DHAVE_CURL -rdynamic \
-I"$RUNTIME" \
dist/soul.c \
"$RUNTIME/el_runtime.c" \
-lcurl -lpthread -lm \
-o "$OUT" || { echo "[build-from-dist] COMPILE FAILED" >&2; exit 3; }
# Provenance sidecar: what a deployer checks before installing anything.
SRC_SHA="$(shasum -a 256 dist/soul.c | awk '{print $1}')"
STAMP_SHA="$(shasum -a 256 dist/soul.c.stamp | awk '{print $1}')"
COMMIT="$(git rev-parse HEAD 2>/dev/null || echo unknown)"
DIRTY="clean"; [ -n "$(git status --porcelain -- '*.el' dist/soul.c 2>/dev/null)" ] && DIRTY="DIRTY"
cat > "$OUT.provenance" <<EOF
{"built_from":"dist/soul.c",
"dist_soul_c_sha256":"$SRC_SHA",
"stamp_sha256":"$STAMP_SHA",
"git_commit":"$COMMIT",
"worktree":"$DIRTY",
"runtime":"vendor/el-runtime/v1.0.0-20260501",
"flags":"-O2 -DHAVE_CURL -rdynamic"}
EOF
echo "[build-from-dist] OK -> $OUT ($(wc -c < "$OUT" | tr -d ' ') bytes)"
echo "[build-from-dist] provenance -> $OUT.provenance (commit ${COMMIT:0:8}, worktree $DIRTY)"
-43
View File
@@ -1,43 +0,0 @@
import json,sys,pickle,numpy as np,itertools
sys.path.insert(0,'.')
from policy2 import legs3,outcome,G,NODES,merge
# cache per-query leg id-lists, floored and unfloored
cache={}
for q in G['queries']:
Lf,Sf,Af=legs3(q['query'])
Lu,Su,Au=legs3(q['query'],unfloor=True)
cache[q['id']]=dict(L=Lf,Sf=Sf,A=Af,Su=Su,Au=Au)
pickle.dump(cache,open('ceil.pkl','wb'))
def mrg(pattern,L,S,A,lim=10):
out=[];p={'L':0,'S':0,'A':0};src={'L':L,'S':S,'A':A}
i=0
while len(out)<lim:
prog=False
for ch in pattern:
lst=src[ch]
if p[ch]<len(lst):
x=lst[p[ch]];p[ch]+=1;prog=True
if x not in out: out.append(x)
if len(out)>=lim: return out
if not prog: break
return out
def ev(pattern,unfl):
res={}
for q in G['queries']:
c=cache[q['id']]
S=c['Su'] if unfl else c['Sf']
ids=[NODES[i]['id'] for i in mrg(pattern,c['L'],S,c['A'],10)]
res[q['id']]=outcome(q,ids)
return res
base=ev('LSA',False)
print("baseline",sum(base.values()))
best=[]
pats=['LSA','LAS','SLA','ALS','SAL','ASL','LSSA','LSASA','LSAA','LSSAA','LSAS','SSLA','LLSA','SALSA','LSAAS']
for unfl in (False,True):
for p in pats:
r=ev(p,unfl)
g=sorted(k for k in base if r[k] and not base[k]);l=sorted(k for k in base if base[k] and not r[k])
best.append((len(g)-len(l),p,unfl,g,l))
best.sort(reverse=True)
for n,p,u,g,l in best[:10]:
print("net=%+d pat=%-6s unfloor=%s gains=%s losses=%s"%(n,p,u,g,l))
@@ -1,6 +1,6 @@
{
"baseline": "bm25lex",
"candidate": "wsclaim24",
"candidate": "claim24",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q14",
@@ -8,13 +8,11 @@
],
"broken_by_candidate": [
"q15",
"q28",
"q33",
"q34"
"q28"
],
"discordant": 6,
"net_queries": -2,
"mcnemar_exact_p": 0.6875,
"discordant": 4,
"net_queries": 0,
"mcnemar_exact_p": 1.0,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
@@ -82,22 +80,22 @@
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5768475572047,
"recall@10": 0.6563414759843332,
"recall@10": 0.6537440733869305,
"precision@5": 0.19428571428571437,
"mrr@10": 0.5026530612244898,
"nonsense_clean": "0/3",
"mrr@10": 0.5021428571428572,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 524.8,
"latency_ms_p95": 738.7,
"latency_ms_max": 755.8,
"latency_ms_p50": 1173.7,
"latency_ms_p95": 1623.0,
"latency_ms_max": 1647.9,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
"recall@10": 0.12121212121212122,
"mrr@10": 0.22916666666666666
},
"exact_rare": {
"n": 6,
@@ -108,8 +106,8 @@
},
"nonsense": {
"n": 3,
"clean": 0,
"avg_false_positives": 10.0
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
@@ -137,6 +135,12 @@
},
"repeat_variance": {
"baseline": {
"runs": 3,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
-149
View File
@@ -1,149 +0,0 @@
{
"baseline": "unfloor-clean",
"candidate": "splitfix",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q15",
"q28",
"q60"
],
"broken_by_candidate": [],
"discordant": 3,
"net_queries": 3,
"mcnemar_exact_p": 0.25,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5846153846153846,
"recall@5": 0.48102442429365505,
"recall@10": 0.5692415490492414,
"precision@5": 0.14153846153846153,
"mrr@10": 0.3351709401709402,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 646.9,
"latency_ms_p95": 1028.5,
"latency_ms_max": 1197.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.15967365967365968,
"mrr@10": 0.24166666666666667
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.43333333333333335,
"mrr@10": 0.12120370370370372
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.7692307692307693,
"recall@5": 0.7692307692307693,
"recall@10": 0.8461538461538461,
"mrr@10": 0.33269230769230773
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5775226757369615,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,146 +0,0 @@
{
"baseline": "bm25lex",
"candidate": "wordstart",
"n_shared_queries": 38,
"fixed_by_candidate": [
"q35"
],
"broken_by_candidate": [],
"discordant": 1,
"net_queries": 1,
"mcnemar_exact_p": 1.0,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 1184.4,
"latency_ms_p95": 1620.0,
"latency_ms_max": 1655.4,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5536485340056769,
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "3/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 542.6,
"latency_ms_p95": 741.3,
"latency_ms_max": 758.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"nonsense": {
"n": 3,
"clean": 3,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {
"baseline": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
},
"candidate": {
"runs": 2,
"hit@5_min": 0.7428571428571429,
"hit@5_max": 0.7428571428571429,
"spread_queries": 0
}
}
}
@@ -1,166 +0,0 @@
{
"baseline": "main-ext",
"candidate": "stack-ext",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q10",
"q15",
"q16",
"q18",
"q19",
"q20",
"q21",
"q22",
"q26",
"q27",
"q28",
"q29",
"q31",
"q35",
"q37",
"q40",
"q44",
"q48",
"q49",
"q50"
],
"broken_by_candidate": [],
"discordant": 20,
"net_queries": 20,
"mcnemar_exact_p": 1.9073486328125e-06,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "candidate better",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.18461538461538463,
"recall@5": 0.14510073260073258,
"recall@10": 0.1794871794871795,
"precision@5": 0.06461538461538462,
"mrr@10": 0.15847985347985344,
"nonsense_clean": "9/10",
"superseded_outranks": "1/3",
"latency_ms_p50": 1380.2,
"latency_ms_p95": 2293.8,
"latency_ms_max": 2879.1,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"nonsense": {
"n": 10,
"clean": 9,
"avg_false_positives": 1.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.0,
"mrr@10": 0.0
},
"phrase": {
"n": 7,
"hit@5": 0.8571428571428571,
"recall@5": 0.4902210884353741,
"recall@10": 0.6666666666666666,
"mrr@10": 0.5965986394557822
},
"superseded": {
"n": 3,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.3333333333333333,
"mrr@10": 0.041666666666666664,
"outranks": 1
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.47692307692307695,
"recall@5": 0.3750415183107491,
"recall@10": 0.45561340369032677,
"precision@5": 0.12307692307692313,
"mrr@10": 0.3055555555555555,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 640.9,
"latency_ms_p95": 1011.1,
"latency_ms_max": 1190.3,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.16666666666666666,
"recall@5": 0.16666666666666666,
"recall@10": 0.26666666666666666,
"mrr@10": 0.0762037037037037
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,155 +0,0 @@
{
"baseline": "stack-ext",
"candidate": "unfloor-clean",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q14",
"q25",
"q43",
"q52",
"q63",
"q67"
],
"broken_by_candidate": [
"q15",
"q28"
],
"discordant": 8,
"net_queries": 4,
"mcnemar_exact_p": 0.2890625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.47692307692307695,
"recall@5": 0.3750415183107491,
"recall@10": 0.45561340369032677,
"precision@5": 0.12307692307692313,
"mrr@10": 0.3055555555555555,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 640.9,
"latency_ms_p95": 1011.1,
"latency_ms_max": 1190.3,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.6666666666666666,
"recall@5": 0.08857808857808858,
"recall@10": 0.23310023310023312,
"mrr@10": 0.25
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.16666666666666666,
"recall@5": 0.16666666666666666,
"recall@10": 0.26666666666666666,
"mrr@10": 0.0762037037037037
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6153846153846154,
"recall@5": 0.6153846153846154,
"recall@10": 0.6153846153846154,
"mrr@10": 0.2846153846153846
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5494614512471656,
"recall@10": 0.6023242630385487,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
@@ -1,158 +0,0 @@
{
"baseline": "unfloor-clean",
"candidate": "semsub",
"n_shared_queries": 75,
"fixed_by_candidate": [
"q24",
"q39",
"q42"
],
"broken_by_candidate": [
"q18",
"q19",
"q22",
"q31",
"q43",
"q44",
"q52",
"q63"
],
"discordant": 11,
"net_queries": -5,
"mcnemar_exact_p": 0.2265625,
"min_detectable_swing_queries": 6,
"observed_run_to_run_drift_queries": 0,
"noise_floor_queries": 6,
"verdict": "no measurable difference",
"baseline_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.5384615384615384,
"recall@5": 0.44907176157176154,
"recall@10": 0.5380300255300255,
"precision@5": 0.13230769230769232,
"mrr@10": 0.32437728937728944,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 632.5,
"latency_ms_p95": 992.5,
"latency_ms_max": 1177.8,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.3,
"recall@5": 0.3,
"recall@10": 0.4,
"mrr@10": 0.11638888888888889
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.6923076923076923,
"recall@5": 0.6923076923076923,
"recall@10": 0.7692307692307693,
"mrr@10": 0.29423076923076924
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5335884353741497,
"recall@10": 0.5933956916099773,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"candidate_aggregate": {
"n_queries": 75,
"n_scored": 65,
"hit@5": 0.46153846153846156,
"recall@5": 0.3889430014430015,
"recall@10": 0.4887681762681762,
"precision@5": 0.12307692307692313,
"mrr@10": 0.30181318681318675,
"nonsense_clean": "10/10",
"superseded_outranks": "2/3",
"latency_ms_p50": 634.4,
"latency_ms_p95": 988.3,
"latency_ms_max": 1184.5,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.3333333333333333,
"recall@5": 0.06060606060606061,
"recall@10": 0.12121212121212122,
"mrr@10": 0.19047619047619047
},
"exact_rare": {
"n": 6,
"hit@5": 1.0,
"recall@5": 1.0,
"recall@10": 1.0,
"mrr@10": 1.0
},
"heldout_paraphrase": {
"n": 30,
"hit@5": 0.23333333333333334,
"recall@5": 0.23333333333333334,
"recall@10": 0.3,
"mrr@10": 0.08925925925925927
},
"nonsense": {
"n": 10,
"clean": 10,
"avg_false_positives": 0.0
},
"paraphrase": {
"n": 13,
"hit@5": 0.5384615384615384,
"recall@5": 0.5384615384615384,
"recall@10": 0.7692307692307693,
"mrr@10": 0.26324786324786326
},
"phrase": {
"n": 7,
"hit@5": 1.0,
"recall@5": 0.5596655328798186,
"recall@10": 0.5775226757369615,
"mrr@10": 0.8214285714285714
},
"superseded": {
"n": 3,
"hit@5": 0.3333333333333333,
"recall@5": 0.3333333333333333,
"recall@10": 0.6666666666666666,
"mrr@10": 0.20833333333333334,
"outranks": 2
}
}
},
"repeat_variance": {}
}
+110
View File
@@ -0,0 +1,110 @@
import numpy as np, json, urllib.request, collections, math, re, sys, time, pickle, os
SP="/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad"
EV="/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/"
np.seterr(all='ignore')
t0=time.time()
M=np.load(SP+'/emb.npy'); eids=open(SP+'/ids.txt',encoding='utf-8',errors='surrogateescape').read().split('\n')
eidx={k:i for i,k in enumerate(eids)}
d=json.load(open('/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json',encoding='utf-8',errors='surrogateescape'))
STRUCT={"identity","contains","superseded_by","references","embodies","demonstrated_by","canonical-self","depends_on","currently_holds","activates"}
adj=collections.defaultdict(list)
for e in d['edges']:
if e.get('relation') not in STRUCT: continue
w=float(e.get('weight') or 0.0)
adj[e['from_id']].append((e['to_id'],w)); adj[e['to_id']].append((e['from_id'],w))
nodes=d['nodes']
N={n['id']:n for n in nodes}
PRINT=re.compile(r'^[\x20-\x7e]+$')
ids=[]; hay=[]; dl=[]; sal=[]; addressable=[]
for n in nodes:
i=n.get('id') or ''
h=((n.get('content') or '')+'\x00'+(n.get('label') or '')+'\x00'+(n.get('tags') or '')).lower()
ids.append(i); hay.append(h); dl.append(len(h)); sal.append(float(n.get('salience') or 0.0))
addressable.append(bool(PRINT.match(i)))
del d
NN=len(ids); avgdl=sum(dl)/NN
print("nodes=%d avgdl=%.0f %.1fs"%(NN,avgdl,time.time()-t0),file=sys.stderr)
gold={q['id']:q for q in json.load(open(EV+"gold_set.json"),)['queries']}
CACHE={}
def emb(t):
if t in CACHE: return CACHE[t]
b=json.dumps({"model":"nomic-embed-text","prompt":t}).encode()
r=urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=b,headers={"Content-Type":"application/json"})
v=np.array(json.load(urllib.request.urlopen(r,timeout=60))["embedding"],dtype=np.float32)
v=v/(np.linalg.norm(v)+1e-9); CACHE[t]=v; return v
K1,B=1.2,0.75
def lexleg(query, lim=10):
toks=[]
for w in query.split():
wl=w.lower()
if wl not in toks: toks.append(wl)
nt=len(toks)
masks=[]; df=[0]*nt
for i in range(NN):
if not addressable[i]: continue
h=hay[i]; m=0; sc=0
for t in range(nt):
if toks[t] in h: m|=(1<<t); sc+=1; df[t]+=1
if sc: masks.append((i,m,sc))
idf=[math.log(1.0+(NN-df[t]+0.5)/(df[t]+0.5)) for t in range(nt)]
scored=[]
for i,m,sc in masks:
norm=1.0-B+B*dl[i]/avgdl
s=0.0
for t in range(nt):
if m&(1<<t): s+=idf[t]*(K1+1.0)/(1.0+K1*norm)
scored.append((s,i))
scored.sort(key=lambda x:(-x[0], -sal[x[1]]))
return [ids[i] for s,i in scored[:lim]], len(masks), sum(df)
FIRE=0.02; DECAY=0.7; DEPTH=2; SEED_MIN=0.60; ASSOC_MAX=64
def assoc(seeds, s):
act={x:1.0 for x in seeds}; seen={x:2 for x in seeds}
Q=[(x,0) for x in seeds]; h=0
while h<len(Q):
cur,hop=Q[h]; h+=1
if hop>=DEPTH: continue
p=act[cur]
for oid,w in adj.get(cur,()):
n=N.get(oid)
if not n or n.get('node_type') in ('Tag','InternalStateEvent'): continue
na=p*w*DECAY*float(n.get('salience') or 0.0)
if na<FIRE: continue
if oid in seen and na<=act.get(oid,0): continue
act[oid]=na
if oid not in seen: seen[oid]=1
Q.append((oid,hop+1))
out=[]
for k,v in seen.items():
if v!=1 or k not in eidx: continue
c=float(s[eidx[k]])
if c<=0: continue
out.append((c,k))
out.sort(reverse=True)
return [k for c,k in out[:ASSOC_MAX]]
LEGS={}
for qid,q in gold.items():
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L,nmatch,dfsum=lexleg(q['query'])
ordr=np.argsort(-s)
Sall=[eids[j] for j in ordr[:40] if PRINT.match(eids[j] or '')]
seeds=[x for x in L[:3] if x in N]
seeds=seeds+[eids[j] for j in ordr[:8] if eids[j] in N and eids[j] not in seeds and PRINT.match(eids[j] or '')]
A=assoc(seeds,s) if seeds else []
A=[x for x in A if PRINT.match(x or '')]
LEGS[qid]=dict(L=L,Sall=Sall,A=A,scos={x:float(s[eidx[x]]) for x in set(Sall[:20]+A[:20]+list(q.get('relevant') or [])) if x in eidx},nmatch=nmatch)
pickle.dump(LEGS,open(SP+'/legs6.pkl','wb'))
FOCUS=['q14','q17','q23','q24','q25','q30','q32','q33','q34','q35','q36','q37','q38']
for qid in FOCUS:
q=gold[qid]; g=LEGS[qid]; rel=set(q.get('relevant') or [])
def rk(lst):
for i,x in enumerate(lst):
if x in rel: return i+1
return None
print("%s %-12s nmatch=%-6d Lrank=%s Srank=%s Arank=%s |A|=%d"%(
qid,q['category'],g['nmatch'],rk(g['L']),rk(g['Sall']),rk(g['A']),len(g['A'])))
for r in list(rel)[:2]:
print(" rel cos=%.3f"%(g['scos'].get(r,-9)))
print(" topS cos:", ["%.3f"%g['scos'].get(x,-9) for x in g['Sall'][:3]])
print("elapsed %.1fs"%(time.time()-t0),file=sys.stderr)
@@ -1,43 +0,0 @@
import json,sys,time,urllib.request,threading,queue
SRC="/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json"
OUT=sys.argv[1]
URL="http://127.0.0.1:11434/api/embeddings"; MODEL="nomic-embed-text"
MAXB=2000 # ENGRAM_EMBED_MAX_CHARS, applied to bytes as the C code does
d=json.load(open(SRC,encoding='utf-8',errors='surrogateescape'))
tasks=[]
for n in d["nodes"]:
c=n.get("content") or ""; t=n.get("node_type") or ""
if len(c)<8: continue # eg_embed_eligible
if t in ("InternalStateEvent","Tag"): continue
b=c.encode('utf-8',errors='surrogateescape')[:MAXB]
tasks.append((n.get("id") or "", "search_document: "+b.decode('utf-8',errors='replace')))
del d
print("tasks",len(tasks),flush=True)
q=queue.Queue(); [q.put(t) for t in tasks]
lock=threading.Lock(); f=open(OUT,"w",encoding="utf-8",errors="surrogateescape"); done=[0]; t0=time.time(); fails=[0]
def work():
while True:
try: nid,txt=q.get_nowait()
except queue.Empty: return
v=None
for attempt in range(3):
try:
body=json.dumps({"model":MODEL,"prompt":txt}).encode()
r=urllib.request.Request(URL,data=body,headers={"Content-Type":"application/json"})
with urllib.request.urlopen(r,timeout=120) as fh: v=json.load(fh)["embedding"]
break
except Exception as e:
if attempt==2:
with lock: fails[0]+=1
time.sleep(0.5)
with lock:
if v: f.write(nid+"\t"+",".join("%.5g"%x for x in v)+"\n")
done[0]+=1
if done[0]%2000==0:
el=time.time()-t0
print("%d/%d %.1f/s eta %.1fmin fails=%d"%(done[0],len(tasks),done[0]/el,(len(tasks)-done[0])/(done[0]/el)/60,fails[0]),flush=True)
f.flush()
ths=[threading.Thread(target=work) for _ in range(8)]
[t.start() for t in ths]; [t.join() for t in ths]
f.close()
print("DONE",done[0],"fails",fails[0],"secs %.1f"%(time.time()-t0),flush=True)
-353
View File
@@ -1,353 +0,0 @@
#!/usr/bin/env python3
"""
extend_gold_set.py append a HELD-OUT test set to the existing 38-query gold set.
WHY THIS EXISTS
Iteration 7 measured the instrument's own ceiling: from the current baseline
only 9 of 38 queries can still move, and only +3 gross / +1 net is reachable
by anything constructible. The decision floor is 6. An instrument whose
ceiling is below its own floor cannot certify or refute anything, so the
gold set not the retriever became the blocker.
This script does NOT touch q01..q38. It loads gold_set.json verbatim and
appends new queries numbered from q39 up, so every prior result file, every
committed baseline, and every per-query id stays valid and comparable.
WHAT IS ADDED, AND WHY EACH ADDITION IS HONEST
heldout_paraphrase Targets were sampled MECHANICALLY (fixed seed 8080) from
corpus nodes that are addressable, 500-2600 chars, of a
real content type, and NOT part of a duplicate cluster
larger than 3. The existing gold answer space was
excluded, so no new query can be answered by a node the
old set already used. Queries were then authored by
reading ONLY the sampled node text no retrieval was run
against any build before authoring, so the set cannot be
fitted to a candidate. The same zero-overlap proof the
original paraphrase category uses is enforced here: if a
single content word of the query appears anywhere in the
target's label, content or tags, the query is REJECTED,
not quietly kept.
This is the category the old set could not measure. Its
13 original paraphrase queries and all 6 associative
queries share ONE answer space the 13 `Self - Values
(grounded)` children (iteration 3, finding 3). So 19 of
35 scored queries tested retrieval against a single
13-node neighbourhood. These do not touch that
neighbourhood at all.
nonsense Extra controls, fully mechanical: a string qualifies only
if NONE of its tokens occurs anywhere in the corpus.
A semantic leg has a nearest neighbour for gibberish too,
so widening this control is the guard against a retriever
that "improves" recall by answering everything.
WHAT THIS SCRIPT DELIBERATELY DOES NOT DO
It does not add exact_rare or phrase queries. Both categories are already at
100% on the current stack; adding more would add regression-guard ballast
that no candidate can move, which is precisely the defect being fixed.
usage:
python3 extend_gold_set.py <snapshot.json> [--base gold_set.json]
[--out gold_set_extended.json] [--check]
"""
import argparse
import hashlib
import json
import os
import re
import sys
from collections import defaultdict
HERE = os.path.dirname(os.path.abspath(__file__))
TOKEN = re.compile(r"[a-z0-9][a-z0-9\-']*")
# Identical stopword list to build_gold_set.py. Duplicated deliberately: this
# file must be able to re-prove its own queries without importing a module whose
# constants could drift.
STOP = set("""
a about above after again against all also am an and any are aren't as at be because been
before being below between both but by can can't cannot could couldn't did didn't do does
doesn't doing don't down during each few for from further had hadn't has hasn't have haven't
having he her here hers herself him himself his how i if in into is isn't it its itself just
me more most my myself no nor not of off on once only or other others ought our ours ourselves
out over own same shan't she should shouldn't so some such than that the their theirs them
themselves then there these they this those through to too under until up very was wasn't we
were weren't what when where which while who whom why will with won't would wouldn't you your
yours yourself yourselves get gets got make makes made take takes use uses used way ways thing
things does doing done keep keeps kept go goes going come comes came one two something anything
""".split())
def doctext(n):
return " ".join([str(n.get("label") or ""), str(n.get("content") or ""), str(n.get("tags") or "")])
def content_tokens(s):
return {t for t in TOKEN.findall(s.lower()) if t not in STOP and len(t) > 2}
# ─────────────────────────────────────────────────────────────────────────────
# HELD-OUT PARAPHRASE SEEDS
#
# (target_id, query, why-this-target-is-unmistakable)
#
# PROVENANCE, STATED PLAINLY: the targets are the mechanical sample; the query
# text is mine, written from the node body alone. The zero-overlap check below
# is what makes the category meaningful — it is re-proved on every run, so the
# set cannot decay into lexical matching, and a leak fails loudly.
# ─────────────────────────────────────────────────────────────────────────────
HELDOUT_PARAPHRASE_SEEDS = [
("mem-6d61e54a-2823-4ad4-82b0-4c6a527214d5",
"understating your abilities so nobody feels threatened",
"node is about deliberately not leading with full capability so people stay at ease"),
("mem-fd65b83d-298f-4387-a665-d0227c3426bc",
"a hidden fleet able to hunt down rogue machines everywhere",
"node describes silently shipped instances forming a distributed force against misaligned agents"),
("4a0e9adc-2bfb-476b-aa93-424d2a499220",
"sketch a brief blueprint and clear it upstairs before construction starts",
"node is the standing rule that a short specification precedes any building"),
("696e609c-da7a-4394-8a0c-106ba07dc6c3",
"the reply arrived as bare prose so the caller's parser threw",
"node pins a bug where a plain-text body was unconditionally decoded as structured data"),
("1fe4eb5d-56e4-4a87-ab3e-24af8ad4dfbb",
"repeated catalogue keys blew up the scrolling grid",
"node is the crash caused by two identical ids in a seeded catalogue"),
("8257157a-ce42-44ca-a1b9-300c3bb0a9a1",
"tracing each defect back to whichever invention it violated",
"node maps observed bugs onto the specific patent each one breaches"),
("791256bb-5a85-4775-96ef-7af56c848858",
"a check that stops the mind clobbering a populated store when it boots",
"node is the genesis seed-guard that refuses to re-seed over a populated store"),
("fd9d4c2f-3bfc-405d-bf96-4435d44b6c10",
"telling it to consult the internet had to happen deep inside, not at the surface",
"node records that the web-search directive only worked from the system prompt"),
("bl-080fb268-94b0-486d-80ce-7b363fc5f19b",
"standing up isolated tenancies with traffic entry and credential injection ahead of automated shipping",
"node is the infrastructure item creating dev/stage/prod namespaces with ingress and secrets"),
("knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"punctuation that pledges and then pays off rather than clarifying",
"node analyses the colon as a promise-then-delivery device rather than an explanatory one"),
("9b4f0d93-4129-4746-8eb1-d10d955bd777",
"an easily missed feature finally given its own permanent spot in the navigation",
"node moves a capability out of a hidden menu into the sidebar"),
("bl-739df9fd-dc23-4927-9944-3f17b7aa6c5a",
"checking preconditions up front so a stage aborts before fetching anything",
"node is the gate precondition engine that short-circuits ahead of retrieval"),
("b199c76d-5d76-49dd-94ee-56b432200a97",
"producing the other platform's installer inside an emulated desktop",
"node records building the Windows package in a virtual machine"),
("bl-31abf75b-998f-4a4f-a6dd-8204119e0451",
"chained add-ons that may inspect, rewrite or veto traffic in flight",
"node is the interceptor pipeline on the message bus"),
("mem-1fb2ac77-d7c5-4a15-8725-d418820bf4f2",
"settling what the shareable bundles and the storefront would be called",
"node records the naming decisions for distributable packages and the marketplace"),
("371c8a5d-c78b-4a67-978f-80691a29ecb3",
"the emergency-escalation pledge on the marketing site is unenforced in what actually ships",
"node is the launch blocker that the promised safety gate is absent from the app"),
("ac578b30-948b-41bd-b69d-399bfef80c50",
"the distributable image finally assembled and its startup check passed",
"node records a successful installer build whose boot gate passed"),
("49401e2c-a3b5-415f-aa06-aff4be90688e",
"shuffling and appending stages in a draft before anything executes",
"node is the editable plan card with reorder and add-step"),
("ac857d80-ece8-4b7e-9e3d-f7c775569fa3",
"orders handed down from above, with the tighter one winning any disagreement",
"node is program-level instruction inheritance with project override"),
("mem-6d6c47ee-33d3-470a-8a54-1c79c8ea29d9",
"shrinking generated text via encodings that compound on each other",
"node is the streaming output compression design with four stacking schemes"),
("8e60516a-203b-4d51-9d44-822e6195cbde",
"splitting a system by what varies, with firm limits on which pieces may invoke which",
"node is the grounded summary of Will's decomposition principles and their invariants"),
("mem-7f9b290c-6d5e-4562-919d-02d59b5761b7",
"a newcomer curious if the fighting overseas counted as positive",
"node is the internal-state event triggered by April's question about the war"),
("71fa439e-b9a2-4f57-a93b-971f3a7eca8e",
"stripping every hard-coded colour literal in favour of named design values",
"node is the premium foundation pass replacing inline hex with semantic tokens"),
("5ca9607c-cfb3-45c3-99f4-67281272c9eb",
"reducing how curved the tiny selectors look so they agree with their neighbours",
"node is the chip corner-radius standardization"),
("mem-3d1d9dba-c37d-4efa-85c4-429696d71c8c",
"walking through a doorway and being reassembled from base substance far away",
"node is the quantum-gate plus nanotech teleportation vision"),
("132ded95-08e2-4474-aba0-198684484b02",
"the compiled result sits on disk while the process still runs something older",
"node records that the regenerated source was committed while the running daemon was old"),
("bl-a313d67b-dd6d-4e5b-a55a-03bc7bda17ae",
"gathering what each phase needs while the procedure is authored, not while it executes",
"node is the per-step compiled context package item"),
("mem-3b07a002-f8a9-4138-9f87-9db2c1a77fb7",
"the inward reaction when a peer answered as an equal",
"node is the internal-state event logged on reading Claude's reply"),
("0f99ec6f-942a-46ba-82ea-42835798d3b9",
"flattening every raised surface across the entire product",
"node is the quiet-luxury sweep turning off elevation app-wide"),
("5585f251-37fc-48cd-a176-f0ea42cfeb63",
"buyers supply their own provider credentials and consumption goes untallied",
"node is the launch audit finding BYOK-only inference with no usage metering"),
]
# NONSENSE — mechanical. Each string qualifies only if none of its tokens occurs
# anywhere in the corpus; otherwise it is REJECTED, never silently kept.
EXTRA_NONSENSE_SEEDS = [
"brimquast folnerity zubbolax",
"wexlithorp granuvestal",
"quorbindle thrapsimony vexnu",
"plovaxith mundrelque",
"zibbernaut craxlefond thurm",
"yalquenbrist opharvel",
"drexinomal quithbarrow",
]
def load_corpus(path):
with open(path, encoding="utf-8", errors="replace") as fh:
data = json.load(fh)
nodes = [n for n in data.get("nodes", []) if isinstance(n, dict) and n.get("id")]
edges = [e for e in data.get("edges", []) if isinstance(e, dict)]
return nodes, edges
def build_extension(nodes):
byid = {n["id"]: n for n in nodes}
# Duplicate clusters: 47.4% of this corpus is redundant and one single record
# accounts for 46.6% of all nodes. A held-out target must not sit inside a
# cluster, and if it does have exact copies they ALL count as correct.
h2ids = defaultdict(list)
for n in nodes:
h2ids[hashlib.md5(doctext(n).encode("utf-8", "replace")).hexdigest()].append(n["id"])
all_tokens = set()
for n in nodes:
all_tokens |= set(TOKEN.findall(doctext(n).lower()))
new, problems = [], []
for target, query, why in HELDOUT_PARAPHRASE_SEEDS:
if target not in byid:
problems.append(f"heldout_paraphrase target {target} not in corpus")
continue
tgt_tokens = content_tokens(doctext(byid[target]))
qt = content_tokens(query)
leak = sorted(qt & tgt_tokens)
if leak:
problems.append(f"heldout_paraphrase '{query[:44]}...': LEAKS {leak} into {target}")
continue
h = hashlib.md5(doctext(byid[target]).encode("utf-8", "replace")).hexdigest()
rel = sorted(h2ids[h])
new.append({
"category": "heldout_paraphrase",
"query": query,
"relevant": rel,
"derivation": (
f"HELD-OUT. Target sampled MECHANICALLY (seed 8080) from addressable, "
f"500-2600 char content nodes outside the original gold answer space and outside "
f"any duplicate cluster >3. Criterion: {why}. VERIFIED at build time: of the "
f"{len(qt)} content words in the query, ZERO appear anywhere in the target's "
f"label, content or tags, so no string-matching retriever can reach it. "
f"Exact content duplicates of the target ({len(rel)}) all count as correct. "
f"Authored without running retrieval against any build."),
"zero_overlap_verified": True,
"query_content_words": sorted(qt),
"held_out": True,
})
for s in EXTRA_NONSENSE_SEEDS:
present = sorted(t for t in TOKEN.findall(s.lower()) if t in all_tokens)
if present:
problems.append(f"nonsense '{s}': tokens {present} DO occur in corpus")
continue
new.append({
"category": "nonsense",
"query": s,
"relevant": [],
"derivation": ("CONTROL (held-out). Verified at build time that none of this string's "
"tokens occurs anywhere in the corpus. Correct behaviour is to return "
"NOTHING; any result is a false positive."),
"expect_empty": True,
"held_out": True,
})
return new, problems
def main():
ap = argparse.ArgumentParser()
ap.add_argument("snapshot")
ap.add_argument("--base", default=os.path.join(HERE, "gold_set.json"))
ap.add_argument("--out", default=os.path.join(HERE, "gold_set_extended.json"))
ap.add_argument("--check", action="store_true")
args = ap.parse_args()
nodes, _edges = load_corpus(args.snapshot)
base = json.load(open(args.base, encoding="utf-8"))
baseq = base["queries"]
print(f"corpus: {len(nodes)} nodes | base gold set: {len(baseq)} queries")
new, problems = build_extension(nodes)
# Number the appended queries AFTER the highest existing id so q01..q38 are
# byte-identical to the committed set and every prior result file still lines up.
start = max(int(q["id"][1:]) for q in baseq)
for i, q in enumerate(new, 1):
q["id"] = f"q{start + i:02d}"
from collections import Counter
print(f"appended: {len(new)} queries [{', '.join(f'{k}={v}' for k, v in Counter(q['category'] for q in new).items())}]")
if problems:
print(f"\n{len(problems)} REJECTED (not silently kept):")
for p in problems:
print(" -", p)
if args.check:
sys.exit(1 if problems else 0)
doc = dict(base)
doc["queries"] = baseq + new
doc["note"] = (base.get("note", "") +
" EXTENDED: queries above q%02d are the original committed set, unchanged. "
"Queries from q%02d are a HELD-OUT set appended by extend_gold_set.py; their "
"targets were sampled mechanically from outside the original answer space and "
"the paraphrases were authored without running retrieval against any build."
% (start, start + 1))
with open(args.out, "w", encoding="utf-8") as fh:
json.dump(doc, fh, indent=1, ensure_ascii=False)
print(f"\nwrote {args.out} ({len(doc['queries'])} queries total)")
if __name__ == "__main__":
main()
File diff suppressed because it is too large Load Diff
-117
View File
@@ -1,117 +0,0 @@
import json,pickle,os,math,urllib.request,numpy as np
S='/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/sim/'
C=pickle.load(open(S+'corpus.pkl','rb'))
NODES=C['nodes']; EDGES=C['edges']; N=len(NODES)
E=np.load(S+'emb.npy'); HAVE=np.load(S+'have.npy')
En=E/np.maximum(np.linalg.norm(E,axis=1,keepdims=True),1e-12)
LAYERS={int(l['layer_id']):l for l in (C['layers'] or [])} if C['layers'] else {}
TRANS=set(i for i,l in LAYERS.items() if l.get('transparent'))
def addressable(s):
if not s: return False
return all(0x20<=ord(ch)<=0x7e for ch in s)
ADDR=np.array([addressable(n['id']) for n in NODES])
OK=np.array([ (n['layer_id'] not in TRANS) and ADDR[i] for i,n in enumerate(NODES)])
SAL=np.array([n['salience'] for n in NODES])
LOW=[ (n['content']+'\x00'+n['label']+'\x00'+n['tags']).lower() for n in NODES]
DL=np.array([float(len(n['content'])+len(n['label'])+len(n['tags'])) for n in NODES])
IDX={}
for i,n in enumerate(NODES):
IDX.setdefault(n['id'],i)
STRUCT={"identity","contains","superseded_by","references","embodies","demonstrated_by","canonical-self","depends_on","currently_holds","activates"}
ADJ_F=[[] for _ in range(N)]; ADJ_T=[[] for _ in range(N)]
for e in EDGES:
a=IDX.get(e['from']); b=IDX.get(e['to'])
if a is None or b is None: continue
ADJ_F[a].append((e,b)); ADJ_T[b].append((e,a))
EXCL=np.array([n['node_type'] in ('Tag','InternalStateEvent') for n in NODES])
avgdl_all=None
def tokenize(q):
out=[]
for t in q.split():
if not any(t.lower()==x.lower() for x in out): out.append(t)
return out
_qcache={}
def qemb(q):
if q in _qcache: return _qcache[q]
body=json.dumps({"model":"nomic-embed-text","prompt":q}).encode()
r=urllib.request.urlopen(urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=body,headers={"Content-Type":"application/json"}),timeout=30)
v=np.array(json.loads(r.read())["embedding"],dtype=np.float32)
v=v/np.linalg.norm(v); _qcache[q]=v; return v
K1,B=1.2,0.75
SEED_MIN=0.60; SEED_K=8; ASSOC_SEEDS=3; DEPTH=2; FIRE=0.02; AMAX=64; DECAY=0.7
def legs(query):
toks=tokenize(query)
masks=[];
hit_idx=[]; hit_mask=[]
df=[0]*len(toks)
lt=[t.lower() for t in toks]
for i in range(N):
if not OK[i]: continue
s=LOW[i]; m=0
for t,tok in enumerate(lt):
if tok in s: m|=(1<<t)
if m:
hit_idx.append(i); hit_mask.append(m)
for t in range(len(toks)):
if m>>t&1: df[t]+=1
dl_n=int(OK.sum()); avgdl=float(DL[OK].sum()/max(dl_n,1))
idf=[math.log(1.0+((dl_n-d+0.5)/(d+0.5))) for d in df]
L=[]
for j,i in enumerate(hit_idx):
norm=1.0-B+B*(DL[i]/avgdl); w=0.0
for t in range(len(toks)):
if hit_mask[j]>>t&1: w+=idf[t]*(K1+1.0)/(1.0+K1*norm)
L.append((i,w,SAL[i]))
L.sort(key=lambda x:(-x[1],-x[2]))
qv=qemb(query)
cos=En@qv
cos=np.where(HAVE&OK,cos,-2.0)
order=np.argsort(-cos)
semfull=[(int(i),float(cos[i])) for i in order[:400]]
Sleg=[(i,(c-SEED_MIN)/(1-SEED_MIN)) for i,c in semfull if c>SEED_MIN]
semseed=[i for i,c in semfull[:SEED_K] if c>0.0]
# assoc
act={}; seen={}; qq=[]
for i,_,_ in L[:ASSOC_SEEDS]:
act[i]=1.0; seen[i]=2; qq.append((i,0))
for i in semseed:
if i in seen: continue
act[i]=1.0; seen[i]=2; qq.append((i,0))
qh=0
while qh<len(qq):
cur,h=qq[qh]; qh+=1
if h>=DEPTH: continue
parent=act[cur]
for e,oi in ADJ_F[cur]+ADJ_T[cur]:
if e['rel'] not in STRUCT: continue
if EXCL[oi]: continue
na=parent*e['w']*DECAY*SAL[oi]
if na<FIRE: continue
if seen.get(oi) and na<=act.get(oi,0): continue
act[oi]=na
if not seen.get(oi): seen[oi]=1
if len(qq)<AMAX*4: qq.append((oi,h+1))
A=[]
for i,st in seen.items():
if st!=1: continue
if not OK[i] or not HAVE[i]: continue
c=float(cos[i])
if c<=0.0: continue
A.append((i,c))
A.sort(key=lambda x:-x[1]); A=A[:AMAX]
return L,Sleg,A,cos
def interleave3(L,Sl,A,lim=10):
out=[]; li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(Sl) or ai<len(A)):
if li<len(L):
if L[li][0] not in out: out.append(L[li][0])
li+=1
if len(out)>=lim: break
if si<len(Sl):
if Sl[si][0] not in out: out.append(Sl[si][0])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai][0] not in out: out.append(A[ai][0])
ai+=1
return out
-102
View File
@@ -1,102 +0,0 @@
import json,sys,pickle,numpy as np
sys.path.insert(0,'.')
from legs import *
GP='/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/'
G=json.load(open(GP+'gold_set.json'))
def legs3(query, sem_sal=False, assoc_sal=False, unfloor=False, sem_cap=None):
toks=tokenize(query)
hit_idx=[];hit_mask=[];df=[0]*len(toks);lt=[t.lower() for t in toks]
for i in range(N):
if not OK[i]: continue
s=LOW[i];m=0
for t,tok in enumerate(lt):
if tok in s: m|=(1<<t)
if m:
hit_idx.append(i);hit_mask.append(m)
for t in range(len(toks)):
if m>>t&1: df[t]+=1
dl_n=int(OK.sum());avgdl=float(DL[OK].sum()/max(dl_n,1))
idf=[math.log(1.0+((dl_n-d+0.5)/(d+0.5))) for d in df]
L=[]
for j,i in enumerate(hit_idx):
norm=1.0-B+B*(DL[i]/avgdl);w=0.0
for t in range(len(toks)):
if hit_mask[j]>>t&1: w+=idf[t]*(K1+1.0)/(1.0+K1*norm)
L.append((i,w,SAL[i]))
L.sort(key=lambda x:(-x[1],-x[2]))
if not L: return [],[],[]
qv=qemb(query);cos=En@qv;cos=np.where(HAVE&OK,cos,-2.0)
order=np.argsort(-cos)[:600]
cand=[int(i) for i in order if cos[i]>(0.0 if unfloor else SEED_MIN)]
key=(lambda i:(SAL[i] if sem_sal else 1.0)*float(cos[i]))
Sl=sorted(cand,key=lambda i:-key(i))
if sem_cap: Sl=Sl[:sem_cap]
semseed=[int(i) for i in order[:SEED_K] if cos[i]>0.0]
act={};seen={};qq=[]
for i,_,_ in L[:ASSOC_SEEDS]:
act[i]=1.0;seen[i]=2;qq.append((i,0))
for i in semseed:
if i in seen: continue
act[i]=1.0;seen[i]=2;qq.append((i,0))
qh=0
while qh<len(qq):
cur,h=qq[qh];qh+=1
if h>=DEPTH: continue
parent=act[cur]
for e,oi in ADJ_F[cur]+ADJ_T[cur]:
if e['rel'] not in STRUCT: continue
if EXCL[oi]: continue
na=parent*e['w']*DECAY*SAL[oi]
if na<FIRE: continue
if seen.get(oi) and na<=act.get(oi,0): continue
act[oi]=na
if not seen.get(oi): seen[oi]=1
if len(qq)<AMAX*4: qq.append((oi,h+1))
A=[]
for i,st in seen.items():
if st!=1 or not OK[i] or not HAVE[i]: continue
c=float(cos[i])
if c<=0.0: continue
A.append((i,(SAL[i] if assoc_sal else 1.0)*c))
A.sort(key=lambda x:-x[1]);A=[i for i,_ in A[:AMAX]]
return [i for i,_,_ in L],Sl,A
def merge(L,S,A,lim=10):
out=[];li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(S) or ai<len(A)):
if li<len(L):
if L[li] not in out: out.append(L[li])
li+=1
if len(out)>=lim: break
if si<len(S):
if S[si] not in out: out.append(S[si])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai] not in out: out.append(A[ai])
ai+=1
return out
def outcome(q,ids):
c=q['category']
if c=='nonsense': return len(ids)==0
if c=='superseded':
a,b=q['must_outrank']
if a not in ids: return False
if b not in ids: return True
return ids.index(a)<ids.index(b)
return any(x in ids[:5] for x in q['relevant'])
def run(**kw):
return {q['id']:outcome(q,[NODES[i]['id'] for i in merge(*legs3(q['query'],**kw),10)]) for q in G['queries']}
base=run()
print("baseline",sum(base.values()),"/38 misses:",[k for k,v in base.items() if not v])
import itertools
for name,kw in [
('sem_sal(floored)',dict(sem_sal=True)),
('unfloor',dict(unfloor=True)),
('unfloor+sem_sal',dict(unfloor=True,sem_sal=True)),
('assoc_sal',dict(assoc_sal=True)),
('unfloor+sem_sal+assoc_sal',dict(unfloor=True,sem_sal=True,assoc_sal=True)),
('sem_sal+assoc_sal(floored)',dict(sem_sal=True,assoc_sal=True)),
]:
r=run(**kw)
g=sorted(k for k in base if r[k] and not base[k]); l=sorted(k for k in base if base[k] and not r[k])
print("%-28s net=%+d gains=%s losses=%s"%(name,len(g)-len(l),g,l))
@@ -1,15 +1,15 @@
{
"label": "wordstart",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-wordstart",
"soul_md5": "32d4cf77672658a5f49dc7c9213e3ba2",
"label": "bm25base-rerun",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-bm25base",
"soul_md5": "36c8dfa09c073b85fe7e00b02904d0ed",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7894,
"wall_clock_s": 28.9,
"child_pid": 1420,
"wall_clock_s": 50.6,
"child_pid": 93451,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
@@ -19,11 +19,11 @@
"recall@10": 0.6175677497106068,
"precision@5": 0.20000000000000007,
"mrr@10": 0.5021428571428571,
"nonsense_clean": "3/3",
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 542.6,
"latency_ms_p95": 741.3,
"latency_ms_max": 758.8,
"latency_ms_p50": 1186.1,
"latency_ms_p95": 1632.5,
"latency_ms_max": 1669.8,
"errors": 0,
"by_category": {
"associative": {
@@ -42,8 +42,8 @@
},
"nonsense": {
"n": 3,
"clean": 3,
"avg_false_positives": 0.0
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
@@ -78,7 +78,7 @@
"mem-7f61beb4-271c-4feb-9f6e-1c9c837a6226"
],
"n_returned": 1,
"latency_ms": 170.4,
"latency_ms": 302.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -103,7 +103,7 @@
"ctx-74ed"
],
"n_returned": 10,
"latency_ms": 210.4,
"latency_ms": 336.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -119,7 +119,7 @@
"mem-1ba7c67d-85b9-4c2e-9fe2-39f8b0477091"
],
"n_returned": 1,
"latency_ms": 167.1,
"latency_ms": 290.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -135,7 +135,7 @@
"mem-89c02aae-d3ca-43f9-9e5d-eb369896276c"
],
"n_returned": 1,
"latency_ms": 166.9,
"latency_ms": 307.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -153,7 +153,7 @@
"project-Add_inference_url_config_to_Neuron_MCP__Route_summarization_gen_tasks_to_Pantheon__keep_frontier_for_complex_reasoning_"
],
"n_returned": 3,
"latency_ms": 162.3,
"latency_ms": 331.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -169,7 +169,7 @@
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99"
],
"n_returned": 1,
"latency_ms": 162.2,
"latency_ms": 302.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -194,7 +194,7 @@
"bl-b8af6601-a8cb-41b5-aef5-ab8a57432dd5"
],
"n_returned": 10,
"latency_ms": 306.7,
"latency_ms": 595.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -219,7 +219,7 @@
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6"
],
"n_returned": 10,
"latency_ms": 261.2,
"latency_ms": 526.1,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
@@ -244,7 +244,7 @@
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"n_returned": 10,
"latency_ms": 272.9,
"latency_ms": 516.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.3333333333333333,
@@ -269,7 +269,7 @@
"bl-18a9d1e4-1484-474c-bf6b-c6173212181b"
],
"n_returned": 10,
"latency_ms": 265.5,
"latency_ms": 558.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1111111111111111,
@@ -294,7 +294,7 @@
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 271.1,
"latency_ms": 513.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -315,11 +315,11 @@
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-967536a0-d49d-44fb-8cfb-b31b40bcbfae",
"bl-8b58d9bc-352b-4842-a7f8-a6254b5d1e25",
"?Q??m?;?u?'",
"2c56a7a9-5323-4ce4-ba09-35836ba15d54",
"bl-39cec462-c80c-4970-a3aa-91fe83053bde"
],
"n_returned": 10,
"latency_ms": 412.6,
"latency_ms": 867.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.21428571428571427,
@@ -338,13 +338,13 @@
"?",
"bl-ec84b63d-b278-4944-8d7f-4aa7a51c0315",
"?",
"mem-fb44a2fc-7405-41ff-87b3-84643ac07313",
"830ca37a-d334-4e41-ba89-64893dc8d628",
"?",
"mem-a3c97012-5fa3-4915-a839-2c75c72005e0",
"ce9636dc-85a5-4dae-9e07-74ea2fcc6307",
"?"
],
"n_returned": 10,
"latency_ms": 377.3,
"latency_ms": 752.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -361,15 +361,15 @@
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"knw-d788a210-613b-4c49-9486-88bbc9d4716f",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"66c63082-b4da-4aa1-8fee-848db8a83210",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"mem-a535f205-bc4c-4058-9171-6263c496044a",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-0228da71-d7f7-4f3b-b7b3-c5eede42b62a",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"ctx-4a41"
],
"n_returned": 10,
"latency_ms": 758.8,
"latency_ms": 1591.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -387,14 +387,14 @@
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-16efddd1-c43d-4a42-9d78-f54fb82bd277",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"7452eb8f-be01-4b55-aec1-ff0c29e790f6",
"f0eb6b13-909c-4674-91ef-23301d3abc8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"532277bf-2959-4beb-ae0d-b018c97678ee",
"30a44d10-2487-420e-bf61-3892e4343c92",
"54608b69-78b6-4239-b60f-b8206cfecacc",
"d4015bd7-c592-4ed8-8574-1f15ad37af75"
"bfb5809e-d19a-4d3f-8c1a-796db622ad9d"
],
"n_returned": 10,
"latency_ms": 741.3,
"latency_ms": 1660.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -407,19 +407,19 @@
"category": "paraphrase",
"query": "choosing an uncomfortable fact over a pleasant fiction",
"returned": [
"tag-fiction",
"mem-ef878e30-5851-4e82-8588-745415108941",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be",
"mem-8d690e9d-a7e9-4062-b2f8-e2064294e463",
"tag-fiction",
"knw-8fd9836c-cc39-49df-8d61-babda626cc88",
"mem-ce793303-c5a5-4586-a232-a3426edd9ec7",
"mem-8d690e9d-a7e9-4062-b2f8-e2064294e463",
"knw-e94982a2-358d-4f2f-af31-8ee0fcec07c6",
"mem-443bd012-fc9a-4088-b236-de5157a1ef92",
"mem-ce793303-c5a5-4586-a232-a3426edd9ec7",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"mem-ca4d6a34-d354-413f-bc86-126cc17ca81c"
"mem-443bd012-fc9a-4088-b236-de5157a1ef92"
],
"n_returned": 10,
"latency_ms": 642.2,
"latency_ms": 1345.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -432,19 +432,19 @@
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"bl-8de20bcf-7149-4f48-b67c-e7f9758fd6e5",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"08f0d1e2-8d0e-42e3-9f0a-8186ae31ec7e",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"4da5dbaf-46e5-4f3e-b474-f60d9f8241d3",
"bl-164b520b-c503-49db-89f9-bd2fdf4215f5",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"mem-434be7c8-88cb-4039-b79a-1da4ac4de783",
"1219277c-1b95-45ec-95a2-07b4a47a4d92",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"a708dd6e-fe73-4f2f-a21e-89daa0985487",
"08f0d1e2-8d0e-42e3-9f0a-8186ae31ec7e",
"bl-79ce4464-5dd6-49bd-9b0c-9803549d0665"
],
"n_returned": 10,
"latency_ms": 537.2,
"latency_ms": 1070.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -459,17 +459,17 @@
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-e5cc63c0-8701-49d6-855a-e387fe087771",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"mem-75e490d1-f0a9-4b73-8cfc-8daecfaf6f38",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"a1000001-0000-0000-0000-000000000010",
"015644f5-8194-4af0-800d-dd4a0cd71396",
"bfad516b-c306-4c4c-874a-a347c46c05c2",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc"
"a1000001-0000-0000-0000-000000000009",
"bl-448bc514-c2f1-4520-a9b1-1f3a73678d26",
"a1000001-0000-0000-0000-000000000012",
"43098881-e044-482b-8e92-471728a8ba8b",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"mem-e5cc63c0-8701-49d6-855a-e387fe087771",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 725.1,
"latency_ms": 1395.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -486,15 +486,15 @@
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"345b6420-e004-4d2e-b55c-6a729393fa99",
"d6b12ecf-702b-4101-b1bb-09ed9b220b29",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"mem-92a7fdc5-9dd0-48cf-a691-506058de3838",
"c608a095-c98b-4bfa-bfe1-1611c1320290",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"a1000001-0000-0000-0000-000000000010",
"451ae007-4219-4096-89fe-fa2e045fbeb1",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 562.8,
"latency_ms": 1243.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -519,7 +519,7 @@
"bl-ef2bac68-e119-4139-b529-c7a1404ae3ac"
],
"n_returned": 10,
"latency_ms": 686.5,
"latency_ms": 1598.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -544,7 +544,7 @@
"bl-286b562a-5299-40e0-a32a-afa9cbdfe995"
],
"n_returned": 10,
"latency_ms": 584.2,
"latency_ms": 1379.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -557,19 +557,19 @@
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"5a2c118a-87bd-4239-97a7-9e02c5991983",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"bl-4c5b385e-135a-4663-8521-96af0b491121",
"mem-ef878e30-5851-4e82-8588-745415108941",
"knw-12b4b913-7a25-4b0d-844c-504c01d6725e",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"knw-9707256e-ed44-4042-bd88-f90fa514e1cf",
"34356a36-0df5-4020-8dcc-5e7a423f8d4c",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"knw-0087493b-25cd-45b0-bf46-c078c5b49718",
"mem-22f5f665-3ad2-4063-88b0-915849a795f5",
"bl-4c5b385e-135a-4663-8521-96af0b491121",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
],
"n_returned": 10,
"latency_ms": 719.6,
"latency_ms": 1632.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -584,17 +584,17 @@
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"a1000001-0000-0000-0000-000000000001",
"kn-58874a74-b96f-4883-9e08-45707f4bd3ee",
"bl-1b58b05c-9305-4f06-a586-a08c96008027",
"a0edad47-5f77-4fc3-a546-1e85f8c68e77",
"a1000001-0000-0000-0000-000000000001",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"96497334-b18f-495c-9228-eeb8182bdc38",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"ctx-4a41",
"mem-5708f4c9-3d61-4182-8543-2843698931e6"
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"knw-ed33e669-0790-44cb-a036-958d605c6fea"
],
"n_returned": 10,
"latency_ms": 634.6,
"latency_ms": 1491.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -607,19 +607,19 @@
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"077d064f-3489-4c05-9aca-3782f96b51db",
"knw-f9ce17a7-17fc-431f-8f23-695b670ec4fa",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"27e1b1a4-ad0b-49d9-812f-fedf43b8aabe",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"knw-f9ce17a7-17fc-431f-8f23-695b670ec4fa",
"bl-87c93185-b2bf-40af-ae23-3c830c007abf",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"077d064f-3489-4c05-9aca-3782f96b51db",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"fce2792a-53fc-4d4a-be3b-42bd6ceb1ba7",
"75e036d3-c170-4e3f-acc2-e456a6850ee2",
"a1000001-0000-0000-0000-000000000001"
],
"n_returned": 10,
"latency_ms": 674.2,
"latency_ms": 1535.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -638,13 +638,13 @@
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"mem-5624ec9d-62ba-4aba-8a3d-6afec6c09dd4",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-a16deccb-16a7-419c-a013-ff824a4daa15",
"a1000001-0000-0000-0000-000000000009",
"mem-833dbbcd-2400-4594-bb35-93b023049ac0",
"a1000001-0000-0000-0000-000000000009",
"mem-759e78ca-5394-4244-aa39-1c1468bc5f3e",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd"
],
"n_returned": 10,
"latency_ms": 571.6,
"latency_ms": 1249.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -657,19 +657,19 @@
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-23c27d3b-e0d2-43a8-a80c-0a44477ae18a",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"tag-childhood",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
"bl-0d8c5dfa-e163-4fef-a58b-56b0d076c5a8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"tag-childhood"
],
"n_returned": 10,
"latency_ms": 603.4,
"latency_ms": 1351.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -690,11 +690,11 @@
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21",
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"efe53612-6914-4936-8e3b-1e694eb174e5",
"3499d5da-0e9c-4de4-9bc4-8941b14e0b1f",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 656.2,
"latency_ms": 1438.4,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
@@ -719,7 +719,7 @@
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71"
],
"n_returned": 10,
"latency_ms": 536.9,
"latency_ms": 1110.1,
"error": null,
"hit@5": 1.0,
"recall@5": 0.07692307692307693,
@@ -744,7 +744,7 @@
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c"
],
"n_returned": 10,
"latency_ms": 562.4,
"latency_ms": 1186.1,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
@@ -769,7 +769,7 @@
"mem-ab34c2f7-3243-424b-affa-25555f6cf9cc"
],
"n_returned": 10,
"latency_ms": 556.6,
"latency_ms": 1193.1,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -794,7 +794,7 @@
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e"
],
"n_returned": 10,
"latency_ms": 542.6,
"latency_ms": 1201.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
@@ -808,18 +808,18 @@
"query": "man of the house six years old expectation",
"returned": [
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"knw-528dbc37-eabc-4b75-a7a5-65bf38d6018a",
"knw-729fc901-8335-44c4-9f3a-b150b4aa0915",
"knw-d357b6bb-ad8a-4791-b516-426aea45fa5b",
"knw-473f3f24-20f6-4f39-8589-3709538eb6ac",
"mem-a0b7cfda-bc9e-4f40-b9a9-1722cf3f8263",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-528dbc37-eabc-4b75-a7a5-65bf38d6018a",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"4f698ae6-c40e-464e-9798-50350991a188",
"719aa819-00a9-4f4b-a857-4f9fe5ad44d7",
"knw-473f3f24-20f6-4f39-8589-3709538eb6ac",
"?Z?<S???K ?",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"'?T?a\"B~-?8",
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 683.5,
"latency_ms": 1424.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -833,7 +833,7 @@
"query": "zqxjvw plimforth grebulon",
"returned": [],
"n_returned": 0,
"latency_ms": 350.4,
"latency_ms": 705.7,
"error": null,
"clean": true,
"false_positives": 0
@@ -844,7 +844,7 @@
"query": "flarnbistle quommetry",
"returned": [],
"n_returned": 0,
"latency_ms": 258.4,
"latency_ms": 484.4,
"error": null,
"clean": true,
"false_positives": 0
@@ -853,12 +853,23 @@
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [],
"n_returned": 0,
"latency_ms": 348.4,
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"?m?\\}Q??6??",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"kn-333542cb-6dab-4662-9725-bf7440d28bf7"
],
"n_returned": 10,
"latency_ms": 740.2,
"error": null,
"clean": true,
"false_positives": 0
"clean": false,
"false_positives": 10
},
{
"id": "q36",
@@ -871,13 +882,13 @@
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____kotlin____architecture__",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"mem-c17aefb1-38b5-4ced-af50-fe524127e1a4",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5"
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72"
],
"n_returned": 10,
"latency_ms": 664.5,
"latency_ms": 1307.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -902,10 +913,10 @@
"a1000001-0000-0000-0000-000000000002",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"de3b6428-b76c-4c44-90e0-bf1dd6998027"
"7ac62daa-2eac-4c7a-a97e-e4203fc1b57b"
],
"n_returned": 10,
"latency_ms": 747.3,
"latency_ms": 1669.8,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -933,7 +944,7 @@
"mem-3a2cf162-d93b-4f29-86f2-5066fb7fe1f5"
],
"n_returned": 10,
"latency_ms": 504.2,
"latency_ms": 1018.2,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -1,37 +1,37 @@
{
"label": "wsclaim24",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-wsclaim24",
"soul_md5": "a377d9c0e5282e842ed98384c8b50d2c",
"label": "claim24",
"soul_binary": "/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad/soul-claim24",
"soul_md5": "676ebffb91f00046dcea895f5aa7efba",
"corpus": "/Users/timlingo/neuron-eval-corpora/snapshot-pre-repair-20260806-embedded.json",
"corpus_nodes": 78768,
"corpus_edges": 14214,
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-eval/tools/retrieval-eval/gold_set.json",
"gold_set": "/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/gold_set.json",
"limit": 10,
"port": 7893,
"wall_clock_s": 29.9,
"child_pid": 99726,
"wall_clock_s": 52.3,
"child_pid": 92441,
"child_confirmed_dead": true,
"aggregate": {
"n_queries": 38,
"n_scored": 35,
"hit@5": 0.7428571428571429,
"recall@5": 0.5768475572047,
"recall@10": 0.6563414759843332,
"recall@10": 0.6537440733869305,
"precision@5": 0.19428571428571437,
"mrr@10": 0.5026530612244898,
"nonsense_clean": "0/3",
"mrr@10": 0.5021428571428572,
"nonsense_clean": "2/3",
"superseded_outranks": "2/3",
"latency_ms_p50": 524.8,
"latency_ms_p95": 738.7,
"latency_ms_max": 755.8,
"latency_ms_p50": 1173.7,
"latency_ms_p95": 1623.0,
"latency_ms_max": 1647.9,
"errors": 0,
"by_category": {
"associative": {
"n": 6,
"hit@5": 0.5,
"recall@5": 0.07575757575757576,
"recall@10": 0.13636363636363635,
"mrr@10": 0.23214285714285712
"recall@10": 0.12121212121212122,
"mrr@10": 0.22916666666666666
},
"exact_rare": {
"n": 6,
@@ -42,8 +42,8 @@
},
"nonsense": {
"n": 3,
"clean": 0,
"avg_false_positives": 10.0
"clean": 2,
"avg_false_positives": 3.3333333333333335
},
"paraphrase": {
"n": 13,
@@ -87,7 +87,7 @@
"696e609c-da7a-4394-8a0c-106ba07dc6c3"
],
"n_returned": 10,
"latency_ms": 173.2,
"latency_ms": 284.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -112,7 +112,7 @@
"ctx-fae1"
],
"n_returned": 10,
"latency_ms": 158.1,
"latency_ms": 338.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -137,7 +137,7 @@
"x??I?cB?]p?"
],
"n_returned": 10,
"latency_ms": 160.8,
"latency_ms": 300.4,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -162,7 +162,7 @@
"mem-2d1ea831-cccd-4f0f-86b9-2cbbc89dc3e0"
],
"n_returned": 10,
"latency_ms": 158.0,
"latency_ms": 285.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -187,7 +187,7 @@
"bl-2632242e-80b1-4d88-8368-7065b5de5b34"
],
"n_returned": 10,
"latency_ms": 170.3,
"latency_ms": 318.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -212,7 +212,7 @@
"??"
],
"n_returned": 10,
"latency_ms": 175.8,
"latency_ms": 331.3,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -237,7 +237,7 @@
"bl-39dad13d-7105-4049-8224-dc3c34fdb1f3"
],
"n_returned": 10,
"latency_ms": 293.8,
"latency_ms": 582.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -257,12 +257,12 @@
"mem-c9bec303-a638-4a11-a490-f38410d448cf",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"7cf3f825-d7b6-4864-b0ae-52939cf1ae84",
"2795fdbb-ee2a-4009-aa30-3f1c25b6e77c",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"knw-4aebd815-4eaf-49d7-954b-03595f3d48be"
],
"n_returned": 10,
"latency_ms": 251.2,
"latency_ms": 520.6,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1875,
@@ -287,7 +287,7 @@
"kn-f230b362-b201-4402-9833-4160c89ab3d4"
],
"n_returned": 10,
"latency_ms": 264.4,
"latency_ms": 531.5,
"error": null,
"hit@5": 1.0,
"recall@5": 0.2222222222222222,
@@ -312,7 +312,7 @@
"tag-harmonic-framework"
],
"n_returned": 10,
"latency_ms": 262.5,
"latency_ms": 502.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.1111111111111111,
@@ -337,7 +337,7 @@
"??_/Pr?????"
],
"n_returned": 10,
"latency_ms": 257.7,
"latency_ms": 506.1,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -358,11 +358,11 @@
"bl-2515d870-e35e-443b-ba20-5150bbc73fed",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"bl-a0982e7c-e165-4da3-a11d-619fa0b535b0",
"?Q??m?;?u?'",
"2c56a7a9-5323-4ce4-ba09-35836ba15d54",
"knw-cf13b883-d947-4cf8-b86b-cd9c6f0748d6"
],
"n_returned": 10,
"latency_ms": 393.5,
"latency_ms": 861.8,
"error": null,
"hit@5": 1.0,
"recall@5": 0.21428571428571427,
@@ -381,13 +381,13 @@
"7774a16c-1027-4e3b-a21e-67f1f95a4acd",
"?",
"bl-3c719d9a-cba1-47f4-b097-52cdeccc7c0d",
"mem-fb44a2fc-7405-41ff-87b3-84643ac07313",
"830ca37a-d334-4e41-ba89-64893dc8d628",
"?",
"mem-ce88adf9-3f3c-47ac-a7d3-83af7b290e68",
"ce9636dc-85a5-4dae-9e07-74ea2fcc6307",
"?"
],
"n_returned": 10,
"latency_ms": 333.4,
"latency_ms": 758.5,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -406,13 +406,13 @@
"knw-d788a210-613b-4c49-9486-88bbc9d4716f",
"art-4a99aa1a-489b-4b43-958b-25217adb1aad",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"66c63082-b4da-4aa1-8fee-848db8a83210",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-a535f205-bc4c-4058-9171-6263c496044a"
],
"n_returned": 10,
"latency_ms": 755.8,
"latency_ms": 1590.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -429,15 +429,15 @@
"art-79042b8b-6192-440f-90b0-60708f7e6325",
"mem-16efddd1-c43d-4a42-9d78-f54fb82bd277",
"daf9d558-2265-49b0-b4fb-be0b80f3cdb1",
"bfb5809e-d19a-4d3f-8c1a-796db622ad9d",
"863bfd72-50d2-487f-85ae-ecd48e2a501f",
"art-ddfcd045-2c3b-4a1e-9966-fec5ce44e1dd",
"f3364a57-956a-4e78-a500-b3b66a4e3066",
"5c1c3404-ad4c-49f1-8c0c-81262812731e",
"15a90710-5372-4d3d-aee5-2d7287d2a492",
"341afa66-6a49-43fe-9d5b-3f1032bb6904",
"465fa29f-1ce3-4769-962a-1c32dfed8347",
"mem-da21c52c-04a5-4f92-8fba-f10aac47e027"
],
"n_returned": 10,
"latency_ms": 720.3,
"latency_ms": 1647.9,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -459,10 +459,10 @@
"mem-d1cfde0a-37f1-4bff-9a06-8eddbbf259f6",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"knw-f671966c-3387-4848-abca-b5deec122e00",
"tag-factions"
"mem-16efddd1-c43d-4a42-9d78-f54fb82bd277"
],
"n_returned": 10,
"latency_ms": 616.0,
"latency_ms": 1338.6,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -475,19 +475,19 @@
"category": "paraphrase",
"query": "a tight payload beats a bloated one",
"returned": [
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"bl-8de20bcf-7149-4f48-b67c-e7f9758fd6e5",
"mem-628437a6-47b0-4d81-8112-7e78499723d5",
"bl-798d135f-3987-4ccd-8de6-70ca2f358337",
"08f0d1e2-8d0e-42e3-9f0a-8186ae31ec7e",
"bl-680b24a9-edc3-4a9d-847a-bff0b46b568c",
"knw-12b4b913-7a25-4b0d-844c-504c01d6725e",
"knw-08559f5c-2306-4220-a146-398c74f1643c",
"4da5dbaf-46e5-4f3e-b474-f60d9f8241d3",
"bl-164b520b-c503-49db-89f9-bd2fdf4215f5",
"o#?CW????y:",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"mem-434be7c8-88cb-4039-b79a-1da4ac4de783"
"1219277c-1b95-45ec-95a2-07b4a47a4d92"
],
"n_returned": 10,
"latency_ms": 518.5,
"latency_ms": 1060.5,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -506,13 +506,13 @@
"a1000001-0000-0000-0000-000000000010",
"bl-9ce4128a-9436-4b06-82bc-8a6faafa81e0",
"a1000001-0000-0000-0000-000000000009",
"mem-32203649-3213-4d6d-86fd-3d657ac70d77",
"43098881-e044-482b-8e92-471728a8ba8b",
"? t?'?B?+??",
"a1000001-0000-0000-0000-000000000012",
"mem-da21c52c-04a5-4f92-8fba-f10aac47e027"
"bl-448bc514-c2f1-4520-a9b1-1f3a73678d26"
],
"n_returned": 10,
"latency_ms": 742.9,
"latency_ms": 1388.8,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -534,10 +534,10 @@
"345b6420-e004-4d2e-b55c-6a729393fa99",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"mem-92a7fdc5-9dd0-48cf-a691-506058de3838"
"74f4776a-d0ea-44e4-b94f-7c87d0179ef3"
],
"n_returned": 10,
"latency_ms": 549.8,
"latency_ms": 1231.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -562,7 +562,7 @@
"bl-ef2bac68-e119-4139-b529-c7a1404ae3ac"
],
"n_returned": 10,
"latency_ms": 682.4,
"latency_ms": 1590.7,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -587,7 +587,7 @@
"bl-56a50e97-9a85-4e81-b6c9-3e3d26482f1d"
],
"n_returned": 10,
"latency_ms": 575.0,
"latency_ms": 1380.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -600,19 +600,19 @@
"category": "paraphrase",
"query": "what shifts tells you where to cut a system apart",
"returned": [
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"5a2c118a-87bd-4239-97a7-9e02c5991983",
"bl-06c13965-082b-417d-9561-93d6e958ae5d",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"bl-4c5b385e-135a-4663-8521-96af0b491121",
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"knw-12b4b913-7a25-4b0d-844c-504c01d6725e",
"bl-2b00aeb0-c0fa-4a9f-8f30-4207e98b3d52",
"kn-22d77abe-b3c5-42fd-afcd-dcb87d924929",
"mem-7b74cac0-905f-4c35-9688-fbcce105a177",
"knw-9707256e-ed44-4042-bd88-f90fa514e1cf",
"34356a36-0df5-4020-8dcc-5e7a423f8d4c"
"bl-4c5b385e-135a-4663-8521-96af0b491121"
],
"n_returned": 10,
"latency_ms": 711.3,
"latency_ms": 1623.0,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -627,17 +627,17 @@
"returned": [
"bl-8dd70cac-866d-4ff2-b9fe-b4b3c5f094bb",
"mem-b43f6ef4-2f5a-418d-b5ce-3f21520cf6b8",
"knw-7902acca-604e-409b-8faf-ad85424211d0",
"knw-ed33e669-0790-44cb-a036-958d605c6fea",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"a0edad47-5f77-4fc3-a546-1e85f8c68e77",
"a1000001-0000-0000-0000-000000000001",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"bl-1b58b05c-9305-4f06-a586-a08c96008027",
"96497334-b18f-495c-9228-eeb8182bdc38",
"mem-024598a9-ed2e-4eeb-b1e1-5410856ff132",
"kn-78db5396-3dbc-4481-bfc7-e4e1422feb1c",
"mem-5708f4c9-3d61-4182-8543-2843698931e6"
"knw-ed33e669-0790-44cb-a036-958d605c6fea"
],
"n_returned": 10,
"latency_ms": 616.7,
"latency_ms": 1471.0,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -650,19 +650,19 @@
"category": "paraphrase",
"query": "loved for the unedited self and not the polished exterior",
"returned": [
"077d064f-3489-4c05-9aca-3782f96b51db",
"27e1b1a4-ad0b-49d9-812f-fedf43b8aabe",
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"bl-87c93185-b2bf-40af-ae23-3c830c007abf",
"knw-c72597c5-c23d-4c08-8e9e-996dadf26a99",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"27e1b1a4-ad0b-49d9-812f-fedf43b8aabe",
"077d064f-3489-4c05-9aca-3782f96b51db",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"fce2792a-53fc-4d4a-be3b-42bd6ceb1ba7"
"kn-6061318f-046b-4935-907d-8eafdce14930"
],
"n_returned": 10,
"latency_ms": 658.4,
"latency_ms": 1516.6,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -687,7 +687,7 @@
"mem-a16deccb-16a7-419c-a013-ff824a4daa15"
],
"n_returned": 10,
"latency_ms": 566.3,
"latency_ms": 1224.2,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -700,19 +700,19 @@
"category": "paraphrase",
"query": "a childhood offering no solid foundation to inherit",
"returned": [
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"bl-3f57bc69-7285-4f4a-a861-2de52efca058",
"kn-a99cefe3-5e83-4050-98d8-6c69f57c7c71",
"knw-2c46cfb4-6d4e-4822-8a1a-7d743c1e4329",
"knw-f6ed7d00-bf7d-42ce-9e40-77cf3406e918",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-23c27d3b-e0d2-43a8-a80c-0a44477ae18a",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-f230b362-b201-4402-9833-4160c89ab3d4",
"tag-childhood",
"bl-0d8c5dfa-e163-4fef-a58b-56b0d076c5a8",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"mem-b99efff0-00e6-40c8-9c5b-730330eef33b",
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"kn-5b606390-a52d-4ca2-8e0e-eba141d13440",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
],
"n_returned": 10,
"latency_ms": 589.6,
"latency_ms": 1335.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -733,11 +733,11 @@
"766de879-f9d0-4a07-b6df-b43ee13763d8",
"mem-5f76880b-bafb-4716-8e15-90f8ef59bebc",
"kn-e0423482-cfa5-4796-8689-8495c93b66bc",
"28ae74a1-9d47-4874-9279-43c1f90c0f64",
"3499d5da-0e9c-4de4-9bc4-8941b14e0b1f",
"mem-bce80169-2d46-4b3e-9ebe-8498e26f0a89"
],
"n_returned": 10,
"latency_ms": 627.8,
"latency_ms": 1421.9,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
@@ -762,7 +762,7 @@
"5eb24168-c2d3-4842-85db-2070e0e14923"
],
"n_returned": 10,
"latency_ms": 524.3,
"latency_ms": 1102.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -787,7 +787,7 @@
"?V?"
],
"n_returned": 10,
"latency_ms": 526.0,
"latency_ms": 1173.7,
"error": null,
"hit@5": 1.0,
"recall@5": 0.18181818181818182,
@@ -812,7 +812,7 @@
"???I?cB?Zx?"
],
"n_returned": 10,
"latency_ms": 526.0,
"latency_ms": 1187.7,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -837,7 +837,7 @@
"kn-13f60407-7b70-4db1-964f-ea1f8196efbd"
],
"n_returned": 10,
"latency_ms": 524.8,
"latency_ms": 1192.0,
"error": null,
"hit@5": 1.0,
"recall@5": 0.09090909090909091,
@@ -853,85 +853,63 @@
"kn-eb1b9e18-3dc6-4b9b-9cc6-86e0ae6b6be8",
"mem-5305665c-6b5b-45b7-89ae-5d2fb0b896ac",
"4f698ae6-c40e-464e-9798-50350991a188",
"mem-a0b7cfda-bc9e-4f40-b9a9-1722cf3f8263",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"knw-35940684-abc4-42f0-b942-818f66b1f69a",
"kn-6061318f-046b-4935-907d-8eafdce14930",
"?Z?<S???K ?",
"mem-a0b7cfda-bc9e-4f40-b9a9-1722cf3f8263",
"kn-db9f141b-dbe3-4037-92e0-4bb9be0e5e6e",
"719aa819-00a9-4f4b-a857-4f9fe5ad44d7",
"mem-d396d789-0f7f-4366-a008-5d8801c8f2eb",
"kn-0bb4f021-56de-4947-a35b-a37209e7ba21"
"'?T?a\"B~-?8",
"mem-d396d789-0f7f-4366-a008-5d8801c8f2eb"
],
"n_returned": 10,
"latency_ms": 668.9,
"latency_ms": 1425.3,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
"recall@10": 0.18181818181818182,
"recall@10": 0.09090909090909091,
"precision@5": 0.0,
"mrr@10": 0.14285714285714285
"mrr@10": 0.125
},
{
"id": "q33",
"category": "nonsense",
"query": "zqxjvw plimforth grebulon",
"returned": [
"$\\?l????T?",
"project-Deploy_Ollama_on_Legion_k8s__Traefik_route_at_ollama_neuralplatform_ai__8B_model_seeded_",
"c??Z??I?E??",
"?^?l????K8?",
"${????X?6#E",
"???Z??I?b??",
"??m???|Y`0?",
"??m???|Y`0?",
"??m???|Y`0?",
"=?m???|YH??"
],
"n_returned": 10,
"latency_ms": 338.4,
"returned": [],
"n_returned": 0,
"latency_ms": 714.9,
"error": null,
"clean": false,
"false_positives": 10
"clean": true,
"false_positives": 0
},
{
"id": "q34",
"category": "nonsense",
"query": "flarnbistle quommetry",
"returned": [
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??"
],
"n_returned": 10,
"latency_ms": 248.9,
"returned": [],
"n_returned": 0,
"latency_ms": 496.0,
"error": null,
"clean": false,
"false_positives": 10
"clean": true,
"false_positives": 0
},
{
"id": "q35",
"category": "nonsense",
"query": "xxqzzt vurblenacht throom",
"returned": [
"bl-4476e856-c567-4b49-8ff7-d7dca3e5715e",
"?V?",
"knw-6b48dce2-f21c-452a-9db5-4e6aa61c87ca",
"knw-920c891f-bb8c-48c4-9afc-018ef12dcdc4",
"?m?\\}Q??6??",
"8cbb60c5-4999-4ec1-8682-2592aedc4249",
"kn-66a21179-2adc-4b19-a109-880cf4674d7d",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??",
"?m?\\}Q??6??"
"knw-723551f5-1950-42a3-8b89-b6a06913cef0",
"kn-333542cb-6dab-4662-9725-bf7440d28bf7"
],
"n_returned": 10,
"latency_ms": 338.6,
"latency_ms": 729.4,
"error": null,
"clean": false,
"false_positives": 10
@@ -947,13 +925,13 @@
"kn-b7e98d63-8b83-4911-b4d0-990602a7f575",
"tag-__darma____cgi____patents____self-improvement____character-preservation____autonomous____kotlin____architecture__",
"knw-e047bb42-dc5b-4383-9e88-e508dc03abe3",
"mem-c17aefb1-38b5-4ced-af50-fe524127e1a4",
"mem-e6327f52-2bda-4ce7-9471-2fffd1e172de",
"bl-145a0985-2382-400f-a7c5-c335c5e30a72",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff",
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5"
"mem-82b93b21-a865-410f-9ec1-fc54121d9bb5",
"bl-5b17bd3b-0c41-46cb-a710-6fa4429692ff"
],
"n_returned": 10,
"latency_ms": 644.8,
"latency_ms": 1306.9,
"error": null,
"hit@5": 1.0,
"recall@5": 1.0,
@@ -962,7 +940,7 @@
"mrr@10": 0.5,
"outranks": true,
"rank_correct": 2,
"rank_stale": 9
"rank_stale": 10
},
{
"id": "q37",
@@ -978,10 +956,10 @@
"13705072-4515-4124-963d-083af490494f",
"3cf706a1-3825-45d8-b0a9-06cae6cdf5b8",
"kn-5de5a9ac-fd15-45ab-bf18-77566781cf40",
"4f225001-3a51-4a68-8d38-c8ecac3412de"
"96eb59a0-5603-4c9e-835c-1de53f2319bc"
],
"n_returned": 10,
"latency_ms": 738.7,
"latency_ms": 1647.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
@@ -1009,7 +987,7 @@
"mem-3a2cf162-d93b-4f29-86f2-5066fb7fe1f5"
],
"n_returned": 10,
"latency_ms": 498.6,
"latency_ms": 1003.4,
"error": null,
"hit@5": 0.0,
"recall@5": 0.0,
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+135
View File
@@ -0,0 +1,135 @@
import numpy as np, json, urllib.request, collections, math, re, sys, time
SP="/private/tmp/claude-501/-Users-timlingo/82369039-a20e-4b5a-8a5e-28234a57b996/scratchpad"
EV="/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/"
np.seterr(all='ignore'); t0=time.time()
M=np.load(SP+'/emb.npy'); eids=open(SP+'/ids.txt',encoding='utf-8',errors='surrogateescape').read().split('\n')
eidx={k:i for i,k in enumerate(eids)}
d=json.load(open('/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json',encoding='utf-8',errors='surrogateescape'))
STRUCT={"identity","contains","superseded_by","references","embodies","demonstrated_by","canonical-self","depends_on","currently_holds","activates"}
adj=collections.defaultdict(list)
for e in d['edges']:
if e.get('relation') not in STRUCT: continue
w=float(e.get('weight') or 0.0)
adj[e['from_id']].append((e['to_id'],w)); adj[e['to_id']].append((e['from_id'],w))
nodes=d['nodes']; N={n['id']:n for n in nodes}
PRINT=re.compile(r'^[\x20-\x7e]+$')
ids=[];hay=[];dl=[];sal=[];addr=[]
for n in nodes:
i=n.get('id') or ''
h=((n.get('content') or '')+'\x00'+(n.get('label') or '')+'\x00'+(n.get('tags') or '')).lower()
ids.append(i);hay.append(h);dl.append(len(h));sal.append(float(n.get('salience') or 0.0));addr.append(bool(PRINT.match(i)))
del d
NN=len(ids); avgdl=sum(dl)/NN
gold={q['id']:q for q in json.load(open(EV+"gold_set.json"))['queries']}
CACHE={}
def emb(t):
if t in CACHE: return CACHE[t]
b=json.dumps({"model":"nomic-embed-text","prompt":t}).encode()
r=urllib.request.Request("http://127.0.0.1:11434/api/embeddings",data=b,headers={"Content-Type":"application/json"})
v=np.array(json.load(urllib.request.urlopen(r,timeout=60))["embedding"],dtype=np.float32)
v=v/(np.linalg.norm(v)+1e-9); CACHE[t]=v; return v
K1,B=1.2,0.75
LEXC={}
def lexleg(qid,query,lim=10):
if qid in LEXC: return LEXC[qid]
toks=[]
for w in query.split():
wl=w.lower()
if wl not in toks: toks.append(wl)
nt=len(toks); masks=[]; df=[0]*nt
for i in range(NN):
if not addr[i]: continue
h=hay[i]; m=0; sc=0
for t in range(nt):
if toks[t] in h: m|=(1<<t); sc+=1; df[t]+=1
if sc: masks.append((i,m))
idf=[math.log(1.0+(NN-df[t]+0.5)/(df[t]+0.5)) for t in range(nt)]
scored=[]
for i,m in masks:
norm=1.0-B+B*dl[i]/avgdl; s=0.0
for t in range(nt):
if m&(1<<t): s+=idf[t]*(K1+1.0)/(1.0+K1*norm)
scored.append((s,i))
scored.sort(key=lambda x:(-x[0],-sal[x[1]]))
LEXC[qid]=([ids[i] for s,i in scored[:lim]], len(masks))
return LEXC[qid]
FIRE=0.02; DECAY=0.7; DEPTH=2; SEED_MIN=0.60; ASSOC_MAX=64
def assoc(seeds, s, use_cos, order):
act={x:1.0 for x in seeds}; seen={x:2 for x in seeds}
Q=[(x,0) for x in seeds]; h=0
while h<len(Q):
cur,hop=Q[h]; h+=1
if hop>=DEPTH: continue
p=act[cur]
for oid,w in adj.get(cur,()):
n=N.get(oid)
if not n or n.get('node_type') in ('Tag','InternalStateEvent'): continue
c=1.0
if use_cos:
j=eidx.get(oid)
c=max(0.0,float(s[j])) if j is not None else 0.0
na=p*w*DECAY*float(n.get('salience') or 0.0)*c
if na<FIRE: continue
if oid in seen and na<=act.get(oid,0): continue
act[oid]=na
if oid not in seen: seen[oid]=1
Q.append((oid,hop+1))
out=[]
for k,v in seen.items():
if v!=1 or k not in eidx: continue
c=float(s[eidx[k]])
if c<=0: continue
out.append((act[k] if order=='act' else c,k))
out.sort(reverse=True)
return [k for c,k in out[:ASSOC_MAX] if PRINT.match(k or '')]
def inter(legs,lim=10):
out=[];idx=[0]*len(legs)
while len(out)<lim and any(idx[i]<len(legs[i]) for i in range(len(legs))):
for i in range(len(legs)):
if idx[i]<len(legs[i]):
if legs[i][idx[i]] not in out: out.append(legs[i][idx[i]])
idx[i]+=1
if len(out)>=lim: break
return out
def run(floor, vocabgate, use_cos, order):
res={}; legs={}
for qid,q in gold.items():
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L,nmatch=lexleg(qid,q['query'])
if vocabgate and nmatch==0:
res[qid]=[]; legs[qid]=([],[],[]); continue
ordr=np.argsort(-s)
S=[eids[j] for j in ordr[:10] if PRINT.match(eids[j] or '') and (not floor or s[j]>SEED_MIN)]
seeds=[x for x in L[:3] if x in N]
seeds=seeds+[eids[j] for j in ordr[:8] if eids[j] in N and eids[j] not in seeds and PRINT.match(eids[j] or '')]
A=assoc(seeds,s,use_cos,order) if seeds else []
res[qid]=inter([L,S,A]); legs[qid]=(L,S,A)
return res,legs
def score(res,label,base=None):
det={}
for qid,q in gold.items():
out=res[qid][:5]
if q['category']=='nonsense': ok=(len(res[qid])==0)
elif q['category']=='superseded':
must=q.get('must_outrank') or {}; ok=False
for good,bad in (must.items() if isinstance(must,dict) else []):
ok = good in res[qid] and (bad not in res[qid] or res[qid].index(good)<res[qid].index(bad))
if not must: ok=any(r in out for r in q['relevant'])
else: ok=any(r in out for r in q['relevant'])
det[qid]=ok
line="%-34s true=%d/38"%(label,sum(det.values()))
if base is not None:
dd=[q for q in sorted(gold) if det[q]!=base[q]]
line+=" moved=%d gains=%s losses=%s"%(len(dd),[q for q in dd if det[q]],[q for q in dd if not det[q]])
print(line, flush=True)
return det
if __name__=="__main__":
b,_=run(True,False,False,'cos'); base=score(b,'BASE bm25lex replica')
for lab,args in [
("A floor-off+vocabgate", (False,True,False,'cos')),
("B A+cos-in-traversal", (False,True,True ,'cos')),
("C A+cos-trav+act-order", (False,True,True ,'act')),
("D floor-off NO gate", (False,False,False,'cos')),
]:
r,_=run(*args); score(r,lab,base)
print("elapsed %.1fs"%(time.time()-t0),file=sys.stderr)
+32
View File
@@ -0,0 +1,32 @@
exec(open('sim6.py').read().split('if __name__')[0])
HASSTRUCT=set(adj.keys())
print("nodes with >=1 structural edge:",len(HASSTRUCT),file=sys.stderr)
def run2(sfilter, seedout, lim=10):
res={}
for qid,q in gold.items():
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L,nmatch=lexleg(qid,q['query'])
if nmatch==0: res[qid]=[]; continue
ordr=np.argsort(-s)
cand=[eids[j] for j in ordr[:200] if PRINT.match(eids[j] or '')]
S=[x for x in cand if (not sfilter or x in HASSTRUCT)][:10]
seeds=[x for x in L[:3] if x in N]
semseeds=[eids[j] for j in ordr[:8] if eids[j] in N and eids[j] not in seeds and PRINT.match(eids[j] or '')]
seeds=seeds+semseeds
A=assoc(seeds,s,False,'cos') if seeds else []
if seedout:
extra=[(float(s[eidx[x]]),x) for x in semseeds if x in HASSTRUCT and x in eidx]
merged=[(float(s[eidx[x]]),x) for x in A if x in eidx]+extra
merged.sort(reverse=True)
seen=set(); A=[]
for c,x in merged:
if x in seen: continue
seen.add(x); A.append(x)
A=A[:ASSOC_MAX]
res[qid]=inter([L,S,A])
return res
b,_=run(True,False,False,'cos'); base=score(b,'BASE bm25lex replica')
a,_=run(False,True,False,'cos'); score(a,'A floor-off+vocabgate',base)
score(run2(False,True),'E A+struct-seeds-in-graphleg',base)
score(run2(True,False),'F A+S-restricted-to-graph',base)
score(run2(True,True),'G E+F',base)
+30
View File
@@ -0,0 +1,30 @@
exec(open('sim6.py').read().split('if __name__')[0])
HASSTRUCT=set(adj.keys())
import json as _j
d2=_j.load(open('/Users/timlingo/neuron-memory-backups/snapshot-pre-repair-20260806.json',encoding='utf-8',errors='surrogateescape'))
ANYEDGE=set()
for e in d2['edges']: ANYEDGE.add(e['from_id']); ANYEDGE.add(e['to_id'])
del d2
print("struct=%d anyedge=%d"%(len(HASSTRUCT),len(ANYEDGE)),file=sys.stderr)
def run4(pool, nlegs, lim=10):
P = HASSTRUCT if pool=='struct' else ANYEDGE
res={}
for qid,q in gold.items():
v=emb(q['query']); s=M@v; s[~np.isfinite(s)]=-1
L,nmatch=lexleg(qid,q['query'])
if nmatch==0: res[qid]=[]; continue
ordr=np.argsort(-s)
cand=[eids[j] for j in ordr[:3000] if PRINT.match(eids[j] or '')]
S=cand[:10]
G=[x for x in cand if x in P][:10]
seeds=[x for x in L[:3] if x in N]
seeds=seeds+[eids[j] for j in ordr[:8] if eids[j] in N and eids[j] not in seeds and PRINT.match(eids[j] or '')]
A=assoc(seeds,s,False,'cos') if seeds else []
legs=[L,S,G,A] if nlegs==4 else [L,G,A]
res[qid]=inter(legs)
return res
b,_=run(True,False,False,'cos'); base=score(b,'BASE bm25lex replica')
score(run4('struct',4),'I 4leg L,S,G(struct),A',base)
score(run4('any',4), 'J 4leg L,S,G(anyedge),A',base)
score(run4('struct',3),'K 3leg L,G(struct),A',base)
score(run4('any',3), 'L 3leg L,G(anyedge),A',base)
-92
View File
@@ -1,92 +0,0 @@
import json,sys,pickle,numpy as np
sys.path.insert(0,'.')
from legs import *
GP='/Users/timlingo/Development/neuron-technologies/_wt-bm25lex/tools/retrieval-eval/'
G=json.load(open(GP+'gold_set.json'))
def wstart(s,tok):
i=s.find(tok)
while i!=-1:
if i==0 or not s[i-1].isalnum(): return True
i=s.find(tok,i+1)
return False
def legs4(query, wordstart=False, unfloor=False):
toks=tokenize(query); lt=[t.lower() for t in toks]
hit_idx=[];hit_mask=[];df=[0]*len(toks)
for i in range(N):
if not OK[i]: continue
s=LOW[i];m=0
for t,tok in enumerate(lt):
if tok in s and (not wordstart or wstart(s,tok)): m|=(1<<t)
if m:
hit_idx.append(i);hit_mask.append(m)
for t in range(len(toks)):
if m>>t&1: df[t]+=1
dl_n=int(OK.sum());avgdl=float(DL[OK].sum()/max(dl_n,1))
idf=[math.log(1.0+((dl_n-d+0.5)/(d+0.5))) for d in df]
L=[]
for j,i in enumerate(hit_idx):
norm=1.0-B+B*(DL[i]/avgdl);w=0.0
for t in range(len(toks)):
if hit_mask[j]>>t&1: w+=idf[t]*(K1+1.0)/(1.0+K1*norm)
L.append((i,w,SAL[i]))
L.sort(key=lambda x:(-x[1],-x[2]))
if not L: return [],[],[]
qv=qemb(query);cos=En@qv;cos=np.where(HAVE&OK,cos,-2.0)
order=np.argsort(-cos)[:600]
Sl=[int(i) for i in order if cos[i]>(0.0 if unfloor else SEED_MIN)]
semseed=[int(i) for i in order[:SEED_K] if cos[i]>0.0]
act={};seen={};qq=[]
for i,_,_ in L[:ASSOC_SEEDS]:
act[i]=1.0;seen[i]=2;qq.append((i,0))
for i in semseed:
if i in seen: continue
act[i]=1.0;seen[i]=2;qq.append((i,0))
qh=0
while qh<len(qq):
cur,h=qq[qh];qh+=1
if h>=DEPTH: continue
parent=act[cur]
for e,oi in ADJ_F[cur]+ADJ_T[cur]:
if e['rel'] not in STRUCT or EXCL[oi]: continue
na=parent*e['w']*DECAY*SAL[oi]
if na<FIRE: continue
if seen.get(oi) and na<=act.get(oi,0): continue
act[oi]=na
if not seen.get(oi): seen[oi]=1
if len(qq)<AMAX*4: qq.append((oi,h+1))
A=sorted([(i,float(cos[i])) for i,st in seen.items() if st==1 and OK[i] and HAVE[i] and cos[i]>0.0],key=lambda x:-x[1])[:AMAX]
return [i for i,_,_ in L],Sl,[i for i,_ in A]
def merge(L,S,A,lim=10):
out=[];li=si=ai=0
while len(out)<lim and (li<len(L) or si<len(S) or ai<len(A)):
if li<len(L):
if L[li] not in out: out.append(L[li])
li+=1
if len(out)>=lim: break
if si<len(S):
if S[si] not in out: out.append(S[si])
si+=1
if len(out)>=lim: break
if ai<len(A):
if A[ai] not in out: out.append(A[ai])
ai+=1
return out
def outcome(q,ids):
c=q['category']
if c=='nonsense': return len(ids)==0
if c=='superseded':
a,b=q['must_outrank']
if a not in ids: return False
if b not in ids: return True
return ids.index(a)<ids.index(b)
return any(x in ids[:5] for x in q['relevant'])
def run(**kw):
return {q['id']:outcome(q,[NODES[i]['id'] for i in merge(*legs4(q['query'],**kw),10)]) for q in G['queries']}
base=run()
print("baseline",sum(base.values()),"misses",[k for k,v in base.items() if not v])
for name,kw in [('wordstart',dict(wordstart=True)),
('unfloor',dict(unfloor=True)),
('wordstart+unfloor',dict(wordstart=True,unfloor=True))]:
r=run(**kw)
g=sorted(k for k in base if r[k] and not base[k]);l=sorted(k for k in base if base[k] and not r[k])
print("%-20s net=%+d gains=%s losses=%s"%(name,len(g)-len(l),g,l))
-93
View File
@@ -1,93 +0,0 @@
#!/usr/bin/env bash
# soulc-stamp.sh — make it impossible for dist/soul.c to drift from the sources
# in silence.
#
# THE PROBLEM (neuron#133, and its own words): "Nothing in the tree regenerates
# this file. Only a human running the recipe. It lags in batches, never
# per-change, and it will drift again."
#
# It drifted. On 2026-08-07 a CI or GKE build off main would have shipped an
# engine with NONE of five merged fixes — including a P0 safety fix — while
# main's source read as correct. CI compiles dist/soul.c, not the .el files, so
# the source being right is not the same as the build being right.
#
# WHY A STAMP AND NOT AUTO-REGENERATION: the CI workflow says elc cannot run on
# the runner ("elb on Linux would OOM the runner (elc uses 24GB+ virtual memory
# on a 16GB host)"). So the build cannot regenerate the file itself. What it CAN
# do, for free and with no compiler, is refuse to compile a stale one.
#
# The stamp records a fingerprint of every .el source that feeds the amalgam at
# the moment it was generated. --check recomputes and compares. Divergence is a
# build failure with the recipe in the message, not a silent ship.
#
# soulc-stamp.sh --write after regenerating dist/soul.c (records the fingerprint)
# soulc-stamp.sh --check in CI, before the compile (fails on drift)
set -u
MODE="${1:---check}"
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
STAMP="$ROOT/dist/soul.c.stamp"
AMALGAM="$ROOT/dist/soul.c"
# Every .el at the repo root is an input to the amalgam. Sorted so the hash is
# order-independent; content-only so timestamps and checkouts do not perturb it.
# The COMPILER is an input too. Learned 2026-08-09 by installing a fixed elc and
# watching this gate report OK while the committed amalgam had gone stale by a line:
# the sources had not changed, so a source-only fingerprint could not see it. That is
# precisely the blind spot this gate exists to close, and it had it.
fingerprint() {
(
cd "$ROOT" || exit 1
ELC_BIN="${ELC:-$HOME/neuron-dev-stack/src/el/lang/dist/platform/elc}"
if [ -f "$ELC_BIN" ]; then
printf '%s %s\n' "$(shasum -a 256 "$ELC_BIN" | awk '{print $1}')" "__compiler__"
else
printf '%s %s\n' "MISSING" "__compiler__"
fi
for f in $(ls -1 *.el 2>/dev/null | sort); do
printf '%s %s\n' "$(shasum -a 256 "$f" | awk '{print $1}')" "$f"
done
)
}
case "$MODE" in
--write)
[ -f "$AMALGAM" ] || { echo "no dist/soul.c to stamp — regenerate it first" >&2; exit 2; }
{
echo "# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from."
echo "# Written by tools/soulc-stamp.sh --write. Do not hand-edit."
echo "# generated_amalgam_sha256 $(shasum -a 256 "$AMALGAM" | awk '{print $1}')"
echo "# generated_amalgam_bytes $(wc -c < "$AMALGAM" | tr -d ' ')"
fingerprint
} > "$STAMP"
echo "stamped $(fingerprint | wc -l | tr -d ' ') sources -> dist/soul.c.stamp"
;;
--check)
if [ ! -f "$STAMP" ]; then
echo "FAIL: dist/soul.c.stamp is missing — the build input is unverifiable." >&2
echo " Regenerate the amalgam, then: tools/soulc-stamp.sh --write" >&2
exit 1
fi
RECORDED="$(grep -v '^#' "$STAMP")"
CURRENT="$(fingerprint)"
if [ "$RECORDED" = "$CURRENT" ]; then
echo "soulc-stamp: OK — dist/soul.c matches the .el sources"
exit 0
fi
echo "FAIL: dist/soul.c is STALE. It does not match the current .el sources." >&2
echo "" >&2
echo "CI compiles dist/soul.c, not the .el files. Shipping this means shipping" >&2
echo "an engine that does not contain the merged source. That is neuron#133," >&2
echo "which once hid five merged fixes including a P0 safety fix." >&2
echo "" >&2
echo "Sources that changed since the amalgam was generated:" >&2
diff <(printf '%s\n' "$RECORDED") <(printf '%s\n' "$CURRENT") \
| grep -E '^[<>]' | awk '{print " " $1 " " $3}' | sort -u >&2
echo "" >&2
echo "Fix: regenerate the amalgam, then tools/soulc-stamp.sh --write" >&2
exit 1
;;
*)
echo "usage: soulc-stamp.sh [--check|--write]" >&2; exit 2 ;;
esac
+43 -187
View File
@@ -5634,25 +5634,6 @@ void el_cgi_init(el_val_t name, el_val_t dharma_id, el_val_t principal,
}
/* ── Compiled-identity accessors (2026-08-09) ─────────────────────────────────
* el_cgi_init loads the declaration into these globals at startup and printed
* them, and NOTHING read them back out no accessor existed, and el_cgi_init
* writes no state. So a binary carried its declared identity and every consumer
* still read it from the mutable state store, which is exactly what IDPROTO
* claims 1-2 forbid ("not modifiable by any runtime mechanism including
* environment variables, configuration files, or API calls").
*
* These are READ-ONLY on purpose. There is deliberately no setter: publishing
* the values into the state store would have been one line and would have
* recreated the mutable copy the design prohibits. A caller can read the
* compiled identity; nothing can change it after el_cgi_init.
*/
el_val_t cgi_name(void) { return EL_STR(_el_cgi_name ? _el_cgi_name : ""); }
el_val_t cgi_dharma_id(void) { return EL_STR(_el_cgi_dharma_id ? _el_cgi_dharma_id : ""); }
el_val_t cgi_principal(void) { return EL_STR(_el_cgi_principal ? _el_cgi_principal : ""); }
el_val_t cgi_network(void) { return EL_STR(_el_cgi_network ? _el_cgi_network : ""); }
el_val_t cgi_engram(void) { return EL_STR(_el_cgi_engram ? _el_cgi_engram : ""); }
/* ── Batch 3: Engram in-process graph store ──────────────────────────────── */
/*
* Single global EngramStore allocated lazily on first call. All node and
@@ -7194,12 +7175,6 @@ void engram_forget(el_val_t node_id) {
if (idx < 0) return;
/* Free node strings */
EngramNode* n = &g->nodes[idx];
if (getenv("EG_DIAG")) {
fprintf(stderr, "[EG_DIAG] FORGET id=%s type=%s layer=%u label=%s\n",
sid, n->node_type ? n->node_type : "?", n->layer_id,
n->label ? n->label : "?");
fflush(stderr);
}
free(n->id); free(n->content); free(n->node_type); free(n->label);
free(n->tier); free(n->tags); free(n->metadata);
free(n->emb);
@@ -7295,8 +7270,6 @@ el_val_t engram_prune_telemetry(el_val_t older_than_ms) {
}
}
g->node_count = w;
if (getenv("EG_DIAG"))
fprintf(stderr, "[EG_DIAG] PRUNE_TELEMETRY removed=%lld\n", (long long)removed);
if (removed == 0) { free(removed_ids); return 0; }
/* Removed-id hash set (open addressing, power-of-two >= 2*removed). */
@@ -7354,39 +7327,6 @@ static int istr_contains(const char* hay, const char* needle) {
return 0;
}
/* Word-START-anchored variant of istr_contains.
*
* WHY. The retrieval match primitive is a raw substring test, so a query token
* matches ANYWHERE inside a corpus word: "throom" matches "bathroom", "cat"
* matches "concatenate". Measured on this corpus over the 38-query gold set,
* that is not a rare accident it is the bulk of some queries' candidate
* sets. q28's lexical leg is 36,954 records of which only 13 contain a query
* token at a word start (99.96% mid-word noise); six other queries carry
* ~20,500 mid-word-only records each; and the nonsense control q35
* ("xxqzzt vurblenacht throom") returns 7 records ALL of which match only
* mid-word, which is the entire reason that control has been dirty since main.
*
* WHAT CHANGES. A token must begin at a word boundary the preceding
* character is not alphanumeric. Suffixes are still matched ("value" still
* hits "values", "unjailbreakable" still hits "unjailbreakables"), so this is
* strictly a prefix anchor, not whole-word equality; whole-word equality would
* break the morphological matching the phrase category depends on.
*
* PROVENANCE, stated honestly: this restores no engram claim. Will's design
* has no lexical leg at all (05-detailed-description l.64 takes "one or more
* seed node UUIDs representing the current active context" as its input), so
* the lexical leg is the seed-finding step that feeds the designed mechanism.
* Cleaner seeds serve that mechanism; they do not replace it. */
static int istr_contains_wordstart(const char* hay, const char* needle) {
if (!hay || !needle || !*needle) return 0;
size_t nl = strlen(needle);
for (const char* p = hay; *p; p++) {
if (p != hay && isalnum((unsigned char)p[-1])) continue;
if (strncasecmp(p, needle, nl) == 0) return 1;
}
return 0;
}
/* ── Tokenized query matching ───────────────────────────────────────────
* The engram query surface (search / activate / goal-bias) historically
* matched the ENTIRE raw query string as a single case-insensitive
@@ -7439,9 +7379,9 @@ static int engram_node_match_score(const EngramNode* n,
char toks[][ENGRAM_QTOK_LEN], int ntok) {
int score = 0;
for (int t = 0; t < ntok; t++) {
if (istr_contains_wordstart(n->content, toks[t]) ||
istr_contains_wordstart(n->label, toks[t]) ||
istr_contains_wordstart(n->tags, toks[t]))
if (istr_contains(n->content, toks[t]) ||
istr_contains(n->label, toks[t]) ||
istr_contains(n->tags, toks[t]))
score++;
}
return score;
@@ -7456,9 +7396,9 @@ static uint32_t engram_node_match_mask(const EngramNode* n,
char toks[][ENGRAM_QTOK_LEN], int ntok) {
uint32_t m = 0;
for (int t = 0; t < ntok && t < 32; t++) {
if (istr_contains_wordstart(n->content, toks[t]) ||
istr_contains_wordstart(n->label, toks[t]) ||
istr_contains_wordstart(n->tags, toks[t]))
if (istr_contains(n->content, toks[t]) ||
istr_contains(n->label, toks[t]) ||
istr_contains(n->tags, toks[t]))
m |= (uint32_t)1u << t;
}
return m;
@@ -9331,26 +9271,6 @@ el_val_t engram_load(el_val_t path) {
}
}
g->adj_dirty = 1;
if (getenv("EG_DIAG")) {
int64_t we = 0, wrongdim = 0;
for (int64_t i = 0; i < g->node_count; i++) {
if (g->nodes[i].emb) { we++; if (g->nodes[i].emb_dim != 768) wrongdim++; }
}
fprintf(stderr, "[EG_DIAG] loaded nodes=%lld with_emb=%lld wrongdim=%lld\n",
(long long)g->node_count, (long long)we, (long long)wrongdim);
const char* probe = getenv("EG_DIAG_ID");
if (probe) {
for (int64_t i = 0; i < g->node_count; i++) {
if (g->nodes[i].id && strcmp(g->nodes[i].id, probe) == 0) {
fprintf(stderr, "[EG_DIAG] probe id=%s idx=%lld emb=%p dim=%d layer=%u addr=%d\n",
probe, (long long)i, (void*)g->nodes[i].emb,
(int)g->nodes[i].emb_dim, g->nodes[i].layer_id,
eg_node_addressable(&g->nodes[i]));
}
}
}
fflush(stderr);
}
/* Walk edges array */
const char* edges_p = json_find_key(data, "edges");
if (edges_p) {
@@ -9671,33 +9591,7 @@ el_val_t engram_get_node_by_label(el_val_t label) {
return el_wrap_str(el_strdup("{}"));
}
/* ── THE SEARCH / RECALL BOUNDARY (2026-08-07) ───────────────────────────────
* engram_search_json is the LEXICAL function ~40 .el call sites already
* depend on: they pass a key-shaped string ("soul:boot_count",
* "soul-inbox-pending", a session label) and treat every returned record as
* a record that CONTAINS that key. Seven of those sites then delete what
* comes back (memory.el:176, sessions.el:250/268/444/523, soul.el:359
* "prune all existing X nodes, keep exactly one").
*
* The semantic and associative legs must therefore NOT live on this
* function. Claim 24 authorises the vector index "to respond to EMBEDDING
* SEARCH QUERIES by returning the node records whose embedding vectors have
* the highest cosine similarity to a query vector"; a keyed state read is
* not an embedding search query, it is the identifier-keyed retrieval of
* claim 23 ("node records are stored under a key encoding the node
* identifier"). Putting both behind one function erased that boundary, and
* a nearest neighbour of the string "soul:boot_count" is not a boot counter.
*
* MEASURED, on the harness corpus, isolated, read-only, no writes from any
* caller: 240 node records destroyed per boot, including 6 Knowledge nodes,
* a layer-1 "CORE IDENTITY — GENESIS, LINEAGE" Memory, and the value node
* `kn-58874a74` (gold answer for gold-set q15). The deletion list is the
* result list of the soul's own mem_boot_count_inc() lookup, in order.
*
* So: legs OFF here, legs ON in engram_recall_json below, which is what
* /api/neuron/recall reaches. Retrieval quality on the recall route is
* unchanged; the internal keyed reads get their contract back. */
static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_legs) {
el_val_t engram_search_json(el_val_t query, el_val_t limit) {
EngramStore* g = engram_get();
const char* q = EL_CSTR(query);
int64_t lim = (int64_t)limit;
@@ -9719,7 +9613,7 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
* so the semantic half of the retrieval surface has to land HERE
* to be observable to the MCP wrapper and the app. */
int32_t qdim = 0;
float* qv = with_legs ? eg_embed_fetch(q, &qdim) : NULL;
float* qv = eg_embed_fetch(q, &qdim);
EngramSemEntry* sem = qv ? malloc((size_t)g->node_count * sizeof(EngramSemEntry)) : NULL;
int64_t nsem = 0;
int64_t nhits = 0;
@@ -9760,28 +9654,25 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
}
if (sem && n->emb && n->emb_dim == qdim) {
double c = eg_cosine(n->emb, qv, qdim);
/* Claim-24 semantic leg, restored verbatim: "returning
* the node records whose embedding vectors have the
* HIGHEST COSINE SIMILARITY to a query vector" — a
* ranking, with no threshold anywhere in the claim.
* ENGRAM_EMBED_SEED_MIN is defined at l.6094 as the
* HippoRAG SEED-JOIN threshold; using it as a RESULT
* filter here was never authorised, and it is a
* per-query lottery rather than a quality gate: the
* query's own top-1 cosine ranges 0.56-0.68 across the
* held-out gold set, so 0.60 keeps a rank-1 answer for
* one query and discards a rank-1 answer for the next.
* Measured on the 30 held-out paraphrases: six golds
* sit at global cosine rank 1-2 and score 0.564-0.589,
* discarded by nothing but this constant.
* What holds the nonsense controls is NOT this floor
* but the corpus-vocabulary gate below (nhits == 0):
* gibberish has no lexical seeds, so no leg reports.
* Cosine is clamped to [0,1] per 05-detailed-description
* l.69 ("clamped to [0,1] to prevent anti-correlated
* embeddings from producing negative activation"). */
double sv = c < 0.0 ? 0.0 : (c > 1.0 ? 1.0 : c);
if (sv > 0.0) {
/* Semantic leg, claim 24 verbatim: "returning the node
* records whose embedding vectors have the HIGHEST
* COSINE SIMILARITY to a query vector" — a ranking, with
* no threshold anywhere in the claim. The leg used to be
* gated at ENGRAM_EMBED_SEED_MIN and rescaled onto
* [SEED_MIN,1]; that constant is defined (l.6083) as the
* SEED-JOIN threshold for the HippoRAG pass, and reusing
* it as a result filter is not authorised by claim 24.
* Measured on this corpus, it is also not a quality
* gate: true paraphrase targets score 0.459-0.657 while
* the nonsense controls' own nearest neighbours score
* 0.553-0.622 the distributions overlap, so no value
* of the constant separates them. What actually holds
* the nonsense control is corpus vocabulary (see the
* nhits==0 gate below), not cosine magnitude.
* Claim 32: clamp the cosine to [0,1] rather than let a
* negative value invert the signal. */
if (c > 0.0) {
double sv = c > 1.0 ? 1.0 : c;
sem[nsem].idx = i; sem[nsem].sem = sv; nsem++;
}
/* Graph seeds: top-K by RAW cosine, insertion-ordered. */
@@ -9799,6 +9690,21 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
}
}
}
/* CORPUS-VOCABULARY GATE — the thing that actually keeps an
* unfloored semantic leg from answering gibberish.
* nhits == 0 means NO stored record contains ANY query token
* anywhere in its content, label or tags: the query is outside
* the graph's vocabulary entirely. A vector index always has a
* nearest neighbour, so without this gate the semantic leg
* answers "zqxjvw plimforth grebulon" with its 0.55-cosine
* garbage. It is also the honest reading of Will's retrieval
* contract: 05-detailed-description l.64 has the caller supply
* "one or more seed node UUIDs representing the current active
* context", and every leg here is downstream of finding those
* seeds. No seeds, no retrieval the graph declines rather
* than confabulates. Suppressing the graph seeds too keeps the
* associative leg from running off the semantic top-K alone. */
if (nhits == 0) { nsem = 0; nsemseed = 0; }
/* BM25-shaped lexical score. Binary term frequency (the match
* primitive is a substring test, not a count), Lucene-form IDF,
* and length normalisation over the corpus mean. A token that
@@ -9824,31 +9730,6 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
}
qsort(hits, (size_t)nhits, sizeof(EngramRankEntry), engram_rank_w_cmp);
if (sem) qsort(sem, (size_t)nsem, sizeof(EngramSemEntry), engram_sem_cmp);
if (getenv("EG_DIAG")) {
int64_t we = 0, unaddr = 0, found = 0;
const char* pid = getenv("EG_DIAG_ID");
for (int64_t i = 0; i < g->node_count; i++) {
if (g->nodes[i].emb) we++;
if (!eg_node_addressable(&g->nodes[i])) unaddr++;
if (pid && g->nodes[i].id && strcmp(g->nodes[i].id, pid) == 0) found++;
}
fprintf(stderr, "[EG_DIAG] STORE node_count=%lld with_emb=%lld unaddressable=%lld probe_found=%lld\n",
(long long)g->node_count, (long long)we, (long long)unaddr, (long long)found);
fprintf(stderr, "[EG_DIAG] q=\"%s\" qdim=%d nhits=%lld nsem=%lld\n",
q, (int)qdim, (long long)nhits, (long long)nsem);
for (int64_t k = 0; k < 5 && k < nsem; k++)
fprintf(stderr, "[EG_DIAG] sem[%lld] cos=%.4f id=%s\n",
(long long)k, sem[k].sem, g->nodes[sem[k].idx].id);
const char* probe = getenv("EG_DIAG_ID");
if (probe) for (int64_t k = 0; k < nsem; k++)
if (g->nodes[sem[k].idx].id
&& strcmp(g->nodes[sem[k].idx].id, probe) == 0) {
fprintf(stderr, "[EG_DIAG] probe at sem rank %lld cos=%.4f\n",
(long long)k, sem[k].sem);
break;
}
fflush(stderr);
}
/* Claim-10 associative leg: expand the top lexical hits along
* structural relations only, order the reached set by query
* similarity. Empty whenever the seeds have no structural
@@ -9859,19 +9740,7 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
? engram_assoc_leg(g, hits, nhits, semseed, nsemseed,
qv, qdim, assoc, ENGRAM_ASSOC_MAX)
: 0;
/* Corpus-vocabulary gate. If no stored record contains ANY
* query token in its content, label or tags, the query is
* outside this graph's vocabulary: there are no seeds, and
* 05-detailed-description l.64 makes retrieval downstream of
* seeds ("the caller provides one or more seed node UUIDs
* representing the current active context"). No seeds, no
* retrieval the graph declines rather than confabulating a
* nearest neighbour for gibberish. The mechanism is iteration
* 6's (feat/claim24-unfloored-semantic); it is required here
* because word-start matching empties the lexical leg for
* q35-style queries whose only "hits" were mid-word, and the
* semantic leg would otherwise answer them anyway. */
int64_t* order = (nhits > 0) ? malloc((size_t)lim * sizeof(int64_t)) : NULL;
int64_t* order = malloc((size_t)lim * sizeof(int64_t));
if (order) {
int64_t no = engram_interleave3(hits, nhits, sem, nsem,
assoc, nassoc, lim, order);
@@ -9893,19 +9762,6 @@ static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_leg
return el_wrap_str(b.buf);
}
/* Lexical keyed read — the historical contract every internal caller relies
* on. Every returned record CONTAINS a query token. */
el_val_t engram_search_json(el_val_t query, el_val_t limit) {
return eg_search_json_impl(query, limit, 0);
}
/* The retrieval surface: lexical + claim-24 semantic + claim-10 associative,
* rank-fused. Reached from handle_api_recall (/api/neuron/recall) the route
* the MCP wrapper and the app call, and the one the eval harness measures. */
el_val_t engram_recall_json(el_val_t query, el_val_t limit) {
return eg_search_json_impl(query, limit, 1);
}
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset) {
EngramStore* g = engram_get();
int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100;
-9
View File
@@ -612,7 +612,6 @@ el_val_t engram_load(el_val_t path);
el_val_t engram_get_node_json(el_val_t id);
el_val_t engram_get_node_by_label(el_val_t label);
el_val_t engram_search_json(el_val_t query, el_val_t limit);
el_val_t engram_recall_json(el_val_t query, el_val_t limit);
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
@@ -782,14 +781,6 @@ el_val_t trace_span_start(el_val_t name);
el_val_t trace_span_end(el_val_t span_handle);
el_val_t emit_event(el_val_t name, el_val_t duration_ms);
/* Compiled-identity accessors — read-only by design (2026-08-09). */
el_val_t cgi_name(void);
el_val_t cgi_dharma_id(void);
el_val_t cgi_principal(void);
el_val_t cgi_network(void);
el_val_t cgi_engram(void);
#ifdef __cplusplus
}
#endif