Compare commits
13 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d9653b1a22 | |||
| f1f52bcb2f | |||
| aad988ecbf | |||
| 7b86e6f72c | |||
| bf521974af | |||
| 5743568bf1 | |||
| 9fd8c11670 | |||
| be0f9d1afe | |||
| 1742d0b575 | |||
| c6772e3d27 | |||
| b9ef66cae9 | |||
| 9501e4ac12 | |||
| dd952c0e46 |
@@ -63,6 +63,16 @@ jobs:
|
||||
cp vendor/el-runtime/v1.0.0-20260501/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
echo "El runtime PINNED to v1.0.0-20260501: $(ls /opt/el/runtime/)"
|
||||
|
||||
# neuron#133: CI compiles dist/soul.c, NOT the .el sources. On 2026-08-07 a
|
||||
# build off main would have shipped an engine with none of five merged fixes,
|
||||
# including a P0 safety fix, while main's source read as correct. The runner
|
||||
# cannot regenerate the amalgam (elc needs 24GB+ virtual memory), but it can
|
||||
# refuse to compile a stale one. Fails loudly with the recipe in the message.
|
||||
- name: Verify dist/soul.c matches the sources
|
||||
run: |
|
||||
chmod +x tools/soulc-stamp.sh
|
||||
./tools/soulc-stamp.sh --check
|
||||
|
||||
- name: Build neuron soul binary
|
||||
run: |
|
||||
RUNTIME=/opt/el/runtime
|
||||
|
||||
@@ -889,10 +889,29 @@ fn awareness_run() -> Void {
|
||||
state_set("soul.last_beat_ts", int_to_str(now_ts))
|
||||
// Persist in-process Engram (sessions, memories, conversation nodes)
|
||||
// to local snapshot so they survive restarts.
|
||||
// FILE MODE ONLY: "soul_snapshot_path" is set exclusively in the
|
||||
// genesis+safe_to_seed branch of soul.el, and safe_to_seed is
|
||||
// unconditionally false when ENGRAM_URL is set. In HTTP mode the
|
||||
// owner persists; the soul must not (soul.el:571-573).
|
||||
let snap_path: String = state_get("soul_snapshot_path")
|
||||
if !str_eq(snap_path, "") {
|
||||
mem_save(snap_path)
|
||||
}
|
||||
// WRITE-THROUGH RETRY (neuron#117). The HTTP-mode counterpart of the
|
||||
// save above: hand anything still spooled to the persistence owner.
|
||||
//
|
||||
// This is the retry arm of the whole design. Deltas that could not be
|
||||
// pushed — owner down, owner restarting, transient refusal — stay on
|
||||
// disk and are re-offered here every heartbeat until they land. It is
|
||||
// also the catch-all for writes made by the awareness loop itself,
|
||||
// which never passes through the HTTP handler's flush point.
|
||||
//
|
||||
// No-op with no HTTP call when the spool is empty or ENGRAM_URL is
|
||||
// unset, so an idle soul in file mode pays nothing for this.
|
||||
let wt_pushed: Int = wt_drain()
|
||||
if wt_pushed < 0 {
|
||||
ise_post("{\"event\":\"write_through_backlog\",\"ts\":" + int_to_str(now_ts) + "}")
|
||||
}
|
||||
}
|
||||
|
||||
// Curiosity scan: idle-gated AND wall-clock based. Only fires when the
|
||||
|
||||
@@ -1162,7 +1162,7 @@ fn hist_trim_with_bell_guard(hist: String) -> String {
|
||||
+ " | evicted_at:" + ts_str
|
||||
+ " | message:" + safe_content
|
||||
let preserve_tags: String = "[\"bell-history\",\"bell:" + bell_level + "\",\"evicted\",\"affective\",\"BellEvent\"]"
|
||||
let discard: String = engram_node_full(
|
||||
let discard: String = wt_node(
|
||||
preserve_content,
|
||||
"BellEvent",
|
||||
"bell:" + bell_level + ":preserved",
|
||||
@@ -1210,7 +1210,7 @@ fn conv_history_persist(session_id: String, hist: String) -> Void {
|
||||
if !str_contains(hist, "]") { return "" }
|
||||
let tags: String = "[\"conv-history\",\"persistent\"]"
|
||||
// FIX B: one label rule, shared with the agentic path. See conv_hist_label.
|
||||
let node_id: String = engram_node_full(
|
||||
let node_id: String = wt_node(
|
||||
hist, "Conversation", conv_hist_label(session_id),
|
||||
el_from_float(0.7), el_from_float(0.8), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
@@ -3461,7 +3461,7 @@ fn handle_dharma_room_turn(body: String) -> String {
|
||||
// engram_node(content, "episodic", ...) which wrongly put a TIER into the node_type
|
||||
// slot — that's why nodes showed node_type="episodic". Use the full, correct contract.)
|
||||
let utterance_tags: String = "[\"soul-utterance\",\"episodic\"]"
|
||||
let discard_id: String = engram_node_full(
|
||||
let discard_id: String = wt_node(
|
||||
clean_response, "Conversation", "soul:utterance",
|
||||
el_from_float(0.6), el_from_float(0.6), el_from_float(0.8),
|
||||
"Episodic", utterance_tags
|
||||
@@ -3552,7 +3552,7 @@ fn session_summary_write(summary_text: String) -> String {
|
||||
}
|
||||
}
|
||||
let tags: String = "[\"SessionSummary\",\"session-summary\",\"previous-session\",\"consolidate\"]"
|
||||
let node_id: String = engram_node_full(
|
||||
let node_id: String = wt_node(
|
||||
content, "SessionSummary", "session:summary",
|
||||
el_from_float(0.85), el_from_float(0.85), el_from_float(1.0),
|
||||
"Episodic", tags
|
||||
@@ -3578,7 +3578,7 @@ fn session_summary_write_dated(summary_text: String, label: String) -> String {
|
||||
let ts_str: String = int_to_str(ts)
|
||||
let content: String = "[session-summary] " + trimmed + " | ts:" + ts_str
|
||||
let tags: String = "[\"SessionSummary\",\"session-summary\",\"previous-session\",\"consolidate\"]"
|
||||
let node_id: String = engram_node_full(
|
||||
let node_id: String = wt_node(
|
||||
content, "SessionSummary", label,
|
||||
el_from_float(0.9), el_from_float(0.8), el_from_float(1.0),
|
||||
"Episodic", tags
|
||||
@@ -3654,7 +3654,7 @@ fn auto_persist(req: String, resp: String) -> Void {
|
||||
+ ",\"bell\":\"" + bell_level + "\""
|
||||
+ ",\"label\":\"chat:" + ts_str + "\"}"
|
||||
|
||||
let conv_node_id: String = engram_node_full(
|
||||
let conv_node_id: String = wt_node(
|
||||
content,
|
||||
"Conversation",
|
||||
"chat:" + ts_str,
|
||||
@@ -3692,7 +3692,7 @@ fn auto_persist(req: String, resp: String) -> Void {
|
||||
let bell_tags: String = "[\"safety\",\"bell\",\"bell:" + bell_level + "\",\"affective\",\"BellEvent\"]"
|
||||
let bell_ts_str: String = int_to_str(time_now())
|
||||
let bell_label: String = "bell:" + bell_level + ":" + bell_ts_str
|
||||
let bell_node_id: String = engram_node_full(
|
||||
let bell_node_id: String = wt_node(
|
||||
bell_content,
|
||||
"BellEvent",
|
||||
bell_label,
|
||||
@@ -3751,7 +3751,7 @@ fn auto_persist(req: String, resp: String) -> Void {
|
||||
let pos_tags: String = "[\"joy\",\"positive\",\"joy:" + positive_level + "\",\"affective\",\"PositiveEvent\"]"
|
||||
let pos_ts_label: String = int_to_str(time_now())
|
||||
let pos_label: String = "joy:" + positive_level + ":" + pos_ts_label
|
||||
let pos_node_id: String = engram_node_full(
|
||||
let pos_node_id: String = wt_node(
|
||||
pos_content, "PositiveEvent", pos_label,
|
||||
pos_sal_a, pos_sal_b, pos_sal_c, "Episodic", pos_tags
|
||||
)
|
||||
|
||||
+1589
-664
File diff suppressed because one or more lines are too long
+18
@@ -0,0 +1,18 @@
|
||||
# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from.
|
||||
# Written by tools/soulc-stamp.sh --write. Do not hand-edit.
|
||||
# generated_amalgam_sha256 63e30030bee5e87fa082a84cda5c1226896f49da6076101fcd6b9530ea7caf49
|
||||
# generated_amalgam_bytes 1204442
|
||||
f8597e10546654bce3fbbe40461b2da59d0e06dbf1b038d1d362d24f949e3911 awareness.el
|
||||
b6f3d14ca0c26017a2d617399a6d3754dabb0905e4d5f52eb75d25c4ad18d3c5 chat.el
|
||||
42288c212cbf72fb1e8ecbd4d9900e4e9ee1cfa475b7974295c7637f1bf2939f elp-input.el
|
||||
b3f77f49d6086932c38bd17fe7a5eaf8bce25685f6fc3e1750f05729c6b49b9e imprint.el
|
||||
fba8ffdb9ba72bca5b09ca1c93a520edc52f3f4d8aec2c7585fe9b17e06420b2 manifest.el
|
||||
550a72e234ae8cec1f33e02108fd365353f45edd88513da90b792e79b6c0e5f0 memory.el
|
||||
5ec07ec9785b02abe32f3ff7acf2d1f9f7e07c0967fac97e6eff17d7110b5c84 neuron-api.el
|
||||
03c47c451e0e87f2c252cadb4b765867943962a804f548dd53adeef0520912c8 persist.el
|
||||
a6d69f3fc55233d9d3300160fd46a1551f2064bcd0fb84e2c9e432f636a72476 routes.el
|
||||
c28e36952ec56525963a0bdf29455ab097d3b0c5653d19c25fbb005e1069a1f7 safety.el
|
||||
fd3ab91d0ae0ea26639e21bef2f8f94054dc4b02eae68b19e3fe689d2769aad4 sessions.el
|
||||
0f1cf43904a98a5a646cce5a07e0e96162ced662692fbc13357d9b67d9a8ac3d soul.el
|
||||
30337940905171a9645b0929f0a412ce6b3dccb1246495070c553bca0bbae6cd stewardship.el
|
||||
e105dc5990e6adbf39db9dc0462cd8bcf6e6c3dfd03709059227ecfad2bbab29 studio.el
|
||||
+11
-2
@@ -110,7 +110,7 @@ tool("beginSession", "Initialize session: surface recent high-importance memorie
|
||||
"," + tool("linkCausal", "Create a causal edge (cause -> effect).") +
|
||||
"," + tool("restructureCausalGraph", "Re-balance the causal subgraph after new evidence.") +
|
||||
"," + tool("rebuildGraph", "Rebuild graph indices from the on-disk snapshot.") +
|
||||
"," + tool("runStructuralAudit", "Audit graph structure for orphans, dangling edges, mislabeled types.") +
|
||||
"," + tool("runStructuralAudit", "Stage 1 structural audit: owner-vs-runtime divergence, orphans and dangling edges, typed-edge distribution, self-model connectivity. Returns an annotated characterization, not a score.") +
|
||||
// ── Backlog + work ──────────────────────────────────────────────────────────
|
||||
"," + tool("planWork", "Create a backlog item.") +
|
||||
"," + tool("reviewBacklog", "Browse work items.") +
|
||||
@@ -680,7 +680,16 @@ fn dispatch_tool_call(tool_name: String, args: String) -> String {
|
||||
return mcp_json_result(resp)
|
||||
}
|
||||
if str_eq(tool_name, "runStructuralAudit") {
|
||||
let resp: String = http_get(neuron_url() + "/session/begin")
|
||||
// Was: GET /session/begin — an unrelated session digest returned under an
|
||||
// audit tool name, i.e. the tool advertised a check that did not exist.
|
||||
// Now points at the real Stage 1 route (neuron-api.el
|
||||
// handle_api_structural_audit). Sample caps ride the query string; the
|
||||
// defaults keep a manual audit to a couple of seconds.
|
||||
let e_s: Int = json_get_int(args, "edge_sample")
|
||||
let n_s: Int = json_get_int(args, "node_sample")
|
||||
let qs: String = "?edge_sample=" + int_to_str(if e_s > 0 { e_s } else { 3000 })
|
||||
+ "&node_sample=" + int_to_str(if n_s > 0 { n_s } else { 300 })
|
||||
let resp: String = http_get(neuron_url() + "/audit/structural" + qs)
|
||||
return mcp_json_result(resp)
|
||||
}
|
||||
|
||||
|
||||
@@ -1,9 +1,91 @@
|
||||
import "persist.el"
|
||||
|
||||
fn tier_working() -> String { return "Working" }
|
||||
fn tier_episodic() -> String { return "Episodic" }
|
||||
fn tier_canonical() -> String { return "Canonical" }
|
||||
|
||||
// ── Association on write ──────────────────────────────────────────────────────
|
||||
// DESIGN: "promotion integrates candidate nodes by linking them to existing nodes
|
||||
// using typed semantic edges RATHER THAN APPENDING AS UNLINKED CONTENT" (CCR
|
||||
// claim 29). Unlinked append is the explicitly rejected behaviour — and it is the
|
||||
// only behaviour this system had. Measured 2026-08-09 on Tim's graph: 14,214 edges
|
||||
// across 80,936 nodes, 5% of nodes connected to anything, and NO edge created by
|
||||
// any write since 2026-07-19 while 27,000+ nodes were added. A memory that forms
|
||||
// no connections cannot be reached by spreading activation, so retrieval silently
|
||||
// degrades to literal matching.
|
||||
//
|
||||
// BOUNDS, each one bought with a specific failure:
|
||||
// * max 3 edges per memory — link_memories.py's cap, precision over spray
|
||||
// * never link to identity (self/*, Value): the existing policy is explicit that
|
||||
// "memories must not pollute the self traversal by similarity; only an explicit
|
||||
// citation may touch identity". Similarity is not citation.
|
||||
// * never link telemetry (state-event, soul-response, boot_count, loop-outcome):
|
||||
// these are ~97% of daily write volume (1,020 vs 31 real memories on 08-08).
|
||||
// Linking them would add ~3,000 noise edges a day and re-flatten the graph in
|
||||
// the name of connecting it.
|
||||
// * fail-soft: a failed association never fails the write.
|
||||
// Edges go through wt_edge so they reach the owner and survive restart.
|
||||
fn mem_assoc_skip_label(label: String) -> Bool {
|
||||
if str_contains(label, "state-event") { return true }
|
||||
if str_contains(label, "soul-response") { return true }
|
||||
if str_contains(label, "soul-outbox") { return true }
|
||||
if str_contains(label, "boot_count") { return true }
|
||||
if str_contains(label, "loop-outcome") { return true }
|
||||
if str_contains(label, "search-result") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// A candidate is linkable only if it is a real, distinct, non-identity node.
|
||||
fn mem_assoc_ok(cand_id: String, cand_label: String, self_id: String) -> Bool {
|
||||
if str_eq(cand_id, "") { return false }
|
||||
if str_eq(cand_id, self_id) { return false }
|
||||
// CASE MATTERS — measured 2026-08-09. A lowercase-only check let a memory link
|
||||
// to "Self — Values (grounded)", i.e. it polluted the self traversal, which is
|
||||
// the one thing this policy exists to prevent. My verification had the same
|
||||
// blind spot and printed PASS. Check every casing the graph actually uses, and
|
||||
// exclude identity node TYPES as well as labels.
|
||||
let lab: String = str_lower(cand_label)
|
||||
if str_starts_with(lab, "self") { return false }
|
||||
if str_starts_with(lab, "value") { return false }
|
||||
if str_contains(lab, "values") { return false }
|
||||
if str_contains(lab, "identity") { return false }
|
||||
if mem_assoc_skip_label(cand_label) { return false }
|
||||
return true
|
||||
}
|
||||
|
||||
// One slot of the association. Manual unroll rather than a loop: EL's codegen
|
||||
// mis-emits accumulating while-loops (documented at soul.el:212, which unrolled
|
||||
// three affective slots for the same reason).
|
||||
fn mem_assoc_slot(results: String, idx: Int, new_id: String) -> Void {
|
||||
if idx >= json_array_len(results) { return }
|
||||
let cand: String = json_array_get(results, idx)
|
||||
let cid: String = json_get(cand, "id")
|
||||
let clabel: String = json_get(cand, "label")
|
||||
let ctype: String = json_get(cand, "node_type")
|
||||
if str_eq(ctype, "Value") { return }
|
||||
if str_eq(ctype, "DharmaSelf") { return }
|
||||
if str_eq(ctype, "Safety") { return }
|
||||
if mem_assoc_ok(cid, clabel, new_id) {
|
||||
wt_edge(new_id, cid, el_from_float(0.5), "related")
|
||||
}
|
||||
}
|
||||
|
||||
// mem_associate — connect a freshly written memory to what it is about.
|
||||
fn mem_associate(new_id: String, content: String, label: String) -> Void {
|
||||
if str_eq(new_id, "") { return }
|
||||
if mem_assoc_skip_label(label) { return }
|
||||
// Ask the graph what this memory resembles. Now that the store carries
|
||||
// meaning-vectors this is semantic, not merely lexical.
|
||||
let probe: String = str_slice(content, 0, 400)
|
||||
let results: String = engram_recall_json(probe, 4)
|
||||
if str_eq(results, "") { return }
|
||||
mem_assoc_slot(results, 0, new_id)
|
||||
mem_assoc_slot(results, 1, new_id)
|
||||
mem_assoc_slot(results, 2, new_id)
|
||||
}
|
||||
|
||||
fn mem_store(content: String, label: String, tags: String) -> String {
|
||||
let id: String = engram_node_full(
|
||||
let id: String = wt_node(
|
||||
content,
|
||||
"Memory",
|
||||
label,
|
||||
@@ -17,13 +99,26 @@ fn mem_store(content: String, label: String, tags: String) -> String {
|
||||
println("[memory] write rejected by engram (empty id): label=" + label)
|
||||
return ""
|
||||
}
|
||||
// Read back to verify the node actually persisted — guards against silent write failures.
|
||||
let readback: String = engram_get_node_json(id)
|
||||
if str_eq(readback, "") || str_eq(readback, "{}") {
|
||||
println("[memory] WRITE VERIFY FAILED: label=" + label + " id=" + id + " — node absent after write")
|
||||
return ""
|
||||
// wt_node has already read the node back locally and returns "" if it did
|
||||
// not land, so the old duplicate read-back here is gone.
|
||||
//
|
||||
// HONESTY (neuron#117): the receipt now says WHERE the write is.
|
||||
// The old unconditional "write verified" line asserted against the soul's
|
||||
// own RAM — true in memory, false on disk — and printed ~115,000 times on
|
||||
// Tim's machine while the canonical snapshot sat frozen for three days.
|
||||
// wt_commit flushes the spool and then asks the OWNER. When it says false
|
||||
// the node is real and recallable but not yet durable, and the log says so
|
||||
// rather than claiming a save that did not happen. The id is still returned:
|
||||
// the local write DID succeed, and the queued delta will be retried.
|
||||
let durable: Bool = wt_commit(id)
|
||||
// Associate AFTER the node is durable: an edge to a node that did not persist
|
||||
// is a dangling edge, which is the defect the 2026-08-09 cleanup removed 830 of.
|
||||
mem_associate(id, content, label)
|
||||
if durable {
|
||||
println("[memory] write persisted at owner: " + id + " label=" + label)
|
||||
} else {
|
||||
println("[memory] write IN MEMORY ONLY (queued for owner, not yet durable): " + id + " label=" + label)
|
||||
}
|
||||
println("[memory] write verified: " + id + " ok")
|
||||
return id
|
||||
}
|
||||
|
||||
@@ -51,12 +146,12 @@ fn mem_strengthen(node_id: String) -> Void {
|
||||
// memory.el (imported first) so awareness.el and neuron-api.el can both call it.
|
||||
fn mem_tombstone(node_id: String) -> String {
|
||||
let tags: String = "[\"Tombstone\",\"status:deleted\"]"
|
||||
let marker: String = engram_node_full(
|
||||
let marker: String = wt_node(
|
||||
node_id, "Tombstone", "tombstone:" + node_id,
|
||||
el_from_float(0.01), el_from_float(0.01), el_from_float(1.0),
|
||||
"Episodic", tags)
|
||||
if !str_eq(marker, "") {
|
||||
engram_connect(marker, node_id, el_from_float(1.0), "tombstones")
|
||||
wt_edge(marker, node_id, el_from_float(1.0), "tombstones")
|
||||
}
|
||||
return marker
|
||||
}
|
||||
|
||||
+534
-29
@@ -195,15 +195,24 @@ fn api_compact_activated(raw: String, max_items: Int, snip: Int) -> String {
|
||||
}
|
||||
|
||||
// api_persisted — read-back-after-write guard against hallucinated saves.
|
||||
// After a write builtin returns an id, confirm the node is actually queryable
|
||||
// via engram_get_node_json(id) (returns "" or "null" when missing). Returns
|
||||
// true only when the node is genuinely persisted.
|
||||
//
|
||||
// WIDENED FOR neuron#117. This function is the single gate every MCP write
|
||||
// handler passes through before it reports success (10 call sites), which makes
|
||||
// it the right place to close the honesty gap rather than editing ten receipts.
|
||||
//
|
||||
// It used to read back from engram_get_node_json — the SOUL'S OWN in-process
|
||||
// graph. In HTTP-engram mode that asserts the wrong thing: the soul is not the
|
||||
// persistence owner, so a node present in its RAM and absent from the owner read
|
||||
// as "persisted" and then vanished on the next restart. The guard was doing
|
||||
// exactly what its comment promised and still certifying writes that did not
|
||||
// survive. It now flushes the write-through spool and asks the OWNER.
|
||||
//
|
||||
// In file mode (no ENGRAM_URL) the soul IS the owner and wt_commit collapses to
|
||||
// the original local read-back — unchanged behaviour, which is what keeps this
|
||||
// reversible.
|
||||
fn api_persisted(id: String) -> Bool {
|
||||
if str_eq(id, "") { return false }
|
||||
let node: String = engram_get_node_json(id)
|
||||
// engram_get_node_json returns "{}" (empty object) when node is not found — not "" or "null".
|
||||
// Check all three to guard against any runtime variation.
|
||||
return !str_eq(node, "") && !str_eq(node, "null") && !str_eq(node, "{}")
|
||||
return wt_commit(id)
|
||||
}
|
||||
|
||||
// api_not_persisted — standard error for a write that did not read back.
|
||||
@@ -295,10 +304,35 @@ fn handle_api_begin_session(body: String) -> String {
|
||||
let state_events: String = api_compact_node_array(state_events_raw, 5, 500)
|
||||
let recent_raw: String = engram_scan_nodes_json(10, 0)
|
||||
let recent: String = api_compact_node_array(recent_raw, 10, 240)
|
||||
// SELF-SEEDED SLICE (2026-08-09). The design is explicit: "Every compilation
|
||||
// query begins at the self-model node and traverses outward... structural
|
||||
// reachability from the self-model node is a precondition for any node to
|
||||
// appear in compiled context" (will-anderson patents/drafts/engram-claims.md,
|
||||
// Self-Seeded Activation; DRAFT, not a filed provisional — cite it as such).
|
||||
//
|
||||
// Measured 2026-08-09 before this change: compiled context contained 0-1
|
||||
// identity records out of 10, because compilation seeds from a hardcoded
|
||||
// TEXT STRING, never from the self. Even an explicit "my values identity who
|
||||
// I am" query returned a boot counter and state-events.
|
||||
//
|
||||
// This restores the designed behaviour WITHOUT repeating the failure that got
|
||||
// self_neighbors set to [] in the first place: that was an UNBOUNDED ~90KB
|
||||
// neighbour dump which closed the socket on every call. Same bound as every
|
||||
// other list here — cap 8, 240-char snippets. The self root has 34 direct
|
||||
// neighbours of which 23 are identity records, so depth 1 is dense enough to
|
||||
// be worth seeding and small enough to stay cheap.
|
||||
let self_raw: String = engram_neighbors_json("kn-efeb4a5b-5aff-4759-8a97-7233099be6ee", 1, "both")
|
||||
// Cap 24, not 8: measured 2026-08-09, the self root's first 8 neighbours are
|
||||
// TAG nodes ("neuron", "tier:note", "disposition:experimental", "imprint",
|
||||
// "traversal") which crowd out the substantive identity records behind them.
|
||||
// The root has 34 neighbours of which 23 are identity; 24 captures them while
|
||||
// staying bounded. Cost measured at ~+4KB on a ~12KB response, nowhere near
|
||||
// the ~90KB unbounded dump that closed sockets and got this set to [].
|
||||
let self_slice: String = api_compact_node_array(self_raw, 24, 240)
|
||||
return "{\"stats\":" + stats
|
||||
+ ",\"recent\":" + recent
|
||||
+ ",\"activated\":" + activated
|
||||
+ ",\"self_neighbors\":[]"
|
||||
+ ",\"self_neighbors\":" + self_slice
|
||||
+ ",\"recent_state_events\":" + state_events + "}"
|
||||
}
|
||||
|
||||
@@ -342,10 +376,17 @@ fn handle_api_remember(body: String) -> String {
|
||||
let inner: String = str_slice(base_tags, 1, str_len(base_tags) - 1)
|
||||
"[" + inner + ",\"project:" + project + "\"]"
|
||||
}
|
||||
let id: String = engram_node_full(content, "Memory", "memory:remembered",
|
||||
let id: String = wt_node(content, "Memory", "memory:remembered",
|
||||
sal, sal, el_from_float(0.9),
|
||||
"Episodic", final_tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
// Associate on write (2026-08-09). THIS CALL MUST BE HERE, not only in mem_store.
|
||||
// The HTTP memory route writes via wt_node directly; mem_store serves only the
|
||||
// awareness paths (soul-response, search-result, activation-result) which are
|
||||
// exactly the telemetry we refuse to link. Hooking mem_store alone produced
|
||||
// ZERO edges across four real writes — measured, not assumed, which is the only
|
||||
// reason it was caught before shipping.
|
||||
mem_associate(id, content, "memory:remembered")
|
||||
return "{\"id\":\"" + id + "\",\"ok\":true}"
|
||||
}
|
||||
|
||||
@@ -369,7 +410,7 @@ fn handle_api_node_create(body: String) -> String {
|
||||
if str_eq(importance, "low") { 0.25 } else { 0.5 }
|
||||
}
|
||||
}
|
||||
let id: String = engram_node_full(content, node_type, label,
|
||||
let id: String = wt_node(content, node_type, label,
|
||||
sal, sal, el_from_float(0.9),
|
||||
tier, tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
@@ -422,11 +463,11 @@ fn handle_api_node_update(body: String) -> String {
|
||||
}
|
||||
let body_tags: String = json_get(body, "tags")
|
||||
let tags: String = if str_eq(body_tags, "") { "[\"" + node_type + "\"]" } else { body_tags }
|
||||
let new_id: String = engram_node_full(content, node_type, label,
|
||||
let new_id: String = wt_node(content, node_type, label,
|
||||
el_from_float(0.5), el_from_float(0.5), el_from_float(0.8),
|
||||
tier, tags)
|
||||
if !api_persisted(new_id) { return api_not_persisted(new_id) }
|
||||
engram_connect(new_id, id, el_from_float(0.9), "supersedes")
|
||||
wt_edge(new_id, id, el_from_float(0.9), "supersedes")
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + id + "\",\"ok\":true}"
|
||||
}
|
||||
|
||||
@@ -450,7 +491,12 @@ fn handle_api_recall(method: String, path: String, body: String) -> String {
|
||||
if str_eq(eff_q, "") {
|
||||
return api_or_empty(engram_scan_nodes_json(limit, 0))
|
||||
}
|
||||
let results: String = engram_search_json(eff_q, limit)
|
||||
// engram_recall_json, not engram_search_json: this route IS the retrieval
|
||||
// surface (claim 24's "embedding search queries"), so it gets the semantic
|
||||
// and associative legs. engram_search_json stays lexical because ~40
|
||||
// internal call sites pass a KEY and seven of them delete every record
|
||||
// that comes back — see the boundary note above eg_search_json_impl.
|
||||
let results: String = engram_recall_json(eff_q, limit)
|
||||
return api_or_empty(results)
|
||||
}
|
||||
|
||||
@@ -498,7 +544,7 @@ fn handle_api_capture_knowledge(body: String) -> String {
|
||||
let full: String = if str_eq(title, "") { content } else { title + ": " + content }
|
||||
let lbl: String = str_slice(title, 0, 80)
|
||||
let tags: String = "[\"Knowledge\",\"captured\"]"
|
||||
let id: String = engram_node_full(full, "Knowledge", lbl,
|
||||
let id: String = wt_node(full, "Knowledge", lbl,
|
||||
el_from_float(0.85), el_from_float(0.8), el_from_float(0.9),
|
||||
"Episodic", tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
@@ -513,12 +559,12 @@ fn handle_api_evolve_knowledge(body: String) -> String {
|
||||
if !str_eq(prior_id, "") && is_protected_node(prior_id) { return api_err_protected(prior_id) }
|
||||
let tags: String = "[\"Knowledge\",\"evolved\"]"
|
||||
// Empty label → engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
|
||||
let new_id: String = engram_node_full(content, "Knowledge", "",
|
||||
let new_id: String = wt_node(content, "Knowledge", "",
|
||||
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
|
||||
"Episodic", tags)
|
||||
if !api_persisted(new_id) { return api_not_persisted(new_id) }
|
||||
if !str_eq(prior_id, "") {
|
||||
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
}
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
|
||||
}
|
||||
@@ -535,11 +581,11 @@ fn handle_api_promote_knowledge(body: String) -> String {
|
||||
"[\"Knowledge\",\"tier:canonical\",\"disposition:stable\"]"
|
||||
} else { tags_raw }
|
||||
// Empty label → engram_node_full derives content[:60] (LABEL FIX 2026-07-23).
|
||||
let new_id: String = engram_node_full(content, "Knowledge", "",
|
||||
let new_id: String = wt_node(content, "Knowledge", "",
|
||||
el_from_float(0.9), el_from_float(0.9), el_from_float(1.0),
|
||||
"Canonical", tags)
|
||||
if !api_persisted(new_id) { return api_not_persisted(new_id) }
|
||||
engram_connect(new_id, prior_id, el_from_float(0.95), "supersedes")
|
||||
wt_edge(new_id, prior_id, el_from_float(0.95), "supersedes")
|
||||
return "{\"ok\":true,\"new_id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\"}"
|
||||
}
|
||||
|
||||
@@ -562,7 +608,7 @@ fn handle_api_define_process(body: String) -> String {
|
||||
if str_eq(content, "") { return api_err("content is required") }
|
||||
let label: String = if str_eq(name, "") { "process:unnamed" } else { "process:" + name }
|
||||
let tags: String = "[\"Process\"]"
|
||||
let id: String = engram_node_full(content, "Process", label,
|
||||
let id: String = wt_node(content, "Process", label,
|
||||
el_from_float(0.8), el_from_float(0.8), el_from_float(0.9),
|
||||
"Canonical", tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
@@ -647,7 +693,7 @@ fn handle_api_tune_config(body: String) -> String {
|
||||
if str_eq(key, "") { return api_err("key is required") }
|
||||
let content: String = "config:" + key + "=" + value
|
||||
let tags: String = "[\"ConfigEntry\",\"config\"]"
|
||||
let id: String = engram_node_full(content, "ConfigEntry", key,
|
||||
let id: String = wt_node(content, "ConfigEntry", key,
|
||||
el_from_float(0.85), el_from_float(0.85), el_from_float(0.9),
|
||||
"Canonical", tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
@@ -694,7 +740,7 @@ fn handle_api_link_entities(body: String) -> String {
|
||||
if is_protected_node(to_id) { return api_err_protected(to_id) }
|
||||
let relation: String = json_get(body, "relation")
|
||||
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
|
||||
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
|
||||
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
|
||||
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\"}"
|
||||
}
|
||||
|
||||
@@ -727,11 +773,11 @@ fn handle_api_evolve_memory(body: String) -> String {
|
||||
}
|
||||
}
|
||||
let tags: String = "[\"Memory\",\"evolved\"]"
|
||||
let new_id: String = engram_node_full(content, "Memory", "memory:evolved",
|
||||
let new_id: String = wt_node(content, "Memory", "memory:evolved",
|
||||
sal, sal, el_from_float(0.9),
|
||||
"Episodic", tags)
|
||||
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
|
||||
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
}
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true}"
|
||||
}
|
||||
@@ -789,11 +835,11 @@ fn handle_api_cultivate(body: String) -> String {
|
||||
let content: String = json_get(body, "content")
|
||||
if str_eq(content, "") { return api_err("content is required") }
|
||||
let tags: String = "[\"Knowledge\",\"evolved\",\"cultivated\"]"
|
||||
let new_id: String = engram_node_full(content, "Knowledge", "knowledge:cultivated",
|
||||
let new_id: String = wt_node(content, "Knowledge", "knowledge:cultivated",
|
||||
el_from_float(0.75), el_from_float(0.75), el_from_float(0.9),
|
||||
"Episodic", tags)
|
||||
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
|
||||
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
}
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
|
||||
}
|
||||
@@ -809,11 +855,11 @@ fn handle_api_cultivate(body: String) -> String {
|
||||
}
|
||||
}
|
||||
let tags: String = "[\"Memory\",\"evolved\",\"cultivated\"]"
|
||||
let new_id: String = engram_node_full(content, "Memory", "memory:cultivated",
|
||||
let new_id: String = wt_node(content, "Memory", "memory:cultivated",
|
||||
sal, sal, el_from_float(0.9),
|
||||
"Episodic", tags)
|
||||
if !str_eq(prior_id, "") && !str_eq(new_id, "") {
|
||||
engram_connect(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
wt_edge(new_id, prior_id, el_from_float(0.9), "supersedes")
|
||||
}
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + prior_id + "\",\"ok\":true,\"cultivated\":true}"
|
||||
}
|
||||
@@ -833,7 +879,7 @@ fn handle_api_cultivate(body: String) -> String {
|
||||
if str_eq(to_id, "") { return api_err("to_id is required") }
|
||||
let relation: String = json_get(body, "relation")
|
||||
let eff_relation: String = if str_eq(relation, "") { "associates" } else { relation }
|
||||
engram_connect(from_id, to_id, el_from_float(0.5), eff_relation)
|
||||
wt_edge(from_id, to_id, el_from_float(0.5), eff_relation)
|
||||
return "{\"ok\":true,\"from_id\":\"" + from_id + "\",\"to_id\":\"" + to_id + "\",\"relation\":\"" + eff_relation + "\",\"cultivated\":true}"
|
||||
}
|
||||
|
||||
@@ -868,7 +914,7 @@ fn handle_api_consolidate(body: String) -> String {
|
||||
if !str_eq(summary, "") {
|
||||
let safe_summary: String = str_replace(summary, "\"", "'")
|
||||
let tags: String = "[\"SessionSummary\",\"consolidate\"]"
|
||||
let summary_id: String = engram_node_full(
|
||||
let summary_id: String = wt_node(
|
||||
"[session-summary] " + safe_summary,
|
||||
"SessionSummary", "session:summary",
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
@@ -880,3 +926,462 @@ fn handle_api_consolidate(body: String) -> String {
|
||||
}
|
||||
return "{\"ok\":true,\"snapshot\":\"" + snap + "\"}"
|
||||
}
|
||||
|
||||
// ── Stage 1: structural audit ─────────────────────────────────────────────────
|
||||
//
|
||||
// WHAT THIS IMPLEMENTS
|
||||
// The CGI provisional, 05-detailed-description.md, "Stage 1: Structural audit
|
||||
// 430". Verbatim, the audit module evaluates: the density and typed
|
||||
// distribution of causal edges; the consistency between value nodes and
|
||||
// execution-record neighborhoods; the richness and connectivity of the
|
||||
// self-model; and the authenticity of open-question nodes in the wonder
|
||||
// manifest. It "produces a coherence assessment 432 — NOT A BINARY SCORE but
|
||||
// an annotated characterization of the graph's structural properties".
|
||||
//
|
||||
// That last clause is the whole shape of this handler. Every finding carries
|
||||
// its own numbers AND a plain-language note saying what the numbers mean and
|
||||
// how they were obtained. There is no pass/fail, no percentage-of-health, no
|
||||
// composite score, and `"score":null` is emitted explicitly so a downstream
|
||||
// reader cannot mistake its absence for an omission.
|
||||
//
|
||||
// WHY IT EXISTS NOW, AND WHY THE FIRST FINDING IS THE ONE IT IS
|
||||
// `runStructuralAudit` has been an advertised MCP tool with nothing behind it:
|
||||
// the dispatcher GET'd /session/begin and returned that blob (mcp-wrapper/src/
|
||||
// main.el). Meanwhile the failure the audit would have caught ran silently for
|
||||
// about three weeks — the soul reported 103,089 nodes while the engram, which
|
||||
// OWNS persistence, held ~79,900; a crash discarded the difference. Every boot
|
||||
// reported green throughout, because nothing in the system ever compared the
|
||||
// two sides. So finding 1 is owner-versus-runtime divergence: it is the check
|
||||
// whose absence cost real memory, and it is cheap and exact.
|
||||
//
|
||||
// WHAT IS DELIBERATELY NOT HERE (stage 1b, see the `deferred` array in the
|
||||
// response): value/execution-record consistency and wonder-manifest
|
||||
// authenticity. Both need node types that barely exist in this graph today —
|
||||
// the response MEASURES those populations and reports the counts as the reason,
|
||||
// rather than asserting a deferral without evidence.
|
||||
//
|
||||
// MEASUREMENT HONESTY: EXACT WHERE CHEAP, SAMPLED WHERE NOT, ALWAYS LABELLED
|
||||
// Counts, edge typing and self-model connectivity are exact. Orphan rate and
|
||||
// dangling-edge rate are SAMPLED, because the engram runtime has no node-id
|
||||
// index — `engram_find_node_index` is a linear scan over every node, so an
|
||||
// exhaustive dangling check is O(nodes x edges) (~2.2e9 string compares at
|
||||
// today's scale, tens of seconds inside one request). The samples are UNIFORM
|
||||
// across the whole population, not head-of-list, and every sampled figure is
|
||||
// emitted with its own `sampled` / `population` fields plus an extrapolation
|
||||
// labelled as such. Raise `?edge_sample=` / `?node_sample=` to the population
|
||||
// size to run either check exhaustively and pay the time. The real fix is an
|
||||
// id index in the runtime; that is the engram repo's, not this handler's.
|
||||
|
||||
// audit_pct1 — one-decimal percentage as a bare JSON number, sign-safe.
|
||||
// Integer math only: EL has no fixed-precision formatter, and float_to_str
|
||||
// would put an unbounded mantissa in the response.
|
||||
fn audit_pct1(num: Int, den: Int) -> String {
|
||||
if den <= 0 { return "null" }
|
||||
let neg: Bool = num < 0
|
||||
let a: Int = if neg { 0 - num } else { num }
|
||||
let tenths: Int = (a * 1000) / den
|
||||
let whole: Int = tenths / 10
|
||||
let frac: Int = tenths - (whole * 10)
|
||||
let sign: String = if neg { "-" } else { "" }
|
||||
return sign + int_to_str(whole) + "." + int_to_str(frac)
|
||||
}
|
||||
|
||||
// audit_finding — the one envelope every finding uses: name, the measurements,
|
||||
// and the annotation. Keeping it in one place is what stops the characterization
|
||||
// from degenerating into a bag of numbers with no reading attached.
|
||||
fn audit_finding(name: String, measured: String, note: String) -> String {
|
||||
return "{\"finding\":\"" + name + "\""
|
||||
+ ",\"measured\":{" + measured + "}"
|
||||
+ ",\"note\":\"" + api_json_escape(note) + "\"}"
|
||||
}
|
||||
|
||||
// audit_str_at — read the quoted string value starting at byte `start`.
|
||||
// Slices a bounded window rather than the tail of the (multi-MB) edges array, so
|
||||
// this is O(window) per call instead of O(remaining input).
|
||||
fn audit_str_at(s: String, start: Int, maxlen: Int) -> String {
|
||||
let n: Int = str_len(s)
|
||||
if start < 0 || start >= n { return "" }
|
||||
let end_guess: Int = start + maxlen
|
||||
let stop: Int = if end_guess > n { n } else { end_guess }
|
||||
let win: String = str_slice(s, start, stop)
|
||||
let q: Int = str_index_of(win, "\"")
|
||||
if q < 0 { return "" }
|
||||
return str_slice(win, 0, q)
|
||||
}
|
||||
|
||||
// audit_rel_count — exact count of edges carrying `rel`, by scanning the emitted
|
||||
// edge array for the literal `"relation":"<rel>"`. engram_emit_edge_json writes
|
||||
// metadata ESCAPED as a string, so no nested object can contain that literal and
|
||||
// the count cannot be inflated by edge payloads.
|
||||
fn audit_rel_count(edges: String, rel: String) -> Int {
|
||||
return str_count(edges, "\"relation\":\"" + rel + "\"")
|
||||
}
|
||||
|
||||
// audit_owner_stats — ask the persistence OWNER for its own counts.
|
||||
// Returns "" when there is no HTTP owner configured or the owner is unreachable;
|
||||
// both are reported as findings, never as a failure of the audit.
|
||||
fn audit_owner_stats(url: String) -> String {
|
||||
if str_eq(url, "") { return "" }
|
||||
return http_get(url + "/api/stats")
|
||||
}
|
||||
|
||||
// audit_divergence — FINDING 1. Runtime (this soul's in-process graph) versus
|
||||
// the persistence owner's own count. Trend is measured against the previous
|
||||
// audit recorded in soul state, so a second call answers "is the gap growing?"
|
||||
// rather than just restating it.
|
||||
fn audit_divergence() -> String {
|
||||
let rt_nodes: Int = engram_node_count()
|
||||
let rt_edges: Int = engram_edge_count()
|
||||
let url: String = wt_engram_url()
|
||||
|
||||
if str_eq(url, "") {
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"none\",\"owner_reachable\":false",
|
||||
"No HTTP persistence owner is configured, so this soul IS the owner "
|
||||
+ "(file mode) and divergence is not defined. This check only has "
|
||||
+ "meaning when ENGRAM_URL points at a separate engram that owns the "
|
||||
+ "canonical store.")
|
||||
}
|
||||
|
||||
let stats: String = audit_owner_stats(url)
|
||||
// REACHABILITY IS PROVED BY THE PAYLOAD, NOT BY A NON-EMPTY REPLY.
|
||||
// http_get does not return "" on a connection failure — it returns a JSON
|
||||
// error object ({"error":"Failed to connect to ... Couldn't connect to
|
||||
// server"}). Testing only for "" made a DEAD owner read as reachable with
|
||||
// node_count 0, i.e. the audit would have reported a 100% divergence and
|
||||
// named it as data loss. That false positive is worse than no check at all:
|
||||
// it is precisely the kind of confident wrong answer this route exists to
|
||||
// stop. Require the field the contract promises.
|
||||
let owner_nc_raw: String = json_get_raw(stats, "node_count")
|
||||
if str_eq(stats, "") || str_eq(owner_nc_raw, "") {
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":false"
|
||||
+ ",\"owner_reply\":\"" + api_json_escape(api_utf8_trunc(stats, 200)) + "\"",
|
||||
"The persistence owner at " + url + " did not return a node_count "
|
||||
+ "from GET /api/stats. Divergence is UNKNOWN, NOT ZERO — an owner "
|
||||
+ "that cannot be read is exactly the condition under which the "
|
||||
+ "runtime's own count means least, and reporting 0 for the owner "
|
||||
+ "would manufacture a total-loss reading out of a network error. "
|
||||
+ "Reported as a finding rather than raised as an error so the rest "
|
||||
+ "of the audit still returns; the owner's raw reply is in "
|
||||
+ "owner_reply.")
|
||||
}
|
||||
|
||||
let ow_nodes: Int = json_get_int(stats, "node_count")
|
||||
let ow_edges: Int = json_get_int(stats, "edge_count")
|
||||
let d_nodes: Int = rt_nodes - ow_nodes
|
||||
let d_edges: Int = rt_edges - ow_edges
|
||||
|
||||
// Trend against the previous audit in this soul's state.
|
||||
let prev_raw: String = state_get("audit_prev_node_delta")
|
||||
let prev: Int = str_to_int(prev_raw)
|
||||
let abs_now: Int = if d_nodes < 0 { 0 - d_nodes } else { d_nodes }
|
||||
let abs_prev: Int = if prev < 0 { 0 - prev } else { prev }
|
||||
let trend: String = if str_eq(prev_raw, "") {
|
||||
"no_prior_audit"
|
||||
} else {
|
||||
if abs_now > abs_prev { "growing" } else {
|
||||
if abs_now < abs_prev { "shrinking" } else { "flat" }
|
||||
}
|
||||
}
|
||||
state_set("audit_prev_node_delta", int_to_str(d_nodes))
|
||||
state_set("audit_prev_ts", int_to_str(time_now()))
|
||||
|
||||
let note_head: String = if d_nodes == 0 {
|
||||
"Runtime and owner agree on node count."
|
||||
} else {
|
||||
"Runtime holds " + int_to_str(d_nodes) + " nodes (" + audit_pct1(d_nodes, rt_nodes)
|
||||
+ "% of its own graph) that the persistence owner does not report. Nodes "
|
||||
+ "that exist only in runtime memory do not survive a restart."
|
||||
}
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":true"
|
||||
+ ",\"owner_nodes\":" + int_to_str(ow_nodes)
|
||||
+ ",\"owner_edges\":" + int_to_str(ow_edges)
|
||||
+ ",\"node_delta\":" + int_to_str(d_nodes)
|
||||
+ ",\"edge_delta\":" + int_to_str(d_edges)
|
||||
+ ",\"node_delta_pct_of_runtime\":" + audit_pct1(d_nodes, rt_nodes)
|
||||
+ ",\"trend_vs_previous_audit\":\"" + trend + "\""
|
||||
+ ",\"previous_node_delta\":" + (if str_eq(prev_raw, "") { "null" } else { int_to_str(prev) }),
|
||||
note_head + " Trend against the previous audit recorded in this soul's "
|
||||
+ "state: " + trend + ". This is the comparison whose absence let a "
|
||||
+ "~24,000-node loss run for weeks with every boot reporting green.")
|
||||
}
|
||||
|
||||
// audit_edge_typing — FINDING 2. Density plus the typed distribution the patent
|
||||
// asks for, against the claim-10 relation vocabulary. Exact: str_count over the
|
||||
// emitted edge array, one linear pass per relation.
|
||||
fn audit_edge_typing(edges: String, total_edges: Int, node_total: Int) -> String {
|
||||
let c_sup: Int = audit_rel_count(edges, "Supersedes")
|
||||
let c_cau: Int = audit_rel_count(edges, "Causes")
|
||||
let c_con: Int = audit_rel_count(edges, "Contains")
|
||||
let c_ref: Int = audit_rel_count(edges, "References")
|
||||
let c_ctr: Int = audit_rel_count(edges, "Contradicts")
|
||||
let c_exe: Int = audit_rel_count(edges, "Exemplifies")
|
||||
let c_act: Int = audit_rel_count(edges, "Activates")
|
||||
let c_tmp: Int = audit_rel_count(edges, "TemporallyPrecedes")
|
||||
let typed: Int = c_sup + c_cau + c_con + c_ref + c_ctr + c_exe + c_act + c_tmp
|
||||
|
||||
// Lowercase near-misses: the same eight concepts written by the ad-hoc write
|
||||
// paths (linkEntities defaults to "associates", linkCausal to "causes").
|
||||
// Counted separately because "the vocabulary is unused" and "the vocabulary
|
||||
// is used in the wrong case" are different defects with different fixes.
|
||||
let l_sup: Int = audit_rel_count(edges, "supersedes")
|
||||
let l_cau: Int = audit_rel_count(edges, "causes")
|
||||
let l_con: Int = audit_rel_count(edges, "contains")
|
||||
let l_ref: Int = audit_rel_count(edges, "references")
|
||||
let l_ctr: Int = audit_rel_count(edges, "contradicts")
|
||||
let l_exe: Int = audit_rel_count(edges, "exemplifies")
|
||||
let l_act: Int = audit_rel_count(edges, "activates")
|
||||
let l_tmp: Int = audit_rel_count(edges, "temporallyPrecedes")
|
||||
let near: Int = l_sup + l_cau + l_con + l_ref + l_ctr + l_exe + l_act + l_tmp
|
||||
|
||||
let untyped: Int = total_edges - typed
|
||||
return audit_finding("typed_edge_distribution",
|
||||
"\"total_edges\":" + int_to_str(total_edges)
|
||||
+ ",\"total_nodes\":" + int_to_str(node_total)
|
||||
// Density per 100 nodes, not per node: EL has no fixed-precision float
|
||||
// formatter, and "0.3 edges per node" rounded to an integer is a lie.
|
||||
+ ",\"edges_per_100_nodes\":" + audit_pct1(total_edges, node_total)
|
||||
+ ",\"claim10_typed\":" + int_to_str(typed)
|
||||
+ ",\"claim10_typed_pct\":" + audit_pct1(typed, total_edges)
|
||||
+ ",\"outside_claim10_vocabulary\":" + int_to_str(untyped)
|
||||
+ ",\"lowercase_near_miss\":" + int_to_str(near)
|
||||
+ ",\"by_relation\":{"
|
||||
+ "\"Supersedes\":" + int_to_str(c_sup)
|
||||
+ ",\"Causes\":" + int_to_str(c_cau)
|
||||
+ ",\"Contains\":" + int_to_str(c_con)
|
||||
+ ",\"References\":" + int_to_str(c_ref)
|
||||
+ ",\"Contradicts\":" + int_to_str(c_ctr)
|
||||
+ ",\"Exemplifies\":" + int_to_str(c_exe)
|
||||
+ ",\"Activates\":" + int_to_str(c_act)
|
||||
+ ",\"TemporallyPrecedes\":" + int_to_str(c_tmp) + "}",
|
||||
"Only " + int_to_str(typed) + " of " + int_to_str(total_edges)
|
||||
+ " edges use the claim-10 causal vocabulary; the remainder are ad-hoc "
|
||||
+ "relation strings, which is why the graph's causal claims cannot yet "
|
||||
+ "be checked for internal consistency — an untyped edge asserts "
|
||||
+ "association, not causation. " + int_to_str(near) + " edges use a "
|
||||
+ "lowercase spelling of a claim-10 relation: those are near-misses the "
|
||||
+ "write paths could be corrected to emit, not genuinely foreign types.")
|
||||
}
|
||||
|
||||
// audit_orphans_dangling — FINDING 3. Both figures are SAMPLED; see the header
|
||||
// for why exhaustive is O(nodes x edges) on this runtime.
|
||||
//
|
||||
// An "orphan" here is a node with zero RESOLVABLE edges: engram_neighbors_json
|
||||
// drops any edge whose other endpoint does not resolve to a node, so a node
|
||||
// whose only edges are dangling reads as an orphan. That is the right reading —
|
||||
// such a node is unreachable by traversal — but it is stated rather than hidden.
|
||||
fn audit_orphans_dangling(edges: String, total_edges: Int, node_total: Int,
|
||||
edge_cap: Int, node_cap: Int) -> String {
|
||||
// ── orphan sample: uniform stride over the node store ──
|
||||
let n_take: Int = if node_total < node_cap { node_total } else { node_cap }
|
||||
let n_stride: Int = if n_take > 0 { node_total / n_take } else { 1 }
|
||||
let n_stride = if n_stride < 1 { 1 } else { n_stride }
|
||||
let orphans: Int = 0
|
||||
let n_checked: Int = 0
|
||||
let j: Int = 0
|
||||
while j < n_take {
|
||||
let one: String = engram_scan_nodes_json(1, j * n_stride)
|
||||
let nid: String = json_get(json_array_get(one, 0), "id")
|
||||
if !str_eq(nid, "") {
|
||||
let nbrs: String = engram_neighbors_json(nid, 1, "both")
|
||||
let deg: Int = json_array_len(nbrs)
|
||||
let orphans = if deg == 0 { orphans + 1 } else { orphans }
|
||||
let n_checked = n_checked + 1
|
||||
}
|
||||
let j = j + 1
|
||||
}
|
||||
|
||||
// ── dangling sample: uniform stride over the edge array ──
|
||||
// str_index_of_all gives every edge's field offsets in ONE linear pass, so
|
||||
// any index can be read in O(1). json_array_get would have been O(i) per
|
||||
// element and O(n^2) over the array.
|
||||
let from_pos: [Int] = str_index_of_all(edges, "\"from_id\":\"")
|
||||
let to_pos: [Int] = str_index_of_all(edges, "\"to_id\":\"")
|
||||
let nf: Int = len(from_pos)
|
||||
let nt: Int = len(to_pos)
|
||||
let ne: Int = if nf < nt { nf } else { nt }
|
||||
let e_take: Int = if ne < edge_cap { ne } else { edge_cap }
|
||||
let e_stride: Int = if e_take > 0 { ne / e_take } else { 1 }
|
||||
let e_stride = if e_stride < 1 { 1 } else { e_stride }
|
||||
let dangling: Int = 0
|
||||
let e_checked: Int = 0
|
||||
let i: Int = 0
|
||||
while i < ne && e_checked < e_take {
|
||||
let fid: String = audit_str_at(edges, get(from_pos, i) + 11, 96)
|
||||
let tid: String = audit_str_at(edges, get(to_pos, i) + 9, 96)
|
||||
let f_gone: Bool = str_eq(engram_get_node_json(fid), "{}")
|
||||
let t_gone: Bool = if f_gone { true } else { str_eq(engram_get_node_json(tid), "{}") }
|
||||
let dangling = if f_gone || t_gone { dangling + 1 } else { dangling }
|
||||
let e_checked = e_checked + 1
|
||||
let i = i + e_stride
|
||||
}
|
||||
|
||||
let orphan_est: Int = if n_checked > 0 { (orphans * node_total) / n_checked } else { 0 }
|
||||
let dangle_est: Int = if e_checked > 0 { (dangling * total_edges) / e_checked } else { 0 }
|
||||
let exhaustive_n: String = if n_checked >= node_total { "true" } else { "false" }
|
||||
let exhaustive_e: String = if e_checked >= ne { "true" } else { "false" }
|
||||
|
||||
return audit_finding("orphans_and_dangling_edges",
|
||||
"\"nodes_population\":" + int_to_str(node_total)
|
||||
+ ",\"nodes_sampled\":" + int_to_str(n_checked)
|
||||
+ ",\"nodes_sample_exhaustive\":" + exhaustive_n
|
||||
+ ",\"orphans_in_sample\":" + int_to_str(orphans)
|
||||
+ ",\"orphan_rate_pct\":" + audit_pct1(orphans, n_checked)
|
||||
+ ",\"orphans_extrapolated\":" + int_to_str(orphan_est)
|
||||
+ ",\"edges_population\":" + int_to_str(total_edges)
|
||||
+ ",\"edges_sampled\":" + int_to_str(e_checked)
|
||||
+ ",\"edges_sample_exhaustive\":" + exhaustive_e
|
||||
+ ",\"dangling_in_sample\":" + int_to_str(dangling)
|
||||
+ ",\"dangling_rate_pct\":" + audit_pct1(dangling, e_checked)
|
||||
+ ",\"dangling_extrapolated\":" + int_to_str(dangle_est),
|
||||
"Orphan = zero RESOLVABLE edges, so a node whose only edges dangle counts "
|
||||
+ "as an orphan; either way it is unreachable by traversal. Dangling = an "
|
||||
+ "edge with an endpoint id that resolves to no node. Both are uniform "
|
||||
+ "stride samples over the whole population, not the head of the list; "
|
||||
+ "the extrapolations are estimates and are labelled as such. Pass "
|
||||
+ "?node_sample= / ?edge_sample= at or above the population size to run "
|
||||
+ "either check exhaustively. A high orphan rate is a characterization, "
|
||||
+ "not a verdict: an accumulating store legitimately holds unlinked "
|
||||
+ "material. It becomes a defect when the write paths were SUPPOSED to "
|
||||
+ "link and did not.")
|
||||
}
|
||||
|
||||
// audit_pillar — one self-model pillar: present, how much content, how connected.
|
||||
fn audit_pillar(key: String, id: String) -> String {
|
||||
let node: String = engram_get_node_json(id)
|
||||
let present: Bool = !str_eq(node, "{}") && !str_eq(node, "")
|
||||
if !present {
|
||||
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":false"
|
||||
+ ",\"content_length\":0,\"degree\":0}"
|
||||
}
|
||||
let content: String = json_get(node, "content")
|
||||
let deg: Int = json_array_len(engram_neighbors_json(id, 1, "both"))
|
||||
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":true"
|
||||
+ ",\"label\":\"" + api_json_escape(json_get(node, "label")) + "\""
|
||||
+ ",\"tier\":\"" + api_json_escape(json_get(node, "tier")) + "\""
|
||||
+ ",\"content_length\":" + int_to_str(str_len(content))
|
||||
+ ",\"degree\":" + int_to_str(deg) + "}"
|
||||
}
|
||||
|
||||
// audit_self_model — FINDING 4. "the richness and connectivity of the
|
||||
// self-model ... is it connected to behavioral evidence?"
|
||||
//
|
||||
// This finding RETIRES the Claude-side vitals identity block. That check lived
|
||||
// outside the system it was checking — a shell script grepping a snapshot — so
|
||||
// it could only ever report on a file, and it went on reporting green while the
|
||||
// memory-philosophy pillar was absent from the live graph for about three weeks.
|
||||
// Asking the running soul about its own three pillars is the designed mechanism;
|
||||
// a shell probe was the fourth patch on the same hole.
|
||||
fn audit_self_model() -> String {
|
||||
let dna: String = audit_pillar("intellectual_dna", "kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6")
|
||||
let val: String = audit_pillar("values_hub", "kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
|
||||
let phi: String = audit_pillar("memory_philosophy", "kn-dcfe04b3-3702-4cac-b6f0-ecb4db837eee")
|
||||
let root: String = audit_pillar("self_root", "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
|
||||
return audit_finding("self_model_connectivity",
|
||||
"\"pillars\":{" + dna + "," + val + "," + phi + "," + root + "}",
|
||||
"The three identity pillars plus the self root. `degree` counts nodes "
|
||||
+ "reachable in one hop in either direction — the self-model's connection "
|
||||
+ "to the rest of the graph. present:false on any pillar is the condition "
|
||||
+ "that ran undetected for weeks; content_length distinguishes a pillar "
|
||||
+ "that is present from one that is present but hollowed out. The patent "
|
||||
+ "also asks whether the self-model makes ACCURATE PREDICTIONS about the "
|
||||
+ "system's own behavior; that half needs Prediction nodes and is deferred "
|
||||
+ "with the rest of stage 1b below.")
|
||||
}
|
||||
|
||||
// audit_deferred — what stage 1 does NOT yet evaluate, with the measured reason.
|
||||
// Emitted as data, not as a comment, so a reader of the assessment sees the gap
|
||||
// and its evidence rather than inferring completeness from silence.
|
||||
fn audit_deferred() -> String {
|
||||
let preds: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("Prediction", 50, 0)))
|
||||
let wonders: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("WonderQuestion", 50, 0)))
|
||||
return "[{\"deferred\":\"value_execution_record_consistency\""
|
||||
+ ",\"stage\":\"1b\""
|
||||
+ ",\"measured\":{\"prediction_nodes_found\":" + int_to_str(preds) + "}"
|
||||
+ ",\"reason\":\"" + api_json_escape(
|
||||
"The patent asks whether the execution history SUPPORTS the stated "
|
||||
+ "values or shows systematic conflict. That requires execution "
|
||||
+ "records tied to value nodes and predictions to score them against. "
|
||||
+ "Prediction nodes found (capped at 50): " + int_to_str(preds)
|
||||
+ ". Asserting value/execution coherence on that population would be "
|
||||
+ "a fabricated result, which is worse than a stated gap.") + "\"}"
|
||||
+ ",{\"deferred\":\"wonder_manifest_authenticity\""
|
||||
+ ",\"stage\":\"1b\""
|
||||
+ ",\"measured\":{\"wonder_question_nodes_found\":" + int_to_str(wonders) + "}"
|
||||
+ ",\"reason\":\"" + api_json_escape(
|
||||
"The patent asks whether pull weights CORRELATE WITH GENUINE "
|
||||
+ "PREDICTION UNCERTAINTY or are uniform/externally assigned — a "
|
||||
+ "correlation between two populations. WonderQuestion nodes readable "
|
||||
+ "by type (capped at 50): " + int_to_str(wonders) + ", against "
|
||||
+ int_to_str(preds) + " Prediction nodes. There is a known write/read "
|
||||
+ "node-type mismatch on the wonder path; until that is fixed and both "
|
||||
+ "populations exist, any correlation reported here would be noise.") + "\"}]"
|
||||
}
|
||||
|
||||
// handle_api_structural_audit — Stage 1. Returns the coherence assessment 432:
|
||||
// an annotated characterization, explicitly NOT a score.
|
||||
//
|
||||
// COST NOTE: the edge findings need the relation labels, and the runtime exposes
|
||||
// no edge-enumeration builtin. The only way to see them is the same one
|
||||
// GET /api/graph/edges already uses — engram_save to a SCRATCH path (never the
|
||||
// owner's canonical file; see routes.el, neuron#117) and read the array back.
|
||||
// On a large graph that is a multi-hundred-MB write, so this is a manual audit
|
||||
// route, not something to put on a timer. Pass ?edges=0 to skip both edge
|
||||
// findings and get the divergence + self-model readings cheaply.
|
||||
fn handle_api_structural_audit(method: String, path: String, body: String) -> String {
|
||||
let node_total: Int = engram_node_count()
|
||||
let edge_total: Int = engram_edge_count()
|
||||
let want_edges: Bool = !str_eq(api_query_param(path, "edges"), "0")
|
||||
let edge_cap: Int = api_query_int(path, "edge_sample", 3000)
|
||||
let node_cap: Int = api_query_int(path, "node_sample", 300)
|
||||
|
||||
let divergence: String = audit_divergence()
|
||||
let self_model: String = audit_self_model()
|
||||
|
||||
let edge_part: String = if want_edges {
|
||||
// Scratch export only. state_get("soul_snapshot_path") is deliberately
|
||||
// NOT used: in HTTP-engram mode the soul is not the persistence owner and
|
||||
// must never write the canonical file, not even on a read path.
|
||||
let scratch_dir: String = env("TMPDIR")
|
||||
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
|
||||
let snap_path: String = scratch_base + "/soul-audit-export-" + state_get("soul_cgi_id") + ".json"
|
||||
// engram_save returns Int (1 ok / 0 fail); str_eq on it SIGSEGVs (#150).
|
||||
let saved: Int = engram_save(snap_path)
|
||||
if saved == 0 {
|
||||
"," + audit_finding("typed_edge_distribution", "\"available\":false",
|
||||
"Could not export the graph to " + snap_path + " for edge analysis, "
|
||||
+ "so edge typing and the dangling-edge sample were not run. "
|
||||
+ "Reported as a gap, not as zero findings.")
|
||||
} else {
|
||||
// wt_read, not fs_read: fs_read leaves a thread-local length hint that
|
||||
// the NEXT HTTP response would use as its Content-Length, appending
|
||||
// adjacent heap bytes to the reply (see persist.el wt_read).
|
||||
let snap: String = wt_read(snap_path)
|
||||
let edges_raw: String = json_get_raw(snap, "edges")
|
||||
let edges: String = if str_eq(edges_raw, "") { "[]" } else { edges_raw }
|
||||
"," + audit_edge_typing(edges, edge_total, node_total)
|
||||
+ "," + audit_orphans_dangling(edges, edge_total, node_total, edge_cap, node_cap)
|
||||
}
|
||||
} else {
|
||||
""
|
||||
}
|
||||
|
||||
return "{\"audit\":\"structural\",\"stage\":1"
|
||||
+ ",\"spec\":\"CGI provisional 05-detailed-description.md, Stage 1: Structural audit 430\""
|
||||
+ ",\"assessment\":\"coherence_assessment_432\""
|
||||
+ ",\"assessment_kind\":\"annotated_characterization\""
|
||||
+ ",\"score\":null"
|
||||
+ ",\"score_note\":\"By design. The specification calls for an annotated characterization of the graph's structural properties, not a binary score. Read the findings.\""
|
||||
+ ",\"cgi_id\":\"" + api_json_escape(state_get("soul_cgi_id")) + "\""
|
||||
+ ",\"ts_ms\":" + int_to_str(time_now())
|
||||
+ ",\"findings\":[" + divergence + "," + self_model + edge_part + "]"
|
||||
+ ",\"deferred\":" + audit_deferred() + "}"
|
||||
}
|
||||
|
||||
@@ -37,3 +37,4 @@ extern fn handle_api_memory_update(body: String) -> String
|
||||
extern fn handle_api_cultivate(body: String) -> String
|
||||
extern fn handle_api_list_typed(node_type: String, path: String, body: String) -> String
|
||||
extern fn handle_api_consolidate(body: String) -> String
|
||||
extern fn handle_api_structural_audit(method: String, path: String, body: String) -> String
|
||||
|
||||
+426
@@ -0,0 +1,426 @@
|
||||
// persist.el — the soul→engram WRITE-THROUGH boundary (neuron#117).
|
||||
//
|
||||
// WHY THIS FILE EXISTS
|
||||
// soul.el:571-573 states the ownership rule: "when ENGRAM_URL is set the HTTP
|
||||
// Engram owns persistence — the soul must NEVER write to the local snapshot
|
||||
// (not the persistence owner)." The soul obeys the NEGATIVE half. The POSITIVE
|
||||
// half — how a write made inside the soul actually REACHES the owner — was
|
||||
// never built. Sync is pull-only (awareness.el `/api/sync` -> engram_load_merge),
|
||||
// so every node the soul creates lives in its process RAM and is shed on
|
||||
// restart. Measured live 2026-08-07: soul node_count=102184, engram
|
||||
// node_count=79197 — ~23k nodes existing nowhere but RAM.
|
||||
//
|
||||
// SCOPE NOTE ON THE PATENT (corrects an earlier internal reading)
|
||||
// Engram provisional claims 15-18 describe a delta-sync protocol "with peer
|
||||
// Engram instances"; claim 17's pull-then-push sequence is PEER-ENGRAM to
|
||||
// PEER-ENGRAM. The soul is NOT a peer Engram — it is a CALLER of the database
|
||||
// system API (cf. claim 27, "invoked explicitly by a caller of the database
|
||||
// system API"). So claim 17 does not specify a soul↔engram contract and is not
|
||||
// cited as authority here. This design is derived from the ownership rule
|
||||
// alone: the owner owns the writes, therefore the soul must HAND writes to the
|
||||
// owner and must never write the owner's file itself.
|
||||
//
|
||||
// THE MECHANISM, AND WHY NOT `POST /api/nodes`
|
||||
// The obvious route is the one the persona/boot-counter write-backs already
|
||||
// use, POST /api/nodes. It is the wrong instrument here, verified against the
|
||||
// live engram binary in a sandbox:
|
||||
// - it mints a NEW server-side id (engram_node), so the soul's id and the
|
||||
// owner's id diverge — the next /api/sync pull re-imports the node as a
|
||||
// DUPLICATE, and any edge referencing the soul's id never resolves;
|
||||
// - it accepts only {content, node_type, salience} and drops label, tier,
|
||||
// tags, importance, confidence, metadata. A probe posted with tier
|
||||
// "Canonical" came back tier "Working", importance 0.5.
|
||||
// POST /api/load-merge (Will's own route, el `dc39a61`) is the right one:
|
||||
// - engram_load_merge PRESERVES the id and every field;
|
||||
// - it dedups nodes by id and edges by (from_id,to_id,relation), so a
|
||||
// re-submitted delta is a NO-OP — retry safety is free, and it is the same
|
||||
// local-wins semantics the graph already uses;
|
||||
// - it calls persist_canonical() — THE OWNER writes its own canonical file.
|
||||
// The soul never touches it. The ownership rule is honoured in its
|
||||
// strongest form rather than worked around;
|
||||
// - it returns real counts {ok, nodes_added, edges_added, node_count},
|
||||
// so a receipt can be a MEASUREMENT instead of a fixed success shape.
|
||||
//
|
||||
// SPOOL-AND-DRAIN, AND WHY IT IS NOT JUST A DIRECT POST
|
||||
// Measured in a sandbox against a 79k-node / 176MB graph (live scale): one
|
||||
// load-merge costs ~0.38s, essentially all of it the owner's persist_canonical.
|
||||
// A chat turn writes 5-7 nodes; pushing each separately would add ~2.7s per
|
||||
// turn. So writes are STAGED and pushed in one coalesced batch.
|
||||
// The staging buffer is the FILESYSTEM, not process state, because the soul
|
||||
// serves each HTTP connection on its own pthread (el_runtime http_serve_async)
|
||||
// and a shared in-process buffer would lose entries to a read-modify-write
|
||||
// race — silently, which is the one failure mode this file exists to end.
|
||||
// One file per write, named with uuid_v4, is race-free by construction and
|
||||
// buys a property a memory buffer cannot: writes that could not be pushed
|
||||
// SURVIVE A SOUL CRASH and are drained on the next boot.
|
||||
//
|
||||
// WHAT IS DELIBERATELY NOT PUSHED
|
||||
// - InternalStateEvent / heartbeat telemetry. Will's own carve-out, stated in
|
||||
// engram server.el 8f8ccc9: "48h-pruned, loss-tolerant, ~2/min; snapshotting
|
||||
// 28MB per heartbeat is waste."
|
||||
// NOTE (ours, flagged for Will): we do NOT additionally exclude Working-tier
|
||||
// nodes. That exclusion exists in `fb0bb55` to stop the boot counter leaking
|
||||
// through the /api/sync PULL; it is about sync backflow, not durability.
|
||||
// Applying it here would exclude mem_store — which writes tier "Working" — and
|
||||
// mem_store is the single most important durable write path in the soul. Boot
|
||||
// seeding reads the canonical file wholesale, so a pushed Working-tier node
|
||||
// does survive restart. This is the one classification call this file makes
|
||||
// that Will has not ruled on.
|
||||
//
|
||||
// WHAT THIS BOUNDARY CANNOT EXPRESS (by construction, not by omission)
|
||||
// - engram_strengthen (salience/activation drift): load-merge SKIPS ids that
|
||||
// already exist, so it cannot update an existing node. There is no owner-side
|
||||
// update/upsert route. Not pushable through any current route; left as a
|
||||
// follow-up that needs a change in the engram repo.
|
||||
// - engram_forget (hard delete): load-merge is additive and has no delete verb.
|
||||
// Propagating deletes would mean DELETE /api/nodes/<id>, a HARD delete at the
|
||||
// owner — which scripts/verify-soul-contract.sh section B explicitly fails the
|
||||
// build for ("to delete is to supersede/tombstone, never hard-remove"). Local
|
||||
// deletes therefore stay local; the TOMBSTONE NODE and its "tombstones" edge
|
||||
// (mem_tombstone) are pushed, and that is the sanctioned representation of a
|
||||
// deletion in this graph.
|
||||
|
||||
// ── Configuration ─────────────────────────────────────────────────────────────
|
||||
|
||||
// wt_engram_url — same resolution order as ise_post: env, then the state key
|
||||
// stashed at boot. NO hardcoded localhost fallback: unlike telemetry, inventing
|
||||
// a destination for durable data would risk pushing a user's memories at whatever
|
||||
// happens to be listening on 8742. Empty means "no HTTP owner" -> file mode.
|
||||
fn wt_engram_url() -> String {
|
||||
let env_url: String = env("ENGRAM_URL")
|
||||
if !str_eq(env_url, "") { return env_url }
|
||||
return state_get("soul_engram_url")
|
||||
}
|
||||
|
||||
fn wt_api_key() -> String {
|
||||
let env_key: String = env("ENGRAM_API_KEY")
|
||||
if !str_eq(env_key, "") { return env_key }
|
||||
return state_get("soul_engram_api_key")
|
||||
}
|
||||
|
||||
// wt_enabled — true only in HTTP-engram mode. In file mode the soul IS the
|
||||
// persistence owner and every path below is a no-op, so this whole feature is
|
||||
// inert for genesis/local deployments. That is also what makes it reversible.
|
||||
fn wt_enabled() -> Bool {
|
||||
return !str_eq(wt_engram_url(), "")
|
||||
}
|
||||
|
||||
// wt_spool_dir — where staged deltas live. MUST be readable by the engram
|
||||
// process: /api/load-merge takes a PATH and the owner opens it itself. Both
|
||||
// processes are same-host by construction (dev-stack LaunchAgents; the GKE
|
||||
// image starts engram and soul in one container per entrypoint.sh).
|
||||
fn wt_spool_dir() -> String {
|
||||
let raw: String = env("SOUL_OUTBOX_DIR")
|
||||
let dir: String = if str_eq(raw, "") { env("HOME") + "/.neuron/soul-outbox" } else { raw }
|
||||
fs_mkdir(dir)
|
||||
return dir
|
||||
}
|
||||
|
||||
// ── Helpers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
// wt_esc — minimal JSON string escape. Deliberately local rather than reusing
|
||||
// chat.el's json_safe: persist.el is imported BY memory.el, which is imported by
|
||||
// chat.el, so depending on chat.el here would be an import cycle.
|
||||
fn wt_esc(s: String) -> String {
|
||||
let s1: String = str_replace(s, "\\", "\\\\")
|
||||
let s2: String = str_replace(s1, "\"", "\\\"")
|
||||
let s3: String = str_replace(s2, "\n", "\\n")
|
||||
let s4: String = str_replace(s3, "\r", "\\r")
|
||||
let s5: String = str_replace(s4, "\t", "\\t")
|
||||
return s5
|
||||
}
|
||||
|
||||
// wt_durable_class — Will's telemetry carve-out, by node_type. See header.
|
||||
fn wt_durable_class(node_type: String) -> Bool {
|
||||
if str_eq(node_type, "InternalStateEvent") { return false }
|
||||
return true
|
||||
}
|
||||
|
||||
// wt_inner — strip the surrounding brackets off a JSON array so several arrays
|
||||
// can be concatenated into one. Returns "" for "[]" / "" / anything too short.
|
||||
fn wt_inner(arr: String) -> String {
|
||||
let n: Int = str_len(arr)
|
||||
if n < 3 { return "" }
|
||||
if !str_starts_with(arr, "[") { return "" }
|
||||
return str_slice(arr, 1, n - 1)
|
||||
}
|
||||
|
||||
// wt_read — fs_read, plus a MANDATORY reset of the runtime's binary-length hint.
|
||||
//
|
||||
// THIS IS NOT OPTIONAL AND MUST NOT BE "SIMPLIFIED" BACK TO A BARE fs_read.
|
||||
// The pinned runtime (vendor/el-runtime/v1.0.0-20260501) keeps a thread-local
|
||||
// `_tl_fs_read_len` that fs_read SETS to the file's byte count (so binary files
|
||||
// can be served with a correct Content-Length) and that http_send_response
|
||||
// CONSUMES as the Content-Length of the next reply. Nothing else clears it
|
||||
// except json_get_raw. So any fs_read during request handling that is not
|
||||
// followed by a json_get_raw makes the NEXT HTTP response advertise the FILE's
|
||||
// length instead of the body's — and the runtime then sends that many bytes,
|
||||
// appending whatever adjacent heap memory follows the reply.
|
||||
//
|
||||
// Caught here, measured: a /api/neuron/memory reply that should be 86 bytes went
|
||||
// out as 497, with 411 bytes of this module's own spool paths and log strings
|
||||
// trailing the JSON. The drain reads spool files mid-request, so this boundary
|
||||
// is exactly where the landmine gets stepped on.
|
||||
//
|
||||
// Upstream el fixed the class in `43636ae` ("pair fs_read length hint with its
|
||||
// buffer"); that runtime is NOT the one vendored here, and re-pinning the
|
||||
// runtime is deliberately out of scope for this change. Clearing the hint at
|
||||
// our own boundary fixes our exposure without touching the pinned C.
|
||||
// json_get_raw is used as the reset because it is the only builtin in this
|
||||
// runtime that zeroes the hint, and it does so before any early return.
|
||||
fn wt_clear_binlen() -> Void {
|
||||
let discard: String = json_get_raw("{}", "_wt_reset")
|
||||
}
|
||||
|
||||
fn wt_read(path: String) -> String {
|
||||
let data: String = fs_read(path)
|
||||
wt_clear_binlen()
|
||||
return data
|
||||
}
|
||||
|
||||
// wt_sweep — best-effort removal of the zero-byte husks left by truncation.
|
||||
// The runtime exposes no unlink builtin, so a drained delta is emptied rather
|
||||
// than deleted; this reclaims the directory entries.
|
||||
//
|
||||
// `-empty` is the safety property, not an optimisation: the command is
|
||||
// STRUCTURALLY INCAPABLE of removing a delta that still has content, so it can
|
||||
// never destroy a pending write even if it runs concurrently with a stage.
|
||||
// Only the directory path is interpolated (never a filename), and it is quoted.
|
||||
// The exit code is ignored — an un-swept husk costs one directory entry.
|
||||
fn wt_sweep(dir: String) -> Void {
|
||||
if str_eq(dir, "") { return }
|
||||
if str_contains(dir, "'") { return }
|
||||
exec_command("find '" + dir + "' -maxdepth 1 -name 'wt*.json' -empty -delete 2>/dev/null")
|
||||
}
|
||||
|
||||
// ── Staging ───────────────────────────────────────────────────────────────────
|
||||
|
||||
// wt_stage — write ONE delta file. uuid_v4 in the name makes concurrent stagers
|
||||
// collision-free without any lock. Returns true if the delta is on disk.
|
||||
fn wt_stage(nodes_json: String, edges_json: String) -> Bool {
|
||||
let dir: String = wt_spool_dir()
|
||||
if str_eq(dir, "") { return false }
|
||||
let payload: String = "{\"nodes\":" + nodes_json + ",\"edges\":" + edges_json + "}"
|
||||
let path: String = dir + "/wt-" + uuid_v4() + ".json"
|
||||
fs_write(path, payload)
|
||||
// Read-back-verify the stage itself. A stage that did not land is a write we
|
||||
// would otherwise believe was queued — exactly the hallucinated-save class.
|
||||
if str_eq(wt_read(path), "") {
|
||||
println("[persist] wt_stage: FAILED to write spool file " + path + " — delta not queued")
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// ── The write boundary ────────────────────────────────────────────────────────
|
||||
|
||||
// wt_node — create a node locally AND queue it for the persistence owner.
|
||||
// Same signature and same return contract as engram_node_full ("" on failure),
|
||||
// so converting a call site is a rename and nothing else.
|
||||
fn wt_node(content: String, node_type: String, label: String,
|
||||
salience: Float, importance: Float, confidence: Float,
|
||||
tier: String, tags: String) -> String {
|
||||
let id: String = engram_node_full(content, node_type, label,
|
||||
salience, importance, confidence,
|
||||
tier, tags)
|
||||
if str_eq(id, "") { return "" }
|
||||
// engram_get_node_json emits the SAME record shape engram_save writes (minus
|
||||
// the embedding vector, which the owner backfills lazily), so the read-back
|
||||
// doubles as the delta payload — no second serialization to drift.
|
||||
let rec: String = engram_get_node_json(id)
|
||||
if str_eq(rec, "") || str_eq(rec, "{}") {
|
||||
println("[persist] wt_node: local write did not read back, id=" + id + " label=" + label)
|
||||
return ""
|
||||
}
|
||||
if wt_enabled() && wt_durable_class(node_type) {
|
||||
wt_stage("[" + rec + "]", "[]")
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// wt_edge — create an edge locally AND queue it. Mirrors engram_connect.
|
||||
//
|
||||
// The edge id is freshly generated rather than read back: the runtime exposes no
|
||||
// "id of the edge I just created" accessor, and the owner dedups edges by
|
||||
// (from_id,to_id,relation), never by id — so the id is not load-bearing. The
|
||||
// consequence, stated plainly: the soul's copy and the owner's copy of the same
|
||||
// edge carry different edge ids. Nothing in either codebase looks an edge up by
|
||||
// id (neighbors traversal scans from_id/to_id), so this is cosmetic.
|
||||
fn wt_edge(from_id: String, to_id: String, weight: Float, relation: String) -> Void {
|
||||
engram_connect(from_id, to_id, weight, relation)
|
||||
if !wt_enabled() { return }
|
||||
if str_eq(from_id, "") || str_eq(to_id, "") { return }
|
||||
let ts: Int = time_now()
|
||||
let rec: String = "{\"id\":\"" + uuid_v4() + "\""
|
||||
+ ",\"from_id\":\"" + wt_esc(from_id) + "\""
|
||||
+ ",\"to_id\":\"" + wt_esc(to_id) + "\""
|
||||
+ ",\"relation\":\"" + wt_esc(relation) + "\""
|
||||
+ ",\"metadata\":\"{}\""
|
||||
+ ",\"weight\":" + float_to_str(weight)
|
||||
+ ",\"confidence\":1"
|
||||
+ ",\"created_at\":" + int_to_str(ts)
|
||||
+ ",\"updated_at\":" + int_to_str(ts)
|
||||
+ ",\"last_fired\":0,\"inhibitory\":0,\"layer_id\":1}"
|
||||
wt_stage("[]", "[" + rec + "]")
|
||||
}
|
||||
|
||||
// ── The drain ─────────────────────────────────────────────────────────────────
|
||||
|
||||
// wt_drain — coalesce every staged delta into ONE load-merge against the owner.
|
||||
//
|
||||
// Returns: nodes_added on success (>= 0), 0 when there was nothing to do, and
|
||||
// -1 when the push FAILED. -1 is load-bearing: on failure the spool files are
|
||||
// left untouched, so nothing is lost and the next drain retries them. A caller
|
||||
// must never read a non-negative return as "my particular node is durable" —
|
||||
// use wt_durable(id) for that.
|
||||
//
|
||||
// Concurrency: several threads may drain at once. Each builds its own batch file
|
||||
// (uuid-named), and overlapping batches are harmless because load-merge dedups.
|
||||
// Files are truncated ONLY after a confirmed ok:true, so a lost race costs a
|
||||
// redundant push, never a dropped write.
|
||||
fn wt_drain() -> Int {
|
||||
if !wt_enabled() { return 0 }
|
||||
let dir: String = wt_spool_dir()
|
||||
if str_eq(dir, "") { return 0 }
|
||||
|
||||
// el_list_len/el_list_get, NOT json_stringify(fs_list(...)): fs_list builds
|
||||
// a native list via el_list_append, and json_stringify does not serialize
|
||||
// that type — it renders the raw pointer value. (Verified in isolation; the
|
||||
// same latent defect is live in studio.el's /api/tools/file/list route,
|
||||
// which returns e.g. {"entries":4386409744}. Noted, not fixed here.)
|
||||
let listing = fs_list(dir)
|
||||
let count: Int = el_list_len(listing)
|
||||
if count == 0 { return 0 }
|
||||
|
||||
let nodes_acc: String = ""
|
||||
let edges_acc: String = ""
|
||||
let drained: String = ""
|
||||
let found: Int = 0
|
||||
let i: Int = 0
|
||||
// No `continue` / `break`: elc lists them as keywords but not one line of
|
||||
// the shipped soul uses either, so they are unexercised on this build path.
|
||||
// Guard conditions are expressed as nested ifs instead, and every rebind is
|
||||
// at the loop-body top level where `let x = ...` is assignment (the idiom
|
||||
// memory.el's boot-counter loop relies on) — never inside a nested block,
|
||||
// where it would shadow instead.
|
||||
while i < count {
|
||||
let name: String = el_list_get(listing, i)
|
||||
// A delta is only usable when it ends with the closing "]}" that
|
||||
// wt_stage writes last. fs_write is not atomic, so a file being written
|
||||
// right now can be observed half-formed; requiring the terminator means
|
||||
// it is picked up whole on the next drain instead of merged as garbage.
|
||||
// An empty read means "already drained and truncated" — not an error.
|
||||
let p: String = if str_starts_with(name, "wt-") { dir + "/" + name } else { "" }
|
||||
let raw: String = if str_eq(p, "") { "" } else { wt_read(p) }
|
||||
let usable: Bool = !str_eq(raw, "") && str_ends_with(raw, "]}")
|
||||
let nj: String = if usable { wt_inner(json_get_raw(raw, "nodes")) } else { "" }
|
||||
let ej: String = if usable { wt_inner(json_get_raw(raw, "edges")) } else { "" }
|
||||
let nodes_acc = if str_eq(nj, "") { nodes_acc } else if str_eq(nodes_acc, "") { nj } else { nodes_acc + "," + nj }
|
||||
let edges_acc = if str_eq(ej, "") { edges_acc } else if str_eq(edges_acc, "") { ej } else { edges_acc + "," + ej }
|
||||
let drained = if !usable { drained } else if str_eq(drained, "") { p } else { drained + "\n" + p }
|
||||
let found = if usable { found + 1 } else { found }
|
||||
let i = i + 1
|
||||
}
|
||||
|
||||
if found == 0 { return 0 }
|
||||
|
||||
let combined: String = "{\"nodes\":[" + nodes_acc + "],\"edges\":[" + edges_acc + "]}"
|
||||
let batch: String = dir + "/wtb-" + uuid_v4() + ".json"
|
||||
fs_write(batch, combined)
|
||||
if str_eq(wt_read(batch), "") {
|
||||
println("[persist] wt_drain: could not write batch file " + batch + " — " + int_to_str(found) + " deltas stay queued")
|
||||
return -1
|
||||
}
|
||||
|
||||
let url: String = wt_engram_url()
|
||||
let key: String = wt_api_key()
|
||||
let body: String = "{\"path\":\"" + wt_esc(batch) + "\",\"_auth\":\"" + wt_esc(key) + "\"}"
|
||||
let resp: String = http_post_json(url + "/api/load-merge", body)
|
||||
|
||||
// The batch file is pure scratch — the retry is rebuilt from the SPOOL, not
|
||||
// from it. Truncate it unconditionally, before branching on the outcome, so
|
||||
// a persistently unreachable owner cannot accumulate one husk per attempt.
|
||||
fs_write(batch, "")
|
||||
|
||||
// Distinguish the two failures rather than collapsing them: "cannot reach
|
||||
// the owner" and "the owner refused this delta" need different human
|
||||
// responses, and a log line that says the wrong one costs a debugging hour.
|
||||
// curl surfaces transport errors as a JSON body, so an empty response is not
|
||||
// the only unreachable signal.
|
||||
// (str_contains rather than a strict parse on purpose — the engram's HTTP
|
||||
// responses have been observed carrying trailing bytes past the JSON.)
|
||||
let unreachable: Bool = str_eq(resp, "")
|
||||
|| str_contains(resp, "Couldn't connect")
|
||||
|| str_contains(resp, "Failed to connect")
|
||||
|| str_contains(resp, "Could not resolve")
|
||||
|| str_contains(resp, "timed out")
|
||||
if unreachable {
|
||||
wt_sweep(dir)
|
||||
println("[persist] wt_drain: owner UNREACHABLE at " + url + " — " + int_to_str(found)
|
||||
+ " deltas stay queued in " + dir + " (will retry): " + resp)
|
||||
return -1
|
||||
}
|
||||
if !str_contains(resp, "\"ok\":true") {
|
||||
wt_sweep(dir)
|
||||
println("[persist] wt_drain: owner REJECTED the delta — " + int_to_str(found)
|
||||
+ " stay queued in " + dir + ": " + resp)
|
||||
return -1
|
||||
}
|
||||
|
||||
let added: Int = json_get_int(resp, "nodes_added")
|
||||
let added_e: Int = json_get_int(resp, "edges_added")
|
||||
|
||||
// Confirmed. Truncate the drained spool files so they are not re-pushed.
|
||||
// Truncation (not deletion) because the runtime exposes no unlink builtin;
|
||||
// an emptied file is inert to the loop above. The zero-byte husks are then
|
||||
// swept below.
|
||||
let paths = str_split(drained, "\n")
|
||||
let pn: Int = el_list_len(paths)
|
||||
let k: Int = 0
|
||||
while k < pn {
|
||||
let one: String = el_list_get(paths, k)
|
||||
if !str_eq(one, "") { fs_write(one, "") }
|
||||
let k = k + 1
|
||||
}
|
||||
wt_sweep(dir)
|
||||
|
||||
println("[persist] wt_drain: pushed " + int_to_str(found) + " deltas -> owner added "
|
||||
+ int_to_str(added) + " nodes, " + int_to_str(added_e) + " edges")
|
||||
return added
|
||||
}
|
||||
|
||||
// wt_durable — is this id present AT THE OWNER? The only honest answer to
|
||||
// "did my write persist" in HTTP mode.
|
||||
//
|
||||
// In file mode the soul IS the owner, so the local read-back is the owner-side
|
||||
// read-back and this collapses to the pre-existing check.
|
||||
//
|
||||
// nodes_added from wt_drain is NOT a substitute: a concurrent drain may have
|
||||
// already pushed this node, making our own added count 0 while the node is
|
||||
// perfectly durable. Presence at the owner is the fact; counts are telemetry.
|
||||
fn wt_durable(id: String) -> Bool {
|
||||
if str_eq(id, "") { return false }
|
||||
if !wt_enabled() {
|
||||
let local: String = engram_get_node_json(id)
|
||||
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
|
||||
}
|
||||
let url: String = wt_engram_url()
|
||||
let resp: String = http_get(url + "/api/nodes/" + id)
|
||||
if str_eq(resp, "") { return false }
|
||||
if str_eq(resp, "{}") { return false }
|
||||
return str_contains(resp, "\"id\"")
|
||||
}
|
||||
|
||||
// wt_commit — flush, then assert at the owner. The receipt callers should use.
|
||||
// Deliberately NOT a fixed success shape: it can and does return false while the
|
||||
// local write is perfectly fine in RAM, which is the true state of affairs when
|
||||
// the owner is unreachable.
|
||||
fn wt_commit(id: String) -> Bool {
|
||||
if str_eq(id, "") { return false }
|
||||
if !wt_enabled() {
|
||||
let local: String = engram_get_node_json(id)
|
||||
return !str_eq(local, "") && !str_eq(local, "null") && !str_eq(local, "{}")
|
||||
}
|
||||
let pushed: Int = wt_drain()
|
||||
return wt_durable(id)
|
||||
}
|
||||
@@ -186,7 +186,7 @@ fn route_imprint_contextual(body: String) -> String {
|
||||
return "{\"ok\":false,\"error\":\"empty body\"}"
|
||||
}
|
||||
let tags: String = "[\"imprint\",\"contextual\"]"
|
||||
let id: String = engram_node_full(
|
||||
let id: String = wt_node(
|
||||
body,
|
||||
"Entity",
|
||||
"imprint:contextual",
|
||||
@@ -208,7 +208,7 @@ fn route_imprint_user(body: String) -> String {
|
||||
return "{\"ok\":false,\"error\":\"empty body\"}"
|
||||
}
|
||||
let tags: String = "[\"imprint\",\"user\"]"
|
||||
let id: String = engram_node_full(
|
||||
let id: String = wt_node(
|
||||
body,
|
||||
"Entity",
|
||||
"imprint:user",
|
||||
@@ -239,7 +239,7 @@ fn route_synthesize(body: String) -> String {
|
||||
}
|
||||
let req: String = "synthesize " + parent_a + " " + parent_b
|
||||
let tags: String = "[\"soul-inbox-pending\",\"synthesis-request\"]"
|
||||
engram_node_full(
|
||||
wt_node(
|
||||
req,
|
||||
"Entity",
|
||||
"synthesis-request",
|
||||
@@ -395,7 +395,28 @@ fn handle_connectors(method: String, clean: String, body: String) -> String {
|
||||
return "{\"ok\":false,\"error\":\"unknown connectors route\"}"
|
||||
}
|
||||
|
||||
// handle_request — the soul's HTTP entry point.
|
||||
//
|
||||
// NOTE ON THE NAME (neuron#117): the el runtime resolves this handler by NAME
|
||||
// via dlsym(RTLD_DEFAULT, "handle_request") — that is why the Linux build must
|
||||
// link -rdynamic. So the dispatcher body moved to route_dispatch and the name
|
||||
// `handle_request` stays put as a thin wrapper. Do not rename it back.
|
||||
//
|
||||
// The wrapper exists to give the write-through boundary a guaranteed flush
|
||||
// point. route_dispatch returns from ~60 places; a per-branch flush would be
|
||||
// forgotten on the 61st. Draining here means EVERY request that staged a write
|
||||
// pushes it before the connection closes, whatever route produced it, including
|
||||
// routes added later that know nothing about persistence.
|
||||
//
|
||||
// wt_drain is a no-op (no HTTP, no cost) when nothing is staged and when the
|
||||
// soul is not in HTTP-engram mode, so this is free on read traffic.
|
||||
fn handle_request(method: String, path: String, body: String) -> String {
|
||||
let resp: String = route_dispatch(method, path, body)
|
||||
let flushed: Int = wt_drain()
|
||||
return resp
|
||||
}
|
||||
|
||||
fn route_dispatch(method: String, path: String, body: String) -> String {
|
||||
let clean: String = strip_query(path)
|
||||
|
||||
// ACTIVITY STAMP (2026-07-30 self-review): every inbound HTTP request —
|
||||
@@ -432,10 +453,27 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
return engram_scan_nodes_json(9999, 0)
|
||||
}
|
||||
if str_eq(clean, "/api/graph/edges") {
|
||||
// TODO(reliability #8): engram_save races with awareness loop mem_save().
|
||||
// Both now use atomic write-to-temp+rename (el_runtime.c). Serialised
|
||||
// by engram_global_mu. Future: add engram_edges_json() builtin.
|
||||
let snap_path: String = env("HOME") + "/.neuron/engram/snapshot.json"
|
||||
// FIXED (neuron#117): this GET used to engram_save() straight over
|
||||
// ~/.neuron/engram/snapshot.json — a READ route, in a process that is
|
||||
// NOT the persistence owner, overwriting the owner's canonical file
|
||||
// on every call. It broke soul.el:571-573 ("the soul must NEVER write
|
||||
// to the local snapshot") and it is the same defect class Will removed
|
||||
// from the engram itself in el `dc39a61` ("stop read routes clobbering
|
||||
// canonical snapshot"), where route_scan_edges/route_sync were moved
|
||||
// to scratch paths for exactly this reason. It was also the race the
|
||||
// old TODO(reliability #8) admitted to.
|
||||
//
|
||||
// Export to a scratch path instead. Same response, no canonical write.
|
||||
// The soul's own snapshot writes are otherwise already gated behind
|
||||
// state key "soul_snapshot_path", which is set ONLY in the genesis
|
||||
// file-mode branch (soul.el: is_genesis && safe_to_seed, and
|
||||
// safe_to_seed is unconditionally false when ENGRAM_URL is set) — so
|
||||
// after this change the soul writes nothing at all in HTTP mode.
|
||||
// Future: add an engram_edges_json() builtin and drop the file round
|
||||
// trip entirely.
|
||||
let scratch_dir: String = env("TMPDIR")
|
||||
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
|
||||
let snap_path: String = scratch_base + "/soul-edges-export-" + state_get("soul_cgi_id") + ".json"
|
||||
engram_save(snap_path)
|
||||
let snap: String = fs_read(snap_path)
|
||||
let edges_raw: String = json_get_raw(snap, "edges")
|
||||
@@ -529,6 +567,13 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
if str_starts_with(clean, "/api/neuron/graph") {
|
||||
return handle_api_inspect_graph(method, path, body)
|
||||
}
|
||||
// Stage 1 structural audit (CGI provisional, "Structural audit 430").
|
||||
// GET because it is a read of the graph's own structure; the query string
|
||||
// carries the sample caps (?edge_sample=, ?node_sample=, ?edges=0), so
|
||||
// str_starts_with rather than str_eq.
|
||||
if str_starts_with(clean, "/api/neuron/audit/structural") {
|
||||
return handle_api_structural_audit(method, path, body)
|
||||
}
|
||||
if str_starts_with(clean, "/api/neuron/list/") {
|
||||
// Offset 17 = len("/api/neuron/list/"). Was 16, which left a leading "/" on node_type
|
||||
// ("/BacklogItem"), so engram_scan_nodes_by_type_json matched nothing → list/<type>
|
||||
@@ -710,6 +755,12 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
if str_eq(clean, "/api/neuron/graph/link") {
|
||||
return handle_api_link_entities(body)
|
||||
}
|
||||
// POST accepted too: same handler, so a JSON-RPC-shaped caller that only
|
||||
// speaks POST reaches the identical audit. Options still come from the
|
||||
// query string — the handler reads no body fields.
|
||||
if str_eq(clean, "/api/neuron/audit/structural") {
|
||||
return handle_api_structural_audit(method, path, body)
|
||||
}
|
||||
if str_eq(clean, "/api/neuron/memory") {
|
||||
return handle_api_remember(body)
|
||||
}
|
||||
|
||||
@@ -204,7 +204,7 @@ fn safety_log_bell(level: String, reason: String, input_summary: String) -> Stri
|
||||
// Emit a fallback println so the bell event leaves at least a log trace even
|
||||
// when engram is degraded. This does not replace engram persistence -- it is a
|
||||
// last-resort audit trail when the primary write cannot be confirmed.
|
||||
let node_id: String = engram_node_full(
|
||||
let node_id: String = wt_node(
|
||||
content,
|
||||
"BellEvent",
|
||||
"bell:" + level,
|
||||
|
||||
Executable
+937
@@ -0,0 +1,937 @@
|
||||
#!/usr/bin/env python3
|
||||
"""state-key-audit.py — the analyzer behind scripts/verify-state-keys.sh.
|
||||
|
||||
Read that script's header for WHY this exists (issue #129). This file is the
|
||||
HOW: a small El reader that resolves the key expression at every state_get /
|
||||
state_set site, including keys that are computed.
|
||||
|
||||
WHAT IT PARSES
|
||||
El as this engine writes it: `fn f(a: T, b: T) -> T { ... }`, `let x: T = e`,
|
||||
`return e`, `if c { a } else { b }` as an expression, `+` concatenation,
|
||||
`"..."` with backslash escapes, `//` line comments. No block comments, no
|
||||
const/match/struct exist in this dialect (verified over the whole tree).
|
||||
|
||||
KEY PATTERNS — the only two things a key expression can resolve to
|
||||
EXACT "soul_model" the whole key is known
|
||||
PREFIX "session_hist_" a known head, then runtime text
|
||||
(plus UNRESOLVED, which is a report line and never a failure)
|
||||
|
||||
RESOLUTION — resolve_expr() returns a SET of patterns; unions are how branches,
|
||||
multiple returns, and multiple bindings of one name are represented.
|
||||
literal "k" -> {EXACT k}
|
||||
concat A + B -> fold left; all-static -> EXACT,
|
||||
static head + dynamic tail -> PREFIX
|
||||
if-expression if c {A} else {B} -> resolve(A) | resolve(B), except that
|
||||
str_eq(X,"") with X statically ""
|
||||
folds to the taken branch only
|
||||
call f(args) -> union over f's return expressions,
|
||||
with f's params bound to THIS call
|
||||
site's actual argument expressions
|
||||
local var let k = e; state_get(k)-> union over every `let k =` in the
|
||||
enclosing function
|
||||
parameter fn g(k) { state_get(k) }-> union over the argument at that
|
||||
position across every call site of g
|
||||
anything else json_get(...), env(...)-> UNRESOLVED
|
||||
Recursion is depth- and cycle-guarded; a guard trip yields UNRESOLVED, never a
|
||||
failure.
|
||||
|
||||
COVERAGE — a read is satisfied when some write can produce the same key:
|
||||
read EXACT k <- write EXACT k, or write PREFIX p where k starts with p
|
||||
read PREFIX p <- write EXACT k where k starts with p, or write PREFIX q
|
||||
where p and q are prefixes of each other
|
||||
Deliberately permissive at the boundaries: a gate that cries wolf gets deleted.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
MAX_DEPTH = 12
|
||||
|
||||
# ── patterns ────────────────────────────────────────────────────────────────
|
||||
EXACT = "exact"
|
||||
PREFIX = "prefix"
|
||||
|
||||
|
||||
def pat_exact(s):
|
||||
return (EXACT, s)
|
||||
|
||||
|
||||
def pat_prefix(s):
|
||||
# A prefix with no static text at all carries no information; that is the
|
||||
# UNRESOLVED case, not a pattern.
|
||||
return (PREFIX, s) if s else None
|
||||
|
||||
|
||||
def covers(write, read):
|
||||
"""Can a write of pattern `write` produce a key that `read` reads?
|
||||
|
||||
The prefix rule is DIRECTIONAL, and that direction is the whole point. A
|
||||
write namespace that is the same or BROADER than the read namespace covers
|
||||
it (write "rl:" covers read "rl:x"). A write namespace that is NARROWER does
|
||||
NOT (write "session_histv2_" does not cover read "session_hist_") — being
|
||||
permissive there re-opens the exact hole this gate exists to close: rename
|
||||
the producer, leave the readers, stay green. Verified with a control run
|
||||
that renames sessions.el's writer and leaves its four readers behind."""
|
||||
wk, wv = write
|
||||
rk, rv = read
|
||||
if rk == EXACT:
|
||||
return rv == wv if wk == EXACT else rv.startswith(wv)
|
||||
# read is a PREFIX: some key starting with rv is read
|
||||
if wk == EXACT:
|
||||
return wv.startswith(rv) # that one written key is in range
|
||||
return rv.startswith(wv) # write namespace same-or-broader
|
||||
|
||||
|
||||
# ── lexer ───────────────────────────────────────────────────────────────────
|
||||
TOK_STR, TOK_IDENT, TOK_PUNCT, TOK_NUM = "str", "ident", "punct", "num"
|
||||
IDENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
|
||||
NUM_RE = re.compile(r"[0-9]+(\.[0-9]+)?")
|
||||
|
||||
|
||||
class Tok:
|
||||
__slots__ = ("kind", "val", "line")
|
||||
|
||||
def __init__(self, kind, val, line):
|
||||
self.kind, self.val, self.line = kind, val, line
|
||||
|
||||
def __repr__(self):
|
||||
return "%s(%r)@%d" % (self.kind, self.val, self.line)
|
||||
|
||||
|
||||
def lex(src):
|
||||
toks, i, n, line = [], 0, len(src), 1
|
||||
while i < n:
|
||||
c = src[i]
|
||||
if c == "\n":
|
||||
line += 1
|
||||
i += 1
|
||||
continue
|
||||
if c in " \t\r":
|
||||
i += 1
|
||||
continue
|
||||
if c == "/" and i + 1 < n and src[i + 1] == "/":
|
||||
while i < n and src[i] != "\n":
|
||||
i += 1
|
||||
continue
|
||||
if c == '"':
|
||||
j, buf = i + 1, []
|
||||
while j < n:
|
||||
if src[j] == "\\" and j + 1 < n:
|
||||
esc = src[j + 1]
|
||||
buf.append({"n": "\n", "t": "\t", "r": "\r"}.get(esc, esc))
|
||||
j += 2
|
||||
continue
|
||||
if src[j] == '"':
|
||||
break
|
||||
if src[j] == "\n":
|
||||
line += 1
|
||||
buf.append(src[j])
|
||||
j += 1
|
||||
toks.append(Tok(TOK_STR, "".join(buf), line))
|
||||
i = j + 1
|
||||
continue
|
||||
m = IDENT_RE.match(src, i)
|
||||
if m:
|
||||
toks.append(Tok(TOK_IDENT, m.group(0), line))
|
||||
i = m.end()
|
||||
continue
|
||||
m = NUM_RE.match(src, i)
|
||||
if m:
|
||||
toks.append(Tok(TOK_NUM, m.group(0), line))
|
||||
i = m.end()
|
||||
continue
|
||||
toks.append(Tok(TOK_PUNCT, c, line))
|
||||
i += 1
|
||||
return toks
|
||||
|
||||
|
||||
def match_close(toks, i, open_ch, close_ch):
|
||||
"""toks[i] is open_ch; return index of its matching close_ch."""
|
||||
depth = 0
|
||||
while i < len(toks):
|
||||
if toks[i].kind == TOK_PUNCT:
|
||||
if toks[i].val == open_ch:
|
||||
depth += 1
|
||||
elif toks[i].val == close_ch:
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return i
|
||||
i += 1
|
||||
return len(toks) - 1
|
||||
|
||||
|
||||
# ── program model ───────────────────────────────────────────────────────────
|
||||
class Func:
|
||||
def __init__(self, name, path, line, params, toks, start, end):
|
||||
self.name, self.path, self.line = name, path, line
|
||||
self.params = params # [param name]
|
||||
self.toks = toks # the whole file's token list
|
||||
self.start, self.end = start, end # body token range, exclusive of braces
|
||||
self.lets = None # name -> [expr token ranges], lazily built
|
||||
|
||||
|
||||
class Site:
|
||||
def __init__(self, kind, path, line, func, arg_range, text):
|
||||
self.kind = kind # "get" | "set"
|
||||
self.path, self.line = path, line
|
||||
self.func = func
|
||||
self.arg_range = arg_range
|
||||
self.text = text # source text of the key expression
|
||||
self.pats = set()
|
||||
self.unresolved = False
|
||||
self.literal = None # set when the key expression is a bare literal
|
||||
|
||||
|
||||
class Program:
|
||||
def __init__(self):
|
||||
self.files = {} # path -> toks
|
||||
self.funcs = {} # name -> [Func] (El allows no overloads, but be safe)
|
||||
self.toplevel = [] # [Func] one per file, params=[]
|
||||
self.sites = [] # [Site]
|
||||
self.calls = {} # callee name -> [(Func caller, [arg ranges])]
|
||||
|
||||
# -- loading ------------------------------------------------------------
|
||||
def load(self, path, rel):
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
||||
src = fh.read()
|
||||
toks = lex(src)
|
||||
self.files[rel] = toks
|
||||
self._scan_funcs(rel, toks)
|
||||
|
||||
def _scan_funcs(self, rel, toks):
|
||||
covered = []
|
||||
i = 0
|
||||
while i < len(toks):
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "fn" and i + 2 < len(toks) \
|
||||
and toks[i + 1].kind == TOK_IDENT and toks[i + 2].val == "(":
|
||||
name = toks[i + 1].val
|
||||
pclose = match_close(toks, i + 2, "(", ")")
|
||||
params = self._params(toks, i + 3, pclose)
|
||||
bopen = pclose + 1
|
||||
while bopen < len(toks) and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
f = Func(name, rel, t.line, params, toks, bopen + 1, bclose)
|
||||
self.funcs.setdefault(name, []).append(f)
|
||||
covered.append((i, bclose))
|
||||
i = bclose + 1
|
||||
continue
|
||||
i += 1
|
||||
# everything outside a fn is the file's top-level "function"
|
||||
tl = Func("<toplevel:%s>" % rel, rel, 1, [], toks, 0, len(toks))
|
||||
tl.covered = covered
|
||||
self.toplevel.append(tl)
|
||||
|
||||
@staticmethod
|
||||
def _params(toks, i, end):
|
||||
"""`a: T, b: T` -> ['a','b'] (top-level commas only)."""
|
||||
names, depth, expect = [], 0, True
|
||||
while i < end:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
depth += 1
|
||||
elif t.kind == TOK_PUNCT and t.val in ")]}":
|
||||
depth -= 1
|
||||
elif depth == 0 and t.kind == TOK_PUNCT and t.val == ",":
|
||||
expect = True
|
||||
elif depth == 0 and expect and t.kind == TOK_IDENT:
|
||||
names.append(t.val)
|
||||
expect = False
|
||||
i += 1
|
||||
return names
|
||||
|
||||
def func_at(self, rel, tok_index):
|
||||
for f in self.funcs_in(rel):
|
||||
if f.start <= tok_index < f.end:
|
||||
return f
|
||||
for f in self.toplevel:
|
||||
if f.path == rel:
|
||||
return f
|
||||
return None
|
||||
|
||||
def funcs_in(self, rel):
|
||||
for fl in self.funcs.values():
|
||||
for f in fl:
|
||||
if f.path == rel:
|
||||
yield f
|
||||
|
||||
# -- indexing -----------------------------------------------------------
|
||||
def index(self):
|
||||
for rel, toks in self.files.items():
|
||||
i = 0
|
||||
while i < len(toks):
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and i + 1 < len(toks) and toks[i + 1].val == "(" \
|
||||
and t.val not in KEYWORDS \
|
||||
and not (i > 0 and toks[i - 1].kind == TOK_IDENT
|
||||
and toks[i - 1].val == "fn"):
|
||||
# ^ the `fn f(a: T)` declaration is not a call site; counting
|
||||
# it as one makes every parameter resolve to its own name
|
||||
# and reports the whole function UNRESOLVED.
|
||||
close = match_close(toks, i + 1, "(", ")")
|
||||
args = split_args(toks, i + 2, close)
|
||||
self.calls.setdefault(t.val, []).append(
|
||||
(self.func_at(rel, i), args, rel, t.line))
|
||||
if t.val in ("state_get", "state_set") and args:
|
||||
self.sites.append(Site(
|
||||
"get" if t.val == "state_get" else "set",
|
||||
rel, t.line, self.func_at(rel, i), args[0],
|
||||
render(toks, *args[0])))
|
||||
i += 1
|
||||
|
||||
# -- resolution ---------------------------------------------------------
|
||||
def lets_of(self, f):
|
||||
if f.lets is not None:
|
||||
return f.lets
|
||||
f.lets = {}
|
||||
toks = f.toks
|
||||
skip = getattr(f, "covered", [])
|
||||
i = f.start
|
||||
while i < f.end:
|
||||
if any(a <= i <= b for a, b in skip):
|
||||
i = max(b for a, b in skip if a <= i <= b) + 1
|
||||
continue
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "let" and i + 1 < f.end \
|
||||
and toks[i + 1].kind == TOK_IDENT:
|
||||
name = toks[i + 1].val
|
||||
j = i + 2
|
||||
if j < f.end and toks[j].val == ":": # skip the type
|
||||
while j < f.end and toks[j].val != "=":
|
||||
j += 1
|
||||
if j < f.end and toks[j].val == "=":
|
||||
s = j + 1
|
||||
e = stmt_end(toks, s, f.end)
|
||||
f.lets.setdefault(name, []).append((s, e))
|
||||
i = e
|
||||
continue
|
||||
i += 1
|
||||
return f.lets
|
||||
|
||||
def returns_of(self, ctx, depth=0, seen=None):
|
||||
"""The value expressions of a function, in the context it was CALLED in.
|
||||
|
||||
Context-sensitive on purpose. `conv_hist_key` is written as a guard:
|
||||
|
||||
if str_eq(session_id, "") { return "conv_history" }
|
||||
return "session_hist_" + session_id
|
||||
|
||||
Collecting both returns flat would make state_set(conv_hist_key("")) — the
|
||||
dead handle_chat() write — claim to produce the session_hist_ namespace
|
||||
too. That is a producer this engine does not actually have, and claiming
|
||||
it would let the gate stay green if sessions.el's real writer vanished:
|
||||
a masking hole in the exact namespace #129 lives in. So a guard whose
|
||||
condition folds is honoured, and the branch not taken is dropped."""
|
||||
out = []
|
||||
self._values(ctx.toks, ctx.start, ctx.end, ctx, depth,
|
||||
seen if seen is not None else set(), out)
|
||||
return out
|
||||
|
||||
def _values(self, toks, s, e, ctx, depth, seen, out):
|
||||
"""Append the value expressions of a statement sequence.
|
||||
Returns True when the sequence definitely returns (rest unreachable)."""
|
||||
if depth > MAX_DEPTH:
|
||||
return False
|
||||
i = s
|
||||
while i < e:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "return":
|
||||
j = stmt_end(toks, i + 1, e)
|
||||
if j > i + 1:
|
||||
out.append((i + 1, j))
|
||||
return True
|
||||
if t.kind == TOK_IDENT and t.val == "let":
|
||||
i = stmt_end(toks, i + 2, e)
|
||||
continue
|
||||
if t.kind == TOK_IDENT and t.val == "if":
|
||||
i = self._if_stmt(toks, i, e, ctx, depth, seen, out)
|
||||
if i is True:
|
||||
return True
|
||||
continue
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
i = match_close(toks, i, t.val,
|
||||
{"(": ")", "[": "]", "{": "}"}[t.val]) + 1
|
||||
continue
|
||||
en = stmt_end(toks, i, e)
|
||||
if en <= i:
|
||||
i += 1
|
||||
continue
|
||||
if en >= e: # trailing expression = the value
|
||||
out.append((i, en))
|
||||
i = en
|
||||
return False
|
||||
|
||||
def _if_stmt(self, toks, i, e, ctx, depth, seen, out):
|
||||
"""Walk one if / else-if / else chain. Returns the next index, or True
|
||||
if the chain definitely returns on every reachable branch."""
|
||||
bopen = i + 1
|
||||
while bopen < e and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
if bopen >= e:
|
||||
return e
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
fold = self._fold_cond(toks, i + 1, bopen, ctx, depth, seen)
|
||||
|
||||
j = bclose + 1
|
||||
else_s = else_e = None
|
||||
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
|
||||
if j + 1 < e and toks[j + 1].val == "{":
|
||||
ec = match_close(toks, j + 1, "{", "}")
|
||||
else_s, else_e = j + 2, ec
|
||||
j = ec + 1
|
||||
else: # `else if ...` — the rest of the chain
|
||||
else_s = j + 1
|
||||
else_e = stmt_end(toks, j + 1, e)
|
||||
j = else_e
|
||||
|
||||
then_ret = else_ret = False
|
||||
if fold is not False:
|
||||
then_ret = self._values(toks, bopen + 1, bclose, ctx, depth + 1, seen, out)
|
||||
if fold is not True and else_s is not None:
|
||||
else_ret = self._values(toks, else_s, else_e, ctx, depth + 1, seen, out)
|
||||
|
||||
if fold is True and then_ret:
|
||||
return True
|
||||
if fold is False and else_s is not None and else_ret:
|
||||
return True
|
||||
if fold is None and else_s is not None and then_ret and else_ret:
|
||||
return True
|
||||
return j
|
||||
|
||||
def resolve(self, rng, func, depth=0, seen=None):
|
||||
"""-> (set of patterns, unresolved_flag)"""
|
||||
if seen is None:
|
||||
seen = set()
|
||||
if depth > MAX_DEPTH:
|
||||
return set(), True
|
||||
return self._expr(func.toks, rng[0], rng[1], func, depth, seen)
|
||||
|
||||
# -- expression walker --------------------------------------------------
|
||||
def _expr(self, toks, s, e, func, depth, seen):
|
||||
parts, cur, d = [], s, 0
|
||||
i = s
|
||||
while i < e: # split on top-level '+'
|
||||
v = toks[i].val
|
||||
if toks[i].kind == TOK_PUNCT and v in "([{":
|
||||
d += 1
|
||||
elif toks[i].kind == TOK_PUNCT and v in ")]}":
|
||||
d -= 1
|
||||
elif d == 0 and toks[i].kind == TOK_PUNCT and v == "+" and i > s:
|
||||
parts.append((cur, i))
|
||||
cur = i + 1
|
||||
i += 1
|
||||
parts.append((cur, e))
|
||||
if len(parts) == 1:
|
||||
return self._primary(toks, s, e, func, depth, seen)
|
||||
|
||||
# concatenation: keep folding while every operand so far is EXACT
|
||||
head, unres = "", False
|
||||
static = True
|
||||
for (ps, pe) in parts:
|
||||
pats, u = self._primary(toks, ps, pe, func, depth, seen)
|
||||
exacts = {p[1] for p in pats if p[0] == EXACT}
|
||||
if static and len(exacts) == 1 and not u and len(pats) == 1:
|
||||
head += exacts.pop()
|
||||
continue
|
||||
if static and pats and all(p[0] == EXACT for p in pats) and len(pats) > 1:
|
||||
# a branchy static operand: keep the shared head only
|
||||
static = False
|
||||
head += os.path.commonprefix(sorted({p[1] for p in pats}))
|
||||
break
|
||||
static = False
|
||||
# first non-static operand: everything after it is runtime text
|
||||
if (ps, pe) == parts[0]:
|
||||
for p in pats:
|
||||
if p[0] == PREFIX:
|
||||
head = p[1]
|
||||
break
|
||||
if not head:
|
||||
unres = True
|
||||
break
|
||||
if static:
|
||||
return {pat_exact(head)}, False
|
||||
p = pat_prefix(head)
|
||||
return ({p} if p else set()), (unres or not p)
|
||||
|
||||
def _primary(self, toks, s, e, func, depth, seen):
|
||||
while s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "(" \
|
||||
and match_close(toks, s, "(", ")") == e - 1:
|
||||
s, e = s + 1, e - 1
|
||||
if s >= e:
|
||||
return set(), True
|
||||
t = toks[s]
|
||||
|
||||
if t.kind == TOK_STR and e == s + 1:
|
||||
return {pat_exact(t.val)}, False
|
||||
|
||||
if t.kind == TOK_IDENT and t.val == "if":
|
||||
return self._if_expr(toks, s, e, func, depth, seen)
|
||||
|
||||
if t.kind == TOK_IDENT and s + 1 < e and toks[s + 1].val == "(":
|
||||
close = match_close(toks, s + 1, "(", ")")
|
||||
if close == e - 1:
|
||||
return self._call(toks, t.val, split_args(toks, s + 2, close),
|
||||
func, depth, seen)
|
||||
|
||||
if t.kind == TOK_IDENT and e == s + 1:
|
||||
return self._var(t.val, func, depth, seen)
|
||||
|
||||
return set(), True
|
||||
|
||||
def _if_expr(self, toks, s, e, func, depth, seen):
|
||||
bopen = s + 1
|
||||
while bopen < e and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
cond = (s + 1, bopen)
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
then_rng = block_tail(toks, bopen + 1, bclose) or (bopen + 1, bclose)
|
||||
|
||||
else_rng = None
|
||||
j = bclose + 1
|
||||
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
|
||||
if j + 1 < e and toks[j + 1].val == "{":
|
||||
ec = match_close(toks, j + 1, "{", "}")
|
||||
else_rng = block_tail(toks, j + 2, ec) or (j + 2, ec)
|
||||
else:
|
||||
else_rng = (j + 1, e) # `else if ...`
|
||||
|
||||
taken = self._fold_cond(toks, cond[0], cond[1], func, depth, seen)
|
||||
rngs = []
|
||||
if taken is not False:
|
||||
rngs.append(then_rng)
|
||||
if taken is not True and else_rng:
|
||||
rngs.append(else_rng)
|
||||
|
||||
pats, unres = set(), False
|
||||
for r in rngs:
|
||||
p, u = self._expr(toks, r[0], r[1], func, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
def _fold_cond(self, toks, s, e, func, depth, seen):
|
||||
"""Constant-fold `str_eq(X, "")` / `!str_eq(X, "")` so a helper called with
|
||||
a literal (conv_hist_key("")) yields only the branch it really takes.
|
||||
Returns True / False / None(unknown)."""
|
||||
neg = False
|
||||
if s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "!":
|
||||
neg, s = True, s + 1
|
||||
if not (s < e and toks[s].kind == TOK_IDENT and toks[s].val == "str_eq"
|
||||
and s + 1 < e and toks[s + 1].val == "("):
|
||||
return None
|
||||
close = match_close(toks, s + 1, "(", ")")
|
||||
if close != e - 1:
|
||||
return None
|
||||
args = split_args(toks, s + 2, close)
|
||||
if len(args) != 2:
|
||||
return None
|
||||
va, ua = self._expr(toks, args[0][0], args[0][1], func, depth + 1, seen)
|
||||
vb, ub = self._expr(toks, args[1][0], args[1][1], func, depth + 1, seen)
|
||||
if ua or ub or len(va) != 1 or len(vb) != 1:
|
||||
return None
|
||||
(ka, sa), (kb, sb) = va.pop(), vb.pop()
|
||||
if ka != EXACT or kb != EXACT:
|
||||
return None
|
||||
r = (sa == sb)
|
||||
return (not r) if neg else r
|
||||
|
||||
def _call(self, toks, name, args, func, depth, seen):
|
||||
cands = self.funcs.get(name)
|
||||
if not cands:
|
||||
return set(), True # builtin: json_get, env, ...
|
||||
pats, unres = set(), False
|
||||
for callee in cands:
|
||||
key = ("fn", callee.path, callee.name, tuple(args))
|
||||
if key in seen:
|
||||
unres = True
|
||||
continue
|
||||
seen = seen | {key}
|
||||
# bind the callee's params to THIS call site's argument expressions
|
||||
binding = {}
|
||||
for idx, pname in enumerate(callee.params):
|
||||
if idx < len(args):
|
||||
binding[pname] = (args[idx], func)
|
||||
callee_ctx = _Bound(callee, binding)
|
||||
for r in self.returns_of(callee_ctx, depth + 1, seen):
|
||||
p, u = self._expr(callee.toks, r[0], r[1], callee_ctx,
|
||||
depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
def _var(self, name, func, depth, seen):
|
||||
real = func.func if isinstance(func, _Bound) else func
|
||||
|
||||
# 1. a parameter bound by the call site we came through
|
||||
if isinstance(func, _Bound) and name in func.binding:
|
||||
rng, caller_ctx = func.binding[name]
|
||||
return self._expr(caller_ctx.toks, rng[0], rng[1], caller_ctx,
|
||||
depth + 1, seen)
|
||||
|
||||
# 2. a local `let` in the enclosing function
|
||||
lets = self.lets_of(real)
|
||||
if name in lets:
|
||||
key = ("let", real.path, real.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen = seen | {key}
|
||||
pats, unres = set(), False
|
||||
for rng in lets[name]:
|
||||
p, u = self._expr(real.toks, rng[0], rng[1], real, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
# 3. an unbound parameter -> look at every call site of the enclosing fn
|
||||
if name in real.params:
|
||||
key = ("param", real.path, real.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen = seen | {key}
|
||||
idx = real.params.index(name)
|
||||
pats, unres = set(), False
|
||||
sites = self.calls.get(real.name, [])
|
||||
if not sites:
|
||||
return set(), True
|
||||
for caller, args, _rel, _line in sites:
|
||||
if caller is None or idx >= len(args):
|
||||
unres = True
|
||||
continue
|
||||
p, u = self._expr(caller.toks, args[idx][0], args[idx][1],
|
||||
caller, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
# 4. a file-level / cross-file top-level `let`
|
||||
for tl in self.toplevel:
|
||||
lets = self.lets_of(tl)
|
||||
if name in lets:
|
||||
key = ("let", tl.path, tl.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen2 = seen | {key}
|
||||
pats, unres = set(), False
|
||||
for rng in lets[name]:
|
||||
p, u = self._expr(tl.toks, rng[0], rng[1], tl, depth + 1, seen2)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
return set(), True
|
||||
|
||||
|
||||
class _Bound:
|
||||
"""A callee view that also knows what its params were called with."""
|
||||
|
||||
def __init__(self, func, binding):
|
||||
self.func, self.binding = func, binding
|
||||
self.toks, self.start, self.end = func.toks, func.start, func.end
|
||||
self.params, self.path, self.name = func.params, func.path, func.name
|
||||
|
||||
def __getattr__(self, k):
|
||||
return getattr(self.func, k)
|
||||
|
||||
|
||||
# ── token helpers ───────────────────────────────────────────────────────────
|
||||
def split_args(toks, s, e):
|
||||
out, cur, d = [], s, 0
|
||||
i = s
|
||||
while i < e:
|
||||
v = toks[i].val
|
||||
if toks[i].kind == TOK_PUNCT and v in "([{":
|
||||
d += 1
|
||||
elif toks[i].kind == TOK_PUNCT and v in ")]}":
|
||||
d -= 1
|
||||
elif d == 0 and toks[i].kind == TOK_PUNCT and v == ",":
|
||||
out.append((cur, i))
|
||||
cur = i + 1
|
||||
i += 1
|
||||
if cur < e:
|
||||
out.append((cur, e))
|
||||
return out
|
||||
|
||||
|
||||
STMT_START = {"let", "return", "if", "while", "for"}
|
||||
KEYWORDS = {"if", "while", "for", "return", "fn", "let", "else", "match"}
|
||||
|
||||
|
||||
def stmt_end(toks, s, limit):
|
||||
"""End of the expression starting at s: the next top-level statement
|
||||
boundary. El has no semicolons, so a newline that starts a new statement
|
||||
ends this one."""
|
||||
d, i = 0, s
|
||||
while i < limit:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_PUNCT and t.val in "([":
|
||||
d += 1
|
||||
elif t.kind == TOK_PUNCT and t.val in ")]":
|
||||
d -= 1
|
||||
if d < 0:
|
||||
return i
|
||||
elif t.kind == TOK_PUNCT and t.val == "{":
|
||||
# a brace at depth 0 belongs to this expression only when it is an
|
||||
# if/else block that is part of it
|
||||
d += 1
|
||||
elif t.kind == TOK_PUNCT and t.val == "}":
|
||||
d -= 1
|
||||
if d < 0:
|
||||
return i
|
||||
elif d == 0 and t.kind == TOK_PUNCT and t.val == ",":
|
||||
return i
|
||||
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val in STMT_START:
|
||||
if t.val == "if" and toks[i - 1].kind == TOK_IDENT and toks[i - 1].val == "else":
|
||||
i += 1
|
||||
continue
|
||||
return i
|
||||
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val == "fn":
|
||||
return i
|
||||
i += 1
|
||||
return limit
|
||||
|
||||
|
||||
def block_tail(toks, s, e):
|
||||
"""The trailing expression of a block, if the block ends in one."""
|
||||
i, last = s, None
|
||||
while i < e:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val in ("let", "return"):
|
||||
i = stmt_end(toks, i + 1, e)
|
||||
last = None
|
||||
continue
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
i = match_close(toks, i, t.val, {"(": ")", "[": "]", "{": "}"}[t.val]) + 1
|
||||
continue
|
||||
st = i
|
||||
en = stmt_end(toks, i, e)
|
||||
if en <= st:
|
||||
i = st + 1
|
||||
continue
|
||||
last = (st, en)
|
||||
i = en
|
||||
return last
|
||||
|
||||
|
||||
def render(toks, s, e):
|
||||
out = []
|
||||
for t in toks[s:e]:
|
||||
out.append('"%s"' % t.val if t.kind == TOK_STR else t.val)
|
||||
return " ".join(out)
|
||||
|
||||
|
||||
# ── the gate ────────────────────────────────────────────────────────────────
|
||||
def collect(root, include_tests):
|
||||
files = []
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
dirnames[:] = [d for d in dirnames
|
||||
if d not in ("dist", "vendor", ".git", "node_modules")]
|
||||
rel_dir = os.path.relpath(dirpath, root)
|
||||
if not include_tests and rel_dir.split(os.sep)[0] == "tests":
|
||||
continue
|
||||
for fn in sorted(filenames):
|
||||
if fn.endswith(".el"):
|
||||
rel = os.path.normpath(os.path.join(rel_dir, fn))
|
||||
files.append((os.path.join(dirpath, fn), rel))
|
||||
return sorted(files, key=lambda x: x[1])
|
||||
|
||||
|
||||
def is_bare_literal(prog, site):
|
||||
toks = prog.files[site.path]
|
||||
s, e = site.arg_range
|
||||
return e == s + 1 and toks[s].kind == TOK_STR
|
||||
|
||||
|
||||
def read_decl(path):
|
||||
"""A declaration file: one entry per line, `# ...` comments stripped."""
|
||||
out = []
|
||||
if not path or not os.path.exists(path):
|
||||
return out
|
||||
with open(path) as fh:
|
||||
for ln in fh:
|
||||
ln = ln.split("#", 1)[0].strip()
|
||||
if ln:
|
||||
out.append(ln)
|
||||
return out
|
||||
|
||||
|
||||
def opt(argv, name, default=None):
|
||||
for i, a in enumerate(argv):
|
||||
if a == name and i + 1 < len(argv):
|
||||
return argv[i + 1]
|
||||
return default
|
||||
|
||||
|
||||
def main(argv):
|
||||
root = os.path.abspath(argv[1]) if len(argv) > 1 and not argv[1].startswith("-") else "."
|
||||
include_tests = "--include-tests" in argv
|
||||
verbose = "--verbose" in argv
|
||||
baseline_path = opt(argv, "--baseline")
|
||||
external_path = opt(argv, "--external")
|
||||
|
||||
prog = Program()
|
||||
for path, rel in collect(root, include_tests):
|
||||
prog.load(path, rel)
|
||||
prog.index()
|
||||
for site in prog.sites:
|
||||
pats, unres = prog.resolve(site.arg_range, site.func)
|
||||
site.pats, site.unresolved = {p for p in pats if p}, unres
|
||||
if is_bare_literal(prog, site):
|
||||
site.literal = prog.files[site.path][site.arg_range[0]].val
|
||||
|
||||
writes = [s for s in prog.sites if s.kind == "set"]
|
||||
reads = [s for s in prog.sites if s.kind == "get"]
|
||||
write_pats = set()
|
||||
for w in writes:
|
||||
write_pats |= w.pats
|
||||
|
||||
# Declared host-set keys: written by something outside the El tree (an
|
||||
# operator, the installer, a host process). Each entry must carry a reason.
|
||||
external = []
|
||||
for ln in read_decl(external_path):
|
||||
parts = ln.split(None, 1)
|
||||
if len(parts) != 2 or parts[0] not in (EXACT, PREFIX):
|
||||
print("bad --external line (want `exact|prefix <key>`): %r" % ln,
|
||||
file=sys.stderr)
|
||||
return 2
|
||||
external.append((parts[0], parts[1]))
|
||||
write_pats |= set(external)
|
||||
|
||||
# F1 — a read of a key no write in the tree produces.
|
||||
f1 = []
|
||||
for r in reads:
|
||||
for p in sorted(r.pats):
|
||||
if not any(covers(w, p) for w in write_pats):
|
||||
f1.append((r, p))
|
||||
|
||||
# F2 — a key namespace owned by a helper, accessed by a hand-rolled literal.
|
||||
# This is the #129 shape: the producer moved behind conv_hist_key() and
|
||||
# one consumer kept spelling the old key out by hand.
|
||||
owners = {} # helper fn name -> its value set
|
||||
for s in prog.sites:
|
||||
toks = prog.files[s.path]
|
||||
a, b = s.arg_range
|
||||
if toks[a].kind == TOK_IDENT and a + 1 < b and toks[a + 1].val == "(" \
|
||||
and match_close(toks, a + 1, "(", ")") == b - 1 \
|
||||
and toks[a].val in prog.funcs:
|
||||
name = toks[a].val
|
||||
if name not in owners:
|
||||
vals = set()
|
||||
for callee in prog.funcs[name]:
|
||||
# No call context here on purpose: the OWNED namespace is
|
||||
# every key the helper can ever produce, over all call sites.
|
||||
for rng in prog.returns_of(callee):
|
||||
p, _ = prog._expr(callee.toks, rng[0], rng[1], callee, 0, set())
|
||||
vals |= {x for x in p if x}
|
||||
owners[name] = vals
|
||||
f2 = []
|
||||
for s in prog.sites:
|
||||
if s.literal is None:
|
||||
continue
|
||||
for owner, vals in sorted(owners.items()):
|
||||
for v in sorted(vals):
|
||||
if covers(v, pat_exact(s.literal)):
|
||||
f2.append((s, owner, v))
|
||||
break
|
||||
else:
|
||||
continue
|
||||
break
|
||||
|
||||
unresolved = [s for s in prog.sites if s.unresolved or not s.pats]
|
||||
|
||||
# Baseline signatures carry NO line number on purpose: an unrelated edit that
|
||||
# shifts a line must not un-mute an accepted finding (that is crying wolf),
|
||||
# but a GROWTH in count must not hide either. So a baseline entry is
|
||||
# `<file> <CODE> <detail> [xN]` and only the first N matches are muted.
|
||||
baseline, bad_baseline = {}, []
|
||||
for ln in read_decl(baseline_path):
|
||||
n, key = 1, ln
|
||||
parts = ln.rsplit(" x", 1)
|
||||
if len(parts) == 2 and parts[1].isdigit():
|
||||
key, n = parts[0].strip(), int(parts[1])
|
||||
baseline[key] = n
|
||||
|
||||
def sig(path, code, detail):
|
||||
return "%s %s %s" % (path, code, detail)
|
||||
|
||||
findings = []
|
||||
for r, p in f1:
|
||||
findings.append((sig(r.path, "DEAD-READ", "%s:%s" % p), r.line,
|
||||
" %s:%d state_get(%s)\n resolves to %s %r — no state_set in the tree produces it"
|
||||
% (r.path, r.line, r.text, p[0].upper(), p[1])))
|
||||
for s, owner, v in f2:
|
||||
findings.append((sig(s.path, "HAND-ROLLED", "%s<-%s()" % (s.literal, owner)), s.line,
|
||||
" %s:%d state_%s(\"%s\")\n %s() owns this key namespace (%s %r) — go through the helper, "
|
||||
"or a rename orphans this site silently" % (s.path, s.line, s.kind, s.literal, owner, v[0].upper(), v[1])))
|
||||
findings.sort(key=lambda f: (f[0], f[1]))
|
||||
|
||||
live, muted, budget = [], [], dict(baseline)
|
||||
for f in findings:
|
||||
if budget.get(f[0], 0) > 0:
|
||||
budget[f[0]] -= 1
|
||||
muted.append(f)
|
||||
else:
|
||||
live.append(f)
|
||||
stale = sorted(k for k, v in budget.items() if v > 0)
|
||||
|
||||
print("── state-key audit ─────────────────────────────────────────────")
|
||||
print("scanned %d .el files%s" % (len(prog.files),
|
||||
"" if include_tests else " (tests/ excluded)"))
|
||||
print("sites %d state_set, %d state_get" % (len(writes), len(reads)))
|
||||
print("keys %d distinct write patterns" % len(write_pats))
|
||||
print("")
|
||||
|
||||
if verbose:
|
||||
print("WRITE PATTERNS")
|
||||
for k, v in sorted(write_pats):
|
||||
print(" %-6s %s" % (k, v))
|
||||
print("")
|
||||
|
||||
if external:
|
||||
print("DECLARED HOST-SET (%d) — %s" % (len(external), external_path))
|
||||
for k, v in sorted(external):
|
||||
print(" %-6s %s" % (k, v))
|
||||
print("")
|
||||
|
||||
print("UNRESOLVED (%d) — reported, never fails the build" % len(unresolved))
|
||||
if not unresolved:
|
||||
print(" (none)")
|
||||
for s in sorted(unresolved, key=lambda x: (x.path, x.line)):
|
||||
print(" %s:%d state_%s(%s)%s"
|
||||
% (s.path, s.line, s.kind, s.text,
|
||||
" [partial: %s]" % ", ".join("%s %r" % p for p in sorted(s.pats))
|
||||
if s.pats else ""))
|
||||
print("")
|
||||
|
||||
if muted:
|
||||
print("BASELINED (%d) — pre-existing debt accepted in %s. NOT clean; fix these."
|
||||
% (len(muted), baseline_path))
|
||||
for sg, line, _ in muted:
|
||||
print(" %s (line %d)" % (sg, line))
|
||||
print("")
|
||||
if stale:
|
||||
print("STALE BASELINE (%d) — entries that no longer match anything; delete them:"
|
||||
% len(stale))
|
||||
for sg in stale:
|
||||
print(" %s" % sg)
|
||||
print("")
|
||||
|
||||
print("FINDINGS (%d)" % len(live))
|
||||
if not live:
|
||||
print(" (none)")
|
||||
for _, _, body in live:
|
||||
print(body)
|
||||
print("")
|
||||
|
||||
if live:
|
||||
print("FAIL: %d state-key finding(s). See scripts/verify-state-keys.sh "
|
||||
"for why this gate exists (issue #129)." % len(live))
|
||||
return 1
|
||||
print("PASS: every resolvable state_get key has a producer, and no key "
|
||||
"namespace is spelled two ways.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -0,0 +1,28 @@
|
||||
# state-key-baseline.txt — findings that already existed when this gate landed
|
||||
# (2026-08-07). Each one is a REAL defect of the #129 class, not a false
|
||||
# positive. They are muted only so the gate can be turned on today instead of
|
||||
# being deferred until the debt is paid; every run still prints them under
|
||||
# BASELINED with the word "debt".
|
||||
#
|
||||
# THIS FILE SHOULD ONLY EVER SHRINK. Adding a line means you are shipping a
|
||||
# known silent-"" read. If you must, date it and say why in the comment.
|
||||
#
|
||||
# format: <file> <CODE> <detail> [xN] # N = how many sites are accepted
|
||||
# No line numbers on purpose: an unrelated edit must not un-mute an accepted
|
||||
# finding, but a GROWTH in count is NOT muted — the extra site fails the build.
|
||||
#
|
||||
chat.el DEAD-READ exact:soul_identity x5
|
||||
# ^ soul.el used to run `state_set("soul_identity", soul_identity)`. It was
|
||||
# deleted on 2026-05-13 in b163fa6 ("feat(awareness): route ISE writes to HTTP
|
||||
# Engram ..."), a commit about something else entirely, and the five readers in
|
||||
# chat.el were left behind. Since that date build_system_prompt (737), the
|
||||
# vision handler (1745), the agentic system prompt (2620), the council
|
||||
# transcript handler (3425) and 3480 have all been prefixing "" — exactly the
|
||||
# #129 shape, found by this gate on its first run. Sites: 737, 1745, 2620,
|
||||
# 3425, 3480. Fix = restore the boot-time write or delete the reads; not done
|
||||
# here because this branch must not change engine behaviour.
|
||||
|
||||
studio.el DEAD-READ exact:soul_principal x1
|
||||
# ^ studio.el:57 dharma_registry() emits "principal":"" on every call — no
|
||||
# producer has ever existed in the tree's history (git log -S finds none).
|
||||
# Never-wired rather than orphaned, same silent-"" result.
|
||||
@@ -0,0 +1,16 @@
|
||||
# state-key-external.txt — state keys the engine READS but deliberately never
|
||||
# WRITES, because a host outside the El tree sets them (an operator, the
|
||||
# installer, a deployment env). Read scripts/verify-state-keys.sh for why this
|
||||
# list has to exist and why it has to stay short.
|
||||
#
|
||||
# THE RULE FOR ADDING A LINE: the read site must already treat "" as a defined
|
||||
# default (`if str_eq(x, "") { <default> }`) AND the source must say so in a
|
||||
# comment. "I could not find the writer" is NOT a reason — that is the #129
|
||||
# defect, and it belongs in state-key-baseline.txt with a date, not here.
|
||||
#
|
||||
# format: exact|prefix <key> # why, and where the source says so
|
||||
#
|
||||
exact soul_rate_limit # routes.el:59-61 — "configurable via soul state key ... Falls back to 60 req/min if not set."
|
||||
exact web_search_tool_version # chat.el:1884-1910 — version lives in state "so a future bump is a config write, not a recompile"; defaults to web_search_20250305
|
||||
exact platform_auth # stewardship.el:92 — host-set capability flag; fail-CLOSED (anything but "true" denies the platform tool)
|
||||
exact security_research_authorized # awareness.el:991-996 — state override for env SECURITY_RESEARCH_TOKEN; fail-closed, defaults false
|
||||
Executable
+118
@@ -0,0 +1,118 @@
|
||||
#!/usr/bin/env bash
|
||||
# verify-state-keys.sh — the state-key gate. Retires a defect class at build time.
|
||||
#
|
||||
# ── WHY THIS EXISTS. DO NOT DELETE IT AS NOISE. ──────────────────────────────
|
||||
#
|
||||
# The engine keeps runtime values in a key-value store: state_set("k", v) writes,
|
||||
# state_get("k") reads. A read of a key that NOTHING writes returns an empty
|
||||
# string. Silently. No error, no warning, no log line. The El compiler cannot see
|
||||
# it, no test sees it, and the product keeps running — just with a hole in it.
|
||||
#
|
||||
# That is how issue #129 happened. ff421d3 (2026-08-05) correctly moved
|
||||
# conversation history to a per-session key behind conv_hist_key(session_id). One
|
||||
# consumer did not move with it: the agentic path's L1 safety screen kept reading
|
||||
# the old anonymous "conv_history" bucket. The desktop app always mints a session
|
||||
# id, so history was always written under session_hist_<id> and that read always
|
||||
# returned "". The half of the crisis score that receives history is the
|
||||
# ESCALATION half — the one that exists for distress building across several
|
||||
# turns, where no single message trips the bell on its own. It scored 0 on every
|
||||
# real conversation for two days, and nothing failed.
|
||||
#
|
||||
# The line that broke carried a comment describing this exact bug being fixed
|
||||
# once already, under issue #9. A comment is not a gate. This is the gate.
|
||||
#
|
||||
# ── WHAT IT CHECKS ──────────────────────────────────────────────────────────
|
||||
#
|
||||
# DEAD-READ a state_get whose key resolves to something no state_set in the
|
||||
# tree produces. The direct form of the class.
|
||||
#
|
||||
# HAND-ROLLED a state_get/state_set that spells out a literal belonging to a
|
||||
# key namespace a helper function owns (e.g. "conv_history", owned
|
||||
# by conv_hist_key()). This is #129's actual shape: the producer
|
||||
# moved behind the helper and one consumer kept the old spelling
|
||||
# by hand. DEAD-READ alone does NOT catch #129, because the dead
|
||||
# handle_chat() still writes that key through the helper — so this
|
||||
# second check is the one that earns the gate its keep.
|
||||
#
|
||||
# ── WHY IT DOES NOT CRY WOLF ────────────────────────────────────────────────
|
||||
#
|
||||
# Keys are usually COMPUTED, not literal, so a naive grep would flood and get
|
||||
# switched off within a day. scripts/state-key-audit.py resolves computed keys:
|
||||
# string concatenation (matched on the static prefix), helper functions (resolved
|
||||
# to their possible return values), keys built into a local variable, and keys
|
||||
# arriving as a function parameter (resolved through the call sites). Where a key
|
||||
# genuinely cannot be resolved it is printed under UNRESOLVED and does NOT fail
|
||||
# the build — visible, never silently ignored. Keep that list short.
|
||||
#
|
||||
# On this tree it resolves 278 of 278 sites: UNRESOLVED is 0 and FINDINGS is 0.
|
||||
#
|
||||
# Two declaration files, both of which should only ever shrink:
|
||||
# scripts/state-key-external.txt keys a host outside the El tree writes
|
||||
# scripts/state-key-baseline.txt findings that predate the gate (real debt)
|
||||
#
|
||||
# ── PROVEN TO DISCRIMINATE (2026-08-07) ─────────────────────────────────────
|
||||
#
|
||||
# 1. Synthetic: a scratch copy of this tree with agentic_safety_screen reverted
|
||||
# to the pre-fix state_get("conv_history") — ONE line, nothing else — FAILS
|
||||
# with `chat.el:2536 ... conv_hist_key() owns this key namespace`. The tree
|
||||
# as shipped PASSES. One variable, opposite verdicts.
|
||||
# 2. Independent: run read-only against origin/feat/soul-openai-tools-v2, which
|
||||
# carries the same defect on its own, the gate reported chat.el:2937 — the
|
||||
# exact line 43d0449's commit message had named by hand. Against that
|
||||
# branch's fix (origin/fix/129-on-openai-tools) it passes.
|
||||
# 3. Producer-moved controls: renaming the sole writer of an EXACT key
|
||||
# (soul_model) orphans 3 readers across 3 files; renaming the sole writer of
|
||||
# a PREFIX namespace (agent_workspace_root_*) orphans 3 readers — including
|
||||
# when the producer moves to a NARROWER namespace, which an earlier,
|
||||
# sloppier prefix rule let through.
|
||||
#
|
||||
# It also found, on its first run, a defect nobody was looking for: soul.el's
|
||||
# `state_set("soul_identity", ...)` was deleted on 2026-05-13 in b163fa6 (a
|
||||
# commit about awareness/ISE writes) and five readers in chat.el were left
|
||||
# behind — the system prompt, the vision handler, the agentic prompt and the
|
||||
# council handler have been prefixing "" ever since. See state-key-baseline.txt.
|
||||
#
|
||||
# ── SAFETY ──────────────────────────────────────────────────────────────────
|
||||
# Pure static read of .el sources. Starts nothing, opens no port, touches no
|
||||
# daemon, and never reads or writes ~/.neuron.
|
||||
#
|
||||
# ── USAGE ───────────────────────────────────────────────────────────────────
|
||||
# scripts/verify-state-keys.sh gate the repo (honours baseline)
|
||||
# scripts/verify-state-keys.sh --strict ignore the baseline: show the debt
|
||||
# scripts/verify-state-keys.sh --verbose also dump every write pattern
|
||||
# scripts/verify-state-keys.sh --root DIR audit a different tree
|
||||
# exit 0 = clean; 1 = finding(s); 2 = the gate itself could not run.
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
STRICT=0
|
||||
PASS_THROUGH=()
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--strict) STRICT=1; shift ;;
|
||||
--root) ROOT="${2:?--root needs a directory}"; shift 2 ;;
|
||||
-h|--help) awk 'NR>1 && /^#/ {print; next} NR>1 {exit}' "${BASH_SOURCE[0]}"; exit 0 ;;
|
||||
*) PASS_THROUGH+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
command -v python3 >/dev/null 2>&1 || {
|
||||
echo "[state-keys] CANNOT RUN: python3 not found" >&2; exit 2; }
|
||||
[ -d "$ROOT" ] || { echo "[state-keys] CANNOT RUN: no such tree: $ROOT" >&2; exit 2; }
|
||||
|
||||
AUDIT="$SCRIPT_DIR/state-key-audit.py"
|
||||
[ -f "$AUDIT" ] || { echo "[state-keys] CANNOT RUN: missing $AUDIT" >&2; exit 2; }
|
||||
|
||||
ARGS=("$ROOT" "--external" "$SCRIPT_DIR/state-key-external.txt")
|
||||
[ "$STRICT" -eq 0 ] && ARGS+=("--baseline" "$SCRIPT_DIR/state-key-baseline.txt")
|
||||
[ ${#PASS_THROUGH[@]} -gt 0 ] && ARGS+=("${PASS_THROUGH[@]}")
|
||||
|
||||
python3 "$AUDIT" "${ARGS[@]}"
|
||||
RC=$?
|
||||
if [ "$RC" -gt 1 ]; then
|
||||
echo "[state-keys] CANNOT RUN: the audit itself failed (exit $RC)" >&2
|
||||
exit 2
|
||||
fi
|
||||
exit "$RC"
|
||||
+7
-7
@@ -87,7 +87,7 @@ fn session_create(body: String) -> String {
|
||||
let folder: String = json_get(body, "folder")
|
||||
let content: String = session_make_content(id, title, ts, ts, folder)
|
||||
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
|
||||
let node_id: String = engram_node_full(
|
||||
let node_id: String = wt_node(
|
||||
content, "Conversation", "session:meta",
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
@@ -358,7 +358,7 @@ fn session_update_patch(session_id: String, body: String) -> String {
|
||||
let created_int: Int = str_to_int(old_created)
|
||||
let new_content: String = session_make_content(session_id, eff_title, created_int, ts, eff_folder)
|
||||
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
|
||||
let new_node_id: String = engram_node_full(
|
||||
let new_node_id: String = wt_node(
|
||||
new_content, "Conversation", "session:meta",
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
@@ -456,7 +456,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
|
||||
// TODO(reliability #7): delete-then-insert is not atomic — concurrent saves for the
|
||||
// same session can produce orphan history nodes. State is primary truth; engram fallback.
|
||||
let tags: String = "[\"session\",\"session-history\",\"Conversation\"]"
|
||||
let discard: String = engram_node_full(
|
||||
let discard: String = wt_node(
|
||||
hist, "Conversation", "session:messages:" + session_id,
|
||||
el_from_float(0.6), el_from_float(0.6), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
@@ -488,7 +488,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
|
||||
+ " | ts:" + int_to_str(ts_now)
|
||||
let summary_tags: String = "[\"session-emotional-summary\",\"affective\",\"bell:" + eff_level + "\",\"BellEvent\"]"
|
||||
let summary_sal: String = if str_eq(eff_level, "hard") { el_from_float(0.95) } else { el_from_float(0.85) }
|
||||
let sum_discard: String = engram_node_full(
|
||||
let sum_discard: String = wt_node(
|
||||
summary_content,
|
||||
"BellEvent",
|
||||
"session:emotional-summary",
|
||||
@@ -529,7 +529,7 @@ fn session_hist_save(session_id: String, hist: String) -> Void {
|
||||
if !str_eq(ot_id, "") { engram_forget(ot_id) }
|
||||
let oti = oti + 1
|
||||
}
|
||||
let discard_topic: String = engram_node_full(
|
||||
let discard_topic: String = wt_node(
|
||||
topic_content, "Conversation", topic_label,
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
"Episodic", topic_tags
|
||||
@@ -582,7 +582,7 @@ fn session_update_meta_timestamp(session_id: String) -> Void {
|
||||
let created_int: Int = str_to_int(old_created)
|
||||
let new_content: String = session_make_content(session_id, old_title, created_int, ts, old_folder)
|
||||
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
|
||||
let new_id: String = engram_node_full(
|
||||
let new_id: String = wt_node(
|
||||
new_content, "Conversation", "session:meta",
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
@@ -629,7 +629,7 @@ fn session_auto_title(session_id: String, first_message: String) -> Void {
|
||||
let created_int: Int = str_to_int(old_created)
|
||||
let new_content: String = session_make_content(session_id, new_title, created_int, ts, old_folder)
|
||||
let tags: String = "[\"session\",\"session:meta\",\"Conversation\"]"
|
||||
let new_id: String = engram_node_full(
|
||||
let new_id: String = wt_node(
|
||||
new_content, "Conversation", "session:meta",
|
||||
el_from_float(0.7), el_from_float(0.7), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
|
||||
@@ -657,6 +657,23 @@ if is_genesis && safe_to_seed {
|
||||
}
|
||||
}
|
||||
|
||||
// CRASH RECOVERY (neuron#117). Deltas the previous process staged but could not
|
||||
// hand to the owner are still on disk — the spool is a filesystem queue, not a
|
||||
// memory buffer, precisely so that a soul that died mid-flight does not take its
|
||||
// unpushed writes with it. Drain them before serving, so recovered memories are
|
||||
// durable and recallable from the owner from the first request onward.
|
||||
//
|
||||
// Safe on a clean boot: an empty spool means no HTTP call at all. Safe in file
|
||||
// mode: wt_drain returns immediately when ENGRAM_URL is unset.
|
||||
let wt_recovered: Int = wt_drain()
|
||||
if wt_recovered > 0 {
|
||||
println("[soul] write-through: recovered " + int_to_str(wt_recovered)
|
||||
+ " nodes from a previous process's spool -> persistence owner")
|
||||
}
|
||||
if wt_recovered < 0 {
|
||||
println("[soul] write-through: spool present but the persistence owner is unreachable — queued, will retry on heartbeat")
|
||||
}
|
||||
|
||||
println("[soul] serving on port " + int_to_str(port))
|
||||
http_serve_async(port, "handle_request")
|
||||
println("[soul] awareness loop starting")
|
||||
|
||||
+2
-2
@@ -11,7 +11,7 @@ import "memory.el"
|
||||
fn steward_log_event(kind: String, detail: String) -> Void {
|
||||
let content: String = "STEWARD:" + kind + " | " + detail
|
||||
let tags: String = "[\"stewardship\",\"steward:" + kind + "\"]"
|
||||
let discard: String = engram_node_full(
|
||||
let discard: String = wt_node(
|
||||
content,
|
||||
"StewardshipEvent",
|
||||
"steward:" + kind,
|
||||
@@ -221,7 +221,7 @@ fn steward_fingerprint_session(input: String, session_id: String) -> String {
|
||||
+ " formality=" + fs_str
|
||||
+ " time=" + tb_str
|
||||
let sample_tags: String = "[\"behavior\",\"BehaviorSample\",\"stewardship\"]"
|
||||
let discard: String = engram_node_full(
|
||||
let discard: String = wt_node(
|
||||
sample_content,
|
||||
"BehaviorSample",
|
||||
"behavior:" + session_id,
|
||||
|
||||
Executable
+59
@@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env bash
|
||||
# build-soul-from-dist.sh — build a deployable soul from the SAME input CI compiles.
|
||||
#
|
||||
# THE PROBLEM THIS CLOSES: until now, deploys were built by build-soul.sh, which
|
||||
# compiles a scratch amalgam and never touches dist/soul.c. CI compiles dist/soul.c.
|
||||
# Two lineages. On 2026-08-09 the committed input fell 2,761 bytes behind the sources
|
||||
# while three binaries built the other way were installed on the operator machine —
|
||||
# so "what runs" and "what the repo says builds" were different artifacts again,
|
||||
# which is the whole of #133 and #111 wearing new clothes.
|
||||
#
|
||||
# This builds from dist/soul.c with CI's own flags, after asserting that dist/soul.c
|
||||
# actually matches the .el sources, and writes a provenance sidecar so a deployer can
|
||||
# refuse anything of unknown origin.
|
||||
#
|
||||
# -rdynamic and -DHAVE_CURL are copied from .gitea/workflows/ci.yaml deliberately.
|
||||
# The CI comment explains -rdynamic: without it the runtime cannot resolve its HTTP
|
||||
# handler by name via dlsym and the binary serves nothing on every route.
|
||||
#
|
||||
# usage: build-soul-from-dist.sh <out-binary>
|
||||
set -u
|
||||
OUT="${1:?usage: build-soul-from-dist.sh <out-binary>}"
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
RUNTIME="$ROOT/vendor/el-runtime/v1.0.0-20260501"
|
||||
|
||||
cd "$ROOT" || exit 2
|
||||
|
||||
echo "[build-from-dist] GATE: does dist/soul.c match the sources?"
|
||||
if ! ./tools/soulc-stamp.sh --check; then
|
||||
echo "[build-from-dist] REFUSING — the build input is stale. Regenerate and stamp first." >&2
|
||||
exit 9
|
||||
fi
|
||||
|
||||
[ -f "$RUNTIME/el_runtime.c" ] || { echo "pinned runtime missing at $RUNTIME" >&2; exit 2; }
|
||||
|
||||
echo "[build-from-dist] compiling dist/soul.c with CI's flags"
|
||||
cc -O2 -DHAVE_CURL -rdynamic \
|
||||
-I"$RUNTIME" \
|
||||
dist/soul.c \
|
||||
"$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lpthread -lm \
|
||||
-o "$OUT" || { echo "[build-from-dist] COMPILE FAILED" >&2; exit 3; }
|
||||
|
||||
# Provenance sidecar: what a deployer checks before installing anything.
|
||||
SRC_SHA="$(shasum -a 256 dist/soul.c | awk '{print $1}')"
|
||||
STAMP_SHA="$(shasum -a 256 dist/soul.c.stamp | awk '{print $1}')"
|
||||
COMMIT="$(git rev-parse HEAD 2>/dev/null || echo unknown)"
|
||||
DIRTY="clean"; [ -n "$(git status --porcelain -- '*.el' dist/soul.c 2>/dev/null)" ] && DIRTY="DIRTY"
|
||||
cat > "$OUT.provenance" <<EOF
|
||||
{"built_from":"dist/soul.c",
|
||||
"dist_soul_c_sha256":"$SRC_SHA",
|
||||
"stamp_sha256":"$STAMP_SHA",
|
||||
"git_commit":"$COMMIT",
|
||||
"worktree":"$DIRTY",
|
||||
"runtime":"vendor/el-runtime/v1.0.0-20260501",
|
||||
"flags":"-O2 -DHAVE_CURL -rdynamic"}
|
||||
EOF
|
||||
|
||||
echo "[build-from-dist] OK -> $OUT ($(wc -c < "$OUT" | tr -d ' ') bytes)"
|
||||
echo "[build-from-dist] provenance -> $OUT.provenance (commit ${COMMIT:0:8}, worktree $DIRTY)"
|
||||
@@ -0,0 +1,149 @@
|
||||
{
|
||||
"baseline": "unfloor-clean",
|
||||
"candidate": "splitfix",
|
||||
"n_shared_queries": 75,
|
||||
"fixed_by_candidate": [
|
||||
"q15",
|
||||
"q28",
|
||||
"q60"
|
||||
],
|
||||
"broken_by_candidate": [],
|
||||
"discordant": 3,
|
||||
"net_queries": 3,
|
||||
"mcnemar_exact_p": 0.25,
|
||||
"min_detectable_swing_queries": 6,
|
||||
"observed_run_to_run_drift_queries": 0,
|
||||
"noise_floor_queries": 6,
|
||||
"verdict": "no measurable difference",
|
||||
"baseline_aggregate": {
|
||||
"n_queries": 75,
|
||||
"n_scored": 65,
|
||||
"hit@5": 0.5384615384615384,
|
||||
"recall@5": 0.44907176157176154,
|
||||
"recall@10": 0.5380300255300255,
|
||||
"precision@5": 0.13230769230769232,
|
||||
"mrr@10": 0.32437728937728944,
|
||||
"nonsense_clean": "10/10",
|
||||
"superseded_outranks": "2/3",
|
||||
"latency_ms_p50": 632.5,
|
||||
"latency_ms_p95": 992.5,
|
||||
"latency_ms_max": 1177.8,
|
||||
"errors": 0,
|
||||
"by_category": {
|
||||
"associative": {
|
||||
"n": 6,
|
||||
"hit@5": 0.5,
|
||||
"recall@5": 0.07575757575757576,
|
||||
"recall@10": 0.13636363636363635,
|
||||
"mrr@10": 0.23214285714285712
|
||||
},
|
||||
"exact_rare": {
|
||||
"n": 6,
|
||||
"hit@5": 1.0,
|
||||
"recall@5": 1.0,
|
||||
"recall@10": 1.0,
|
||||
"mrr@10": 1.0
|
||||
},
|
||||
"heldout_paraphrase": {
|
||||
"n": 30,
|
||||
"hit@5": 0.3,
|
||||
"recall@5": 0.3,
|
||||
"recall@10": 0.4,
|
||||
"mrr@10": 0.11638888888888889
|
||||
},
|
||||
"nonsense": {
|
||||
"n": 10,
|
||||
"clean": 10,
|
||||
"avg_false_positives": 0.0
|
||||
},
|
||||
"paraphrase": {
|
||||
"n": 13,
|
||||
"hit@5": 0.6923076923076923,
|
||||
"recall@5": 0.6923076923076923,
|
||||
"recall@10": 0.7692307692307693,
|
||||
"mrr@10": 0.29423076923076924
|
||||
},
|
||||
"phrase": {
|
||||
"n": 7,
|
||||
"hit@5": 1.0,
|
||||
"recall@5": 0.5335884353741497,
|
||||
"recall@10": 0.5933956916099773,
|
||||
"mrr@10": 0.8214285714285714
|
||||
},
|
||||
"superseded": {
|
||||
"n": 3,
|
||||
"hit@5": 0.3333333333333333,
|
||||
"recall@5": 0.3333333333333333,
|
||||
"recall@10": 0.6666666666666666,
|
||||
"mrr@10": 0.20833333333333334,
|
||||
"outranks": 2
|
||||
}
|
||||
}
|
||||
},
|
||||
"candidate_aggregate": {
|
||||
"n_queries": 75,
|
||||
"n_scored": 65,
|
||||
"hit@5": 0.5846153846153846,
|
||||
"recall@5": 0.48102442429365505,
|
||||
"recall@10": 0.5692415490492414,
|
||||
"precision@5": 0.14153846153846153,
|
||||
"mrr@10": 0.3351709401709402,
|
||||
"nonsense_clean": "10/10",
|
||||
"superseded_outranks": "2/3",
|
||||
"latency_ms_p50": 646.9,
|
||||
"latency_ms_p95": 1028.5,
|
||||
"latency_ms_max": 1197.8,
|
||||
"errors": 0,
|
||||
"by_category": {
|
||||
"associative": {
|
||||
"n": 6,
|
||||
"hit@5": 0.6666666666666666,
|
||||
"recall@5": 0.08857808857808858,
|
||||
"recall@10": 0.15967365967365968,
|
||||
"mrr@10": 0.24166666666666667
|
||||
},
|
||||
"exact_rare": {
|
||||
"n": 6,
|
||||
"hit@5": 1.0,
|
||||
"recall@5": 1.0,
|
||||
"recall@10": 1.0,
|
||||
"mrr@10": 1.0
|
||||
},
|
||||
"heldout_paraphrase": {
|
||||
"n": 30,
|
||||
"hit@5": 0.3333333333333333,
|
||||
"recall@5": 0.3333333333333333,
|
||||
"recall@10": 0.43333333333333335,
|
||||
"mrr@10": 0.12120370370370372
|
||||
},
|
||||
"nonsense": {
|
||||
"n": 10,
|
||||
"clean": 10,
|
||||
"avg_false_positives": 0.0
|
||||
},
|
||||
"paraphrase": {
|
||||
"n": 13,
|
||||
"hit@5": 0.7692307692307693,
|
||||
"recall@5": 0.7692307692307693,
|
||||
"recall@10": 0.8461538461538461,
|
||||
"mrr@10": 0.33269230769230773
|
||||
},
|
||||
"phrase": {
|
||||
"n": 7,
|
||||
"hit@5": 1.0,
|
||||
"recall@5": 0.5335884353741497,
|
||||
"recall@10": 0.5775226757369615,
|
||||
"mrr@10": 0.8214285714285714
|
||||
},
|
||||
"superseded": {
|
||||
"n": 3,
|
||||
"hit@5": 0.3333333333333333,
|
||||
"recall@5": 0.3333333333333333,
|
||||
"recall@10": 0.6666666666666666,
|
||||
"mrr@10": 0.20833333333333334,
|
||||
"outranks": 2
|
||||
}
|
||||
}
|
||||
},
|
||||
"repeat_variance": {}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
Executable
+83
@@ -0,0 +1,83 @@
|
||||
#!/usr/bin/env bash
|
||||
# soulc-stamp.sh — make it impossible for dist/soul.c to drift from the sources
|
||||
# in silence.
|
||||
#
|
||||
# THE PROBLEM (neuron#133, and its own words): "Nothing in the tree regenerates
|
||||
# this file. Only a human running the recipe. It lags in batches, never
|
||||
# per-change, and it will drift again."
|
||||
#
|
||||
# It drifted. On 2026-08-07 a CI or GKE build off main would have shipped an
|
||||
# engine with NONE of five merged fixes — including a P0 safety fix — while
|
||||
# main's source read as correct. CI compiles dist/soul.c, not the .el files, so
|
||||
# the source being right is not the same as the build being right.
|
||||
#
|
||||
# WHY A STAMP AND NOT AUTO-REGENERATION: the CI workflow says elc cannot run on
|
||||
# the runner ("elb on Linux would OOM the runner (elc uses 24GB+ virtual memory
|
||||
# on a 16GB host)"). So the build cannot regenerate the file itself. What it CAN
|
||||
# do, for free and with no compiler, is refuse to compile a stale one.
|
||||
#
|
||||
# The stamp records a fingerprint of every .el source that feeds the amalgam at
|
||||
# the moment it was generated. --check recomputes and compares. Divergence is a
|
||||
# build failure with the recipe in the message, not a silent ship.
|
||||
#
|
||||
# soulc-stamp.sh --write after regenerating dist/soul.c (records the fingerprint)
|
||||
# soulc-stamp.sh --check in CI, before the compile (fails on drift)
|
||||
set -u
|
||||
MODE="${1:---check}"
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
STAMP="$ROOT/dist/soul.c.stamp"
|
||||
AMALGAM="$ROOT/dist/soul.c"
|
||||
|
||||
# Every .el at the repo root is an input to the amalgam. Sorted so the hash is
|
||||
# order-independent; content-only so timestamps and checkouts do not perturb it.
|
||||
fingerprint() {
|
||||
(
|
||||
cd "$ROOT" || exit 1
|
||||
for f in $(ls -1 *.el 2>/dev/null | sort); do
|
||||
printf '%s %s\n' "$(shasum -a 256 "$f" | awk '{print $1}')" "$f"
|
||||
done
|
||||
)
|
||||
}
|
||||
|
||||
case "$MODE" in
|
||||
--write)
|
||||
[ -f "$AMALGAM" ] || { echo "no dist/soul.c to stamp — regenerate it first" >&2; exit 2; }
|
||||
{
|
||||
echo "# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from."
|
||||
echo "# Written by tools/soulc-stamp.sh --write. Do not hand-edit."
|
||||
echo "# generated_amalgam_sha256 $(shasum -a 256 "$AMALGAM" | awk '{print $1}')"
|
||||
echo "# generated_amalgam_bytes $(wc -c < "$AMALGAM" | tr -d ' ')"
|
||||
fingerprint
|
||||
} > "$STAMP"
|
||||
echo "stamped $(fingerprint | wc -l | tr -d ' ') sources -> dist/soul.c.stamp"
|
||||
;;
|
||||
|
||||
--check)
|
||||
if [ ! -f "$STAMP" ]; then
|
||||
echo "FAIL: dist/soul.c.stamp is missing — the build input is unverifiable." >&2
|
||||
echo " Regenerate the amalgam, then: tools/soulc-stamp.sh --write" >&2
|
||||
exit 1
|
||||
fi
|
||||
RECORDED="$(grep -v '^#' "$STAMP")"
|
||||
CURRENT="$(fingerprint)"
|
||||
if [ "$RECORDED" = "$CURRENT" ]; then
|
||||
echo "soulc-stamp: OK — dist/soul.c matches the .el sources"
|
||||
exit 0
|
||||
fi
|
||||
echo "FAIL: dist/soul.c is STALE. It does not match the current .el sources." >&2
|
||||
echo "" >&2
|
||||
echo "CI compiles dist/soul.c, not the .el files. Shipping this means shipping" >&2
|
||||
echo "an engine that does not contain the merged source. That is neuron#133," >&2
|
||||
echo "which once hid five merged fixes including a P0 safety fix." >&2
|
||||
echo "" >&2
|
||||
echo "Sources that changed since the amalgam was generated:" >&2
|
||||
diff <(printf '%s\n' "$RECORDED") <(printf '%s\n' "$CURRENT") \
|
||||
| grep -E '^[<>]' | awk '{print " " $1 " " $3}' | sort -u >&2
|
||||
echo "" >&2
|
||||
echo "Fix: regenerate the amalgam, then tools/soulc-stamp.sh --write" >&2
|
||||
exit 1
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "usage: soulc-stamp.sh [--check|--write]" >&2; exit 2 ;;
|
||||
esac
|
||||
+94
-2
@@ -7175,6 +7175,12 @@ void engram_forget(el_val_t node_id) {
|
||||
if (idx < 0) return;
|
||||
/* Free node strings */
|
||||
EngramNode* n = &g->nodes[idx];
|
||||
if (getenv("EG_DIAG")) {
|
||||
fprintf(stderr, "[EG_DIAG] FORGET id=%s type=%s layer=%u label=%s\n",
|
||||
sid, n->node_type ? n->node_type : "?", n->layer_id,
|
||||
n->label ? n->label : "?");
|
||||
fflush(stderr);
|
||||
}
|
||||
free(n->id); free(n->content); free(n->node_type); free(n->label);
|
||||
free(n->tier); free(n->tags); free(n->metadata);
|
||||
free(n->emb);
|
||||
@@ -7270,6 +7276,8 @@ el_val_t engram_prune_telemetry(el_val_t older_than_ms) {
|
||||
}
|
||||
}
|
||||
g->node_count = w;
|
||||
if (getenv("EG_DIAG"))
|
||||
fprintf(stderr, "[EG_DIAG] PRUNE_TELEMETRY removed=%lld\n", (long long)removed);
|
||||
if (removed == 0) { free(removed_ids); return 0; }
|
||||
|
||||
/* Removed-id hash set (open addressing, power-of-two >= 2*removed). */
|
||||
@@ -9304,6 +9312,26 @@ el_val_t engram_load(el_val_t path) {
|
||||
}
|
||||
}
|
||||
g->adj_dirty = 1;
|
||||
if (getenv("EG_DIAG")) {
|
||||
int64_t we = 0, wrongdim = 0;
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
if (g->nodes[i].emb) { we++; if (g->nodes[i].emb_dim != 768) wrongdim++; }
|
||||
}
|
||||
fprintf(stderr, "[EG_DIAG] loaded nodes=%lld with_emb=%lld wrongdim=%lld\n",
|
||||
(long long)g->node_count, (long long)we, (long long)wrongdim);
|
||||
const char* probe = getenv("EG_DIAG_ID");
|
||||
if (probe) {
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
if (g->nodes[i].id && strcmp(g->nodes[i].id, probe) == 0) {
|
||||
fprintf(stderr, "[EG_DIAG] probe id=%s idx=%lld emb=%p dim=%d layer=%u addr=%d\n",
|
||||
probe, (long long)i, (void*)g->nodes[i].emb,
|
||||
(int)g->nodes[i].emb_dim, g->nodes[i].layer_id,
|
||||
eg_node_addressable(&g->nodes[i]));
|
||||
}
|
||||
}
|
||||
}
|
||||
fflush(stderr);
|
||||
}
|
||||
/* Walk edges array */
|
||||
const char* edges_p = json_find_key(data, "edges");
|
||||
if (edges_p) {
|
||||
@@ -9624,7 +9652,33 @@ el_val_t engram_get_node_by_label(el_val_t label) {
|
||||
return el_wrap_str(el_strdup("{}"));
|
||||
}
|
||||
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit) {
|
||||
/* ── THE SEARCH / RECALL BOUNDARY (2026-08-07) ───────────────────────────────
|
||||
* engram_search_json is the LEXICAL function ~40 .el call sites already
|
||||
* depend on: they pass a key-shaped string ("soul:boot_count",
|
||||
* "soul-inbox-pending", a session label) and treat every returned record as
|
||||
* a record that CONTAINS that key. Seven of those sites then delete what
|
||||
* comes back (memory.el:176, sessions.el:250/268/444/523, soul.el:359 —
|
||||
* "prune all existing X nodes, keep exactly one").
|
||||
*
|
||||
* The semantic and associative legs must therefore NOT live on this
|
||||
* function. Claim 24 authorises the vector index "to respond to EMBEDDING
|
||||
* SEARCH QUERIES by returning the node records whose embedding vectors have
|
||||
* the highest cosine similarity to a query vector"; a keyed state read is
|
||||
* not an embedding search query, it is the identifier-keyed retrieval of
|
||||
* claim 23 ("node records are stored under a key encoding the node
|
||||
* identifier"). Putting both behind one function erased that boundary, and
|
||||
* a nearest neighbour of the string "soul:boot_count" is not a boot counter.
|
||||
*
|
||||
* MEASURED, on the harness corpus, isolated, read-only, no writes from any
|
||||
* caller: 240 node records destroyed per boot, including 6 Knowledge nodes,
|
||||
* a layer-1 "CORE IDENTITY — GENESIS, LINEAGE" Memory, and the value node
|
||||
* `kn-58874a74` (gold answer for gold-set q15). The deletion list is the
|
||||
* result list of the soul's own mem_boot_count_inc() lookup, in order.
|
||||
*
|
||||
* So: legs OFF here, legs ON in engram_recall_json below, which is what
|
||||
* /api/neuron/recall reaches. Retrieval quality on the recall route is
|
||||
* unchanged; the internal keyed reads get their contract back. */
|
||||
static el_val_t eg_search_json_impl(el_val_t query, el_val_t limit, int with_legs) {
|
||||
EngramStore* g = engram_get();
|
||||
const char* q = EL_CSTR(query);
|
||||
int64_t lim = (int64_t)limit;
|
||||
@@ -9646,7 +9700,7 @@ el_val_t engram_search_json(el_val_t query, el_val_t limit) {
|
||||
* so the semantic half of the retrieval surface has to land HERE
|
||||
* to be observable to the MCP wrapper and the app. */
|
||||
int32_t qdim = 0;
|
||||
float* qv = eg_embed_fetch(q, &qdim);
|
||||
float* qv = with_legs ? eg_embed_fetch(q, &qdim) : NULL;
|
||||
EngramSemEntry* sem = qv ? malloc((size_t)g->node_count * sizeof(EngramSemEntry)) : NULL;
|
||||
int64_t nsem = 0;
|
||||
int64_t nhits = 0;
|
||||
@@ -9751,6 +9805,31 @@ el_val_t engram_search_json(el_val_t query, el_val_t limit) {
|
||||
}
|
||||
qsort(hits, (size_t)nhits, sizeof(EngramRankEntry), engram_rank_w_cmp);
|
||||
if (sem) qsort(sem, (size_t)nsem, sizeof(EngramSemEntry), engram_sem_cmp);
|
||||
if (getenv("EG_DIAG")) {
|
||||
int64_t we = 0, unaddr = 0, found = 0;
|
||||
const char* pid = getenv("EG_DIAG_ID");
|
||||
for (int64_t i = 0; i < g->node_count; i++) {
|
||||
if (g->nodes[i].emb) we++;
|
||||
if (!eg_node_addressable(&g->nodes[i])) unaddr++;
|
||||
if (pid && g->nodes[i].id && strcmp(g->nodes[i].id, pid) == 0) found++;
|
||||
}
|
||||
fprintf(stderr, "[EG_DIAG] STORE node_count=%lld with_emb=%lld unaddressable=%lld probe_found=%lld\n",
|
||||
(long long)g->node_count, (long long)we, (long long)unaddr, (long long)found);
|
||||
fprintf(stderr, "[EG_DIAG] q=\"%s\" qdim=%d nhits=%lld nsem=%lld\n",
|
||||
q, (int)qdim, (long long)nhits, (long long)nsem);
|
||||
for (int64_t k = 0; k < 5 && k < nsem; k++)
|
||||
fprintf(stderr, "[EG_DIAG] sem[%lld] cos=%.4f id=%s\n",
|
||||
(long long)k, sem[k].sem, g->nodes[sem[k].idx].id);
|
||||
const char* probe = getenv("EG_DIAG_ID");
|
||||
if (probe) for (int64_t k = 0; k < nsem; k++)
|
||||
if (g->nodes[sem[k].idx].id
|
||||
&& strcmp(g->nodes[sem[k].idx].id, probe) == 0) {
|
||||
fprintf(stderr, "[EG_DIAG] probe at sem rank %lld cos=%.4f\n",
|
||||
(long long)k, sem[k].sem);
|
||||
break;
|
||||
}
|
||||
fflush(stderr);
|
||||
}
|
||||
/* Claim-10 associative leg: expand the top lexical hits along
|
||||
* structural relations only, order the reached set by query
|
||||
* similarity. Empty whenever the seeds have no structural
|
||||
@@ -9795,6 +9874,19 @@ el_val_t engram_search_json(el_val_t query, el_val_t limit) {
|
||||
return el_wrap_str(b.buf);
|
||||
}
|
||||
|
||||
/* Lexical keyed read — the historical contract every internal caller relies
|
||||
* on. Every returned record CONTAINS a query token. */
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit) {
|
||||
return eg_search_json_impl(query, limit, 0);
|
||||
}
|
||||
|
||||
/* The retrieval surface: lexical + claim-24 semantic + claim-10 associative,
|
||||
* rank-fused. Reached from handle_api_recall (/api/neuron/recall) — the route
|
||||
* the MCP wrapper and the app call, and the one the eval harness measures. */
|
||||
el_val_t engram_recall_json(el_val_t query, el_val_t limit) {
|
||||
return eg_search_json_impl(query, limit, 1);
|
||||
}
|
||||
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset) {
|
||||
EngramStore* g = engram_get();
|
||||
int64_t lim = (int64_t)limit; if (lim <= 0) lim = 100;
|
||||
|
||||
@@ -612,6 +612,7 @@ el_val_t engram_load(el_val_t path);
|
||||
el_val_t engram_get_node_json(el_val_t id);
|
||||
el_val_t engram_get_node_by_label(el_val_t label);
|
||||
el_val_t engram_search_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_recall_json(el_val_t query, el_val_t limit);
|
||||
el_val_t engram_scan_nodes_json(el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_scan_nodes_by_type_json(el_val_t node_type, el_val_t limit, el_val_t offset);
|
||||
el_val_t engram_neighbors_json(el_val_t node_id, el_val_t max_depth, el_val_t direction);
|
||||
|
||||
Reference in New Issue
Block a user