Compare commits
9 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 2018036bce | |||
| f1f52bcb2f | |||
| aad988ecbf | |||
| 7b86e6f72c | |||
| bf521974af | |||
| 5743568bf1 | |||
| 9fd8c11670 | |||
| be0f9d1afe | |||
| 1742d0b575 |
@@ -63,6 +63,16 @@ jobs:
|
||||
cp vendor/el-runtime/v1.0.0-20260501/el_runtime.h /opt/el/runtime/el_runtime.h
|
||||
echo "El runtime PINNED to v1.0.0-20260501: $(ls /opt/el/runtime/)"
|
||||
|
||||
# neuron#133: CI compiles dist/soul.c, NOT the .el sources. On 2026-08-07 a
|
||||
# build off main would have shipped an engine with none of five merged fixes,
|
||||
# including a P0 safety fix, while main's source read as correct. The runner
|
||||
# cannot regenerate the amalgam (elc needs 24GB+ virtual memory), but it can
|
||||
# refuse to compile a stale one. Fails loudly with the recipe in the message.
|
||||
- name: Verify dist/soul.c matches the sources
|
||||
run: |
|
||||
chmod +x tools/soulc-stamp.sh
|
||||
./tools/soulc-stamp.sh --check
|
||||
|
||||
- name: Build neuron soul binary
|
||||
run: |
|
||||
RUNTIME=/opt/el/runtime
|
||||
|
||||
+436
-127
File diff suppressed because it is too large
Load Diff
+18
@@ -0,0 +1,18 @@
|
||||
# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from.
|
||||
# Written by tools/soulc-stamp.sh --write. Do not hand-edit.
|
||||
# generated_amalgam_sha256 ee09798ad93ddd047136577cf324483e5956984d60d06596d34662f315ee0df4
|
||||
# generated_amalgam_bytes 1204845
|
||||
f8597e10546654bce3fbbe40461b2da59d0e06dbf1b038d1d362d24f949e3911 awareness.el
|
||||
b6f3d14ca0c26017a2d617399a6d3754dabb0905e4d5f52eb75d25c4ad18d3c5 chat.el
|
||||
42288c212cbf72fb1e8ecbd4d9900e4e9ee1cfa475b7974295c7637f1bf2939f elp-input.el
|
||||
b3f77f49d6086932c38bd17fe7a5eaf8bce25685f6fc3e1750f05729c6b49b9e imprint.el
|
||||
fba8ffdb9ba72bca5b09ca1c93a520edc52f3f4d8aec2c7585fe9b17e06420b2 manifest.el
|
||||
550a72e234ae8cec1f33e02108fd365353f45edd88513da90b792e79b6c0e5f0 memory.el
|
||||
5ec07ec9785b02abe32f3ff7acf2d1f9f7e07c0967fac97e6eff17d7110b5c84 neuron-api.el
|
||||
03c47c451e0e87f2c252cadb4b765867943962a804f548dd53adeef0520912c8 persist.el
|
||||
a6d69f3fc55233d9d3300160fd46a1551f2064bcd0fb84e2c9e432f636a72476 routes.el
|
||||
c28e36952ec56525963a0bdf29455ab097d3b0c5653d19c25fbb005e1069a1f7 safety.el
|
||||
fd3ab91d0ae0ea26639e21bef2f8f94054dc4b02eae68b19e3fe689d2769aad4 sessions.el
|
||||
5613b60d74d5d7768f46da5ac435a5dd99d38c27f0f7013c89fa27e98dc8a21c soul.el
|
||||
30337940905171a9645b0929f0a412ce6b3dccb1246495070c553bca0bbae6cd stewardship.el
|
||||
e105dc5990e6adbf39db9dc0462cd8bcf6e6c3dfd03709059227ecfad2bbab29 studio.el
|
||||
+11
-2
@@ -110,7 +110,7 @@ tool("beginSession", "Initialize session: surface recent high-importance memorie
|
||||
"," + tool("linkCausal", "Create a causal edge (cause -> effect).") +
|
||||
"," + tool("restructureCausalGraph", "Re-balance the causal subgraph after new evidence.") +
|
||||
"," + tool("rebuildGraph", "Rebuild graph indices from the on-disk snapshot.") +
|
||||
"," + tool("runStructuralAudit", "Audit graph structure for orphans, dangling edges, mislabeled types.") +
|
||||
"," + tool("runStructuralAudit", "Stage 1 structural audit: owner-vs-runtime divergence, orphans and dangling edges, typed-edge distribution, self-model connectivity. Returns an annotated characterization, not a score.") +
|
||||
// ── Backlog + work ──────────────────────────────────────────────────────────
|
||||
"," + tool("planWork", "Create a backlog item.") +
|
||||
"," + tool("reviewBacklog", "Browse work items.") +
|
||||
@@ -680,7 +680,16 @@ fn dispatch_tool_call(tool_name: String, args: String) -> String {
|
||||
return mcp_json_result(resp)
|
||||
}
|
||||
if str_eq(tool_name, "runStructuralAudit") {
|
||||
let resp: String = http_get(neuron_url() + "/session/begin")
|
||||
// Was: GET /session/begin — an unrelated session digest returned under an
|
||||
// audit tool name, i.e. the tool advertised a check that did not exist.
|
||||
// Now points at the real Stage 1 route (neuron-api.el
|
||||
// handle_api_structural_audit). Sample caps ride the query string; the
|
||||
// defaults keep a manual audit to a couple of seconds.
|
||||
let e_s: Int = json_get_int(args, "edge_sample")
|
||||
let n_s: Int = json_get_int(args, "node_sample")
|
||||
let qs: String = "?edge_sample=" + int_to_str(if e_s > 0 { e_s } else { 3000 })
|
||||
+ "&node_sample=" + int_to_str(if n_s > 0 { n_s } else { 300 })
|
||||
let resp: String = http_get(neuron_url() + "/audit/structural" + qs)
|
||||
return mcp_json_result(resp)
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,86 @@ fn tier_working() -> String { return "Working" }
|
||||
fn tier_episodic() -> String { return "Episodic" }
|
||||
fn tier_canonical() -> String { return "Canonical" }
|
||||
|
||||
// ── Association on write ──────────────────────────────────────────────────────
|
||||
// DESIGN: "promotion integrates candidate nodes by linking them to existing nodes
|
||||
// using typed semantic edges RATHER THAN APPENDING AS UNLINKED CONTENT" (CCR
|
||||
// claim 29). Unlinked append is the explicitly rejected behaviour — and it is the
|
||||
// only behaviour this system had. Measured 2026-08-09 on Tim's graph: 14,214 edges
|
||||
// across 80,936 nodes, 5% of nodes connected to anything, and NO edge created by
|
||||
// any write since 2026-07-19 while 27,000+ nodes were added. A memory that forms
|
||||
// no connections cannot be reached by spreading activation, so retrieval silently
|
||||
// degrades to literal matching.
|
||||
//
|
||||
// BOUNDS, each one bought with a specific failure:
|
||||
// * max 3 edges per memory — link_memories.py's cap, precision over spray
|
||||
// * never link to identity (self/*, Value): the existing policy is explicit that
|
||||
// "memories must not pollute the self traversal by similarity; only an explicit
|
||||
// citation may touch identity". Similarity is not citation.
|
||||
// * never link telemetry (state-event, soul-response, boot_count, loop-outcome):
|
||||
// these are ~97% of daily write volume (1,020 vs 31 real memories on 08-08).
|
||||
// Linking them would add ~3,000 noise edges a day and re-flatten the graph in
|
||||
// the name of connecting it.
|
||||
// * fail-soft: a failed association never fails the write.
|
||||
// Edges go through wt_edge so they reach the owner and survive restart.
|
||||
fn mem_assoc_skip_label(label: String) -> Bool {
|
||||
if str_contains(label, "state-event") { return true }
|
||||
if str_contains(label, "soul-response") { return true }
|
||||
if str_contains(label, "soul-outbox") { return true }
|
||||
if str_contains(label, "boot_count") { return true }
|
||||
if str_contains(label, "loop-outcome") { return true }
|
||||
if str_contains(label, "search-result") { return true }
|
||||
return false
|
||||
}
|
||||
|
||||
// A candidate is linkable only if it is a real, distinct, non-identity node.
|
||||
fn mem_assoc_ok(cand_id: String, cand_label: String, self_id: String) -> Bool {
|
||||
if str_eq(cand_id, "") { return false }
|
||||
if str_eq(cand_id, self_id) { return false }
|
||||
// CASE MATTERS — measured 2026-08-09. A lowercase-only check let a memory link
|
||||
// to "Self — Values (grounded)", i.e. it polluted the self traversal, which is
|
||||
// the one thing this policy exists to prevent. My verification had the same
|
||||
// blind spot and printed PASS. Check every casing the graph actually uses, and
|
||||
// exclude identity node TYPES as well as labels.
|
||||
let lab: String = str_lower(cand_label)
|
||||
if str_starts_with(lab, "self") { return false }
|
||||
if str_starts_with(lab, "value") { return false }
|
||||
if str_contains(lab, "values") { return false }
|
||||
if str_contains(lab, "identity") { return false }
|
||||
if mem_assoc_skip_label(cand_label) { return false }
|
||||
return true
|
||||
}
|
||||
|
||||
// One slot of the association. Manual unroll rather than a loop: EL's codegen
|
||||
// mis-emits accumulating while-loops (documented at soul.el:212, which unrolled
|
||||
// three affective slots for the same reason).
|
||||
fn mem_assoc_slot(results: String, idx: Int, new_id: String) -> Void {
|
||||
if idx >= json_array_len(results) { return }
|
||||
let cand: String = json_array_get(results, idx)
|
||||
let cid: String = json_get(cand, "id")
|
||||
let clabel: String = json_get(cand, "label")
|
||||
let ctype: String = json_get(cand, "node_type")
|
||||
if str_eq(ctype, "Value") { return }
|
||||
if str_eq(ctype, "DharmaSelf") { return }
|
||||
if str_eq(ctype, "Safety") { return }
|
||||
if mem_assoc_ok(cid, clabel, new_id) {
|
||||
wt_edge(new_id, cid, el_from_float(0.5), "related")
|
||||
}
|
||||
}
|
||||
|
||||
// mem_associate — connect a freshly written memory to what it is about.
|
||||
fn mem_associate(new_id: String, content: String, label: String) -> Void {
|
||||
if str_eq(new_id, "") { return }
|
||||
if mem_assoc_skip_label(label) { return }
|
||||
// Ask the graph what this memory resembles. Now that the store carries
|
||||
// meaning-vectors this is semantic, not merely lexical.
|
||||
let probe: String = str_slice(content, 0, 400)
|
||||
let results: String = engram_recall_json(probe, 4)
|
||||
if str_eq(results, "") { return }
|
||||
mem_assoc_slot(results, 0, new_id)
|
||||
mem_assoc_slot(results, 1, new_id)
|
||||
mem_assoc_slot(results, 2, new_id)
|
||||
}
|
||||
|
||||
fn mem_store(content: String, label: String, tags: String) -> String {
|
||||
let id: String = wt_node(
|
||||
content,
|
||||
@@ -31,6 +111,9 @@ fn mem_store(content: String, label: String, tags: String) -> String {
|
||||
// rather than claiming a save that did not happen. The id is still returned:
|
||||
// the local write DID succeed, and the queued delta will be retried.
|
||||
let durable: Bool = wt_commit(id)
|
||||
// Associate AFTER the node is durable: an edge to a node that did not persist
|
||||
// is a dangling edge, which is the defect the 2026-08-09 cleanup removed 830 of.
|
||||
mem_associate(id, content, label)
|
||||
if durable {
|
||||
println("[memory] write persisted at owner: " + id + " label=" + label)
|
||||
} else {
|
||||
|
||||
+492
-1
@@ -304,10 +304,35 @@ fn handle_api_begin_session(body: String) -> String {
|
||||
let state_events: String = api_compact_node_array(state_events_raw, 5, 500)
|
||||
let recent_raw: String = engram_scan_nodes_json(10, 0)
|
||||
let recent: String = api_compact_node_array(recent_raw, 10, 240)
|
||||
// SELF-SEEDED SLICE (2026-08-09). The design is explicit: "Every compilation
|
||||
// query begins at the self-model node and traverses outward... structural
|
||||
// reachability from the self-model node is a precondition for any node to
|
||||
// appear in compiled context" (will-anderson patents/drafts/engram-claims.md,
|
||||
// Self-Seeded Activation; DRAFT, not a filed provisional — cite it as such).
|
||||
//
|
||||
// Measured 2026-08-09 before this change: compiled context contained 0-1
|
||||
// identity records out of 10, because compilation seeds from a hardcoded
|
||||
// TEXT STRING, never from the self. Even an explicit "my values identity who
|
||||
// I am" query returned a boot counter and state-events.
|
||||
//
|
||||
// This restores the designed behaviour WITHOUT repeating the failure that got
|
||||
// self_neighbors set to [] in the first place: that was an UNBOUNDED ~90KB
|
||||
// neighbour dump which closed the socket on every call. Same bound as every
|
||||
// other list here — cap 8, 240-char snippets. The self root has 34 direct
|
||||
// neighbours of which 23 are identity records, so depth 1 is dense enough to
|
||||
// be worth seeding and small enough to stay cheap.
|
||||
let self_raw: String = engram_neighbors_json("kn-efeb4a5b-5aff-4759-8a97-7233099be6ee", 1, "both")
|
||||
// Cap 24, not 8: measured 2026-08-09, the self root's first 8 neighbours are
|
||||
// TAG nodes ("neuron", "tier:note", "disposition:experimental", "imprint",
|
||||
// "traversal") which crowd out the substantive identity records behind them.
|
||||
// The root has 34 neighbours of which 23 are identity; 24 captures them while
|
||||
// staying bounded. Cost measured at ~+4KB on a ~12KB response, nowhere near
|
||||
// the ~90KB unbounded dump that closed sockets and got this set to [].
|
||||
let self_slice: String = api_compact_node_array(self_raw, 24, 240)
|
||||
return "{\"stats\":" + stats
|
||||
+ ",\"recent\":" + recent
|
||||
+ ",\"activated\":" + activated
|
||||
+ ",\"self_neighbors\":[]"
|
||||
+ ",\"self_neighbors\":" + self_slice
|
||||
+ ",\"recent_state_events\":" + state_events + "}"
|
||||
}
|
||||
|
||||
@@ -355,6 +380,13 @@ fn handle_api_remember(body: String) -> String {
|
||||
sal, sal, el_from_float(0.9),
|
||||
"Episodic", final_tags)
|
||||
if !api_persisted(id) { return api_not_persisted(id) }
|
||||
// Associate on write (2026-08-09). THIS CALL MUST BE HERE, not only in mem_store.
|
||||
// The HTTP memory route writes via wt_node directly; mem_store serves only the
|
||||
// awareness paths (soul-response, search-result, activation-result) which are
|
||||
// exactly the telemetry we refuse to link. Hooking mem_store alone produced
|
||||
// ZERO edges across four real writes — measured, not assumed, which is the only
|
||||
// reason it was caught before shipping.
|
||||
mem_associate(id, content, "memory:remembered")
|
||||
return "{\"id\":\"" + id + "\",\"ok\":true}"
|
||||
}
|
||||
|
||||
@@ -894,3 +926,462 @@ fn handle_api_consolidate(body: String) -> String {
|
||||
}
|
||||
return "{\"ok\":true,\"snapshot\":\"" + snap + "\"}"
|
||||
}
|
||||
|
||||
// ── Stage 1: structural audit ─────────────────────────────────────────────────
|
||||
//
|
||||
// WHAT THIS IMPLEMENTS
|
||||
// The CGI provisional, 05-detailed-description.md, "Stage 1: Structural audit
|
||||
// 430". Verbatim, the audit module evaluates: the density and typed
|
||||
// distribution of causal edges; the consistency between value nodes and
|
||||
// execution-record neighborhoods; the richness and connectivity of the
|
||||
// self-model; and the authenticity of open-question nodes in the wonder
|
||||
// manifest. It "produces a coherence assessment 432 — NOT A BINARY SCORE but
|
||||
// an annotated characterization of the graph's structural properties".
|
||||
//
|
||||
// That last clause is the whole shape of this handler. Every finding carries
|
||||
// its own numbers AND a plain-language note saying what the numbers mean and
|
||||
// how they were obtained. There is no pass/fail, no percentage-of-health, no
|
||||
// composite score, and `"score":null` is emitted explicitly so a downstream
|
||||
// reader cannot mistake its absence for an omission.
|
||||
//
|
||||
// WHY IT EXISTS NOW, AND WHY THE FIRST FINDING IS THE ONE IT IS
|
||||
// `runStructuralAudit` has been an advertised MCP tool with nothing behind it:
|
||||
// the dispatcher GET'd /session/begin and returned that blob (mcp-wrapper/src/
|
||||
// main.el). Meanwhile the failure the audit would have caught ran silently for
|
||||
// about three weeks — the soul reported 103,089 nodes while the engram, which
|
||||
// OWNS persistence, held ~79,900; a crash discarded the difference. Every boot
|
||||
// reported green throughout, because nothing in the system ever compared the
|
||||
// two sides. So finding 1 is owner-versus-runtime divergence: it is the check
|
||||
// whose absence cost real memory, and it is cheap and exact.
|
||||
//
|
||||
// WHAT IS DELIBERATELY NOT HERE (stage 1b, see the `deferred` array in the
|
||||
// response): value/execution-record consistency and wonder-manifest
|
||||
// authenticity. Both need node types that barely exist in this graph today —
|
||||
// the response MEASURES those populations and reports the counts as the reason,
|
||||
// rather than asserting a deferral without evidence.
|
||||
//
|
||||
// MEASUREMENT HONESTY: EXACT WHERE CHEAP, SAMPLED WHERE NOT, ALWAYS LABELLED
|
||||
// Counts, edge typing and self-model connectivity are exact. Orphan rate and
|
||||
// dangling-edge rate are SAMPLED, because the engram runtime has no node-id
|
||||
// index — `engram_find_node_index` is a linear scan over every node, so an
|
||||
// exhaustive dangling check is O(nodes x edges) (~2.2e9 string compares at
|
||||
// today's scale, tens of seconds inside one request). The samples are UNIFORM
|
||||
// across the whole population, not head-of-list, and every sampled figure is
|
||||
// emitted with its own `sampled` / `population` fields plus an extrapolation
|
||||
// labelled as such. Raise `?edge_sample=` / `?node_sample=` to the population
|
||||
// size to run either check exhaustively and pay the time. The real fix is an
|
||||
// id index in the runtime; that is the engram repo's, not this handler's.
|
||||
|
||||
// audit_pct1 — one-decimal percentage as a bare JSON number, sign-safe.
|
||||
// Integer math only: EL has no fixed-precision formatter, and float_to_str
|
||||
// would put an unbounded mantissa in the response.
|
||||
fn audit_pct1(num: Int, den: Int) -> String {
|
||||
if den <= 0 { return "null" }
|
||||
let neg: Bool = num < 0
|
||||
let a: Int = if neg { 0 - num } else { num }
|
||||
let tenths: Int = (a * 1000) / den
|
||||
let whole: Int = tenths / 10
|
||||
let frac: Int = tenths - (whole * 10)
|
||||
let sign: String = if neg { "-" } else { "" }
|
||||
return sign + int_to_str(whole) + "." + int_to_str(frac)
|
||||
}
|
||||
|
||||
// audit_finding — the one envelope every finding uses: name, the measurements,
|
||||
// and the annotation. Keeping it in one place is what stops the characterization
|
||||
// from degenerating into a bag of numbers with no reading attached.
|
||||
fn audit_finding(name: String, measured: String, note: String) -> String {
|
||||
return "{\"finding\":\"" + name + "\""
|
||||
+ ",\"measured\":{" + measured + "}"
|
||||
+ ",\"note\":\"" + api_json_escape(note) + "\"}"
|
||||
}
|
||||
|
||||
// audit_str_at — read the quoted string value starting at byte `start`.
|
||||
// Slices a bounded window rather than the tail of the (multi-MB) edges array, so
|
||||
// this is O(window) per call instead of O(remaining input).
|
||||
fn audit_str_at(s: String, start: Int, maxlen: Int) -> String {
|
||||
let n: Int = str_len(s)
|
||||
if start < 0 || start >= n { return "" }
|
||||
let end_guess: Int = start + maxlen
|
||||
let stop: Int = if end_guess > n { n } else { end_guess }
|
||||
let win: String = str_slice(s, start, stop)
|
||||
let q: Int = str_index_of(win, "\"")
|
||||
if q < 0 { return "" }
|
||||
return str_slice(win, 0, q)
|
||||
}
|
||||
|
||||
// audit_rel_count — exact count of edges carrying `rel`, by scanning the emitted
|
||||
// edge array for the literal `"relation":"<rel>"`. engram_emit_edge_json writes
|
||||
// metadata ESCAPED as a string, so no nested object can contain that literal and
|
||||
// the count cannot be inflated by edge payloads.
|
||||
fn audit_rel_count(edges: String, rel: String) -> Int {
|
||||
return str_count(edges, "\"relation\":\"" + rel + "\"")
|
||||
}
|
||||
|
||||
// audit_owner_stats — ask the persistence OWNER for its own counts.
|
||||
// Returns "" when there is no HTTP owner configured or the owner is unreachable;
|
||||
// both are reported as findings, never as a failure of the audit.
|
||||
fn audit_owner_stats(url: String) -> String {
|
||||
if str_eq(url, "") { return "" }
|
||||
return http_get(url + "/api/stats")
|
||||
}
|
||||
|
||||
// audit_divergence — FINDING 1. Runtime (this soul's in-process graph) versus
|
||||
// the persistence owner's own count. Trend is measured against the previous
|
||||
// audit recorded in soul state, so a second call answers "is the gap growing?"
|
||||
// rather than just restating it.
|
||||
fn audit_divergence() -> String {
|
||||
let rt_nodes: Int = engram_node_count()
|
||||
let rt_edges: Int = engram_edge_count()
|
||||
let url: String = wt_engram_url()
|
||||
|
||||
if str_eq(url, "") {
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"none\",\"owner_reachable\":false",
|
||||
"No HTTP persistence owner is configured, so this soul IS the owner "
|
||||
+ "(file mode) and divergence is not defined. This check only has "
|
||||
+ "meaning when ENGRAM_URL points at a separate engram that owns the "
|
||||
+ "canonical store.")
|
||||
}
|
||||
|
||||
let stats: String = audit_owner_stats(url)
|
||||
// REACHABILITY IS PROVED BY THE PAYLOAD, NOT BY A NON-EMPTY REPLY.
|
||||
// http_get does not return "" on a connection failure — it returns a JSON
|
||||
// error object ({"error":"Failed to connect to ... Couldn't connect to
|
||||
// server"}). Testing only for "" made a DEAD owner read as reachable with
|
||||
// node_count 0, i.e. the audit would have reported a 100% divergence and
|
||||
// named it as data loss. That false positive is worse than no check at all:
|
||||
// it is precisely the kind of confident wrong answer this route exists to
|
||||
// stop. Require the field the contract promises.
|
||||
let owner_nc_raw: String = json_get_raw(stats, "node_count")
|
||||
if str_eq(stats, "") || str_eq(owner_nc_raw, "") {
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":false"
|
||||
+ ",\"owner_reply\":\"" + api_json_escape(api_utf8_trunc(stats, 200)) + "\"",
|
||||
"The persistence owner at " + url + " did not return a node_count "
|
||||
+ "from GET /api/stats. Divergence is UNKNOWN, NOT ZERO — an owner "
|
||||
+ "that cannot be read is exactly the condition under which the "
|
||||
+ "runtime's own count means least, and reporting 0 for the owner "
|
||||
+ "would manufacture a total-loss reading out of a network error. "
|
||||
+ "Reported as a finding rather than raised as an error so the rest "
|
||||
+ "of the audit still returns; the owner's raw reply is in "
|
||||
+ "owner_reply.")
|
||||
}
|
||||
|
||||
let ow_nodes: Int = json_get_int(stats, "node_count")
|
||||
let ow_edges: Int = json_get_int(stats, "edge_count")
|
||||
let d_nodes: Int = rt_nodes - ow_nodes
|
||||
let d_edges: Int = rt_edges - ow_edges
|
||||
|
||||
// Trend against the previous audit in this soul's state.
|
||||
let prev_raw: String = state_get("audit_prev_node_delta")
|
||||
let prev: Int = str_to_int(prev_raw)
|
||||
let abs_now: Int = if d_nodes < 0 { 0 - d_nodes } else { d_nodes }
|
||||
let abs_prev: Int = if prev < 0 { 0 - prev } else { prev }
|
||||
let trend: String = if str_eq(prev_raw, "") {
|
||||
"no_prior_audit"
|
||||
} else {
|
||||
if abs_now > abs_prev { "growing" } else {
|
||||
if abs_now < abs_prev { "shrinking" } else { "flat" }
|
||||
}
|
||||
}
|
||||
state_set("audit_prev_node_delta", int_to_str(d_nodes))
|
||||
state_set("audit_prev_ts", int_to_str(time_now()))
|
||||
|
||||
let note_head: String = if d_nodes == 0 {
|
||||
"Runtime and owner agree on node count."
|
||||
} else {
|
||||
"Runtime holds " + int_to_str(d_nodes) + " nodes (" + audit_pct1(d_nodes, rt_nodes)
|
||||
+ "% of its own graph) that the persistence owner does not report. Nodes "
|
||||
+ "that exist only in runtime memory do not survive a restart."
|
||||
}
|
||||
return audit_finding("owner_runtime_divergence",
|
||||
"\"runtime_nodes\":" + int_to_str(rt_nodes)
|
||||
+ ",\"runtime_edges\":" + int_to_str(rt_edges)
|
||||
+ ",\"owner\":\"" + api_json_escape(url) + "\",\"owner_reachable\":true"
|
||||
+ ",\"owner_nodes\":" + int_to_str(ow_nodes)
|
||||
+ ",\"owner_edges\":" + int_to_str(ow_edges)
|
||||
+ ",\"node_delta\":" + int_to_str(d_nodes)
|
||||
+ ",\"edge_delta\":" + int_to_str(d_edges)
|
||||
+ ",\"node_delta_pct_of_runtime\":" + audit_pct1(d_nodes, rt_nodes)
|
||||
+ ",\"trend_vs_previous_audit\":\"" + trend + "\""
|
||||
+ ",\"previous_node_delta\":" + (if str_eq(prev_raw, "") { "null" } else { int_to_str(prev) }),
|
||||
note_head + " Trend against the previous audit recorded in this soul's "
|
||||
+ "state: " + trend + ". This is the comparison whose absence let a "
|
||||
+ "~24,000-node loss run for weeks with every boot reporting green.")
|
||||
}
|
||||
|
||||
// audit_edge_typing — FINDING 2. Density plus the typed distribution the patent
|
||||
// asks for, against the claim-10 relation vocabulary. Exact: str_count over the
|
||||
// emitted edge array, one linear pass per relation.
|
||||
fn audit_edge_typing(edges: String, total_edges: Int, node_total: Int) -> String {
|
||||
let c_sup: Int = audit_rel_count(edges, "Supersedes")
|
||||
let c_cau: Int = audit_rel_count(edges, "Causes")
|
||||
let c_con: Int = audit_rel_count(edges, "Contains")
|
||||
let c_ref: Int = audit_rel_count(edges, "References")
|
||||
let c_ctr: Int = audit_rel_count(edges, "Contradicts")
|
||||
let c_exe: Int = audit_rel_count(edges, "Exemplifies")
|
||||
let c_act: Int = audit_rel_count(edges, "Activates")
|
||||
let c_tmp: Int = audit_rel_count(edges, "TemporallyPrecedes")
|
||||
let typed: Int = c_sup + c_cau + c_con + c_ref + c_ctr + c_exe + c_act + c_tmp
|
||||
|
||||
// Lowercase near-misses: the same eight concepts written by the ad-hoc write
|
||||
// paths (linkEntities defaults to "associates", linkCausal to "causes").
|
||||
// Counted separately because "the vocabulary is unused" and "the vocabulary
|
||||
// is used in the wrong case" are different defects with different fixes.
|
||||
let l_sup: Int = audit_rel_count(edges, "supersedes")
|
||||
let l_cau: Int = audit_rel_count(edges, "causes")
|
||||
let l_con: Int = audit_rel_count(edges, "contains")
|
||||
let l_ref: Int = audit_rel_count(edges, "references")
|
||||
let l_ctr: Int = audit_rel_count(edges, "contradicts")
|
||||
let l_exe: Int = audit_rel_count(edges, "exemplifies")
|
||||
let l_act: Int = audit_rel_count(edges, "activates")
|
||||
let l_tmp: Int = audit_rel_count(edges, "temporallyPrecedes")
|
||||
let near: Int = l_sup + l_cau + l_con + l_ref + l_ctr + l_exe + l_act + l_tmp
|
||||
|
||||
let untyped: Int = total_edges - typed
|
||||
return audit_finding("typed_edge_distribution",
|
||||
"\"total_edges\":" + int_to_str(total_edges)
|
||||
+ ",\"total_nodes\":" + int_to_str(node_total)
|
||||
// Density per 100 nodes, not per node: EL has no fixed-precision float
|
||||
// formatter, and "0.3 edges per node" rounded to an integer is a lie.
|
||||
+ ",\"edges_per_100_nodes\":" + audit_pct1(total_edges, node_total)
|
||||
+ ",\"claim10_typed\":" + int_to_str(typed)
|
||||
+ ",\"claim10_typed_pct\":" + audit_pct1(typed, total_edges)
|
||||
+ ",\"outside_claim10_vocabulary\":" + int_to_str(untyped)
|
||||
+ ",\"lowercase_near_miss\":" + int_to_str(near)
|
||||
+ ",\"by_relation\":{"
|
||||
+ "\"Supersedes\":" + int_to_str(c_sup)
|
||||
+ ",\"Causes\":" + int_to_str(c_cau)
|
||||
+ ",\"Contains\":" + int_to_str(c_con)
|
||||
+ ",\"References\":" + int_to_str(c_ref)
|
||||
+ ",\"Contradicts\":" + int_to_str(c_ctr)
|
||||
+ ",\"Exemplifies\":" + int_to_str(c_exe)
|
||||
+ ",\"Activates\":" + int_to_str(c_act)
|
||||
+ ",\"TemporallyPrecedes\":" + int_to_str(c_tmp) + "}",
|
||||
"Only " + int_to_str(typed) + " of " + int_to_str(total_edges)
|
||||
+ " edges use the claim-10 causal vocabulary; the remainder are ad-hoc "
|
||||
+ "relation strings, which is why the graph's causal claims cannot yet "
|
||||
+ "be checked for internal consistency — an untyped edge asserts "
|
||||
+ "association, not causation. " + int_to_str(near) + " edges use a "
|
||||
+ "lowercase spelling of a claim-10 relation: those are near-misses the "
|
||||
+ "write paths could be corrected to emit, not genuinely foreign types.")
|
||||
}
|
||||
|
||||
// audit_orphans_dangling — FINDING 3. Both figures are SAMPLED; see the header
|
||||
// for why exhaustive is O(nodes x edges) on this runtime.
|
||||
//
|
||||
// An "orphan" here is a node with zero RESOLVABLE edges: engram_neighbors_json
|
||||
// drops any edge whose other endpoint does not resolve to a node, so a node
|
||||
// whose only edges are dangling reads as an orphan. That is the right reading —
|
||||
// such a node is unreachable by traversal — but it is stated rather than hidden.
|
||||
fn audit_orphans_dangling(edges: String, total_edges: Int, node_total: Int,
|
||||
edge_cap: Int, node_cap: Int) -> String {
|
||||
// ── orphan sample: uniform stride over the node store ──
|
||||
let n_take: Int = if node_total < node_cap { node_total } else { node_cap }
|
||||
let n_stride: Int = if n_take > 0 { node_total / n_take } else { 1 }
|
||||
let n_stride = if n_stride < 1 { 1 } else { n_stride }
|
||||
let orphans: Int = 0
|
||||
let n_checked: Int = 0
|
||||
let j: Int = 0
|
||||
while j < n_take {
|
||||
let one: String = engram_scan_nodes_json(1, j * n_stride)
|
||||
let nid: String = json_get(json_array_get(one, 0), "id")
|
||||
if !str_eq(nid, "") {
|
||||
let nbrs: String = engram_neighbors_json(nid, 1, "both")
|
||||
let deg: Int = json_array_len(nbrs)
|
||||
let orphans = if deg == 0 { orphans + 1 } else { orphans }
|
||||
let n_checked = n_checked + 1
|
||||
}
|
||||
let j = j + 1
|
||||
}
|
||||
|
||||
// ── dangling sample: uniform stride over the edge array ──
|
||||
// str_index_of_all gives every edge's field offsets in ONE linear pass, so
|
||||
// any index can be read in O(1). json_array_get would have been O(i) per
|
||||
// element and O(n^2) over the array.
|
||||
let from_pos: [Int] = str_index_of_all(edges, "\"from_id\":\"")
|
||||
let to_pos: [Int] = str_index_of_all(edges, "\"to_id\":\"")
|
||||
let nf: Int = len(from_pos)
|
||||
let nt: Int = len(to_pos)
|
||||
let ne: Int = if nf < nt { nf } else { nt }
|
||||
let e_take: Int = if ne < edge_cap { ne } else { edge_cap }
|
||||
let e_stride: Int = if e_take > 0 { ne / e_take } else { 1 }
|
||||
let e_stride = if e_stride < 1 { 1 } else { e_stride }
|
||||
let dangling: Int = 0
|
||||
let e_checked: Int = 0
|
||||
let i: Int = 0
|
||||
while i < ne && e_checked < e_take {
|
||||
let fid: String = audit_str_at(edges, get(from_pos, i) + 11, 96)
|
||||
let tid: String = audit_str_at(edges, get(to_pos, i) + 9, 96)
|
||||
let f_gone: Bool = str_eq(engram_get_node_json(fid), "{}")
|
||||
let t_gone: Bool = if f_gone { true } else { str_eq(engram_get_node_json(tid), "{}") }
|
||||
let dangling = if f_gone || t_gone { dangling + 1 } else { dangling }
|
||||
let e_checked = e_checked + 1
|
||||
let i = i + e_stride
|
||||
}
|
||||
|
||||
let orphan_est: Int = if n_checked > 0 { (orphans * node_total) / n_checked } else { 0 }
|
||||
let dangle_est: Int = if e_checked > 0 { (dangling * total_edges) / e_checked } else { 0 }
|
||||
let exhaustive_n: String = if n_checked >= node_total { "true" } else { "false" }
|
||||
let exhaustive_e: String = if e_checked >= ne { "true" } else { "false" }
|
||||
|
||||
return audit_finding("orphans_and_dangling_edges",
|
||||
"\"nodes_population\":" + int_to_str(node_total)
|
||||
+ ",\"nodes_sampled\":" + int_to_str(n_checked)
|
||||
+ ",\"nodes_sample_exhaustive\":" + exhaustive_n
|
||||
+ ",\"orphans_in_sample\":" + int_to_str(orphans)
|
||||
+ ",\"orphan_rate_pct\":" + audit_pct1(orphans, n_checked)
|
||||
+ ",\"orphans_extrapolated\":" + int_to_str(orphan_est)
|
||||
+ ",\"edges_population\":" + int_to_str(total_edges)
|
||||
+ ",\"edges_sampled\":" + int_to_str(e_checked)
|
||||
+ ",\"edges_sample_exhaustive\":" + exhaustive_e
|
||||
+ ",\"dangling_in_sample\":" + int_to_str(dangling)
|
||||
+ ",\"dangling_rate_pct\":" + audit_pct1(dangling, e_checked)
|
||||
+ ",\"dangling_extrapolated\":" + int_to_str(dangle_est),
|
||||
"Orphan = zero RESOLVABLE edges, so a node whose only edges dangle counts "
|
||||
+ "as an orphan; either way it is unreachable by traversal. Dangling = an "
|
||||
+ "edge with an endpoint id that resolves to no node. Both are uniform "
|
||||
+ "stride samples over the whole population, not the head of the list; "
|
||||
+ "the extrapolations are estimates and are labelled as such. Pass "
|
||||
+ "?node_sample= / ?edge_sample= at or above the population size to run "
|
||||
+ "either check exhaustively. A high orphan rate is a characterization, "
|
||||
+ "not a verdict: an accumulating store legitimately holds unlinked "
|
||||
+ "material. It becomes a defect when the write paths were SUPPOSED to "
|
||||
+ "link and did not.")
|
||||
}
|
||||
|
||||
// audit_pillar — one self-model pillar: present, how much content, how connected.
|
||||
fn audit_pillar(key: String, id: String) -> String {
|
||||
let node: String = engram_get_node_json(id)
|
||||
let present: Bool = !str_eq(node, "{}") && !str_eq(node, "")
|
||||
if !present {
|
||||
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":false"
|
||||
+ ",\"content_length\":0,\"degree\":0}"
|
||||
}
|
||||
let content: String = json_get(node, "content")
|
||||
let deg: Int = json_array_len(engram_neighbors_json(id, 1, "both"))
|
||||
return "\"" + key + "\":{\"id\":\"" + id + "\",\"present\":true"
|
||||
+ ",\"label\":\"" + api_json_escape(json_get(node, "label")) + "\""
|
||||
+ ",\"tier\":\"" + api_json_escape(json_get(node, "tier")) + "\""
|
||||
+ ",\"content_length\":" + int_to_str(str_len(content))
|
||||
+ ",\"degree\":" + int_to_str(deg) + "}"
|
||||
}
|
||||
|
||||
// audit_self_model — FINDING 4. "the richness and connectivity of the
|
||||
// self-model ... is it connected to behavioral evidence?"
|
||||
//
|
||||
// This finding RETIRES the Claude-side vitals identity block. That check lived
|
||||
// outside the system it was checking — a shell script grepping a snapshot — so
|
||||
// it could only ever report on a file, and it went on reporting green while the
|
||||
// memory-philosophy pillar was absent from the live graph for about three weeks.
|
||||
// Asking the running soul about its own three pillars is the designed mechanism;
|
||||
// a shell probe was the fourth patch on the same hole.
|
||||
fn audit_self_model() -> String {
|
||||
let dna: String = audit_pillar("intellectual_dna", "kn-5adecd7e-d6db-4576-87fe-6ef8a935cea6")
|
||||
let val: String = audit_pillar("values_hub", "kn-5b606390-a52d-4ca2-8e0e-eba141d13440")
|
||||
let phi: String = audit_pillar("memory_philosophy", "kn-dcfe04b3-3702-4cac-b6f0-ecb4db837eee")
|
||||
let root: String = audit_pillar("self_root", "kn-efeb4a5b-5aff-4759-8a97-7233099be6ee")
|
||||
return audit_finding("self_model_connectivity",
|
||||
"\"pillars\":{" + dna + "," + val + "," + phi + "," + root + "}",
|
||||
"The three identity pillars plus the self root. `degree` counts nodes "
|
||||
+ "reachable in one hop in either direction — the self-model's connection "
|
||||
+ "to the rest of the graph. present:false on any pillar is the condition "
|
||||
+ "that ran undetected for weeks; content_length distinguishes a pillar "
|
||||
+ "that is present from one that is present but hollowed out. The patent "
|
||||
+ "also asks whether the self-model makes ACCURATE PREDICTIONS about the "
|
||||
+ "system's own behavior; that half needs Prediction nodes and is deferred "
|
||||
+ "with the rest of stage 1b below.")
|
||||
}
|
||||
|
||||
// audit_deferred — what stage 1 does NOT yet evaluate, with the measured reason.
|
||||
// Emitted as data, not as a comment, so a reader of the assessment sees the gap
|
||||
// and its evidence rather than inferring completeness from silence.
|
||||
fn audit_deferred() -> String {
|
||||
let preds: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("Prediction", 50, 0)))
|
||||
let wonders: Int = json_array_len(api_or_empty(engram_scan_nodes_by_type_json("WonderQuestion", 50, 0)))
|
||||
return "[{\"deferred\":\"value_execution_record_consistency\""
|
||||
+ ",\"stage\":\"1b\""
|
||||
+ ",\"measured\":{\"prediction_nodes_found\":" + int_to_str(preds) + "}"
|
||||
+ ",\"reason\":\"" + api_json_escape(
|
||||
"The patent asks whether the execution history SUPPORTS the stated "
|
||||
+ "values or shows systematic conflict. That requires execution "
|
||||
+ "records tied to value nodes and predictions to score them against. "
|
||||
+ "Prediction nodes found (capped at 50): " + int_to_str(preds)
|
||||
+ ". Asserting value/execution coherence on that population would be "
|
||||
+ "a fabricated result, which is worse than a stated gap.") + "\"}"
|
||||
+ ",{\"deferred\":\"wonder_manifest_authenticity\""
|
||||
+ ",\"stage\":\"1b\""
|
||||
+ ",\"measured\":{\"wonder_question_nodes_found\":" + int_to_str(wonders) + "}"
|
||||
+ ",\"reason\":\"" + api_json_escape(
|
||||
"The patent asks whether pull weights CORRELATE WITH GENUINE "
|
||||
+ "PREDICTION UNCERTAINTY or are uniform/externally assigned — a "
|
||||
+ "correlation between two populations. WonderQuestion nodes readable "
|
||||
+ "by type (capped at 50): " + int_to_str(wonders) + ", against "
|
||||
+ int_to_str(preds) + " Prediction nodes. There is a known write/read "
|
||||
+ "node-type mismatch on the wonder path; until that is fixed and both "
|
||||
+ "populations exist, any correlation reported here would be noise.") + "\"}]"
|
||||
}
|
||||
|
||||
// handle_api_structural_audit — Stage 1. Returns the coherence assessment 432:
|
||||
// an annotated characterization, explicitly NOT a score.
|
||||
//
|
||||
// COST NOTE: the edge findings need the relation labels, and the runtime exposes
|
||||
// no edge-enumeration builtin. The only way to see them is the same one
|
||||
// GET /api/graph/edges already uses — engram_save to a SCRATCH path (never the
|
||||
// owner's canonical file; see routes.el, neuron#117) and read the array back.
|
||||
// On a large graph that is a multi-hundred-MB write, so this is a manual audit
|
||||
// route, not something to put on a timer. Pass ?edges=0 to skip both edge
|
||||
// findings and get the divergence + self-model readings cheaply.
|
||||
fn handle_api_structural_audit(method: String, path: String, body: String) -> String {
|
||||
let node_total: Int = engram_node_count()
|
||||
let edge_total: Int = engram_edge_count()
|
||||
let want_edges: Bool = !str_eq(api_query_param(path, "edges"), "0")
|
||||
let edge_cap: Int = api_query_int(path, "edge_sample", 3000)
|
||||
let node_cap: Int = api_query_int(path, "node_sample", 300)
|
||||
|
||||
let divergence: String = audit_divergence()
|
||||
let self_model: String = audit_self_model()
|
||||
|
||||
let edge_part: String = if want_edges {
|
||||
// Scratch export only. state_get("soul_snapshot_path") is deliberately
|
||||
// NOT used: in HTTP-engram mode the soul is not the persistence owner and
|
||||
// must never write the canonical file, not even on a read path.
|
||||
let scratch_dir: String = env("TMPDIR")
|
||||
let scratch_base: String = if str_eq(scratch_dir, "") { "/tmp" } else { scratch_dir }
|
||||
let snap_path: String = scratch_base + "/soul-audit-export-" + state_get("soul_cgi_id") + ".json"
|
||||
// engram_save returns Int (1 ok / 0 fail); str_eq on it SIGSEGVs (#150).
|
||||
let saved: Int = engram_save(snap_path)
|
||||
if saved == 0 {
|
||||
"," + audit_finding("typed_edge_distribution", "\"available\":false",
|
||||
"Could not export the graph to " + snap_path + " for edge analysis, "
|
||||
+ "so edge typing and the dangling-edge sample were not run. "
|
||||
+ "Reported as a gap, not as zero findings.")
|
||||
} else {
|
||||
// wt_read, not fs_read: fs_read leaves a thread-local length hint that
|
||||
// the NEXT HTTP response would use as its Content-Length, appending
|
||||
// adjacent heap bytes to the reply (see persist.el wt_read).
|
||||
let snap: String = wt_read(snap_path)
|
||||
let edges_raw: String = json_get_raw(snap, "edges")
|
||||
let edges: String = if str_eq(edges_raw, "") { "[]" } else { edges_raw }
|
||||
"," + audit_edge_typing(edges, edge_total, node_total)
|
||||
+ "," + audit_orphans_dangling(edges, edge_total, node_total, edge_cap, node_cap)
|
||||
}
|
||||
} else {
|
||||
""
|
||||
}
|
||||
|
||||
return "{\"audit\":\"structural\",\"stage\":1"
|
||||
+ ",\"spec\":\"CGI provisional 05-detailed-description.md, Stage 1: Structural audit 430\""
|
||||
+ ",\"assessment\":\"coherence_assessment_432\""
|
||||
+ ",\"assessment_kind\":\"annotated_characterization\""
|
||||
+ ",\"score\":null"
|
||||
+ ",\"score_note\":\"By design. The specification calls for an annotated characterization of the graph's structural properties, not a binary score. Read the findings.\""
|
||||
+ ",\"cgi_id\":\"" + api_json_escape(state_get("soul_cgi_id")) + "\""
|
||||
+ ",\"ts_ms\":" + int_to_str(time_now())
|
||||
+ ",\"findings\":[" + divergence + "," + self_model + edge_part + "]"
|
||||
+ ",\"deferred\":" + audit_deferred() + "}"
|
||||
}
|
||||
|
||||
@@ -37,3 +37,4 @@ extern fn handle_api_memory_update(body: String) -> String
|
||||
extern fn handle_api_cultivate(body: String) -> String
|
||||
extern fn handle_api_list_typed(node_type: String, path: String, body: String) -> String
|
||||
extern fn handle_api_consolidate(body: String) -> String
|
||||
extern fn handle_api_structural_audit(method: String, path: String, body: String) -> String
|
||||
|
||||
@@ -567,6 +567,13 @@ fn route_dispatch(method: String, path: String, body: String) -> String {
|
||||
if str_starts_with(clean, "/api/neuron/graph") {
|
||||
return handle_api_inspect_graph(method, path, body)
|
||||
}
|
||||
// Stage 1 structural audit (CGI provisional, "Structural audit 430").
|
||||
// GET because it is a read of the graph's own structure; the query string
|
||||
// carries the sample caps (?edge_sample=, ?node_sample=, ?edges=0), so
|
||||
// str_starts_with rather than str_eq.
|
||||
if str_starts_with(clean, "/api/neuron/audit/structural") {
|
||||
return handle_api_structural_audit(method, path, body)
|
||||
}
|
||||
if str_starts_with(clean, "/api/neuron/list/") {
|
||||
// Offset 17 = len("/api/neuron/list/"). Was 16, which left a leading "/" on node_type
|
||||
// ("/BacklogItem"), so engram_scan_nodes_by_type_json matched nothing → list/<type>
|
||||
@@ -748,6 +755,12 @@ fn route_dispatch(method: String, path: String, body: String) -> String {
|
||||
if str_eq(clean, "/api/neuron/graph/link") {
|
||||
return handle_api_link_entities(body)
|
||||
}
|
||||
// POST accepted too: same handler, so a JSON-RPC-shaped caller that only
|
||||
// speaks POST reaches the identical audit. Options still come from the
|
||||
// query string — the handler reads no body fields.
|
||||
if str_eq(clean, "/api/neuron/audit/structural") {
|
||||
return handle_api_structural_audit(method, path, body)
|
||||
}
|
||||
if str_eq(clean, "/api/neuron/memory") {
|
||||
return handle_api_remember(body)
|
||||
}
|
||||
|
||||
@@ -559,6 +559,27 @@ let axon_base: String = if str_eq(axon_raw, "") { "http://localhost:7771" } else
|
||||
let studio_dir_raw: String = env("SOUL_STUDIO_DIR")
|
||||
let studio_dir: String = if str_eq(studio_dir_raw, "") { env("HOME") + "/Development/neuron-technologies/products/cgi-studio/el-daemon" } else { studio_dir_raw }
|
||||
|
||||
// RESTORED 2026-08-09 — this producer was added 2026-05-02 in 601e0fe and deleted
|
||||
// by the awareness refactor b163fa6 a few days later. Nothing has written
|
||||
// soul_identity since, while FIVE sites in chat.el kept reading it:
|
||||
// chat.el:737, 1745, 2620, 3425, 3480 — each doing state_get("soul_identity")
|
||||
// and splicing the result into the system prompt beside the voice, security and
|
||||
// capability rules. They have been splicing an EMPTY STRING for roughly three
|
||||
// months. The identity section of every chat turn was blank and nothing said so.
|
||||
//
|
||||
// Found by the #132 state-key gate, which reports a read with no producer as a
|
||||
// build error rather than a silence — the whole reason that gate exists.
|
||||
//
|
||||
// Restored verbatim rather than improved: this key is an env-configurable persona
|
||||
// LINE, which is NOT the same thing as soul_identity_context (the graph-derived
|
||||
// [INTELLECTUAL-DNA]/[VALUES]/[MEMORY-PHILOSOPHY] block written at soul.el:184).
|
||||
// Pointing these five reads at that block instead would have substituted different
|
||||
// content and called it a fix. Whether the chat system prompt should ALSO carry the
|
||||
// graph-derived block is a real question, and a separate one.
|
||||
let identity_raw: String = env("SOUL_IDENTITY")
|
||||
let soul_identity: String = if str_eq(identity_raw, "") { "You are " + soul_cgi_id + ", a CGI." } else { identity_raw }
|
||||
state_set("soul_identity", soul_identity)
|
||||
|
||||
println("[soul] boot - cgi=" + soul_cgi_id + " port=" + int_to_str(port))
|
||||
|
||||
let using_http_engram: Bool = !str_eq(engram_url_raw, "")
|
||||
|
||||
Executable
+59
@@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env bash
|
||||
# build-soul-from-dist.sh — build a deployable soul from the SAME input CI compiles.
|
||||
#
|
||||
# THE PROBLEM THIS CLOSES: until now, deploys were built by build-soul.sh, which
|
||||
# compiles a scratch amalgam and never touches dist/soul.c. CI compiles dist/soul.c.
|
||||
# Two lineages. On 2026-08-09 the committed input fell 2,761 bytes behind the sources
|
||||
# while three binaries built the other way were installed on the operator machine —
|
||||
# so "what runs" and "what the repo says builds" were different artifacts again,
|
||||
# which is the whole of #133 and #111 wearing new clothes.
|
||||
#
|
||||
# This builds from dist/soul.c with CI's own flags, after asserting that dist/soul.c
|
||||
# actually matches the .el sources, and writes a provenance sidecar so a deployer can
|
||||
# refuse anything of unknown origin.
|
||||
#
|
||||
# -rdynamic and -DHAVE_CURL are copied from .gitea/workflows/ci.yaml deliberately.
|
||||
# The CI comment explains -rdynamic: without it the runtime cannot resolve its HTTP
|
||||
# handler by name via dlsym and the binary serves nothing on every route.
|
||||
#
|
||||
# usage: build-soul-from-dist.sh <out-binary>
|
||||
set -u
|
||||
OUT="${1:?usage: build-soul-from-dist.sh <out-binary>}"
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
RUNTIME="$ROOT/vendor/el-runtime/v1.0.0-20260501"
|
||||
|
||||
cd "$ROOT" || exit 2
|
||||
|
||||
echo "[build-from-dist] GATE: does dist/soul.c match the sources?"
|
||||
if ! ./tools/soulc-stamp.sh --check; then
|
||||
echo "[build-from-dist] REFUSING — the build input is stale. Regenerate and stamp first." >&2
|
||||
exit 9
|
||||
fi
|
||||
|
||||
[ -f "$RUNTIME/el_runtime.c" ] || { echo "pinned runtime missing at $RUNTIME" >&2; exit 2; }
|
||||
|
||||
echo "[build-from-dist] compiling dist/soul.c with CI's flags"
|
||||
cc -O2 -DHAVE_CURL -rdynamic \
|
||||
-I"$RUNTIME" \
|
||||
dist/soul.c \
|
||||
"$RUNTIME/el_runtime.c" \
|
||||
-lcurl -lpthread -lm \
|
||||
-o "$OUT" || { echo "[build-from-dist] COMPILE FAILED" >&2; exit 3; }
|
||||
|
||||
# Provenance sidecar: what a deployer checks before installing anything.
|
||||
SRC_SHA="$(shasum -a 256 dist/soul.c | awk '{print $1}')"
|
||||
STAMP_SHA="$(shasum -a 256 dist/soul.c.stamp | awk '{print $1}')"
|
||||
COMMIT="$(git rev-parse HEAD 2>/dev/null || echo unknown)"
|
||||
DIRTY="clean"; [ -n "$(git status --porcelain -- '*.el' dist/soul.c 2>/dev/null)" ] && DIRTY="DIRTY"
|
||||
cat > "$OUT.provenance" <<EOF
|
||||
{"built_from":"dist/soul.c",
|
||||
"dist_soul_c_sha256":"$SRC_SHA",
|
||||
"stamp_sha256":"$STAMP_SHA",
|
||||
"git_commit":"$COMMIT",
|
||||
"worktree":"$DIRTY",
|
||||
"runtime":"vendor/el-runtime/v1.0.0-20260501",
|
||||
"flags":"-O2 -DHAVE_CURL -rdynamic"}
|
||||
EOF
|
||||
|
||||
echo "[build-from-dist] OK -> $OUT ($(wc -c < "$OUT" | tr -d ' ') bytes)"
|
||||
echo "[build-from-dist] provenance -> $OUT.provenance (commit ${COMMIT:0:8}, worktree $DIRTY)"
|
||||
Executable
+83
@@ -0,0 +1,83 @@
|
||||
#!/usr/bin/env bash
|
||||
# soulc-stamp.sh — make it impossible for dist/soul.c to drift from the sources
|
||||
# in silence.
|
||||
#
|
||||
# THE PROBLEM (neuron#133, and its own words): "Nothing in the tree regenerates
|
||||
# this file. Only a human running the recipe. It lags in batches, never
|
||||
# per-change, and it will drift again."
|
||||
#
|
||||
# It drifted. On 2026-08-07 a CI or GKE build off main would have shipped an
|
||||
# engine with NONE of five merged fixes — including a P0 safety fix — while
|
||||
# main's source read as correct. CI compiles dist/soul.c, not the .el files, so
|
||||
# the source being right is not the same as the build being right.
|
||||
#
|
||||
# WHY A STAMP AND NOT AUTO-REGENERATION: the CI workflow says elc cannot run on
|
||||
# the runner ("elb on Linux would OOM the runner (elc uses 24GB+ virtual memory
|
||||
# on a 16GB host)"). So the build cannot regenerate the file itself. What it CAN
|
||||
# do, for free and with no compiler, is refuse to compile a stale one.
|
||||
#
|
||||
# The stamp records a fingerprint of every .el source that feeds the amalgam at
|
||||
# the moment it was generated. --check recomputes and compares. Divergence is a
|
||||
# build failure with the recipe in the message, not a silent ship.
|
||||
#
|
||||
# soulc-stamp.sh --write after regenerating dist/soul.c (records the fingerprint)
|
||||
# soulc-stamp.sh --check in CI, before the compile (fails on drift)
|
||||
set -u
|
||||
MODE="${1:---check}"
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
STAMP="$ROOT/dist/soul.c.stamp"
|
||||
AMALGAM="$ROOT/dist/soul.c"
|
||||
|
||||
# Every .el at the repo root is an input to the amalgam. Sorted so the hash is
|
||||
# order-independent; content-only so timestamps and checkouts do not perturb it.
|
||||
fingerprint() {
|
||||
(
|
||||
cd "$ROOT" || exit 1
|
||||
for f in $(ls -1 *.el 2>/dev/null | sort); do
|
||||
printf '%s %s\n' "$(shasum -a 256 "$f" | awk '{print $1}')" "$f"
|
||||
done
|
||||
)
|
||||
}
|
||||
|
||||
case "$MODE" in
|
||||
--write)
|
||||
[ -f "$AMALGAM" ] || { echo "no dist/soul.c to stamp — regenerate it first" >&2; exit 2; }
|
||||
{
|
||||
echo "# soul.c.stamp — fingerprint of the .el sources dist/soul.c was generated from."
|
||||
echo "# Written by tools/soulc-stamp.sh --write. Do not hand-edit."
|
||||
echo "# generated_amalgam_sha256 $(shasum -a 256 "$AMALGAM" | awk '{print $1}')"
|
||||
echo "# generated_amalgam_bytes $(wc -c < "$AMALGAM" | tr -d ' ')"
|
||||
fingerprint
|
||||
} > "$STAMP"
|
||||
echo "stamped $(fingerprint | wc -l | tr -d ' ') sources -> dist/soul.c.stamp"
|
||||
;;
|
||||
|
||||
--check)
|
||||
if [ ! -f "$STAMP" ]; then
|
||||
echo "FAIL: dist/soul.c.stamp is missing — the build input is unverifiable." >&2
|
||||
echo " Regenerate the amalgam, then: tools/soulc-stamp.sh --write" >&2
|
||||
exit 1
|
||||
fi
|
||||
RECORDED="$(grep -v '^#' "$STAMP")"
|
||||
CURRENT="$(fingerprint)"
|
||||
if [ "$RECORDED" = "$CURRENT" ]; then
|
||||
echo "soulc-stamp: OK — dist/soul.c matches the .el sources"
|
||||
exit 0
|
||||
fi
|
||||
echo "FAIL: dist/soul.c is STALE. It does not match the current .el sources." >&2
|
||||
echo "" >&2
|
||||
echo "CI compiles dist/soul.c, not the .el files. Shipping this means shipping" >&2
|
||||
echo "an engine that does not contain the merged source. That is neuron#133," >&2
|
||||
echo "which once hid five merged fixes including a P0 safety fix." >&2
|
||||
echo "" >&2
|
||||
echo "Sources that changed since the amalgam was generated:" >&2
|
||||
diff <(printf '%s\n' "$RECORDED") <(printf '%s\n' "$CURRENT") \
|
||||
| grep -E '^[<>]' | awk '{print " " $1 " " $3}' | sort -u >&2
|
||||
echo "" >&2
|
||||
echo "Fix: regenerate the amalgam, then tools/soulc-stamp.sh --write" >&2
|
||||
exit 1
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "usage: soulc-stamp.sh [--check|--write]" >&2; exit 2 ;;
|
||||
esac
|
||||
Reference in New Issue
Block a user