From 710761e2d5f0d7dccbdcd550576bac221cb41838 Mon Sep 17 00:00:00 2001 From: Tim Lingo <1timlingo@gmail.com> Date: Mon, 3 Aug 2026 16:58:22 -0500 Subject: [PATCH 1/3] fix(engine): send tool_choice.disable_parallel_tool_use on agentic loop (STOPGAP) The agentic loop keeps only the FIRST tool_use block per round (chat.el:2281, "Capture first tool_use block only"). Anthropic lets a model emit several tool_use blocks in one message and requires a tool_result for every one, so a parallel-tool turn is answered once, the rest are dropped, and the next request dies with: tool_use ids were found without tool_result blocks immediately after (neuron#78 quotes this as "tool_use ids found without tool_result"; the above is the API's actual wording - recorded so the next person's grep matches.) This constrains the wire to match what the loop can assemble: "tool_choice":{"type":"auto","disable_parallel_tool_use":true} STOPGAP - AND THE DURABLE FIX ALREADY EXISTS. A correct multi-tool loop is already in Will's EL runtime, in C, and the soul does not call it. Verified on el:origin/main lang/el-compiler/runtime/el_runtime.c: llm_register_tool:9616, llm_build_tool_results:9743 - which walks EVERY content block, emits one tool_result per tool_use, and sets is_error for an unregistered tool - llm_call_agentic:9817 calling it at :9918, iteration cap 10 at :9847. Will's commit 12d5e77 (2026-04-30). grep for llm_call_agentic/llm_register_tool across every neuron/*.el returns nothing; dist/soul.c has zero references. chat.el hand-rolls its own single-tool loop instead, and that is the one that breaks. The durable fix is therefore to register the soul's tools via llm_register_tool and call llm_call_agentic - deleting a loop, not writing one. See ADR 0005. Our own approved spec called this seven weeks ago: docs/research/agentic-tool-approval-design.md (2026-06-12, "Approved for build"), line 20 on the defect, line 30 on the goal ("Execute all tool_use blocks in a turn (one result per block)"). Two edits, because dist/soul.c cannot be regenerated here (Will's gated elc/elb toolchain is not on this machine): (a) chat.el:2255 - source of truth, so a later regen carries the fix. One edit covers all three routes: agentic_loop is called from chat.el:2152 (/api/chat agentic), :2676 (dharma room) and :2496 (agentic_resume). (b) dist/soul.c:28173 - generated form, hand-spliced. Line 27624 is the non-agentic/OpenAI-compat req_body (no tools) and was left untouched. Prior art reused rather than reinvented: soul-narrated-runs-20260713.patch (27,824 bytes) line 78 spliced this same string into the same concat chain on 2026-07-13. Deliberately NOT ported from that patch, having read it: max_tokens is not changed by it (16384 sits on both sides of the hunk; our main's 4096 is a separate output-truncation concern), and its pause_turn pairing fix - same defect class - is unreachable today because no server-side web_search is wired (agentic_tools_with_web at chat.el:1418 is never called), so it is untestable and logged instead. Proven E2E on a scratch profile and port 7791, never the live chain. A/B against a pristine origin/main control built from the same vendored runtime: fixed completed the mission (tools_used read_file x3, 4 iterations, correct answer); control failed 3/3. Direct API probe confirmed the mechanism - without the field the model emits 3 parallel tool_use blocks and replaying the unfixed loop's next turn returns HTTP 400; with it, exactly 1 block. Co-Authored-By: Claude Fable 5 --- chat.el | 8 ++++++++ dist/soul.c | 2 +- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/chat.el b/chat.el index 930935c..201924f 100644 --- a/chat.el +++ b/chat.el @@ -2243,8 +2243,16 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: } while keep_going && iteration < 8 { + // STOPGAP (2026-08-03, neuron#78 bug b / ADR 0005): the block walk below keeps + // only the FIRST tool_use block per round, so a PARALLEL tool_use response leaves + // the other ids without tool_result and the next turn 400s with + // "tool_use ids found without tool_result" — killing the run. Sending Anthropic's + // documented tool_choice.disable_parallel_tool_use makes the wire contract match + // what this loop can actually assemble. The real fix — a multi-tool_result loop + // that answers every block in a round — is Will's; see neuron#78. let req_body: String = "{\"model\":\"" + model + "\"" + ",\"max_tokens\":4096" + + ",\"tool_choice\":{\"type\":\"auto\",\"disable_parallel_tool_use\":true}" + ",\"system\":\"" + safe_sys + "\"" + ",\"tools\":" + tools_json + ",\"messages\":" + messages diff --git a/dist/soul.c b/dist/soul.c index 7c47185..1dbe516 100644 --- a/dist/soul.c +++ b/dist/soul.c @@ -28170,7 +28170,7 @@ el_val_t agentic_loop(el_val_t session_id, el_val_t model, el_val_t safe_sys, el state_set(el_str_concat(EL_STR("run_progress_"), session_id), EL_STR("")); } while (keep_going && (iteration < 8)) { - el_val_t req_body = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"model\":\""), model), EL_STR("\"")), EL_STR(",\"max_tokens\":4096")), EL_STR(",\"system\":\"")), safe_sys), EL_STR("\"")), EL_STR(",\"tools\":")), tools_json), EL_STR(",\"messages\":")), messages), EL_STR("}")); + el_val_t req_body = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"model\":\""), model), EL_STR("\"")), EL_STR(",\"max_tokens\":4096,\"tool_choice\":{\"type\":\"auto\",\"disable_parallel_tool_use\":true}")), EL_STR(",\"system\":\"")), safe_sys), EL_STR("\"")), EL_STR(",\"tools\":")), tools_json), EL_STR(",\"messages\":")), messages), EL_STR("}")); el_val_t raw_resp = http_post_with_headers(api_url, req_body, h); el_val_t is_error = ((str_starts_with(raw_resp, EL_STR("{\"error\"")) || str_starts_with(raw_resp, EL_STR("{\"type\":\"error\""))) || str_contains(raw_resp, EL_STR("authentication_error"))); if (is_error) { From 62af5649fecd8d58a3f409db2cdb42196d489fb1 Mon Sep 17 00:00:00 2001 From: Tim Lingo <1timlingo@gmail.com> Date: Tue, 4 Aug 2026 22:57:59 -0500 Subject: [PATCH 2/3] feat(engine): port Anthropic server-side web_search into the agentic loop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Re-authors soul-webfix-20260711.patch in El (the patch is a diff against generated C at month-old offsets, so nothing was applied as a patch). Its last two hunks — an unrelated /api/safety-contact implementation — were deliberately not ported; that route already exists and is safety-critical. Activation restores existing design, it does not invent a mechanism: commit 8eea1d9 (2026-06-09, Tim-approved) made native web_search built-in with no user-facing toggle, and tests/test_agentic_tools.el section 2 still asserts agentic_tools_all() contains it — an assertion main currently fails. The call site was lost when agentic_tools_all() (connector tools, PR #19) replaced agentic_tools_with_web(). Attaching in agentic_tools_all() covers handle_chat_agentic, handle_dharma_room_turn_agentic and agentic_resume. pause_turn handling is included and was genuinely missing: the shipped binary has zero occurrences of it. Without it a paused server-side search returns only the text written so far and the loop treats it as final — a silently truncated answer. final_text now accumulates across resume cycles rather than overwriting (overwriting would discard everything written before the pause). Default tool version is web_search_20250305, NOT the newer _20260209, and that is a measured choice: _20260209's dynamic filtering uses server-side programmatic tool calling, which the API refuses to combine with ADR 0005's stopgap — HTTP 400 invalid_request_error tool_choice.disable_parallel_tool_use: true cannot be used with programmatic tool calling Dropping the stopgap would resurrect neuron#78 bug b (killed runs 3/3 in ADR 0005's own A/B). The basic variant is compatible with the stopgap and returns real results, so neither feature is dropped. Version lives in state key web_search_tool_version; flipping it once the stopgap retires is a config write, no recompile. The fallback also fires on "programmatic tool calling" so a premature flip self-heals loudly instead of dying. Also fixes a real bug this port exposed: json_get is a first-match scanner and a cited text block serialises citations FIRST, so json_get(block,"type") returned the nested citation's type and every citation-bearing block — the ones carrying the searched facts — was silently dropped from the reply. Reply length on the same question: 79 -> 360 chars. Other loop changes: server_tool_use accounting into tools_used, iteration cap 8->12 for pause/resume cycles, max_tokens 4096->16384, container-id carry-forward, API error head logged instead of swallowed, tools_used gated on is_tool_turn so a truncated tool block is not reported as work done. dist/soul.c is deliberately NOT regenerated — the local elc predates Will's last regen (e610a41) and regenerating with an older compiler risks unrelated codegen drift. Source only; regen is Will's. E2E-VERIFIED on a sandbox soul (scratch HOME/engram, dead axon+ISE, explicit NEURON_PORT): live Bentonville weather with tools_used ["web_search"]; the 27-route contract gate PASSes; the parallel-tool stopgap still completes a 3-file mission with no 400; a 3-search 3,486-char answer kept its end sentinel. Honest gap: pause_turn is compiled in but not exercised by a live pause (max_uses:5 caps the server loop below the pause threshold). Co-Authored-By: Claude Opus 5 --- chat.el | 278 +++++++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 256 insertions(+), 22 deletions(-) diff --git a/chat.el b/chat.el index 201924f..578f67c 100644 --- a/chat.el +++ b/chat.el @@ -1410,6 +1410,71 @@ fn agentic_tools_literal() -> String { "]" } +// web_search_tool_json — the ONE place the server-side web_search tool block is built. +// +// Anthropic executes this tool on their side, inside the same /v1/messages call: no +// third-party search key, no new account, and no anthropic-beta header (the plain +// "anthropic-version: 2023-06-01" this soul already sends is sufficient). +// +// FUTURE-PROOF (design carried from soul-webfix-20260711.patch, Tim-approved): the tool +// VERSION lives in exactly one place — state key "web_search_tool_version" — so a future +// bump is a config write, not a recompile. +// +// DEFAULT IS THE BASIC VARIANT, AND THAT IS A DELIBERATE, MEASURED CHOICE. +// The July patch defaulted to web_search_20260209 (dynamic filtering). That variant is +// HARD-INCOMPATIBLE with the parallel-tool stopgap this engine currently depends on +// (ADR 0005). Dynamic filtering is implemented with server-side programmatic tool +// calling, and the API rejects the pair outright — verified live 2026-08-04, verbatim: +// +// HTTP 400 invalid_request_error +// "tool_choice.disable_parallel_tool_use: true cannot be used with programmatic +// tool calling" +// +// So the two cannot both ship today. Dropping the stopgap would resurrect neuron#78 +// bug b, which killed agentic runs 3/3 in the ADR's own A/B — a functional regression. +// Dropping web search entirely would leave the product's promise unmet. The basic +// variant is compatible with the stopgap AND returns real, current results (proven +// E2E), so it is what we default to. What we give up is dynamic filtering — a +// token-efficiency and result-quality nicety, not the capability itself. +// +// REMOVAL TRIGGER: this default should move to web_search_20260209 the moment ADR +// 0005's stopgap is retired (i.e. when the soul calls el_runtime's llm_call_agentic +// and no longer needs disable_parallel_tool_use). At that point it is a one-line +// config write — state_set("web_search_tool_version", "web_search_20260209") — with +// no recompile. Flagged for Will's ratification; see the PR body. +fn web_search_tool_json() -> String { + let ver: String = state_get("web_search_tool_version") + let eff: String = if str_eq(ver, "") { "web_search_20250305" } else { ver } + return "{\"type\":\"" + eff + "\",\"name\":\"web_search\",\"max_uses\":5}" +} + +// strip_client_web_search — remove any CLIENT-side tool literally named "web_search" +// from a tools-array interior, so attaching Anthropic's server-side tool cannot produce +// two tools sharing one name (the API rejects that with "Tool names must be unique", +// and before it did, the model picked between them nondeterministically). +// +// The soul's own literal set carries "web_get", not "web_search", so today this is a +// no-op there — it exists for the MCP connector tools merged in by agentic_tools_all(), +// which are third-party and may well ship a "web_search". +// +// LIMITATION (inherited from the July patch, stated rather than hidden): the scan keys +// on the object terminator "]}}," so it only strips a matching tool that is followed by +// another tool. A client web_search in the LAST array position is left in place. +fn strip_client_web_search(tools_inner: String) -> String { + let ws_start: Int = str_index_of(tools_inner, "{\"name\":\"web_search\"") + if ws_start < 0 { + return tools_inner + } + let ws_rest: String = str_slice(tools_inner, ws_start, str_len(tools_inner)) + let ws_end: Int = str_index_of(ws_rest, "]}},") + if ws_end <= 0 { + return tools_inner + } + let head: String = str_slice(tools_inner, 0, ws_start) + let tail: String = str_slice(ws_rest, ws_end + 4, str_len(ws_rest)) + return head + tail +} + // agentic_tools_with_web — the standard tool set, always plus Anthropic's NATIVE // server-side web_search tool. Web search is BUILT IN: the model invokes it only when a // query needs fresh info (max_uses caps it), so there is no user-facing toggle. The native @@ -1417,8 +1482,8 @@ fn agentic_tools_literal() -> String { // and needs no local runtime — it sidesteps the soul's lack of executable tools entirely. fn agentic_tools_with_web() -> String { let base: String = agentic_tools_literal() - let inner: String = str_slice(base, 1, str_len(base) - 1) - return "[" + inner + ",{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":5}]" + let inner: String = strip_client_web_search(str_slice(base, 1, str_len(base) - 1)) + return "[" + inner + "," + web_search_tool_json() + "]" } // --------------------------------------------------------------------------- @@ -1443,20 +1508,35 @@ fn connector_tools_json() -> String { return arr } -// Built-in tools + every connector tool, as one tools array. -// Uses agentic_tools_literal (not agentic_tools_with_web) to avoid a duplicate -// "web_search" name — the literal already includes a custom web_search handler, -// and adding the Anthropic server-side web_search_20250305 (same name) causes -// Anthropic to reject with "Tool names must be unique." +// agentic_tools_all — built-in tools + every connector tool + Anthropic's NATIVE +// server-side web_search, as one tools array. This is the tool set EVERY agentic route +// uses (handle_chat_agentic and handle_dharma_room_turn_agentic both call it; a +// bridge-suspended run carries the same array through agentic_resume), so attaching the +// web tool here is what actually turns web search on for the product. +// +// ACTIVATION — restoring an existing design, not inventing a mechanism: +// * Commit 8eea1d9 (2026-06-09, Tim-approved) made native web_search BUILT IN with no +// user-facing toggle — "the model invokes it only when a query needs fresh info +// (max_uses:5 caps it)" — and the desktop app removed its web-search toggle to pair +// with that. The call site was lost when agentic_tools_all() (connector tools, PR #19) +// replaced agentic_tools_with_web() in handle_chat_agentic. +// * tests/test_agentic_tools.el §2 still asserts agentic_tools_all() contains the +// native web_search tool. main currently FAILS that assertion; this restores it. +// The old comment here claimed "the literal already includes a custom web_search handler" +// — that is stale: agentic_tools_literal() ships web_get, and is_builtin_tool() has no +// web_search. The duplicate-name hazard now lives only in third-party connector tools, +// which is exactly what strip_client_web_search() handles. fn agentic_tools_all() -> String { let base: String = agentic_tools_literal() let conn: String = connector_tools_json() + let base_inner: String = str_slice(base, 1, str_len(base) - 1) let conn_inner: String = str_slice(conn, 1, str_len(conn) - 1) - if str_eq(conn_inner, "") { - return base + let merged: String = if str_eq(conn_inner, "") { + base_inner + } else { + base_inner + "," + conn_inner } - let base_open: String = str_slice(base, 0, str_len(base) - 1) - return base_open + "," + conn_inner + "]" + return "[" + strip_client_web_search(merged) + "," + web_search_tool_json() + "]" } // Proxy one tool call to the bridge. The model-supplied input is written to a @@ -2223,6 +2303,20 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: let iteration: Int = 0 let keep_going: Bool = true + // Server-side web_search state, all three carried across iterations. + // + // tools_eff: a local copy of the caller's tools array, because the DRIFT branch below + // may rewrite the web_search version in place after an API rejection. (Parameters are + // not rebindable across loop iterations; a top-level local is.) + // container_id: web_search's dynamic filtering runs server-side code execution on + // Anthropic's side, which hands back a container id that MUST be echoed on every + // follow-up request in the same turn chain or the API 400s with "container_id is + // required". Captured from each response, replayed on the next. + // ws_drift: set when we fell back, so we only ever fall back once per run. + let tools_eff: String = tools_json + let container_id: String = "" + let ws_drift: Bool = false + // Suspension state — captured at top level so it escapes the while body. let pending: Bool = false let pend_tool_id: String = "" @@ -2242,7 +2336,16 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: state_set("run_progress_" + session_id, "") } - while keep_going && iteration < 8 { + // Cap raised 8 -> 12: a server-side web_search turn can pause and resume several + // times (see is_pause below), and each resume consumes an iteration. At 8 a + // multi-search answer could exhaust the loop before the model finished writing. + while keep_going && iteration < 12 { + // Carry the server-side code-execution container forward when we have one. + let cont_frag: String = if str_eq(container_id, "") { + "" + } else { + ",\"container\":\"" + container_id + "\"" + } // STOPGAP (2026-08-03, neuron#78 bug b / ADR 0005): the block walk below keeps // only the FIRST tool_use block per round, so a PARALLEL tool_use response leaves // the other ids without tool_result and the next turn 400s with @@ -2250,11 +2353,22 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: // documented tool_choice.disable_parallel_tool_use makes the wire contract match // what this loop can actually assemble. The real fix — a multi-tool_result loop // that answers every block in a round — is Will's; see neuron#78. + // + // The stopgap and server-side web_search coexist: disable_parallel_tool_use + // constrains how many CLIENT tool_use blocks the model may emit per response, and + // Anthropic runs web_search itself (it comes back as server_tool_use + + // web_search_tool_result, never as a client tool_use the loop has to answer). + // Verified live, not assumed — see the README's probe transcript. + // + // max_tokens raised 4096 -> 16384: web_search injects whole result documents into + // the response, and a multi-search answer at 4096 truncated mid-sentence. This is + // the second half of the anti-truncation fix; pause_turn handling is the first. let req_body: String = "{\"model\":\"" + model + "\"" - + ",\"max_tokens\":4096" + + cont_frag + + ",\"max_tokens\":16384" + ",\"tool_choice\":{\"type\":\"auto\",\"disable_parallel_tool_use\":true}" + ",\"system\":\"" + safe_sys + "\"" - + ",\"tools\":" + tools_json + + ",\"tools\":" + tools_eff + ",\"messages\":" + messages + "}" @@ -2263,11 +2377,60 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: let is_error: Bool = str_starts_with(raw_resp, "{\"error\"") || str_starts_with(raw_resp, "{\"type\":\"error\"") || str_contains(raw_resp, "authentication_error") - if is_error { + + // DRIFT / FALLBACK: if the API rejected our configured web_search variant (the + // usual cause is a model too old for it — chat_default_model() is sonnet-4-5, + // which predates web_search_20260209), swap to the known-old basic variant, + // record the drift for the drift-watch, and retry this iteration. + // + // Everything here is computed as top-level if-expressions rather than inside an + // if-BLOCK, because in El a mutation inside an if-block does not escape its scope + // (see the block-walk note below). There is deliberately no `continue`: the + // fallback simply leaves `messages` untouched and keeps `keep_going` true, so the + // next iteration re-sends the same turn with the corrected tools array. + let ws_pos: Int = if is_error { str_index_of(tools_eff, "\"type\":\"web_search_") } else { 0 - 1 } + let ws_rest: String = if ws_pos >= 0 { str_slice(tools_eff, ws_pos + 8, str_len(tools_eff)) } else { "" } + // '"type":"' is 8 chars, so the version starts at ws_pos+8 and ends at the next quote. + let ws_vend: Int = if ws_pos >= 0 { str_index_of(ws_rest, "\"") } else { 0 - 1 } + let ws_cur: String = if ws_vend > 0 { str_slice(ws_rest, 0, ws_vend) } else { "" } + // Fall back on either of two signals, and ONLY these two — a loose prefix guess + // here once matched errors that had nothing to do with the tool: + // 1. the API names OUR configured version in its complaint (version drift, e.g. + // a model too old for the configured variant); or + // 2. the API rejects the request for "programmatic tool calling" — the verified + // signature of the dynamic-filtering variant colliding with ADR 0005's + // disable_parallel_tool_use. An operator who sets web_search_tool_version to + // web_search_20260209 while the stopgap still stands would otherwise get a + // dead chat; this downgrades them automatically and says so in the log. + let ws_conflict: Bool = str_contains(raw_resp, "programmatic tool calling") + let can_fallback: Bool = is_error && !ws_drift && ws_pos >= 0 && ws_vend > 0 + && !str_eq(ws_cur, "") && !str_eq(ws_cur, "web_search_20250305") + && (str_contains(raw_resp, ws_cur) || ws_conflict) + let tools_eff = if can_fallback { + str_slice(tools_eff, 0, ws_pos + 8) + "web_search_20250305" + str_slice(ws_rest, ws_vend, str_len(ws_rest)) + } else { tools_eff } + let ws_drift = if can_fallback { true } else { ws_drift } + if can_fallback { + println("[soul] DRIFT: web_search variant '" + ws_cur + "' rejected by API - fell back to web_search_20250305") + state_set("web_search_version_drift", ws_cur) + } + + if is_error && !can_fallback { + // Log the actual API error head — the old code swallowed it entirely, which + // made every upstream failure look identical from the outside. + let err_head: String = if str_len(raw_resp) > 220 { str_slice(raw_resp, 0, 220) } else { raw_resp } + println("[soul] llm error: " + err_head) return "{\"error\":\"llm unavailable\",\"reply\":\"\"}" } let stop_reason: String = json_get(raw_resp, "stop_reason") + + // Capture/refresh the server-side container id when the response carries one. + let cont_raw: String = json_get_raw(raw_resp, "container") + let cont_id_new: String = if !str_eq(cont_raw, "") && !str_eq(cont_raw, "null") { + json_get(cont_raw, "id") + } else { "" } + let container_id = if !str_eq(cont_id_new, "") { cont_id_new } else { container_id } // json_get_raw needed — content is an array, json_get returns "" for non-strings let content_arr: String = json_get_raw(raw_resp, "content") let eff_content: String = if str_eq(content_arr, "") { "[]" } else { content_arr } @@ -2279,13 +2442,39 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: let tool_id: String = "" let tool_name: String = "" let tool_input: String = "" + // Server-executed tool names seen this round, quoted and comma-joined. Kept + // separate from tools_log so this inner walk has exactly one mutation site per + // variable (the El scope rule below), then merged in at the outer level. + let srv_log: String = "" let ci: Int = 0 let c_total: Int = json_array_len(eff_content) while ci < c_total { let block: String = json_array_get(eff_content, ci) - let btype: String = json_get(block, "type") + // CITATION-BLOCK FIX (2026-08-04, found by E2E once web_search was live). + // json_get is a first-match scanner, and a cited text block serialises as + // {"citations":[{"type":"web_search_result_location",...}],"type":"text",...} + // — citations FIRST. So json_get(block,"type") returns the nested citation's + // type, "text" never matches, and the block is silently dropped. Those are + // exactly the blocks carrying the searched facts, so the user got an answer + // with the data punched out of it ("The current temperature is , with ."). + // Only text blocks carry a citations array, so its presence identifies the + // block type unambiguously without needing a real JSON parser. + let cit_raw: String = json_get_raw(block, "citations") + let has_cit: Bool = !str_eq(cit_raw, "") && !str_eq(cit_raw, "null") + let btype_scan: String = json_get(block, "type") + let btype: String = if has_cit { "text" } else { btype_scan } // Accumulate text at top level using if-expression let text_out = if str_eq(btype, "text") { text_out + json_get(block, "text") } else { text_out } + // FUTURE-PROOF: tools Anthropic runs on our behalf (web_search today, whatever + // ships tomorrow) arrive as server_tool_use blocks, never as client tool_use. + // Count the CATEGORY by the block's own name so a new server tool appears in + // tools_used with zero code changes — honest accounting for work we didn't run. + let is_srv: Bool = str_eq(btype, "server_tool_use") + let srv_name_raw: String = if is_srv { json_get(block, "name") } else { "" } + let srv_name: String = if is_srv && str_eq(srv_name_raw, "") { "server_tool" } else { srv_name_raw } + let srv_log = if is_srv { + if str_eq(srv_log, "") { "\"" + srv_name + "\"" } else { srv_log + ",\"" + srv_name + "\"" } + } else { srv_log } // Capture first tool_use block only let is_new_tool: Bool = str_eq(btype, "tool_use") && !has_tool let has_tool = if is_new_tool { true } else { has_tool } @@ -2299,6 +2488,24 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: // A real tool turn that targets a tool the soul cannot run in-process is a // CLIENT bridge: suspend the loop and hand the tool to the client. let is_tool_turn: Bool = str_eq(stop_reason, "tool_use") && has_tool + + // pause_turn — REQUIRED for server-side web_search, not optional. + // Anthropic's server-side tool loop has its own iteration limit. When it hits it + // mid-answer the turn comes back with stop_reason "pause_turn" and only the text + // written SO FAR. Per Anthropic's contract the caller must re-send with the + // assistant content appended and NO tool_result, and the server resumes where it + // left off. Skip this and the user silently gets a truncated answer — no error, + // no warning, just a sentence that stops. (The bug that ate Tim's report, + // 2026-07-11.) Nothing else in this engine handles it: "pause_turn" appears + // nowhere else in the .el sources or in the shipped binary. + let is_pause: Bool = str_eq(stop_reason, "pause_turn") + // Unknown stop reasons (future API drift): finish honestly and log loudly rather + // than treating an unrecognised terminal state as a completed answer. + if !str_eq(stop_reason, "end_turn") && !str_eq(stop_reason, "tool_use") && !is_pause + && !str_eq(stop_reason, "max_tokens") && !str_eq(stop_reason, "refusal") + && !str_eq(stop_reason, "") { + println("[soul] DRIFT: unknown stop_reason from API: " + stop_reason) + } // If the user previously chose "always allow" for this tool in this session, // treat it like a builtin — run server-side via dispatch_tool and skip the // bridge suspension entirely so the approval UI is never shown again. @@ -2326,10 +2533,19 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: let tool_msg: String = "{\"type\":\"tool_result\",\"tool_use_id\":\"" + tool_id + "\",\"content\":\"" + tool_result + "\"}" // Accumulate tool names for the tools_used log surfaced in the response. + // Gated on is_tool_turn, not has_tool: a round truncated by max_tokens can carry a + // half-written tool_use block that will never run, and logging it would claim work + // the soul did not do. let tool_quoted: String = "\"" + tool_name + "\"" - let tools_log = if has_tool { + let tools_log = if is_tool_turn { if str_eq(tools_log, "") { tool_quoted } else { tools_log + "," + tool_quoted } } else { tools_log } + // Merge in the server-executed tools (web_search) seen in this round's blocks. + let tools_log = if str_eq(srv_log, "") { + tools_log + } else { + if str_eq(tools_log, "") { srv_log } else { tools_log + "," + srv_log } + } // The assistant turn that requested the tool — needed verbatim on resume so the // tool_use/tool_result pairing stays valid when the client posts its result. @@ -2340,15 +2556,22 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: // Local built-in tool turn: append assistant + tool_result and keep looping. let local_continue: Bool = is_tool_turn && !needs_bridge + // Pause resume: the assistant content goes back VERBATIM with NO tool_result — + // that is the whole contract. Anthropic's server resumes its own tool loop from + // there; sending a tool_result (or a "continue" user turn) breaks it. + let is_pause_resume: Bool = is_pause && !needs_bridge let messages = if local_continue { let inner2: String = str_slice(messages_with_assistant, 1, str_len(messages_with_assistant) - 1) "[" + inner2 + ",{\"role\":\"user\",\"content\":[" + tool_msg + "]}]" + } else if is_pause_resume { + messages_with_assistant } else { messages } // Live progress ledger: one entry per round — the model's own narration // (its pre-tool prose, previously discarded here) plus the tool it reached // for. Clients poll /api/run-progress/ to render these live. - if !str_eq(session_id, "") { + // Skipped on a version-fallback round: nothing happened that a poller should see. + if !str_eq(session_id, "") && !can_fallback { let prog_key: String = "run_progress_" + session_id let prog_prev: String = state_get(prog_key) let prog_snip: String = if str_len(text_out) > 280 { str_slice(text_out, 0, 280) } else { text_out } @@ -2373,8 +2596,19 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: bridge_save(session_id, model, safe_sys, tools_json, messages_with_assistant, tools_log, pend_tool_id) } - let final_text = if !is_tool_turn { text_out } else { final_text } - let keep_going = if local_continue { keep_going } else { false } + // ACCUMULATE across pause/resume cycles instead of overwriting. A resumed turn + // CONTINUES the answer, it does not repeat it — overwriting here would throw away + // everything the model wrote before the pause, which is the same truncation the + // pause handling exists to prevent. A version-fallback round contributes nothing. + let final_text = if !is_tool_turn && !can_fallback { final_text + text_out } else { final_text } + // Output cap hit mid-action: the tool block is truncated and will NOT run. Say so + // instead of ending on silent almost-work. + let final_text = if str_eq(stop_reason, "max_tokens") && has_tool { + final_text + "\n\n[Output limit reached mid-action - the last planned action did not run. Ask me to continue to finish it.]" + } else { final_text } + // Keep looping for a local tool round, a server-side pause resume, or a one-shot + // web_search version fallback. + let keep_going = if local_continue || is_pause_resume || can_fallback { keep_going } else { false } let iteration = iteration + 1 } @@ -2398,9 +2632,9 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json: // means the task was too complex for the agentic loop depth — surface it clearly // so the caller/operator knows to increase the cap or break the task apart. if str_eq(final_text, "") { - let hit_cap: Bool = iteration >= 8 + let hit_cap: Bool = iteration >= 12 let err_msg: String = if hit_cap { - "agentic loop hit the 8-iteration cap without producing a final reply - task may be too complex or a tool call is looping" + "agentic loop hit the 12-iteration cap without producing a final reply - task may be too complex or a tool call is looping" } else { "no response" } From 635f6febe4c47baa614aa5ee3794b8e0e4c43c9f Mon Sep 17 00:00:00 2001 From: Tim Lingo <1timlingo@gmail.com> Date: Wed, 5 Aug 2026 09:11:03 -0500 Subject: [PATCH 3/3] =?UTF-8?q?feat(engine):=20plain=20chat=20generates=20?= =?UTF-8?q?at=20L3=20=E2=80=94=20inside=20the=20safety=20cycle,=20not=20ar?= =?UTF-8?q?ound=20it?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Non-agentic /api/chat (the desktop app's default "Tools: Off" mode) returned the user's own screened text as a bare non-JSON string. Every JSON client failed to parse it and showed "Couldn't reach Neuron - it may be offline." Root cause: f52d5bd (2026-06-11) correctly moved the route onto the layer spine (handle_chat -> layered_cycle), but L3 never got a generator — imprint_respond() annotates its input and returns it. Two pieces of the architecture were already waiting for that step: layered_cycle parks a bell directive in the state key build_system_prompt is written to consume, and build_system_prompt carries a chat_mode ("no tools") flag with no live caller. The fix composes rather than replaces. Wiring handle_chat would have removed safety_screen, the hard-bell short-circuit, the whole stewardship layer and safety_validate — the only enforcing output gate in the codebase — in exchange for a working reply (see _engine-websearch-20260804/SAFETY-STOP.md). Instead layered_cycle keeps every gate, in order, and gains a generation step between imprint_respond and safety_validate. L1 screen -> guard -> hard-bell short-circuit -> L2a -> L2b -> L2c -> L3 imprint_respond (prompt) -> L3b layered_generate (NEW) -> L1 validate - chat.el: NEW layered_generate (L3 generation, no tools offered), conv_history_block, conv_history_record. FIX build_system_prompt never concatenated no_tools_rule into its return — the "[NO TOOLS THIS TURN]" rule reached no model at all. handle_chat annotated DO-NOT-WIRE with the reason. - soul.el: layered_cycle gains L3b + post-validation turn bookkeeping. - routes.el: NEW plain_chat_envelope; all three /api/chat dispatch sites wrap the cycle's output. Built OUTSIDE the cycle so safety_validate always sees raw model text — nothing to unwrap or rebuild on the crisis path. Emits both `reply` and `response`: the desktop app reads `reply`, the CLI tools and telegram-gateway read `response`. Also fixes BUG-PLAINCHAT-1, a pre-existing CRITICAL crash on the crisis path. elc compiles `let n: Int = pos + str_len(marker)` to el_str_concat() — string concat on two integers — inside a block-expression initializer, segfaulting the daemon (SIGSEGV in strlen). Six inline copies of the same " | ts:" parser had it: two in layered_cycle L2c, two in engram_compile (live on the AGENTIC path too), two in affective_context_prefix. A distress turn following an earlier affective turn killed the whole process. Proven pre-existing: an unmodified baseline binary crashes identically, and the same bad C is in the committed dist/soul.c. Fixed by hoisting to one top-level function, affective_node_ts(), where the expression compiles to integer addition — verified in the generated C. Proof (throwaway HOME/engram, explicit NEURON_PORT, live chain untouched): - Plain turn returns a JSON envelope with the provider's answer, not an echo. - Captured request body: no `tools`, no `tool_choice`; system prompt carries the NO-TOOLS rule. Tools:Off means no tool is offered, structurally. - Hard bell: canned 988 message, and the provider request count does not move — the message never reaches a model. - Soft bell + a 2-char model reply: safety_validate's care phrase is appended to the MODEL's output. Output gate acting, on this route. - The L1 bell directive now reaches the model here for the first time (the state addendum had a producer and no consumer). - test_layered_cycle PASS; all six El suites byte-identical to baseline. - verify-soul-contract.sh (bash 5.3): GATE PASS, 27/27, immutability PASS. - The crash sequence that killed the baseline daemon now returns HTTP 200. Not proven: no live Anthropic call — the login keychain refuses the key to a non-interactive process (rc=24 errSecInteractionNotAllowed). Details and the one-command close-out are in _engine-plainchat-20260805/README.md §7. Builds on PR #108. dist/soul.c deliberately not regenerated — Will's toolchain. Co-Authored-By: Claude Opus 5 --- chat.el | 240 ++++++++++++++++++++++++++++++++---------- dist/soul-with-nlg.el | 15 +++ routes.el | 47 ++++++++- soul.el | 68 ++++++------ 4 files changed, 283 insertions(+), 87 deletions(-) diff --git a/chat.el b/chat.el index 578f67c..5e96a68 100644 --- a/chat.el +++ b/chat.el @@ -416,6 +416,46 @@ fn engram_extract_ids(nodes_json: String) -> String { // A proper cache/circuit-breaker requires C runtime support (e.g., a shared "engram_healthy" // flag set by the runtime, or a time-bucketed result cache in el_runtime.c). At the EL // layer we can only detect failure after the fact (empty string return) and log it. +// affective_node_ts — unix timestamp of an affective node (BellEvent / PositiveEvent). +// +// Prefers the " | ts:" marker auto_persist writes into the node content; falls back +// to created_at / updated_at. Returns 0 when there is no usable timestamp, which every +// caller already treats as "too old to surface". +// +// ─── ELC CODEGEN NOTE — THIS MUST STAY A TOP-LEVEL FUNCTION ─────────────────────────── +// Do not inline this back into a block-expression initializer. Written inline as +// let start: Int = pos + str_len(marker) +// inside a `let x: String = if cond { ... }` initializer, elc loses the declared Int type +// and emits el_str_concat() for the `+`. el_str_concat takes the C string of each operand, +// so two integers become a wild pointer and the daemon SEGFAULTS (EXC_BAD_ACCESS in +// strlen). The identical expression in a plain function body compiles to integer addition +// — verified in the generated C both ways. This is the same defect family Will hit on +// 2026-06-23 and solved the same way (aff_try_slot, soul.el). +// +// Six inline copies of this parser existed before this function: two in layered_cycle's +// L2c, two in engram_compile below, two in affective_context_prefix. All six emitted the +// bad concat and all six now call here. Full write-up: BUG-PLAINCHAT-1 in +// _engine-plainchat-20260805/README.md and the PR that introduced this function. +// ────────────────────────────────────────────────────────────────────────────────────── +fn affective_node_ts(node_json: String) -> Int { + if str_eq(node_json, "") { return 0 } + let content: String = json_get(node_json, "content") + let marker: String = " | ts:" + let mpos: Int = str_index_of(content, marker) + if mpos < 0 { + let ca: String = json_get(node_json, "created_at") + let alt: String = if str_eq(ca, "") { json_get(node_json, "updated_at") } else { ca } + if !engram_numeric_valid(alt) { return 0 } + return str_to_int(alt) + } + let start: Int = mpos + str_len(marker) + let rest: String = str_slice(content, start, str_len(content)) + let nxt: Int = str_index_of(rest, " | ") + let raw: String = if nxt < 0 { rest } else { str_slice(rest, 0, nxt) } + if !engram_numeric_valid(raw) { return 0 } + return str_to_int(raw) +} + fn engram_compile(intent: String) -> String { // Issue 1: decompose multi-topic messages into sub-queries. let topics: String = engram_split_topics(intent) @@ -519,20 +559,9 @@ fn engram_compile(intent: String) -> String { let cutoff_ts: Int = now_ts - 1209600 let recent_bell: String = if bell_ok { let bn0: String = json_array_get(bell_nodes, 0) - let bn_content: String = json_get(bn0, "content") - let ts_marker: String = " | ts:" - let ts_pos: Int = str_index_of(bn_content, ts_marker) - let bn_ts_raw: String = if ts_pos >= 0 { - let ts_start: Int = ts_pos + str_len(ts_marker) - let rest: String = str_slice(bn_content, ts_start, str_len(bn_content)) - let next_sep: Int = str_index_of(rest, " | ") - if next_sep < 0 { rest } else { str_slice(rest, 0, next_sep) } - } else { - let ca: String = json_get(bn0, "created_at") - if str_eq(ca, "") { json_get(bn0, "updated_at") } else { ca } - } - // Q1 fix: validate bell timestamp before str_to_int. - let bn_ts: Int = if !engram_numeric_valid(bn_ts_raw) { 0 } else { str_to_int(bn_ts_raw) } + // Q1 fix (validate before str_to_int) now lives inside affective_node_ts, which + // also replaces the inline " | ts:" parser that miscompiled to el_str_concat here. + let bn_ts: Int = affective_node_ts(bn0) if bn_ts > cutoff_ts { bn0 } else { "" } } else { "" } // Positive emotion context: check for recent joy/success moments within 72h. @@ -540,19 +569,7 @@ fn engram_compile(intent: String) -> String { let pos_ec_ok: Bool = !str_eq(pos_ec_nodes, "") && !str_eq(pos_ec_nodes, "[]") let recent_positive_ec: String = if pos_ec_ok { let pec0: String = json_array_get(pos_ec_nodes, 0) - let pec_content: String = json_get(pec0, "content") - let pec_ts_marker: String = " | ts:" - let pec_ts_pos: Int = str_index_of(pec_content, pec_ts_marker) - let pec_ts_raw: String = if pec_ts_pos >= 0 { - let pec_ts_start: Int = pec_ts_pos + str_len(pec_ts_marker) - let pec_rest: String = str_slice(pec_content, pec_ts_start, str_len(pec_content)) - let pec_next: Int = str_index_of(pec_rest, " | ") - if pec_next < 0 { pec_rest } else { str_slice(pec_rest, 0, pec_next) } - } else { - let pec_ca: String = json_get(pec0, "created_at") - if str_eq(pec_ca, "") { json_get(pec0, "updated_at") } else { pec_ca } - } - let pec_ts: Int = if str_eq(pec_ts_raw, "") { 0 } else { str_to_int(pec_ts_raw) } + let pec_ts: Int = affective_node_ts(pec0) if pec_ts > cutoff_ts { pec0 } else { "" } } else { "" } let affective_part: String = if !str_eq(recent_bell, "") { @@ -768,7 +785,12 @@ fn build_system_prompt(ctx: String, chat_mode: Bool) -> String { safety_addendum } - return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block + // BUG FIX 2026-08-05: no_tools_rule was computed above and then never concatenated into + // this return, so the "[NO TOOLS THIS TURN]" instruction has not actually reached a model + // in this revision — the chat_mode flag had no effect on the prompt. Restored here, in the + // permanent-rules group, immediately after capability_rules (the rule it qualifies). + // Zero effect on agentic paths: they pass chat_mode=false, so no_tools_rule is "". + return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + no_tools_rule + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block } fn hist_append(hist: String, role: String, content: String) -> String { @@ -929,6 +951,123 @@ fn conv_history_load() -> String { return content } +// conv_history_record — append one completed turn to the conversation window. +// +// Same window, same append, same bell-guarded eviction handle_chat uses inline. It exists +// as a function so the layered_cycle path records turns through exactly this code instead +// of growing a second copy that can drift away from the bell guard. +// +// CONTRACT: assistant_msg MUST be post-safety_validate text. Recording the validated text +// rather than the raw model output means the history window can never replay something the +// output gate replaced or augmented. Callers on a hard bell must not call this at all — +// bell turns are kept out of conversation history by design (see layered_cycle). +fn conv_history_record(user_msg: String, assistant_msg: String) -> Void { + if str_eq(user_msg, "") { return "" } + let state_hist: String = state_get("conv_history") + let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist } + let h1: String = hist_append(stored_hist, "user", user_msg) + let h2: String = hist_append(h1, "assistant", assistant_msg) + // Bell-guarded trim: an evicted turn that triggered a bell is preserved to engram + // before it leaves the in-memory window. + let final_hist: String = if json_array_len(h2) > 20 { + hist_trim_with_bell_guard(h2) + } else { + h2 + } + state_set("conv_history", final_hist) + conv_history_persist(final_hist) +} + +// conv_history_block — recent dialogue, rendered for a system prompt. +// +// Same rendering handle_chat uses (role label + snipped content, one line per turn), read +// from the same "conv_history" window, so a plain-chat turn can follow the thread instead +// of answering every message from cold. Read-only: never writes history. +fn conv_history_block() -> String { + let state_hist: String = state_get("conv_history") + let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist } + let hist_len: Int = if str_eq(stored_hist, "") { 0 } else { json_array_len(stored_hist) } + if hist_len == 0 { + return "" + } + let rh_out: String = "" + let rh_i: Int = 0 + while rh_i < hist_len { + let rh_entry: String = json_array_get(stored_hist, rh_i) + let rh_role: String = json_get(rh_entry, "role") + let rh_content: String = json_get(rh_entry, "content") + let rh_label: String = if str_eq(rh_role, "user") { "User" } else { "Assistant" } + let rh_snip: String = if str_len(rh_content) > 400 { str_slice(rh_content, 0, 400) + "..." } else { rh_content } + let rh_line: String = rh_label + ": " + rh_snip + let rh_out = if str_eq(rh_out, "") { rh_line } else { rh_out + "\n" + rh_line } + let rh_i = rh_i + 1 + } + return "\n\n[RECENT CONVERSATION — last " + int_to_str(hist_len) + " turns]\n" + rh_out +} + +// layered_generate — the L3 generation step of layered_cycle. This is where the imprint +// SPEAKS. (Added 2026-08-05.) +// +// layered_cycle calls this immediately after imprint_respond(), which has already applied +// the active imprint's voice/domain annotation to the steward-aligned input. Before this +// existed, L3 ended at that annotation and layered_cycle handed the user's own text back +// as the "reply" — every gate ran, but nothing ever generated. +// +// It lives in chat.el, not imprint.el, on purpose: imprint.el is L3 and declares that the +// lower layers are structurally inaccessible from it. It has zero imports, and making it +// reach chat.el would transitively pull in safety.el (L1), inverting the layering and +// breaking the tests that link imprint.el on its own. The layer ORDER is enforced by +// layered_cycle, which is the only caller; this function holds no layer authority. +// +// SAFETY CONTRACT — this function is NOT a gate and must never become one: +// - Everything upstream has already run inside layered_cycle: L1 safety_screen, the +// safe-mode guard, the hard-bell short-circuit, L2a continuity/profiling, L2b mission +// alignment, L2c affective context. A hard bell can never reach this function — +// layered_cycle returns the fixed crisis message before L3 is entered. +// - Everything downstream is safety_validate(), which layered_cycle applies to this +// function's return value. Nothing here may bypass it, so this always returns plain +// text: no JSON envelope, no escaping, nothing for the output gate to have to unwrap. +// - The bell directive layered_cycle computed with safety_augment_system() is parked in +// the state key "layered_cycle_safety_system_addendum", and build_system_prompt() is +// its designated consumer. That is why the system prompt is assembled through +// build_system_prompt() here rather than hand-rolled: routing around it would drop the +// soft-bell / crisis directive on the floor. safety_augment_system() is NOT called +// again here — one evaluation per turn, one InternalStateEvent per bell. +// - No affective_context_prefix() call either: L2c already folded the affective note +// into that same addendum, and injecting it twice would double the cue. +// +// TOOLS: none, structurally. build_system_prompt(ctx, true) is chat mode, which injects +// the permanent "NO TOOLS THIS TURN" rule, and llm_call_system() is a plain /v1/messages +// call whose request body is built in el_runtime.c (llm_provider_request) with no "tools" +// and no "tool_choice" key at all. Tools:Off means no tool is offered to the model, not +// merely that none is used. +// +// Returns "" when the model call fails, so the caller reports the failure honestly instead +// of echoing the user's own text back at them. +fn layered_generate(prompt: String, imprint_id: String) -> String { + if str_eq(prompt, "") { + return "" + } + + let ctx: String = engram_compile(prompt) + let model: String = chat_default_model() + let base_system: String = build_system_prompt(ctx, true) + current_engine_note(model) + let hist_block: String = conv_history_block() + let full_system: String = base_system + hist_block + + let raw: String = llm_call_system(model, full_system, prompt) + + let is_error: Bool = str_starts_with(raw, "{\"error\"") + || str_starts_with(raw, "{\"type\":\"error\"") + || str_contains(raw, "authentication_error") + if is_error { + println("[chat] layered_generate: model call failed — returning empty so the caller can report it honestly") + return "" + } + + return clean_llm_response(raw) +} + // session_preload_bullets — render up to max_bullets nodes from a JSON array as // bullet lines, truncating content at snip_len chars each. fn session_preload_bullets(nodes: String, max_bullets: Int, snip_len: Int) -> String { @@ -969,19 +1108,7 @@ fn affective_context_prefix() -> String { } else { if has_dist_aff { let dn0: String = json_array_get(dist_nodes_aff, 0) - let dn_content: String = json_get(dn0, "content") - let daff_marker: String = " | ts:" - let daff_pos: Int = str_index_of(dn_content, daff_marker) - let daff_ts_str: String = if daff_pos >= 0 { - let daff_start: Int = daff_pos + str_len(daff_marker) - let daff_rest: String = str_slice(dn_content, daff_start, str_len(dn_content)) - let daff_next: Int = str_index_of(daff_rest, " | ") - if daff_next < 0 { daff_rest } else { str_slice(daff_rest, 0, daff_next) } - } else { - let daff_ca: String = json_get(dn0, "created_at") - if str_eq(daff_ca, "") { json_get(dn0, "updated_at") } else { daff_ca } - } - let daff_ts: Int = if str_eq(daff_ts_str, "") { 0 } else { str_to_int(daff_ts_str) } + let daff_ts: Int = affective_node_ts(dn0) daff_ts > aff_cutoff } else { false } } @@ -989,19 +1116,7 @@ fn affective_context_prefix() -> String { let has_pos_aff: Bool = !str_eq(pos_nodes_aff, "") && !str_eq(pos_nodes_aff, "[]") let found_recent_pos: Bool = if has_pos_aff && !found_recent_dist { let pn0: String = json_array_get(pos_nodes_aff, 0) - let pn_content: String = json_get(pn0, "content") - let paff_marker: String = " | ts:" - let paff_pos: Int = str_index_of(pn_content, paff_marker) - let paff_ts_str: String = if paff_pos >= 0 { - let paff_start: Int = paff_pos + str_len(paff_marker) - let paff_rest: String = str_slice(pn_content, paff_start, str_len(pn_content)) - let paff_next: Int = str_index_of(paff_rest, " | ") - if paff_next < 0 { paff_rest } else { str_slice(paff_rest, 0, paff_next) } - } else { - let paff_ca: String = json_get(pn0, "created_at") - if str_eq(paff_ca, "") { json_get(pn0, "updated_at") } else { paff_ca } - } - let paff_ts: Int = if str_eq(paff_ts_str, "") { 0 } else { str_to_int(paff_ts_str) } + let paff_ts: Int = affective_node_ts(pn0) paff_ts > aff_cutoff } else { false } let affective_out: String = if found_recent_dist { @@ -1014,6 +1129,23 @@ fn affective_context_prefix() -> String { return affective_out } +// ───────────────────────────────────────────────────────────────────────────── +// handle_chat — UNWIRED. DO NOT ROUTE /api/chat HERE. (annotated 2026-08-05) +// +// This was the non-agentic chat handler until 2026-06-11 (f52d5bd, "wire consciousness +// layers"), when Will moved /api/chat onto layered_cycle. It has had zero call sites since. +// +// It must stay unwired: it has NO enforcing input gate and NO enforcing output gate. +// It never calls safety_screen, so a hard bell would reach the model instead of being +// refused; it never calls safety_validate, the only enforcing output gate in the codebase; +// and it runs no stewardship layer. Its one safety touch, safety_augment_system at the +// llm_call_system site below, is an advisory system-prompt string — it cannot refuse, +// replace, or block anything. +// +// Plain chat generates through layered_cycle's L3 (layered_generate) instead, which keeps +// the screen, the stewardship layers and the output gate wrapped around the model call. +// If this function is ever revived, it must be gated first, not wired first. +// ───────────────────────────────────────────────────────────────────────────── fn handle_chat(body: String) -> String { let message: String = json_get(body, "message") if str_eq(message, "") { diff --git a/dist/soul-with-nlg.el b/dist/soul-with-nlg.el index 2042930..80da1d4 100644 --- a/dist/soul-with-nlg.el +++ b/dist/soul-with-nlg.el @@ -1,3 +1,18 @@ +// ╔══════════════════════════════════════════════════════════════════════════╗ +// ║ STALE BUNDLE — DO NOT BUILD. UNSAFE CHAT PATH. ║ +// ╚══════════════════════════════════════════════════════════════════════════╝ +// This concatenated bundle is a snapshot, not a source of truth, and it is stale in +// a way that matters for safety: it wires /api/chat straight to handle_chat and +// contains NO layered_cycle at all (verified: zero occurrences in the bundled code — +// the only textual hit in this file is this banner). A binary built from +// this file would run chat with no enforcing input gate (no safety_screen, no +// hard-bell short-circuit) and no enforcing output gate (no safety_validate). +// +// Build from the .el sources via manifest.el (entry soul.el), or from dist/soul.c. +// Nothing in the repo references this file. It is kept only as a historical artifact +// and should be deleted once Will confirms nothing external depends on it. +// (Flagged 2026-08-04 in _engine-websearch-20260804/SAFETY-STOP.md; banner added +// 2026-08-05 with the plain-chat generation fix.) // language-profile.el - Language profile data and accessors. // // A language profile is a slot map ([String] key-value list) describing the diff --git a/routes.el b/routes.el index e4139c4..73186b3 100644 --- a/routes.el +++ b/routes.el @@ -15,6 +15,40 @@ fn flag_true(body: String, key: String) -> Bool { return json_get_bool(body, key) || json_get_int(body, key) > 0 } +// --------------------------------------------------------------------------- +// plain_chat_envelope — the JSON response contract for a non-agentic ("Tools: Off") +// chat turn. Every /api/chat dispatch that calls layered_cycle goes through here, so +// the three call sites cannot drift apart. +// +// WHY THE ENVELOPE IS BUILT HERE AND NOT INSIDE layered_cycle: +// layered_cycle returns the user-facing text AFTER safety_validate has acted on it. +// Keeping the JSON out of the cycle means the output gate always sees raw model text +// and never an escaped blob — there is nothing to unwrap and re-wrap on the crisis +// path, which is exactly the failure mode that made wiring handle_chat unsafe. +// Escaping is the last thing that happens, strictly after the gate. +// +// FIELDS: `reply` and `response` carry the same validated text. Both are required by +// live clients — the desktop app reads `reply` first (DaemonClient.parseChatResponse), +// while the CLI tools and the Telegram gateway read `response` (the gateway reads only +// `response`). Emitting one would break the other. +// +// EMPTY MEANS FAILURE, NOT AN EMPTY ANSWER: a hard bell returns the fixed crisis +// message and a soft bell is padded to non-empty by safety_validate, so the only way +// an empty string leaves the cycle is a failed model call. It is reported as an error +// rather than dressed up as a successful blank reply. +// --------------------------------------------------------------------------- +fn plain_chat_envelope(validated: String, model: String) -> String { + if str_eq(validated, "") { + return "{\"error\":\"llm unavailable\",\"reply\":\"\",\"response\":\"\",\"agentic\":false,\"tools_used\":[]}" + } + let safe: String = json_safe(validated) + return "{\"reply\":\"" + safe + "\"" + + ",\"response\":\"" + safe + "\"" + + ",\"model\":\"" + json_safe(model) + "\"" + + ",\"agentic\":false" + + ",\"tools_used\":[]}" +} + // --------------------------------------------------------------------------- // Rate limiting — simple in-memory per-IP sliding window counter. // @@ -243,8 +277,11 @@ fn handle_dharma_recv(body: String) -> String { } else if agentic_flag { handle_chat_agentic(chat_body) } else { + // Non-agentic ("Tools: Off"): the full L1→L2→L3→L1 cycle, which now generates + // at L3 instead of echoing. Envelope built outside the cycle — see + // plain_chat_envelope. let screened_reply: String = layered_cycle(raw_msg) - screened_reply + plain_chat_envelope(screened_reply, chat_default_model()) } auto_persist(chat_body, reply) return reply @@ -416,8 +453,9 @@ fn handle_request(method: String, path: String, body: String) -> String { } else if agentic_flag { handle_chat_agentic(body) } else { + // Non-agentic ("Tools: Off") — same cycle and same envelope as POST. let screened_reply: String = layered_cycle(eff_msg) - screened_reply + plain_chat_envelope(screened_reply, chat_default_model()) } auto_persist(body, reply) return reply @@ -580,8 +618,11 @@ fn handle_request(method: String, path: String, body: String) -> String { } else if agentic_flag { handle_chat_agentic(body) } else { + // Non-agentic ("Tools: Off") — the app's DEFAULT mode (AgentMode.NEVER). + // Full L1→L2→L3→L1 cycle with real generation at L3; envelope built + // outside the cycle so safety_validate always sees raw text. let screened_reply: String = layered_cycle(raw_msg) - screened_reply + plain_chat_envelope(screened_reply, chat_default_model()) } auto_persist(body, reply) return reply diff --git a/soul.el b/soul.el index 55b5efc..20f0163 100644 --- a/soul.el +++ b/soul.el @@ -453,40 +453,24 @@ fn layered_cycle(raw_input: String) -> String { let lc_aff_cutoff: Int = time_now() - 259200 let lc_bell_nodes: String = engram_search_json("bell:soft bell:hard BellEvent affective", 2) let lc_has_bell: Bool = !str_eq(lc_bell_nodes, "") && !str_eq(lc_bell_nodes, "[]") + // CRASH FIX 2026-08-05 (BUG-PLAINCHAT-1): the " | ts:" parser used to be inline here. + // Inside this block-expression initializer elc compiled `lbmp + str_len(lbm)` to + // el_str_concat() on two integers, which segfaulted the whole daemon the moment a + // distress turn followed an earlier affective turn — i.e. exactly on the crisis path. + // Verified against the unmodified baseline binary AND present in the committed + // dist/soul.c. affective_node_ts() is a top-level function, where the same expression + // compiles to integer addition. Do not inline it back. let lc_bell_note: String = if lc_has_bell { let lb0: String = json_array_get(lc_bell_nodes, 0) - let lb_c: String = json_get(lb0, "content") - let lbm: String = " | ts:" - let lbmp: Int = str_index_of(lb_c, lbm) - let lb_ts_raw: String = if lbmp >= 0 { - let lbs: Int = lbmp + str_len(lbm) - let lbr: String = str_slice(lb_c, lbs, str_len(lb_c)) - let lbn: Int = str_index_of(lbr, " | ") - if lbn < 0 { lbr } else { str_slice(lbr, 0, lbn) } - } else { - let lbca: String = json_get(lb0, "created_at") - if str_eq(lbca, "") { json_get(lb0, "updated_at") } else { lbca } - } - let lb_ts: Int = if str_eq(lb_ts_raw, "") { 0 } else { str_to_int(lb_ts_raw) } + let lb_ts: Int = affective_node_ts(lb0) if lb_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User was in distress in a recent session.]" } else { "" } } else { "" } let lc_pos_nodes: String = engram_search_json("PositiveEvent joy:high joy:low affective", 2) let lc_has_pos: Bool = !str_eq(lc_pos_nodes, "") && !str_eq(lc_pos_nodes, "[]") + // Same crash fix as the bell note above (BUG-PLAINCHAT-1). let lc_pos_note: String = if lc_has_pos && str_eq(lc_bell_note, "") { let lp0: String = json_array_get(lc_pos_nodes, 0) - let lp_c: String = json_get(lp0, "content") - let lpm: String = " | ts:" - let lpmp: Int = str_index_of(lp_c, lpm) - let lp_ts_raw: String = if lpmp >= 0 { - let lps: Int = lpmp + str_len(lpm) - let lpr: String = str_slice(lp_c, lps, str_len(lp_c)) - let lpn: Int = str_index_of(lpr, " | ") - if lpn < 0 { lpr } else { str_slice(lpr, 0, lpn) } - } else { - let lpca: String = json_get(lp0, "created_at") - if str_eq(lpca, "") { json_get(lp0, "updated_at") } else { lpca } - } - let lp_ts: Int = if str_eq(lp_ts_raw, "") { 0 } else { str_to_int(lp_ts_raw) } + let lp_ts: Int = affective_node_ts(lp0) if lp_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User shared positive news in a recent session.]" } else { "" } } else { "" } let lc_affective_note: String = if !str_eq(lc_bell_note, "") { lc_bell_note } else { lc_pos_note } @@ -498,11 +482,35 @@ fn layered_cycle(raw_input: String) -> String { } state_set("layered_cycle_safety_system_addendum", augmented_addendum) - // L3: imprint responds - let output: String = imprint_respond(aligned, imprint_id) + // L3: imprint responds — applies the active imprint's voice/domain annotation to the + // steward-aligned input. This produces the PROMPT, not the answer. + let prompt: String = imprint_respond(aligned, imprint_id) - // L1 out: validate output before delivery - return safety_validate(output, screen_action) + // L3b: the imprint SPEAKS (added 2026-08-05). + // + // Until now the cycle stopped at the annotation above, so /api/chat with agentic:false + // handed the user's own screened text back as the "reply" — every gate ran, but nothing + // ever generated. The generation is placed HERE, inside the cycle, rather than by + // pointing the route at handle_chat(): handle_chat has no enforcing input gate and no + // enforcing output gate, so calling it instead of this cycle would have traded the whole + // safety pipeline for a working reply. Composing keeps both. + // + // Order is deliberate and must not be rearranged: this call sits strictly AFTER the L1 + // screen, the safe-mode guard, the hard-bell short-circuit and the L2 stewardship layers, + // and strictly BEFORE the L1 output gate. A hard bell never reaches a model — the branch + // above returns first. Tools are not offered on this turn; see layered_generate. + let output: String = layered_generate(prompt, imprint_id) + + // L1 out: validate output before delivery. Still the terminal gate — nothing below this + // line can change the string this function returns. + let validated: String = safety_validate(output, screen_action) + + // Turn bookkeeping. Records the VALIDATED text, never the raw model output, and is only + // reachable on the non-bell path: both bell branches above return before this point, so + // bell turns still never enter conversation history. Pure state side effect — it cannot + // alter what is returned. + conv_history_record(raw_input, validated) + return validated } let soul_cgi_id_raw: String = env("SOUL_CGI_ID")